* RE: [PATCH] powerpc/fsl: Add support for pci(e) machine check exception on E500MC / E5500
From: Hongtao Jia @ 2014-10-08 3:10 UTC (permalink / raw)
To: Scott Wood, Guenter Roeck
Cc: linux-kernel@vger.kernel.org, Guenter Roeck, Paul Mackerras,
linuxppc-dev@lists.ozlabs.org, Jojy Varghese
In-Reply-To: <1412124210.13320.330.camel@snotra.buserror.net>
DQoNCj4gLS0tLS1PcmlnaW5hbCBNZXNzYWdlLS0tLS0NCj4gRnJvbTogV29vZCBTY290dC1CMDc0
MjENCj4gU2VudDogV2VkbmVzZGF5LCBPY3RvYmVyIDAxLCAyMDE0IDg6NDQgQU0NCj4gVG86IEd1
ZW50ZXIgUm9lY2sNCj4gQ2M6IEpvankgVmFyZ2hlc2U7IEJlbmphbWluIEhlcnJlbnNjaG1pZHQ7
IFBhdWwgTWFja2VycmFzOyBNaWNoYWVsDQo+IEVsbGVybWFuOyBsaW51eHBwYy1kZXZAbGlzdHMu
b3psYWJzLm9yZzsgbGludXgta2VybmVsQHZnZXIua2VybmVsLm9yZzsNCj4gR3VlbnRlciBSb2Vj
azsgSmlhIEhvbmd0YW8tQjM4OTUxDQo+IFN1YmplY3Q6IFJlOiBbUEFUQ0hdIHBvd2VycGMvZnNs
OiBBZGQgc3VwcG9ydCBmb3IgcGNpKGUpIG1hY2hpbmUgY2hlY2sNCj4gZXhjZXB0aW9uIG9uIEU1
MDBNQyAvIEU1NTAwDQo+IA0KPiBPbiBUdWUsIDIwMTQtMDktMzAgYXQgMDg6NTAgLTA3MDAsIEd1
ZW50ZXIgUm9lY2sgd3JvdGU6DQo+ID4gT24gTW9uLCBTZXAgMjksIDIwMTQgYXQgMDY6MzE6MDZQ
TSAtMDUwMCwgU2NvdHQgV29vZCB3cm90ZToNCj4gPiA+IE9uIE1vbiwgMjAxNC0wOS0yOSBhdCAy
MzowMyArMDAwMCwgSm9qeSBWYXJnaGVzZSB3cm90ZToNCj4gPiA+ID4NCj4gPiA+ID4gT24gOS8y
OS8xNCAxMjowNiBQTSwgIkd1ZW50ZXIgUm9lY2siIDxsaW51eEByb2Vjay11cy5uZXQ+IHdyb3Rl
Og0KPiA+ID4gPg0KPiA+ID4gPiA+VGhvc2UgYXJlIGVycm9ycyByZWxhdGVkIHRvIFBDSWUgaG90
cGx1ZywgYW5kIGFyZSBzZWVuIHdpdGgNCj4gPiA+ID4gPnVuZXhwZWN0ZWQgUENJZSBkZXZpY2Ug
cmVtb3ZhbHMgKHRyaWdnZXJlZCwgZm9yIGV4YW1wbGUsIGJ5DQo+ID4gPiA+ID5yZW1vdmluZyBw
b3dlciBmcm9tIGEgUENJZSBhZGFwdGVyKS4NCj4gPiA+ID4gPlRoZSBiZWhhdmlvciB3ZSBzZWUg
b24gRTU1MDAgaXMgcXVpdGUgc2ltaWxhciB0byB0aGUgc2FtZQ0KPiA+ID4gPiA+YmVoYXZpb3Ig
b24NCj4gPiA+ID4gPkU1MDA6DQo+ID4gPiA+ID5JZiB1bmhhbmRsZWQsIHRoZSBDUFUga2VlcHMg
ZXhlY3V0aW5nIHRoZSBzYW1lIGluc3RydWN0aW9uIG92ZXINCj4gPiA+ID4gPmFuZCBvdmVyIGFn
YWluIGlmIHRoZXJlIGlzIGFuIGVycm9yIG9uIGEgUENJZSBhY2Nlc3MgYW5kIHRodXMNCj4gPiA+
ID4gPnN0YWxscy4gSSBkb24ndCBrbm93IGlmIHRoaXMgaXMgY29uc2lkZXJlZCBhbiBlcnJhdHVt
IG9yIGV4cGVjdGVkDQo+ID4gPiA+ID5iZWhhdmlvciwgYnV0IGl0IGlzIG9uZSB3ZSBoYXZlIHRv
IGFkZHJlc3Mgc2luY2Ugd2UgaGF2ZSB0byBiZQ0KPiA+ID4gPiA+YWJsZSB0byBoYW5kbGUgdGhh
dCBjb25kaXRpb24uDQo+ID4gPg0KPiA+ID4gVGhlIHJlYXNvbiBJIGFzayBpcyB0aGF0IHRoZSBo
YW5kbGluZyBmb3IgZTUwMCB3YXMgZGVzY3JpYmVkIGFzIGFuDQo+ID4gPiBlcnJhdHVtIHdvcmth
cm91bmQuICBJZiBpdCBpcyBhbiBlcnJhdHVtIGl0IHdvdWxkIGJlIG5pY2UgdG8ga25vdw0KPiA+
ID4gdGhlIGVycmF0dW0gbnVtYmVyIGFuZCB0aGUgZnVsbCBsaXN0IG9mIGFmZmVjdGVkIGNoaXBz
Lg0KPiA+ID4NCj4gPiBNeSB1bmRlcnN0YW5kaW5nLCB3aGljaCBtYXkgYmUgd3JvbmcsIHdhcyB0
aGF0IHRoaXMgaXMgZXhwZWN0ZWQNCj4gPiBiZWhhdmlvciwgYXQgbGVhc3QgZm9yIEU1NTAwLiBJ
IGFjdHVhbGx5IHRob3VnaHQgSSBoYWQgc2VlbiBpdA0KPiA+IHNvbWV3aGVyZSBpbiB0aGUgc3Bl
Y2lmaWNhdGlvbiAocmVzcG9uc2UgdG8gUENJZSBlcnJvcnMpLCBidXQgSSBkb24ndA0KPiByZWNh
bGwgd2hlcmUgZXhhY3RseS4NCj4gPg0KPiA+IEF0IGxlYXN0IGZvciBteSBwYXJ0IEkgYW0gbm90
IGF3YXJlIG9mIGFuIGVycmF0dW0uDQo+IA0KPiBKaWEgSG9uZ3RhbywgY2FuIHlvdSBjb21tZW50
IGhlcmU/DQoNCkkgZGlkIG5vdCBmaW5kIGFueSByZWxhdGVkIGVycmF0dW0gZWl0aGVyLg0KDQo+
IA0KPiA+ID4gPiA+VWx0aW1hdGVseSwgd2UnbGwgd2FudA0KPiA+ID4gPiA+dG8NCj4gPiA+ID4g
PmltcGxlbWVudCBQQ0llIGVycm9yIGhhbmRsZXJzIGZvciB0aGUgYWZmZWN0ZWQgZHJpdmVycywg
YnV0IHRoYXQNCj4gPiA+ID4gPndpbGwgYmUgYSBuZXh0IHN0ZXAuDQo+ID4gPg0KPiA+ID4gRm9y
IG5vdyBjYW4gd2UgYXQgbGVhc3QgcHJpbnQgYSByYXRlbGltaXRlZCBlcnJvciBtZXNzYWdlPyAg
SSBkb24ndA0KPiA+ID4gbGlrZSB0aGUgaWRlYSBvZiBzaWxlbnRseSBpZ25vcmluZyB0aGVzZSBl
cnJvcnMuICBJIHN1cHBvc2UgaXQncyBhDQo+ID4gPiBzZXBhcmF0ZSBpc3N1ZSBmcm9tIGV4dGVu
ZGluZyB0aGUgd29ya2Fyb3VuZCB0byBjb3ZlciBlNTAwbWMsIHRob3VnaC4NCj4gPiA+DQo+ID4g
SSBkb24ndCByZWFsbHkgbGlrZSB0aGUgaWRlYSBvZiBwcmludGluZyBhbiBlcnJvciBtZXNzYWdl
IHByZXR0eSBtdWNoDQo+ID4gZWFjaCB0aW1lIHdoZW4gYW4gdW5leHBlY3RlZCBob3RwbHVnIGV2
ZW50IG9jY3Vycy4NCj4gDQo+IFVuZXhwZWN0ZWQgZXZlbnRzIHNlZW0gbGlrZSB0aGUgc29ydCBv
ZiB0aGluZyB5b3UnZCB3YW50IHRvIGxvZywgYnV0IG15DQo+IGNvbmNlcm4gaXMgdGhhdCB0aGlz
IG1pZ2h0IG5vdCBiZSB0aGUgb25seSBjYXVzZSBvZiBQQ0kgZXJyb3JzLg0KPiANCj4gLVNjb3R0
DQo+IA0KDQo=
^ permalink raw reply
* RE: [PATCHv4] clk: ppc-corenet: rename to ppc-qoriq and add CLK_OF_DECLARE support
From: Yuantian Tang @ 2014-10-08 3:28 UTC (permalink / raw)
To: Scott Wood
Cc: linuxppc-dev@lists.ozlabs.org, Mike Turquette,
linux-kernel@vger.kernel.org,
linux-arm-kernel@lists.infradead.org, Jingchang Lu
In-Reply-To: <1412035075.13320.302.camel@snotra.buserror.net>
PiAtLS0tLU9yaWdpbmFsIE1lc3NhZ2UtLS0tLQ0KPiBGcm9tOiBXb29kIFNjb3R0LUIwNzQyMQ0K
PiBTZW50OiBUdWVzZGF5LCBTZXB0ZW1iZXIgMzAsIDIwMTQgNzo1OCBBTQ0KPiBUbzogVGFuZyBZ
dWFudGlhbi1CMjk5ODMNCj4gQ2M6IE1pa2UgVHVycXVldHRlOyBsaW51eHBwYy1kZXZAbGlzdHMu
b3psYWJzLm9yZzsgbGludXgta2VybmVsQHZnZXIua2VybmVsLm9yZzsNCj4gbGludXgtYXJtLWtl
cm5lbEBsaXN0cy5pbmZyYWRlYWQub3JnOyBMdSBKaW5nY2hhbmctQjM1MDgzDQo+IFN1YmplY3Q6
IFJlOiBbUEFUQ0h2NF0gY2xrOiBwcGMtY29yZW5ldDogcmVuYW1lIHRvIHBwYy1xb3JpcSBhbmQg
YWRkDQo+IENMS19PRl9ERUNMQVJFIHN1cHBvcnQNCj4gDQo+IE9uIFNhdCwgMjAxNC0wOS0yNyBh
dCAyMToxOCAtMDUwMCwgVGFuZyBZdWFudGlhbi1CMjk5ODMgd3JvdGU6DQo+ID4gPiAtLS0tLU9y
aWdpbmFsIE1lc3NhZ2UtLS0tLQ0KPiA+ID4gRnJvbTogTGludXhwcGMtZGV2DQo+ID4gPiBbbWFp
bHRvOmxpbnV4cHBjLWRldi1ib3VuY2VzK2IyOTk4Mz1mcmVlc2NhbGUuY29tQGxpc3RzLm96bGFi
cy5vcmddDQo+ID4gPiBPbiBCZWhhbGYgT2YgTWlrZSBUdXJxdWV0dGUNCj4gPiA+IFNlbnQ6IFNh
dHVyZGF5LCBTZXB0ZW1iZXIgMjcsIDIwMTQgNzoyOSBBTQ0KPiA+ID4gVG86IFdvb2QgU2NvdHQt
QjA3NDIxDQo+ID4gPiBDYzogbGludXhwcGMtZGV2QGxpc3RzLm96bGFicy5vcmc7IGxpbnV4LWtl
cm5lbEB2Z2VyLmtlcm5lbC5vcmc7DQo+ID4gPiBsaW51eC1hcm0ta2VybmVsQGxpc3RzLmluZnJh
ZGVhZC5vcmc7IEx1IEppbmdjaGFuZy1CMzUwODMNCj4gPiA+IFN1YmplY3Q6IFJlOiBbUEFUQ0h2
NF0gY2xrOiBwcGMtY29yZW5ldDogcmVuYW1lIHRvIHBwYy1xb3JpcSBhbmQgYWRkDQo+ID4gPiBD
TEtfT0ZfREVDTEFSRSBzdXBwb3J0DQo+ID4gPg0KPiA+ID4gUXVvdGluZyBTY290dCBXb29kICgy
MDE0LTA5LTI1IDE1OjU2OjIwKQ0KPiA+ID4gPiBPbiBUaHUsIDIwMTQtMDktMjUgYXQgMTU6NTQg
LTA3MDAsIE1pa2UgVHVycXVldHRlIHdyb3RlOg0KPiA+ID4gPiA+IFF1b3RpbmcgU2NvdHQgV29v
ZCAoMjAxNC0wOS0yNSAxMzowODowMCkNCj4gPiA+ID4gPiA+IFdlbGwsIGxpa2UgSSBzYWlkLCBJ
J2QgcmF0aGVyIHNlZSB0aGUgQ0xLX09GX0RFQ0xBUkUgc3R1ZmYgYmUNCj4gPiA+ID4gPiA+IG1h
ZGUgdG8gd29yayBvbiBQUEMgcmF0aGVyIHRoYW4gaGF2ZSB0aGUgZHJpdmVyIGNhcnJ5IGFyb3Vu
ZA0KPiA+ID4gPiA+ID4gdHdvIGJpbmRpbmcgbWV0aG9kcy4NCj4gPiA+ID4gPg0KPiA+ID4gPiA+
IEkgZ3Vlc3MgdGhhdCBpcyBhbiBleGlzdGluZyBwcm9ibGVtLCBhbmQgbm90IHJlbGF0ZWQgZGly
ZWN0bHkgdG8NCj4gPiA+ID4gPiB0aGlzIHBhdGNoPyBUaGlzIHBhdGNoIGlzIGVzc2VudGlhbGx5
IGp1c3QgcmVuYW1lcyAodGhvdWdoIHRoZQ0KPiA+ID4gPiA+IFYxLjAvVjIuMCBzdHVmZiBzZWVt
cyB3ZWlyZCkuDQo+ID4gPiA+DQo+ID4gPiA+IFRoaXMgcGF0Y2ggaXMgYWRkaW5nIENMS19PRl9E
RUNMQVJFLg0KPiA+ID4NCj4gPiA+IEknbSBmaW5lIHRha2luZyB0aGlzIHBhdGNoIGJ1dCB5b3Vy
IGNvbW1lbnRzIGFyZSBzdGlsbCB1bnJlc29sdmVkLg0KPiA+ID4gV2hhdCBkbyB5b3UgdGhpbmsg
bmVlZHMgdG8gYmUgZG9uZSB0byBmaXggdGhlIHByb2JsZW1zIHRoYXQgeW91IHNlZT8NCj4gPiA+
DQo+ID4gQ0xLX09GX0RFQ0xBUkUgaXMgdG90YWxseSB3b3JrZWQgb24gUFBDLiBJIHdpbGwgZG8g
aXQgaW4gYSBzZXBhcmF0ZSBwYXRjaC4NCj4gPiBSZWdhcmRpbmcgVjEuMCBhbmQgVjIuMCwgaXQg
aXMgbm90IHdpcmVkIGp1c3Qgc2FtZSBmb3Igbm93LiBCdXQgd2UgYXJlIG5vdCBzdXJlDQo+IGlm
IGl0IGlzIHNhbWUgZm9yIHYzLjAgaW4gdGhlIGZ1dHVyZS4NCj4gPg0KPiA+IEJlc2lkZXMgdXBk
YXRpbmcgZHJpdmVycy9jcHVmcmVxL0tjb25maWcucG93ZXJwYywgdGhlcmUgaXMgb25lIG1vcmUg
dGhpbmcgSQ0KPiBhbSBub3QgY29tZm9ydGFibGUgd2l0aDoNCj4gPiBUaGlzIHBhdGNoIHVzZXMg
IiBmaXhlZC1jbG9jayIgYXMgc3lzY2xrJ3MgY29tcGF0aWJsZSBzdHJpbmcsIHdoaWxlIG9uIFBQ
QyB3ZQ0KPiB0cmVhdGVkIGl0IGFzICIgZnNsLHFvcmlxLXN5c2Nsay1bMS0yXS4wIi4NCj4gPiBU
aGF0J3MgaW5jb25zaXN0ZW50IG9uIGJvdGggQVJNIGFuZCBQUEMgcGxhdGZvcm1zLCBuZWl0aGVy
IGRpZCBvbiBiaW5kaW5ncy4NCj4gDQo+IGZzbCxxb3JpcS1zeXNjbGstWFhYIGlzIHRoZSB3YXkg
aXQgaXMgYmVjYXVzZSBvZiBjb21wYXRpYmlsaXR5IHdpdGggdGhlIGZpeHVwcyBpbg0KPiBleGlz
dGluZyBVLUJvb3RzLiAgSXQgc2hvdWxkbid0IGJlIHVzZWQgYXMgYSBtb2RlbC4NCj4gDQo+IFRo
YXQgc2FpZCwgSSBkb24ndCB0aGluayB5b3UgcmVhbGx5IG1lYW4gInRoaXMgcGF0Y2giLCBhcyBp
dCBkb2Vzbid0IGNvbnRhaW4gdGhlDQo+IGRldmljZSB0cmVlIHVwZGF0ZXMsIGFuZCAiZml4ZWQt
Y2xvY2siIGRvZXMgbm90IGFwcGVhci4NCj4gDQoiZml4ZWQtY2xvY2siIHdpbGwgYXBwZWFyIHdo
ZW4gbHMxMDJ4IHBsYXRmb3JtIERUUyBnZXRzIHVwc3RyZWFtZWQgZXZlbnR1YWxseS4NClRoYXQg
d291bGQgYmUgZmluZSBpZiB5b3UgZG9uJ3QgdGhpbmsgImZzbCxxb3JpcS1zeXNjbGsteHh4IiBo
YXZpbmcgZGlmZmVyZW50IG1lYW5pbmcgb24gQVJNIGFuZCBQb3dlclBDIGlzIGEgaXNzdWUuDQoN
ClRoYW5rcywNCll1YW50aWFuDQo+IC1TY290dA0KPiANCg0K
^ permalink raw reply
* Re: [PATCH tty-next 14/22] tty: Remove tty_wait_until_sent_from_close()
From: Peter Hurley @ 2014-10-08 3:56 UTC (permalink / raw)
To: David Laight, Arnd Bergmann, linuxppc-dev@lists.ozlabs.org,
One Thousand Gnomes
Cc: Greg Kroah-Hartman, Karsten Keil, linux-kernel@vger.kernel.org,
linux-serial@vger.kernel.org
In-Reply-To: <063D6719AE5E284EB5DD2968C1650D6D1725DAF6@AcuExch.aculab.com>
On 06/17/2014 07:03 AM, David Laight wrote:
> From: Peter Hurley
> ...
>>> I don't understand the second half of the changelog, it doesn't seem
>>> to fit here: there deadlock that we are trying to avoid here happens
>>> when the *same* tty needs the lock to complete the function that
>>> sends the pending data. I don't think we do still do that any more,
>>> but it doesn't seem related to the tty lock being system-wide or not.
>>
>> The tty lock is not used in the i/o path; it's purpose is to
>> mutually exclude state changes in open(), close() and hangup().
>>
>> The commit that added this [1] comments that _other_ ttys may wait
>> for this tty to complete, and comments in the code note that this
>> function should be removed when the system-wide tty mutex was removed
>> (which happened with the commit noted in the changelog).
I just wanted to revisit this discussion briefly so I can clarify the
situation regarding holding the tty lock while closing, and how that
affects parallel opens.
I've unnested the tty lock from the tty mutex (which I'm still testing)
but will be submitting after the merge window re-opens for 3.19. So this
is more relevant now.
The original patch that led to this thread is here:
https://lkml.org/lkml/2014/6/16/306
> What happens if another process tries to do a non-blocking open
> while you are sleeping in close waiting for output to drain?
>
> Hopefully this returns before that data has drained.
Current mainline blocks on _any_ racing re-open while this lock is
dropped in tty_wait_until_sent_from_close(); blocking while
ASYNC_CLOSING has been in mainline since at least 2.6.29 and that just
merged existing code together. See tty_port_block_til_ready(); note
the test for O_NONBLOCK is after the wait while ASYNC_CLOSING.
IOW, currently a non-blocking open will sleep for the _entire_ duration
of a parallel hardware shutdown, and when it wakes, the error return will
cause a release of its tty, and it will restart with a fresh attempt
to open. Same with a blocking open that is already waiting; when its
woken the hardware shutdown has already completed so ASYNC_INITIALIZED
is cleared, which forces a release and restart too.
The point being that holding the tty lock across the _entire_ close
is equivalent to the current outcome, regardless of O_NONBLOCK.
I'm reluctant to start returning EGAIN for non-blocking tty opens
because no tty driver does that now, and I don't think userspace will
deal well with new return codes from tty opens.
Regards,
Peter Hurley
^ permalink raw reply
* Re: [RFC PATCH v3 1/3] powerpc: Fix warning reported by verify_cpu_node_mapping()
From: Li Zhong @ 2014-10-08 4:51 UTC (permalink / raw)
To: Nishanth Aravamudan; +Cc: linuxppc-dev, Nathan Fontenot, paulus
In-Reply-To: <20141007153346.GE9339@linux.vnet.ibm.com>
On 二, 2014-10-07 at 08:33 -0700, Nishanth Aravamudan wrote:
> On 07.10.2014 [17:28:38 +1100], Michael Ellerman wrote:
> > On Fri, 2014-10-03 at 16:26 -0700, Nishanth Aravamudan wrote:
> > > On 03.10.2014 [10:50:20 +1000], Michael Ellerman wrote:
> > > > On Thu, 2014-10-02 at 14:13 -0700, Nishanth Aravamudan wrote:
> > > > > Ben & Michael,
> > > > >
> > > > > What's the status of these patches?
> > > >
> > > > Been in my next for a week :)
> > > >
> > > > https://git.kernel.org/cgit/linux/kernel/git/mpe/linux.git/log/?h=next
> > >
> > > Ah ok, thanks -- I wasn't following your tree, my fault.
> >
> > Not really your fault, I hadn't announced my trees existence :)
> >
> > > Do we want these to go back to 3.17-stable, as they fix some annoying splats
> > > during boot (non-fatal afaict, though)?
> >
> > Up to you really, I don't know how often/bad they were. I haven't added CC
> > stable tags to the commits, so if you want them in stable you should send them
> > explicitly.
>
> I think they occur every boot, unconditionally, on pseries. Doesn't
> prevent boot, just really noisy. I think it'd be good to get them into
> -stable.
>
> Li Zhong, can you push them once they get sent upstream?
Ok, I will send these patches to stable mailing list after it is merged.
Thanks, Zhong
>
> Thanks,
> Nish
^ permalink raw reply
* Re: [v3,16/16] cxl: Add documentation for userspace APIs
From: Michael Ellerman @ 2014-10-08 5:36 UTC (permalink / raw)
To: Michael Neuling, greg, arnd, benh
Cc: cbe-oss-dev, mikey, linux-kernel, imunsie, linuxppc-dev,
Aneesh Kumar K.V, anton, jk
In-Reply-To: <1412678902-18672-17-git-send-email-mikey@neuling.org>
On Tue, 2014-07-10 at 10:48:22 UTC, Michael Neuling wrote:
> From: Ian Munsie <imunsie@au1.ibm.com>
>
> This documentation gives an overview of the hardware architecture, userspace
> APIs via /dev/cxl/afu0.0 and the syfs files. It also adds a MAINTAINERS file
Elsewhere you talk about /dev/cxl/afuM.N, please be consistent.
> diff --git a/Documentation/ABI/testing/sysfs-class-cxl b/Documentation/ABI/testing/sysfs-class-cxl
> new file mode 100644
> index 0000000..ca429fc
> --- /dev/null
> +++ b/Documentation/ABI/testing/sysfs-class-cxl
> @@ -0,0 +1,142 @@
> +Slave contexts (eg. /sys/class/cxl/afu0.0):
Don't slave contexts end with 's' ?
> +
> +What: /sys/class/cxl/<afu>/irqs_max
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
We just had to fix up a bunch of these for someone who left IBM. Would it be
better if they just pointed to linuxppc-dev ?
> +Description: read only
> + Maximum number of interrupts that can be requested by userspace.
> + The default on probe is the maximum that hardware can support
> + (eg. 2037). Write values will limit userspace applications to
I thought it was read only?
> + that many userspace interrupts. Must be >= irqs_min.
Decimal, hex?
> +What: /sys/class/cxl/<afu>/irqs_min
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + The minimum number of interrupts that userspace must request
> + on a CXL_START_WORK ioctl. Userspace may omit the
> + num_interrupts field in the START_WORK IOCTL to get this
> + minimum automatically.
> +
> +What: /sys/class/cxl/<afu>/mmio_size
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Size of the MMIO space that may be mmaped by userspace.
Decimal, hex? Bytes, KB ?
> +What: /sys/class/cxl/<afu>/models_supported
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + List of the models this AFU supports.
So there can be more than one? How are multiple values separated?
> + Valid entries are: "dedicated_process" and "afu_directed"
> +
> +What: /sys/class/cxl/<afu>/model
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read/write
> + The current model the AFU is using. Will be one of the models
> + given in models_supported. Writing will change the model
> + provided that no user contexts are attached.
> +
> +
> +What: /sys/class/cxl/<afu>/prefault_mode
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read/write
> + Set the mode for prefaulting in segments into the segment table
> + when performing the START_WORK ioctl. Possible values:
> + none: No prefaulting (default)
> + wed: Treat the wed as an effective address and prefault it
> + all: all segments this process currently maps
"this" process is not entirely clear. You mean "the process that calls START_WORK" ?
> +What: /sys/class/cxl/<afu>/reset
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: write only
> + Reset the AFU.
How does this interact with the chardev API? Do my contexts go away or stop
working or anything?
> +What: /sys/class/cxl/<afu>/api_version
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + List the current version of the kernel/user API.
Show not List. Decimal, hex?
> +What: /sys/class/cxl/<afu>/api_version_com
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + List the lowest version the kernel/user API this
> + kernel is compatible with.
Show not List. Decimal, hex?
And needs to be clearer:
"The lowest version of the userspace API that this kernel supports."
> +Master contexts (eg. /sys/class/cxl/afu0.0m)
> +
> +What: /sys/class/cxl/<afu>m/mmio_size
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Size of the MMIO space that may be mmaped by userspace. This
> + includes all slave contexts space also.
Units.
> +What: /sys/class/cxl/<afu>m/pp_mmio_len
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Per Process MMIO space length.
Units.
> +What: /sys/class/cxl/<afu>m/pp_mmio_off
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Per Process MMIO space offset.
Units.
> +Card info (eg. /sys/class/cxl/card0)
> +
> +What: /sys/class/cxl/<card>/caia_version
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Identifies the CAIA Version the card implements.
> +
> +What: /sys/class/cxl/<card>/psl_version
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Identifies the revision level of the PSL.
> +
> +What: /sys/class/cxl/<card>/base_image
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Identifies the revision level of the base image for devices
> + that support load-able PSLs. For FPGAs this field identifies
loadable is one word.
> + the image contained in the on-adapter flash which is loaded
> + during the initial program load
Full stop missing.
> +What: /sys/class/cxl/<card>/image_loaded
> +Date: September 2014
> +Contact: Ian Munsie <imunsie@au1.ibm.com>,
> + Michael Neuling <mikey@neuling.org>
> +Description: read only
> + Will return "user" or "factory" depending on the image loaded
> + onto the card
Full stop missing.
> diff --git a/Documentation/powerpc/00-INDEX b/Documentation/powerpc/00-INDEX
> index a68784d..116d94d 100644
> --- a/Documentation/powerpc/00-INDEX
> +++ b/Documentation/powerpc/00-INDEX
> @@ -28,3 +28,5 @@ ptrace.txt
> - Information on the ptrace interfaces for hardware debug registers.
> transactional_memory.txt
> - Overview of the Power8 transactional memory support.
> +cxl.txt
> + - Overview of the CXL driver.
This file is not entirely in alphabetical order, but don't make it worse :)
> diff --git a/Documentation/powerpc/cxl.txt b/Documentation/powerpc/cxl.txt
> new file mode 100644
> index 0000000..36f7ba4
> --- /dev/null
> +++ b/Documentation/powerpc/cxl.txt
> @@ -0,0 +1,346 @@
> +Coherent Accelerator Interface (CXL)
> +====================================
> +
> +Introduction
> +============
> +
> + The coherent accelerator interface is designed to allow the
> + coherent connection of FPGA based accelerators (and other devices)
Or "connection of accelerators (FPGAs and other devices)" ?
Below you use FPGA a few times, should they also be "accelerator" or something
more generic?
Also you use "64bit" a bunch, it's "64-bit" :)
> + to a POWER system. These devices need to adhere to the Coherent
> + Accelerator Interface Architecture (CAIA).
It would be good to briefly describe what "coherent" means in this context, you
use it a lot.
> + IBM refers to this as the Coherent Accelerator Processor Interface
> + or CAPI. In the kernel it's referred to by the name CXL to avoid
> + confusion with the ISDN CAPI subsystem.
> +
> +Hardware overview
> +=================
> +
> + POWER8 FPGA
> + +----------+ +---------+
> + | | | |
> + | CPU | | AFU |
> + | | | |
> + | | | |
> + | | | |
> + +----------+ +---------+
> + | | | |
> + | CAPP +--------+ PSL |
> + | | PCIe | |
> + +----------+ +---------+
> +
> + The POWER8 chip has a Coherently Attached Processor Proxy (CAPP)
> + unit which is part of the PCIe Host Bridge (PHB). This is managed
> + by Linux by calls into OPAL. Linux doesn't directly program the
> + CAPP.
But that's not what your diagram shows, how about:
POWER8 FPGA
+----------+ +---------+
| | | |
| CPU | | AFU |
| | | |
| | | |
| | | |
+----------+ +---------+
| PHB | | |
| +------+ | PSL |
| | CAPP |<------>| |
+---+------+ PCIE +---------+
> + The FPGA (or coherently attached device) consists of two parts.
> + The POWER Service Layer (PSL) and the Accelerator Function Unit
> + (AFU). AFU is used to implement specific functionality behind
The AFU
> + the PSL. The PSL, among other things, provides memory address
> + translation services to allow each AFU direct access to userspace
> + memory.
> +
> + The AFU is the core part of the accelerator (eg. the compression,
> + crypto etc function). The kernel has no knowledge of the function
> + of the AFU. Only userspace interacts directly with the AFU.
> +
> + The PSL provides the translation and interrupt services that the
> + AFU needs. This is what the kernel interacts with. For example,
> + if the AFU needs to read a particular virtual address, it sends
> + that address to the PSL, the PSL then translates it, fetches the
> + data from memory and returns it to the AFU. If the PSL has a
> + translation miss, it interrupts the kernel and the kernel services
> + the fault. The context to which this fault is serviced is based
> + on who owns that acceleration function.
> +
> +AFU Models
> +==========
> +
> + There are two programming models supported by the AFU. Dedicated
> + and AFU directed. AFU may support one or both models.
> +
> + In dedicated model only one MMU context is supported. In this
"In dedicated model" is a bit strange.
"When using the dedicated model .." ?
Or "dedicated mode" ?
> + model, only one userspace process can use the accelerator at time.
> +
> + In AFU directed model, up to 16K simultaneous contexts can be
> + supported. This means up to 16K simultaneous userspace
> + applications may use the accelerator (although specific AFUs may
> + support less). In this mode, the AFU sends a 16 bit context ID
fewer not less :)
> + with each of its requests. This tells the PSL which context is
> + associated with this operation. If the PSL can't translate a
with each operation, or with an operation
> + request, the ID can also be accessed by the kernel so it can
> + determine the associated userspace context to service this
> + translation with.
You were talking about operations but now it's translations?
Maybe "the ID can also be accessed by the kernel so it can determine the
userspace context associated with an operation".
> +MMIO space
> +==========
> +
> + A portion of the FPGA MMIO space can be directly mapped from the
> + AFU to userspace. Either the whole space can be mapped (master
> + context), or just a per context portion (slave context). The
> + hardware is self describing, hence the kernel can determine the
> + offset and size of the per context portion.
That's the first mention of master and slave and it's not very clear.
> +
> +Interrupts
> +==========
> +
> + AFUs may generate interrupts that are destined for userspace. These
> + are received by the kernel as hardware interrupts and passed onto
> + userspace.
How? (via read on the context ..)
> + Data storage faults and error interrupts are handled by the kernel
> + driver.
> +
> +Work Element Descriptor (WED)
> +=============================
> +
> + The WED is a 64bit parameter passed to the AFU when a context is
> + started. Its format is up to the AFU hence the kernel has no
> + knowledge of what it represents. Typically it will be a virtual
> + address pointer to a work queue where the AFU and userspace can
Effective address no?
> + share control and status information or work queues.
"point to a work queue .. or work queues" doesn't read that well.
Maybe:
"Typically it will be the effective address of a work queue or status block
where the AFU and userspace can share control and status information."
> +User API
> +========
> +
> + For AFUs operating in the AFU directed model, the driver will
> + create two character devices per AFU under /dev/cxl. One for
> + master and one for slave contexts.
> +
> + The master context (eg. /dev/cxl/afu0.0m), has access to all of
> + the MMIO space that an AFU provides. The slave context
> + (eg. /dev/cxl/afu0.0) has access to only the per process MMIO
> + space an AFU provides (AFU directed only).
There's only one slave context ?
> + For AFUs operating in the dedicated process model, the driver will
> + only create a single character device per AFU (e.g.
> + /dev/cxl/afu0.0), which has access to the entire MMIO space that
> + the AFU provides.
> +
> + The following file operations are supported on both slave and
> + master devices:
> +
> + open
> +
> + Opens the device and allocates a file descriptor to be used
> + with the rest of the API.
This would be better done as a subsection I think, eg:
open
----
Opens the device and allocates a file descriptor to be used
with the rest of the API.
...
Otherwise you end up very indented further down.
> +
> + A dedicated model AFU only has one context and hence only
> + allows this device to be opened once.
Where do we enforce that?
> + An AFU directed model AFU can have many contexts and hence
> + this device can be opened by as many contexts as available.
What happens when all the contexts are opened?
> + Note: IRQs also need to be allocated per context, which may
> + also limit the number of contexts that can be allocated,
> + and hence how many times the device may be opened. The
> + POWER8 CAPP supports 2040 IRQs and 3 are used by the
> + kernel, so 2037 are left. If 1 IRQ is needed per
> + context, then only 2037 contexts can be allocated. If 4
> + IRQs are needed per context, then only 2037/4 = 509
> + contexts can be allocated.
> +
> + ioctl
> +
> + CXL_IOCTL_START_WORK:
> + Starts the AFU context and associates it with the process
> + memory. Once this ioctl is successfully executed, all
"the current process"
> + memory mapped into this process is accessible to this AFU
> + context using the same virtual addresses. No additional
> + calls are required to map/unmap memory. The AFU memory
> + context will be updated as userspace allocates and frees
> + memory. This ioctl returns once the AFU context is
> + started.
> +
> + Takes a pointer to a struct cxl_ioctl_start_work
> + struct cxl_ioctl_start_work {
> + __u64 flags;
> + __u64 wed;
> + __u64 amr;
> + __s16 num_interrupts;
> + __s16 reserved1;
> + __s32 reserved2;
> + __u64 reserved3;
> + __u64 reserved4;
> + __u64 reserved5;
> + __u64 reserved6;
> + };
> +
> + flags:
> + Indicates which optional fields (e.g. amr,
> + num_interrupts) in the structure are valid.
Don't mention the optional fields here, you'll just end up with a stale list
when you add more optional fields in future.
> + wed:
> + The Work Element Descriptor (WED) is a 64bit
> + argument defined by the AFU. Typically this is an
> + virtual address pointing to an AFU specific
> + structure describing what work to perform.
> +
> + amr:
> + Authority Mask Register (AMR), same as the powerpc
> + AMR.
Is it optional? How?
> + num_interrupt:
plural.
> + Number of userspace interrupts to request. If not
> + specified the minimum number required will be
> + automatically allocated. The min and max number
> + can be obtained from sysfs.
Not specified how?
> + reserved fields:
> + For ABI padding and future extensions
> +
> + CXL_IOCTL_GET_PROCESS_ELEMENT:
> + Get info on current context id. This info is returned
> + from the kernel as an int.
It's not "info on the current context id", it *is* the id, isn't it ?
Any good reason it's an int and not a __u32 ?
> + Written by the kernel with the context id (AKA process
> + element) it has allocated. Slave contexts may want to
> + communicate this to a master process.
I don't know what the implied noun for "Written" is. You could probably just
drop that sentence, you've said most of it already.
> +
> + mmap
> +
> + An AFU may have a MMIO space to facilitate communication with
an MMIO
> + the AFU and mmap allows access to this. The size and contents
the AFU. If it does, the MMIO space can be accessed via mmap.
> + of this area are specific to the particular AFU. The size can
> + be discovered via sysfs.
What if there's none? Size of zero?
> + In the AFU directed model, master contexts will get all of the
"get" is a bit vague. You mean master contexts will be allowed to map all of
the MMIO space?
> + MMIO space and slave contexts will get only the per process
> + space associated with its context. In the dedicated process
> + model the entire MMIO space is always mapped.
> +
> + This mmap call must be done after the IOCTL is started.
Which ioctl?
> + Care should be taken when accessing MMIO space. Only 32 and
> + 64bit accesses are supported by POWER8. Also, the AFU will be
> + designed with a specific endian, so all MMIO access should
endian*ness* ?
> + consider endian (recommend endian(3) variants like: le64toh(),
> + be64toh() etc). These endian issues equally apply to shared
> + memory queues the WED may describe.
> +
> + read
> +
> + Reads an event from the AFU. Will return -EINVAL if the user
One or more events?
> + supplied buffer to read into is less than 4096 bytes.
You say that later so drop this one.
> + Blocks
> + if no events are pending (unless O_NONBLOCK is supplied). Will
> + return -EIO in the case of an unrecoverable error or if the
> + card is removed.
"Will return" -> "Returns"
> + A read may return multiple events. A read will return the
> + length of the buffer written and it will be a integral number
> + of events up to the buffer size.
That's a bit confusing, you're starting to rewrite the read(2) manpage, but
from the point of view of the kernel.
The actual text is "On success, the number of bytes read is returned".
I think you should just leave that as implied, because it's read(2). What you
do need to say is that the kernel will return an integral number of events.
> + Users must supply a buffer size of at least 4K bytes.
"The buffer passed to read() must be at least 4K bytes".
> +
> + All events will be return a struct cxl_event which varies in
> + size.
"The result of the read will be a buffer of one or more events, each event is
of type struct cxl_event, of varying size."
> + struct cxl_event {
> + struct cxl_event_header header;
> + union {
> + struct cxl_event_afu_interrupt irq;
> + struct cxl_event_data_storage fault;
> + struct cxl_event_afu_error afu_err;
> + };
> + };
> +
> + A struct cxl_event_header at the start gives:
"The struct cxl_event_header is defined as:"
> + struct cxl_event_header {
> + __u16 type;
> + __u16 size;
> + __u16 process_element;
> + __u16 reserved1;
> + };
> +
> + type:
> + This gives the type of event. The type determines how
s/gives/defines/
> + the rest of the event will be structured. These types
s/will be/is/
> + are shown below.
You don't mention cxl_event_type (the enum). You should.
> +
> + size:
> + This is the size of the event in bytes including the
> + header. The start of the next event can be found at
s/header/struct cxl_event_header/
> + this offset from the start of the current event.
> +
> + process_element:
> + Context ID of the event.
> Currently this will always
> + be the current context. Future work may allow
> + interrupts from one context to be routed to another
> + (eg. a master contexts handling error interrupts on
> + behalf of a slave).
Drop all of that.
> + reserved field:
> + For future extensions and padding.
> +
> + If an AFU interrupt event is received, the full structure received is:
"If the event type is CXL_EVENT_AFU_INTERRUPT then the event structure is defined as"
> + struct cxl_event_afu_interrupt {
> + __u16 flags;
> + __u16 irq; /* Raised AFU interrupt number */
> + __u32 reserved1;
> + };
> +
> + flags:
> + These flags indicate which optional fields are present
> + in this struct. Currently all fields are Mandatory.
mandatory
> + irq:
> + The IRQ number sent by the AFU.
> +
> + reserved field:
> + For future extensions and padding.
> +
> + If a data storage event is received, the full structure received is:
As for cxl_event_afu_interrupt.
> + struct cxl_event_data_storage {
> + __u16 flags;
> + __u16 reserved1;
> + __u32 reserved2;
> + __u64 addr;
> + __u64 dsisr;
> + __u64 reserved3;
> + };
> +
> + flags:
> + These flags indicate which optional fields are present
> + in this struct. Currently all fields are Mandatory.
mandatory
> + address: Mandatory
Drop the Mandatory.
> + Address of the data storage trying to be accessed by
"The address that the AFU unsucessfully attempted to access"
> + the AFU. Valid accesses will handled transparently by
will be handled
> + the kernel but invalid access will generate this
invalid accesses
> + event.
> +
> + dsisr: Manditory
Drop Manditory.
> + These fields give information on the type of
This field
> + fault. Copy of the DSISR from PSL hardware when
> + address fault occured.
Defined in CAIA ?
> + reserved fields:
> + For future extensions
> +
> + If an AFU error event is received, the full structure received is:
As above.
> + struct cxl_event_afu_error {
> + __u16 flags;
> + __u16 reserved1;
> + __u32 reserved2;
> + __u64 err;
> + };
> +
> + flags: Mandatory
Drop Mandatory.
> + These flags indicate which optional fields are present
> + in this struct. Currently all fields are Mandatory.
> +
> + err:
Can we just call it error ?
> + Error status from the AFU. AFU defined.
"Defined by the AFU".
> + reserved fields:
> + For future extensions and padding
> +
> +Sysfs Class
> +===========
> +
> + A cxl sysfs class is added under /sys/class/cxl to facilitate
> + enumeration and tuning of the accelerators. Its layout is
> + described in Documentation/ABI/testing/sysfs-class-cxl
> diff --git a/MAINTAINERS b/MAINTAINERS
> index 809ecd6..c972be3 100644
> --- a/MAINTAINERS
> +++ b/MAINTAINERS
> @@ -2711,6 +2711,13 @@ W: http://www.chelsio.com
> S: Supported
> F: drivers/net/ethernet/chelsio/cxgb4vf/
>
> +CXL (IBM Coherent Accelerator Processor Interface CAPI) DRIVER
> +M: Ian Munsie <imunsie@au1.ibm.com>
> +M: Michael Neuling <mikey@neuling.org>
> +L: linuxppc-dev@lists.ozlabs.org
> +S: Supported
> +F: drivers/misc/cxl/
F: include/misc/cxl.h
F: include/uapi/misc/cxl.h
F: Documentation/powerpc/cxl.txt
F: Documentation/powerpc/cxl.txt
F: Documentation/ABI/testing/sysfs-class-cxl
cheers
^ permalink raw reply
* RE: [PATCH] powerpc: mitigate impact of decrementer reset
From: Heinz Wrobel @ 2014-10-08 5:37 UTC (permalink / raw)
To: Paul Clarke, linuxppc-dev@lists.ozlabs.org
In-Reply-To: <54343B54.4060500@us.ibm.com>
UGF1bCwNCg0Kd2hhdCBpZiB5b3VyIHRiIHdyYXBzIGR1cmluZyB0aGUgIHRlc3Q/DQoNCj4gLS0t
LS1PcmlnaW5hbCBNZXNzYWdlLS0tLS0NCj4gRnJvbTogTGludXhwcGMtZGV2IFttYWlsdG86bGlu
dXhwcGMtZGV2LQ0KPiBib3VuY2VzK2hlaW56Lndyb2JlbD1mcmVlc2NhbGUuY29tQGxpc3RzLm96
bGFicy5vcmddIE9uIEJlaGFsZiBPZiBQYXVsDQo+IENsYXJrZQ0KPiBTZW50OiBUdWVzZGF5LCBP
Y3RvYmVyIDA3LCAyMDE0IDIxOjEzDQo+IFRvOiBsaW51eHBwYy1kZXZAbGlzdHMub3psYWJzLm9y
Zw0KPiBTdWJqZWN0OiBbUEFUQ0hdIHBvd2VycGM6IG1pdGlnYXRlIGltcGFjdCBvZiBkZWNyZW1l
bnRlciByZXNldA0KPiANCj4gVGhlIFBPV0VSIElTQSBkZWZpbmVzIGFuIGFsd2F5cy1ydW5uaW5n
IGRlY3JlbWVudGVyIHdoaWNoIGNhbiBiZSB1c2VkIHRvDQo+IHNjaGVkdWxlIGludGVycnVwdHMg
YWZ0ZXIgYSBjZXJ0YWluIHRpbWUgaW50ZXJ2YWwgaGFzIGVsYXBzZWQuDQo+IFRoZSBkZWNyZW1l
bnRlciBjb3VudHMgZG93biBhdCB0aGUgc2FtZSBmcmVxdWVuY3kgYXMgdGhlIFRpbWUgQmFzZSwg
d2hpY2gNCj4gaXMgNTEyIE1Iei4gIFRoZSBtYXhpbXVtIHZhbHVlIG9mIHRoZSBkZWNyZW1lbnRl
ciBpcyAweDdmZmZmZmZmLg0KPiBUaGlzIHdvcmtzIG91dCB0byBhIG1heGltdW0gaW50ZXJ2YWwg
b2YgYWJvdXQgNC4xOSBzZWNvbmRzLg0KPiANCj4gSWYgYSBsYXJnZXIgaW50ZXJ2YWwgaXMgZGVz
aXJlZCwgdGhlIGtlcm5lbCB3aWxsIHNldCB0aGUgZGVjcmVtZW50ZXIgdG8gaXRzDQo+IG1heGlt
dW0gdmFsdWUgYW5kIHJlc2V0IGl0IGFmdGVyIGl0IGV4cGlyZXMgKHVuZGVyZmxvd3MpIGEgc3Vm
ZmljaWVudCBudW1iZXIgb2YNCj4gdGltZXMgdW50aWwgdGhlIGRlc2lyZWQgaW50ZXJ2YWwgaGFz
IGVsYXBzZWQuDQo+IA0KPiBUaGUgbmVnYXRpdmUgZWZmZWN0IG9mIHRoaXMgaXMgdGhhdCBhbiB1
bndhbnRlZCBsYXRlbmN5IHNwaWtlIHdpbGwgaW1wYWN0IG5vcm1hbA0KPiBwcm9jZXNzaW5nIGF0
IG1vc3QgZXZlcnkgNC4xOSBzZWNvbmRzLiAgT24gYW4gSUJNIFBPV0VSOC1iYXNlZCBzeXN0ZW0s
IHRoaXMNCj4gc3Bpa2Ugd2FzIG1lYXN1cmVkIGF0IGFib3V0IDI1LTMwIG1pY3Jvc2Vjb25kcywg
bXVjaCBvZiB3aGljaCB3YXMgYmFzaWMsDQo+IG9wcG9ydHVuaXN0aWMgaG91c2VrZWVwaW5nIHRh
c2tzIHRoYXQgY291bGQgb3RoZXJ3aXNlIGhhdmUgd2FpdGVkLg0KPiANCj4gVGhpcyBwYXRjaCBz
aG9ydC1jaXJjdWl0cyB0aGUgcmVzZXQgb2YgdGhlIGRlY3JlbWVudGVyLCBleGl0aW5nIGFmdGVy
IHRoZQ0KPiBkZWNyZW1lbnRlciByZXNldCwgYnV0IGJlZm9yZSB0aGUgaG91c2VrZWVwaW5nIHRh
c2tzIGlmIHRoZSBvbmx5IG5lZWQgZm9yIHRoZQ0KPiBpbnRlcnJ1cHQgaXMgc2ltcGx5IHRvIHJl
c2V0IGl0LiAgQWZ0ZXIgdGhpcyBwYXRjaCwgdGhlIGxhdGVuY3kgc3Bpa2Ugd2FzIG1lYXN1cmVk
DQo+IGF0IGFib3V0IDE1MCBuYW5vc2Vjb25kcy4NCj4gDQo+IFNpZ25lZC1vZmYtYnk6IFBhdWwg
QS4gQ2xhcmtlIDxwY0B1cy5pYm0uY29tPg0KPiAtLS0NCj4gICBhcmNoL3Bvd2VycGMva2VybmVs
L3RpbWUuYyB8IDEzICsrKysrKysrKysrKysNCj4gICAxIGZpbGUgY2hhbmdlZCwgMTMgaW5zZXJ0
aW9ucygrKQ0KPiANCj4gZGlmZiAtLWdpdCBhL2FyY2gvcG93ZXJwYy9rZXJuZWwvdGltZS5jIGIv
YXJjaC9wb3dlcnBjL2tlcm5lbC90aW1lLmMgaW5kZXgNCj4gMzY4YWIzNy4uOTYyYTA2YiAxMDA2
NDQNCj4gLS0tIGEvYXJjaC9wb3dlcnBjL2tlcm5lbC90aW1lLmMNCj4gKysrIGIvYXJjaC9wb3dl
cnBjL2tlcm5lbC90aW1lLmMNCj4gQEAgLTUyOCw2ICs1MjgsNyBAQCB2b2lkIHRpbWVyX2ludGVy
cnVwdChzdHJ1Y3QgcHRfcmVncyAqIHJlZ3MpDQo+ICAgew0KPiAgIAlzdHJ1Y3QgcHRfcmVncyAq
b2xkX3JlZ3M7DQo+ICAgCXU2NCAqbmV4dF90YiA9ICZfX2dldF9jcHVfdmFyKGRlY3JlbWVudGVy
c19uZXh0X3RiKTsNCj4gKwl1NjQgbm93Ow0KPiANCj4gICAJLyogRW5zdXJlIGEgcG9zaXRpdmUg
dmFsdWUgaXMgd3JpdHRlbiB0byB0aGUgZGVjcmVtZW50ZXIsIG9yIGVsc2UNCj4gICAJICogc29t
ZSBDUFVzIHdpbGwgY29udGludWUgdG8gdGFrZSBkZWNyZW1lbnRlciBleGNlcHRpb25zLg0KPiBA
QCAtNTUwLDYgKzU1MSwxOCBAQCB2b2lkIHRpbWVyX2ludGVycnVwdChzdHJ1Y3QgcHRfcmVncyAq
IHJlZ3MpDQo+ICAgCSAqLw0KPiAgIAltYXlfaGFyZF9pcnFfZW5hYmxlKCk7DQo+IA0KPiArCS8q
IElmIHRoaXMgaXMgc2ltcGx5IHRoZSBkZWNyZW1lbnRlciBleHBpcmluZyAodW5kZXJmbG93KSBk
dWUgdG8NCj4gKwkgKiB0aGUgbGltaXRlZCBzaXplIG9mIHRoZSBkZWNyZW1lbnRlciwgYW5kIG5v
dCBhIHNldCB0aW1lciwNCj4gKwkgKiByZXNldCAoaWYgbmVlZGVkKSBhbmQgcmV0dXJuDQo+ICsJ
ICovDQo+ICsJbm93ID0gZ2V0X3RiX29yX3J0YygpOw0KPiArCWlmIChub3cgPCAqbmV4dF90Yikg
ew0KDQpXaGF0IGlmICJub3ciIGFuZCAqbmV4dF90YiBhcmUgbm90IG9uIHRoZSBzYW1lIHdyYXAg
Y291bnQ/IFRoZXkgYXJlIGJvdGggbW9kdWxvIHZhbHVlcyBBRkFDUy4NClNob3VsZG4ndCB0aGlz
IGJlIHJpZ2h0IGhlcmUgbW9yZSBsaWtlIGEgImlmICgoKm5leHRfdGIgLSBub3cpIDwgMl42Myki
IHN0eWxlIHRlc3QgdG8gY2hlY2sgZm9yIGRlbHRhcyB3aXRoaW4gdGhlIHJhbmdlIGluc3RlYWQg
b2YgYWJzb2x1dGUgdmFsdWVzPw0KDQo+ICsJCW5vdyA9ICpuZXh0X3RiIC0gbm93Ow0KPiArCQlp
ZiAobm93IDw9IERFQ1JFTUVOVEVSX01BWCkNCj4gKwkJCXNldF9kZWMoKGludClub3cpOw0KPiAr
CQlfX2dldF9jcHVfdmFyKGlycV9zdGF0KS50aW1lcl9pcnFzX290aGVycysrOw0KPiArCQlyZXR1
cm47DQo+ICsJfQ0KPiANCj4gICAjaWYgZGVmaW5lZChDT05GSUdfUFBDMzIpICYmIGRlZmluZWQo
Q09ORklHX1BQQ19QTUFDKQ0KPiAgIAlpZiAoYXRvbWljX3JlYWQoJnBwY19uX2xvc3RfaW50ZXJy
dXB0cykgIT0gMCkNCj4gLS0NCj4gMi4xLjIuMzMwLmc1NjUzMDFlDQoNCkJSLA0KDQpIZWlueg0K
^ permalink raw reply
* Re: [PATCH v3 4/7] sound/radeon: Add quirk for broken 64-bit MSI
From: Alex Deucher @ 2014-10-08 6:23 UTC (permalink / raw)
To: Benjamin Herrenschmidt
Cc: linuxppc-dev, Dave Airlie, Linux PCI, Anton Blanchard, Brian King,
Yijing Wang, Takashi Iwai, Bjorn Helgaas
In-Reply-To: <1412746096.30859.229.camel@pasglop>
On Wed, Oct 8, 2014 at 1:28 AM, Benjamin Herrenschmidt
<benh@kernel.crashing.org> wrote:
> On Tue, 2014-10-07 at 19:47 -0400, Alex Deucher wrote:
>> > This moves the setting of the quirk flag to the audio driver.
>> >
>> > While recent ASICs have that problem fixed, they don't seem to
>> > be listed in the PCI IDs of the current driver, so let's quirk all
>> > the ATI HDMI for now. The consequences are nil on x86 anyway.
>> >
>> > Signed-off-by: Alex Deucher <alexdeucher@gmail.com>
>> > Signed-off-by: Benjamin Herrenschmidt <benh@kernel.crashing.org>
>> > CC: <stable@vger.kernel.org>
>>
>> Further discussion with the hw teams have revealed that this is still
>> an issue on newer asics so I think your original patch is correct
>> after all. Just disable 64 bit MSIs on all AMD audio PCI ids.
>
> Allright, I won't resend the whole series, I can just pickup my previous
> patch. Takashi, Bjorn, Dave, this series covers your 3 areas of
> maintainership, how do you want to proceed ? I'm happy to merge the
> whole lot via powerpc ASAP (since it's all CC'ed stable) if you guys
> send me the appropriate acks, otherwise, let me know.
>
I don't remember if I gave my formal review of your original patch, so if not,
Reviewed-by: Alex Deucher <alexander.deucher@amd.com>
Alex
^ permalink raw reply
* Re: [PATCH v3 4/7] sound/radeon: Add quirk for broken 64-bit MSI
From: Benjamin Herrenschmidt @ 2014-10-08 5:28 UTC (permalink / raw)
To: Bjorn Helgaas, Dave Airlie, Takashi Iwai
Cc: linuxppc-dev, Linux PCI, Anton Blanchard, Yijing Wang, Brian King,
Alex Deucher
In-Reply-To: <CADnq5_NFt8neiOAg896d3-DbAn0rnrE7iXaTFmkNGSg-J9oSkA@mail.gmail.com>
On Tue, 2014-10-07 at 19:47 -0400, Alex Deucher wrote:
> > This moves the setting of the quirk flag to the audio driver.
> >
> > While recent ASICs have that problem fixed, they don't seem to
> > be listed in the PCI IDs of the current driver, so let's quirk all
> > the ATI HDMI for now. The consequences are nil on x86 anyway.
> >
> > Signed-off-by: Alex Deucher <alexdeucher@gmail.com>
> > Signed-off-by: Benjamin Herrenschmidt <benh@kernel.crashing.org>
> > CC: <stable@vger.kernel.org>
>
> Further discussion with the hw teams have revealed that this is still
> an issue on newer asics so I think your original patch is correct
> after all. Just disable 64 bit MSIs on all AMD audio PCI ids.
Allright, I won't resend the whole series, I can just pickup my previous
patch. Takashi, Bjorn, Dave, this series covers your 3 areas of
maintainership, how do you want to proceed ? I'm happy to merge the
whole lot via powerpc ASAP (since it's all CC'ed stable) if you guys
send me the appropriate acks, otherwise, let me know.
Cheers,
Ben.
^ permalink raw reply
* Re: [PATCH v3 4/7] sound/radeon: Add quirk for broken 64-bit MSI
From: Takashi Iwai @ 2014-10-08 6:59 UTC (permalink / raw)
To: Benjamin Herrenschmidt
Cc: linuxppc-dev, Dave Airlie, Linux PCI, Anton Blanchard, Brian King,
Yijing Wang, Bjorn Helgaas, Alex Deucher
In-Reply-To: <1412746096.30859.229.camel@pasglop>
At Wed, 08 Oct 2014 16:28:16 +1100,
Benjamin Herrenschmidt wrote:
>
> On Tue, 2014-10-07 at 19:47 -0400, Alex Deucher wrote:
> > > This moves the setting of the quirk flag to the audio driver.
> > >
> > > While recent ASICs have that problem fixed, they don't seem to
> > > be listed in the PCI IDs of the current driver, so let's quirk all
> > > the ATI HDMI for now. The consequences are nil on x86 anyway.
> > >
> > > Signed-off-by: Alex Deucher <alexdeucher@gmail.com>
> > > Signed-off-by: Benjamin Herrenschmidt <benh@kernel.crashing.org>
> > > CC: <stable@vger.kernel.org>
> >
> > Further discussion with the hw teams have revealed that this is still
> > an issue on newer asics so I think your original patch is correct
> > after all. Just disable 64 bit MSIs on all AMD audio PCI ids.
>
> Allright, I won't resend the whole series, I can just pickup my previous
> patch. Takashi, Bjorn, Dave, this series covers your 3 areas of
> maintainership, how do you want to proceed ? I'm happy to merge the
> whole lot via powerpc ASAP (since it's all CC'ed stable) if you guys
> send me the appropriate acks, otherwise, let me know.
Feel free to merge through your tree.
Reviewed-by: Takashi Iwai <tiwai@suse.de>
thanks,
Takashi
^ permalink raw reply
* Re: [PATCH 08/44] kernel: Move pm_power_off to common code
From: Jesper Nilsson @ 2014-10-08 7:25 UTC (permalink / raw)
To: Guenter Roeck
Cc: linux-m32r-ja@ml.linux-m32r.org, linux-mips@linux-mips.org,
linux-efi@vger.kernel.org, linux-ia64@vger.kernel.org,
Steven Miao, linux-xtensa@linux-xtensa.org, Boris Ostrovsky,
Catalin Marinas, Will Deacon, David Howells, Max Filippov,
Paul Mackerras, Ralf Baechle, Pavel Machek, H. Peter Anvin,
Guan Xuetao, Thomas Gleixner, Lennox Wu, Hans-Christian Egtvedt,
devel@driverdev.osuosl.org, linux-s390@vger.kernel.org,
lguest@lists.ozlabs.org, Russell King,
linux-c6x-dev@linux-c6x.org, Len Brown, David S. Miller,
linux-hexagon@vger.kernel.org, Hirokazu Takata,
linux-sh@vger.kernel.org, James E.J. Bottomley,
linux-acpi@vger.kernel.org, Ingo Molnar, Geert Uytterhoeven,
Mark Salter, xen-devel@lists.xenproject.org, Matt Turner,
Chen Liqin, Jonas Bonn, Haavard Skinnemoen,
devicetree@vger.kernel.org, James Hogan,
user-mode-linux-devel@lists.sourceforge.net,
linux-pm@vger.kernel.org, Aurelien Jacquiot, Heiko Carstens,
Jeff Dike, adi-buildroot-devel@lists.sourceforge.net,
Chris Metcalf, Jesper Nilsson, Mikael Starvik, Richard Weinberger,
linux-m68k@lists.linux-m68k.org, linux-am33-list@redhat.com,
Ivan Kokshaysky, linux-tegra@vger.kernel.org,
openipmi-developer@lists.sourceforge.net,
linux-metag@vger.kernel.org, linux-arm-kernel@lists.infradead.org,
Richard Henderson, Chris Zankel, Michal Simek, Tony Luck,
linux-parisc@vger.kernel.org, linux-cris-kernel, Vineet Gupta,
Rafael J. Wysocki, linux-kernel@vger.kernel.org, Fenghua Yu,
Richard Kuo, David Vrabel, linux-alpha@vger.kernel.org,
Martin Schwidefsky, Konrad Rzeszutek Wilk, Koichi Yasutake,
linuxppc-dev@lists.ozlabs.org, Helge Deller
In-Reply-To: <1412659726-29957-9-git-send-email-linux@roeck-us.net>
On Tue, Oct 07, 2014 at 07:28:10AM +0200, Guenter Roeck wrote:
> pm_power_off is defined for all architectures. Move it to common code.
>
> Have all architectures call do_kernel_poweroff instead of pm_power_off.
> Some architectures point pm_power_off to machine_power_off. For those,
> call do_kernel_poweroff from machine_power_off instead.
For the CRIS parts:
> arch/cris/kernel/process.c | 4 +---
Acked-by: Jesper Nilsson <jesper.nilsson@axis.com>
/^JN - Jesper Nilsson
--
Jesper Nilsson -- jesper.nilsson@axis.com
^ permalink raw reply
* Re: [PATCH] powerpc: Reimplement __get_SP() as a function not a define
From: Li Zhong @ 2014-10-08 8:47 UTC (permalink / raw)
To: Anton Blanchard; +Cc: paulus, linuxppc-dev
In-Reply-To: <20141001151000.0d754938@kryten>
On 三, 2014-10-01 at 15:10 +1000, Anton Blanchard wrote:
> Li Zhong points out an issue with our current __get_SP()
> implementation. If ftrace function tracing is enabled (ie -pg
> profiling using _mcount) we spill a stack frame on 64bit all the
> time.
>
> If a function calls __get_SP() and later calls a function that is
> tail call optimised, we will pop the stack frame and the value
> returned by __get_SP() is no longer valid. An example from Li can
> be found in save_stack_trace -> save_context_stack:
>
> c0000000000432c0 <.save_stack_trace>:
> c0000000000432c0: mflr r0
> c0000000000432c4: std r0,16(r1)
> c0000000000432c8: stdu r1,-128(r1) <-- stack frame for _mcount
> c0000000000432cc: std r3,112(r1)
> c0000000000432d0: bl <._mcount>
> c0000000000432d4: nop
>
> c0000000000432d8: mr r4,r1 <-- __get_SP()
>
> c0000000000432dc: ld r5,632(r13)
> c0000000000432e0: ld r3,112(r1)
> c0000000000432e4: li r6,1
>
> c0000000000432e8: addi r1,r1,128 <-- pop stack frame
>
> c0000000000432ec: ld r0,16(r1)
> c0000000000432f0: mtlr r0
> c0000000000432f4: b <.save_context_stack> <-- tail call optimized
>
> save_context_stack ends up with a stack pointer below the current
> one, and it is likely to be scribbled over.
>
> Fix this by making __get_SP() a function which returns the
> callers stack frame. Also replace inline assembly which grabs
> the stack pointer in save_stack_trace and show_stack with
> __get_SP().
>
> Reported-by: Li Zhong <zhong@linux.vnet.ibm.com>
> Signed-off-by: Anton Blanchard <anton@samba.org>
> ---
> arch/powerpc/include/asm/reg.h | 3 +--
> arch/powerpc/kernel/misc.S | 4 ++++
> arch/powerpc/kernel/process.c | 2 +-
> arch/powerpc/kernel/stacktrace.c | 2 +-
> 4 files changed, 7 insertions(+), 4 deletions(-)
>
> diff --git a/arch/powerpc/include/asm/reg.h b/arch/powerpc/include/asm/reg.h
> index 0c05059..0f973c0 100644
> --- a/arch/powerpc/include/asm/reg.h
> +++ b/arch/powerpc/include/asm/reg.h
> @@ -1264,8 +1264,7 @@ static inline unsigned long mfvtb (void)
>
> #define proc_trap() asm volatile("trap")
>
> -#define __get_SP() ({unsigned long sp; \
> - asm volatile("mr %0,1": "=r" (sp)); sp;})
> +extern unsigned long __get_SP(void);
It seems that some module code is using __get_SP, e.g. xfs in the
example below:
ERROR: ".__get_SP" [fs/xfs/xfs.ko] undefined!
Maybe we need export this symbol in arch/powerpc/kernel/ppc_ksyms.c?
diff --git a/arch/powerpc/kernel/ppc_ksyms.c b/arch/powerpc/kernel/ppc_ksyms.c
index 48d17d6f..eebd4e4 100644
--- a/arch/powerpc/kernel/ppc_ksyms.c
+++ b/arch/powerpc/kernel/ppc_ksyms.c
@@ -207,3 +207,5 @@ EXPORT_SYMBOL_GPL(mmu_psize_defs);
#ifdef CONFIG_EPAPR_PARAVIRT
EXPORT_SYMBOL(epapr_hypercall_start);
#endif
+
+EXPORT_SYMBOL(__get_SP);
With the above compiling error fixed, this patch solved the SP issue I saw, so
Tested-by: Li Zhong <zhong@linux.vnet.ibm.com>
>
> extern unsigned long scom970_read(unsigned int address);
> extern void scom970_write(unsigned int address, unsigned long value);
> diff --git a/arch/powerpc/kernel/misc.S b/arch/powerpc/kernel/misc.S
> index 7ce26d4..120deb7 100644
> --- a/arch/powerpc/kernel/misc.S
> +++ b/arch/powerpc/kernel/misc.S
> @@ -114,3 +114,7 @@ _GLOBAL(longjmp)
> mtlr r0
> mr r3,r4
> blr
> +
> +_GLOBAL(__get_SP)
> + PPC_LL r3,0(r1)
> + blr
> diff --git a/arch/powerpc/kernel/process.c b/arch/powerpc/kernel/process.c
> index aa1df89..3cc6439 100644
> --- a/arch/powerpc/kernel/process.c
> +++ b/arch/powerpc/kernel/process.c
> @@ -1545,7 +1545,7 @@ void show_stack(struct task_struct *tsk, unsigned long *stack)
> tsk = current;
> if (sp == 0) {
> if (tsk == current)
> - asm("mr %0,1" : "=r" (sp));
> + sp = __get_SP();
> else
> sp = tsk->thread.ksp;
> }
> diff --git a/arch/powerpc/kernel/stacktrace.c b/arch/powerpc/kernel/stacktrace.c
> index 3d30ef1..7f65bae 100644
> --- a/arch/powerpc/kernel/stacktrace.c
> +++ b/arch/powerpc/kernel/stacktrace.c
> @@ -50,7 +50,7 @@ void save_stack_trace(struct stack_trace *trace)
> {
> unsigned long sp;
>
> - asm("mr %0,1" : "=r" (sp));
> + sp = __get_SP();
>
> save_context_stack(trace, sp, current, 1);
> }
^ permalink raw reply related
* [PATCH v4 0/16] POWER8 Coherent Accelerator device driver
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
This is the latest version of the cxl driver. Change log below:
v4:
- Updates based on comments from mpe (offline and online).
- Refactor the sstp lock to be an entry lock.
- Fixed error paths on new status_mutex in start_work
- added some missing include files
- moved associating pid/mm from open() to start_work ioctl.
- improved IDR setup and destroy
- fix block comments.
- remove #undef at top of files
- wed -> work_element_descriptor on user visible interfaces
- Lots of documentation updates.
- Device name changes.
- No longer has a default dev name /dev/afuM.N for each mode.
- Dedicated, slave and master all have distinct char devs.
- Prevent AFU reset when contexts active.
- Endian bug fix for find_free_sste().
- Fix locking on reset_store_afu.
- Make CXL_IOCTL_GET_PROCESS_ELEMENT return a __u32 instead of int.
- Rename event.afu_err.err to error
- Fixed master specific sysfs attribute creation
- fix sparse errors with debugfs. Was passing iomem ptrs to userspace.
v3:
- Updates based on comments from mpe, benh, aneesh and offline reviews.
- Fixed bug freeing AFU IRQs that also freed the multiplexed PSL IRQ
- Change copro_flush_all_slbs to a static inline as suggested by mpe
- Implement sanitisation routines to clear out more registers and do full
adapter wide tlbia and slbia when initialising hardware
- Add self testcase to msi_bitmap to test allocations are aligned to a power of
2 and cleanup comment as suggested by mpe
- Clean up cxl_use_count
- Split out detach_process_native into two logical functions
- Improve comment in set_msi_irq_chip as requested by mpe
- Move cxl functions in pci-ioda.c to be under just one #ifdef CONFIG_CXL_BASE
- Cleanup hash_page and hash_page_mm from mpes and Aneesh' reviews
- Remove dead code in cxl_alloc_sst
- Add timeout in afu_slbia_native
- Remove cxl backend and driver ops abstractions
- Removed separate cxl-pci module
- Merged cxl pci module init calls into main driver init
- Refactor afu_read() to be a bit simpler and more closely follow exising
patterns in the kernel
- Userspace API updates from reviews:
- Added ioctl to get the process element number, and removed it as a return
from the start work ioctl
- Alter cxl_event to have one common header struct
- Dropped check error ioctl
- Added current and binary compatible API version numbers to sysfs
- read() now takes a 4K (or greater) buffer
- Pack event structs to reduce unecessary reserved fields
- Event sizes can now differ
- All event sizes are 64bit multiples to allow future event coalescing
- Add flags fields to indicate which fields contain valid data
- Add BUILD_BUG_ONs to protect against inadvertantly changing API without
bumping version number and/or flags
- Update documentation
- Skip CXL SLBIA codepath if CXL is not in use
- Split cxl_slbia_core into two functions to be easier to read
- Refactor copro_data_segment (renamed to copro_calc_slb) since we are no
longer merging with hash_page and cleaned up parameters.
- Some renames:
- struct cxl_t -> struct cxl
- struct cxl_afu_t -> struct cxl_afu
- struct cxl_context_t -> struct cxl_context
- copro_data_segment -> copro_calc_slb
- ctx->ph -> ctx->pe
- Added ctx->status mutex lock around for start and release context
v2:
- Updates based on comments from, Anton, Gavin, Aneesh, jk and offline reviews
- Simplified copro_data_segment() and merged code with hash_page_mm()
(New patch 10/17)
- PCIe code simplifications based on Gavin's review
- Removed redundant comment in msi_bitmap_alloc_hwirqs()
- Fix for locking in idr_remove in core driver
- Ensure PSL is enabled when PHB is flipped to CXL mode
- Added CONFIG_PPC_COPRO_BASE to compile copro_fault.c
- Merged SPU and cxl slb flushing calls into copro_flush_all_slbs()
(New patch 03/17)
- Moved slb_vsid_shift() to static inline from #define
- Don't write paca->context when demoting segments and mm != current
- Fix minor typos in documentation
v1:
- Initial post
This add support for the Coherent Accelerator (cxl) attached to POWER8
processors. This coherent accelerator interface is designed to allow the
coherent connection of FPGA based accelerators (and other devices) to a POWER
systems.
IBM refers to this as the Coherent Accelerator Processor Interface or CAPI. In
this driver it's referred to by the name cxl to avoid confusion with the ISDN
CAPI subsystem.
An overview of the patches:
Patches 1-3: Split some of the old Cell co-processor code out so it can be
reused.
Patches 4-10: Add infrastructure to arch/powerpc needed by cxl.
Patches 11: Add call backs needed for invalidating cxl mm contexts.
Patch 12: Add cxl specific support that needs to be built in to the
kernel (can't be a module).
Patches 13-15: Add the majority of the device driver and API header.
Patch 16: Documentation.
The documentation in this last patch gives an overview of the hardware
architecture as well as the userspace API.
The cxl driver has a user-space interface described in include/uapi/misc/cxl.h
and Documentation/powerpc/cxl.txt. There are two ioctls which can be used to
talk to the driver once the new /dev/cxl/afu0.0 device is opened. This device
can also be read and mmaped.
There's also sysfs entries used to communicate information about the cxl
configuration to userspace. These are documented in
Documentation/ABI/testing/sysfs-class-cxl.
Many contributed to this device driver but Ian Munsie is the principal author.
Driver can also be found here (based on 3.17-rc5):
git://github.com/mikey/linux.git cxl
https://github.com/mikey/linux/commits/cxl
(Series rebases on recent linux-next with one trivial include file conflict)
Please consider for inclusion. Feedback welcome!
Regards,
Mikey
^ permalink raw reply
* [PATCH v4 01/16] powerpc/cell: Move spu_handle_mm_fault() out of cell platform
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
Currently spu_handle_mm_fault() is in the cell platform.
This code is generically useful for other non-cell co-processors on powerpc.
This patch moves this function out of the cell platform into arch/powerpc/mm so
that others may use it.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/Kconfig | 4 ++++
arch/powerpc/include/asm/copro.h | 16 ++++++++++++++++
arch/powerpc/include/asm/spu.h | 5 ++---
arch/powerpc/mm/Makefile | 1 +
.../{platforms/cell/spu_fault.c => mm/copro_fault.c} | 14 ++++++--------
arch/powerpc/platforms/cell/Kconfig | 1 +
arch/powerpc/platforms/cell/Makefile | 2 +-
arch/powerpc/platforms/cell/spufs/fault.c | 4 ++--
8 files changed, 33 insertions(+), 14 deletions(-)
create mode 100644 arch/powerpc/include/asm/copro.h
rename arch/powerpc/{platforms/cell/spu_fault.c => mm/copro_fault.c} (89%)
diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig
index 4bc7b62..8f094e9 100644
--- a/arch/powerpc/Kconfig
+++ b/arch/powerpc/Kconfig
@@ -603,6 +603,10 @@ config PPC_SUBPAGE_PROT
to set access permissions (read/write, readonly, or no access)
on the 4k subpages of each 64k page.
+config PPC_COPRO_BASE
+ bool
+ default n
+
config SCHED_SMT
bool "SMT (Hyperthreading) scheduler support"
depends on PPC64 && SMP
diff --git a/arch/powerpc/include/asm/copro.h b/arch/powerpc/include/asm/copro.h
new file mode 100644
index 0000000..51cae85
--- /dev/null
+++ b/arch/powerpc/include/asm/copro.h
@@ -0,0 +1,16 @@
+/*
+ * Copyright 2014 IBM Corp.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License
+ * as published by the Free Software Foundation; either version
+ * 2 of the License, or (at your option) any later version.
+ */
+
+#ifndef _ASM_POWERPC_COPRO_H
+#define _ASM_POWERPC_COPRO_H
+
+int copro_handle_mm_fault(struct mm_struct *mm, unsigned long ea,
+ unsigned long dsisr, unsigned *flt);
+
+#endif /* _ASM_POWERPC_COPRO_H */
diff --git a/arch/powerpc/include/asm/spu.h b/arch/powerpc/include/asm/spu.h
index 37b7ca3..a6e6e2b 100644
--- a/arch/powerpc/include/asm/spu.h
+++ b/arch/powerpc/include/asm/spu.h
@@ -27,6 +27,8 @@
#include <linux/workqueue.h>
#include <linux/device.h>
#include <linux/mutex.h>
+#include <asm/reg.h>
+#include <asm/copro.h>
#define LS_SIZE (256 * 1024)
#define LS_ADDR_MASK (LS_SIZE - 1)
@@ -277,9 +279,6 @@ void spu_remove_dev_attr(struct device_attribute *attr);
int spu_add_dev_attr_group(struct attribute_group *attrs);
void spu_remove_dev_attr_group(struct attribute_group *attrs);
-int spu_handle_mm_fault(struct mm_struct *mm, unsigned long ea,
- unsigned long dsisr, unsigned *flt);
-
/*
* Notifier blocks:
*
diff --git a/arch/powerpc/mm/Makefile b/arch/powerpc/mm/Makefile
index d0130ff..325e861 100644
--- a/arch/powerpc/mm/Makefile
+++ b/arch/powerpc/mm/Makefile
@@ -34,3 +34,4 @@ obj-$(CONFIG_TRANSPARENT_HUGEPAGE) += hugepage-hash64.o
obj-$(CONFIG_PPC_SUBPAGE_PROT) += subpage-prot.o
obj-$(CONFIG_NOT_COHERENT_CACHE) += dma-noncoherent.o
obj-$(CONFIG_HIGHMEM) += highmem.o
+obj-$(CONFIG_PPC_COPRO_BASE) += copro_fault.o
diff --git a/arch/powerpc/platforms/cell/spu_fault.c b/arch/powerpc/mm/copro_fault.c
similarity index 89%
rename from arch/powerpc/platforms/cell/spu_fault.c
rename to arch/powerpc/mm/copro_fault.c
index 641e727..ba7df14 100644
--- a/arch/powerpc/platforms/cell/spu_fault.c
+++ b/arch/powerpc/mm/copro_fault.c
@@ -1,5 +1,5 @@
/*
- * SPU mm fault handler
+ * CoProcessor (SPU/AFU) mm fault handler
*
* (C) Copyright IBM Deutschland Entwicklung GmbH 2007
*
@@ -23,16 +23,14 @@
#include <linux/sched.h>
#include <linux/mm.h>
#include <linux/export.h>
-
-#include <asm/spu.h>
-#include <asm/spu_csa.h>
+#include <asm/reg.h>
/*
* This ought to be kept in sync with the powerpc specific do_page_fault
* function. Currently, there are a few corner cases that we haven't had
* to handle fortunately.
*/
-int spu_handle_mm_fault(struct mm_struct *mm, unsigned long ea,
+int copro_handle_mm_fault(struct mm_struct *mm, unsigned long ea,
unsigned long dsisr, unsigned *flt)
{
struct vm_area_struct *vma;
@@ -58,12 +56,12 @@ int spu_handle_mm_fault(struct mm_struct *mm, unsigned long ea,
goto out_unlock;
}
- is_write = dsisr & MFC_DSISR_ACCESS_PUT;
+ is_write = dsisr & DSISR_ISSTORE;
if (is_write) {
if (!(vma->vm_flags & VM_WRITE))
goto out_unlock;
} else {
- if (dsisr & MFC_DSISR_ACCESS_DENIED)
+ if (dsisr & DSISR_PROTFAULT)
goto out_unlock;
if (!(vma->vm_flags & (VM_READ | VM_EXEC)))
goto out_unlock;
@@ -91,4 +89,4 @@ out_unlock:
up_read(&mm->mmap_sem);
return ret;
}
-EXPORT_SYMBOL_GPL(spu_handle_mm_fault);
+EXPORT_SYMBOL_GPL(copro_handle_mm_fault);
diff --git a/arch/powerpc/platforms/cell/Kconfig b/arch/powerpc/platforms/cell/Kconfig
index 9978f59..870b6db 100644
--- a/arch/powerpc/platforms/cell/Kconfig
+++ b/arch/powerpc/platforms/cell/Kconfig
@@ -86,6 +86,7 @@ config SPU_FS_64K_LS
config SPU_BASE
bool
default n
+ select PPC_COPRO_BASE
config CBE_RAS
bool "RAS features for bare metal Cell BE"
diff --git a/arch/powerpc/platforms/cell/Makefile b/arch/powerpc/platforms/cell/Makefile
index fe053e7..2d16884 100644
--- a/arch/powerpc/platforms/cell/Makefile
+++ b/arch/powerpc/platforms/cell/Makefile
@@ -20,7 +20,7 @@ spu-manage-$(CONFIG_PPC_CELL_COMMON) += spu_manage.o
obj-$(CONFIG_SPU_BASE) += spu_callbacks.o spu_base.o \
spu_notify.o \
- spu_syscalls.o spu_fault.o \
+ spu_syscalls.o \
$(spu-priv1-y) \
$(spu-manage-y) \
spufs/
diff --git a/arch/powerpc/platforms/cell/spufs/fault.c b/arch/powerpc/platforms/cell/spufs/fault.c
index 8cb6260..e45894a 100644
--- a/arch/powerpc/platforms/cell/spufs/fault.c
+++ b/arch/powerpc/platforms/cell/spufs/fault.c
@@ -138,7 +138,7 @@ int spufs_handle_class1(struct spu_context *ctx)
if (ctx->state == SPU_STATE_RUNNABLE)
ctx->spu->stats.hash_flt++;
- /* we must not hold the lock when entering spu_handle_mm_fault */
+ /* we must not hold the lock when entering copro_handle_mm_fault */
spu_release(ctx);
access = (_PAGE_PRESENT | _PAGE_USER);
@@ -149,7 +149,7 @@ int spufs_handle_class1(struct spu_context *ctx)
/* hashing failed, so try the actual fault handler */
if (ret)
- ret = spu_handle_mm_fault(current->mm, ea, dsisr, &flt);
+ ret = copro_handle_mm_fault(current->mm, ea, dsisr, &flt);
/*
* This is nasty: we need the state_mutex for all the bookkeeping even
--
1.9.1
^ permalink raw reply related
* [PATCH v4 02/16] powerpc/cell: Move data segment faulting code out of cell platform
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
__spu_trap_data_seg() currently contains code to determine the VSID and ESID
required for a particular EA and mm struct.
This code is generically useful for other co-processors. This moves the code of
the cell platform so it can be used by other powerpc code. It also adds 1TB
segment handling which Cell didn't support. The new function is called
copro_calculate_slb().
This also moves the internal struct spu_slb to a generic struct copro_slb which
is now used in the Cell and copro code. We use this new struct instead of
passing around esid and vsid parameters.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/include/asm/copro.h | 7 +++++
arch/powerpc/include/asm/mmu-hash64.h | 7 +++++
arch/powerpc/mm/copro_fault.c | 46 ++++++++++++++++++++++++++++
arch/powerpc/mm/slb.c | 3 --
arch/powerpc/platforms/cell/spu_base.c | 55 ++++++----------------------------
5 files changed, 69 insertions(+), 49 deletions(-)
diff --git a/arch/powerpc/include/asm/copro.h b/arch/powerpc/include/asm/copro.h
index 51cae85..b0e6a18 100644
--- a/arch/powerpc/include/asm/copro.h
+++ b/arch/powerpc/include/asm/copro.h
@@ -10,7 +10,14 @@
#ifndef _ASM_POWERPC_COPRO_H
#define _ASM_POWERPC_COPRO_H
+struct copro_slb
+{
+ u64 esid, vsid;
+};
+
int copro_handle_mm_fault(struct mm_struct *mm, unsigned long ea,
unsigned long dsisr, unsigned *flt);
+int copro_calculate_slb(struct mm_struct *mm, u64 ea, struct copro_slb *slb);
+
#endif /* _ASM_POWERPC_COPRO_H */
diff --git a/arch/powerpc/include/asm/mmu-hash64.h b/arch/powerpc/include/asm/mmu-hash64.h
index d765144..aeabd02 100644
--- a/arch/powerpc/include/asm/mmu-hash64.h
+++ b/arch/powerpc/include/asm/mmu-hash64.h
@@ -190,6 +190,13 @@ static inline unsigned int mmu_psize_to_shift(unsigned int mmu_psize)
#ifndef __ASSEMBLY__
+static inline int slb_vsid_shift(int ssize)
+{
+ if (ssize == MMU_SEGSIZE_256M)
+ return SLB_VSID_SHIFT;
+ return SLB_VSID_SHIFT_1T;
+}
+
static inline int segment_shift(int ssize)
{
if (ssize == MMU_SEGSIZE_256M)
diff --git a/arch/powerpc/mm/copro_fault.c b/arch/powerpc/mm/copro_fault.c
index ba7df14..a15a23e 100644
--- a/arch/powerpc/mm/copro_fault.c
+++ b/arch/powerpc/mm/copro_fault.c
@@ -24,6 +24,7 @@
#include <linux/mm.h>
#include <linux/export.h>
#include <asm/reg.h>
+#include <asm/copro.h>
/*
* This ought to be kept in sync with the powerpc specific do_page_fault
@@ -90,3 +91,48 @@ out_unlock:
return ret;
}
EXPORT_SYMBOL_GPL(copro_handle_mm_fault);
+
+int copro_calculate_slb(struct mm_struct *mm, u64 ea, struct copro_slb *slb)
+{
+ u64 vsid;
+ int psize, ssize;
+
+ slb->esid = (ea & ESID_MASK) | SLB_ESID_V;
+
+ switch (REGION_ID(ea)) {
+ case USER_REGION_ID:
+ pr_devel("%s: 0x%llx -- USER_REGION_ID\n", __func__, ea);
+ psize = get_slice_psize(mm, ea);
+ ssize = user_segment_size(ea);
+ vsid = get_vsid(mm->context.id, ea, ssize);
+ break;
+ case VMALLOC_REGION_ID:
+ pr_devel("%s: 0x%llx -- VMALLOC_REGION_ID\n", __func__, ea);
+ if (ea < VMALLOC_END)
+ psize = mmu_vmalloc_psize;
+ else
+ psize = mmu_io_psize;
+ ssize = mmu_kernel_ssize;
+ vsid = get_kernel_vsid(ea, mmu_kernel_ssize);
+ break;
+ case KERNEL_REGION_ID:
+ pr_devel("%s: 0x%llx -- KERNEL_REGION_ID\n", __func__, ea);
+ psize = mmu_linear_psize;
+ ssize = mmu_kernel_ssize;
+ vsid = get_kernel_vsid(ea, mmu_kernel_ssize);
+ break;
+ default:
+ pr_debug("%s: invalid region access at %016llx\n", __func__, ea);
+ return 1;
+ }
+
+ vsid = (vsid << slb_vsid_shift(ssize)) | SLB_VSID_USER;
+
+ vsid |= mmu_psize_defs[psize].sllp |
+ ((ssize == MMU_SEGSIZE_1T) ? SLB_VSID_B_1T : 0);
+
+ slb->vsid = vsid;
+
+ return 0;
+}
+EXPORT_SYMBOL_GPL(copro_calculate_slb);
diff --git a/arch/powerpc/mm/slb.c b/arch/powerpc/mm/slb.c
index 0399a67..6e450ca 100644
--- a/arch/powerpc/mm/slb.c
+++ b/arch/powerpc/mm/slb.c
@@ -46,9 +46,6 @@ static inline unsigned long mk_esid_data(unsigned long ea, int ssize,
return (ea & slb_esid_mask(ssize)) | SLB_ESID_V | slot;
}
-#define slb_vsid_shift(ssize) \
- ((ssize) == MMU_SEGSIZE_256M? SLB_VSID_SHIFT: SLB_VSID_SHIFT_1T)
-
static inline unsigned long mk_vsid_data(unsigned long ea, int ssize,
unsigned long flags)
{
diff --git a/arch/powerpc/platforms/cell/spu_base.c b/arch/powerpc/platforms/cell/spu_base.c
index 2930d1e..ffcbd24 100644
--- a/arch/powerpc/platforms/cell/spu_base.c
+++ b/arch/powerpc/platforms/cell/spu_base.c
@@ -76,10 +76,6 @@ static LIST_HEAD(spu_full_list);
static DEFINE_SPINLOCK(spu_full_list_lock);
static DEFINE_MUTEX(spu_full_list_mutex);
-struct spu_slb {
- u64 esid, vsid;
-};
-
void spu_invalidate_slbs(struct spu *spu)
{
struct spu_priv2 __iomem *priv2 = spu->priv2;
@@ -149,7 +145,7 @@ static void spu_restart_dma(struct spu *spu)
}
}
-static inline void spu_load_slb(struct spu *spu, int slbe, struct spu_slb *slb)
+static inline void spu_load_slb(struct spu *spu, int slbe, struct copro_slb *slb)
{
struct spu_priv2 __iomem *priv2 = spu->priv2;
@@ -167,45 +163,12 @@ static inline void spu_load_slb(struct spu *spu, int slbe, struct spu_slb *slb)
static int __spu_trap_data_seg(struct spu *spu, unsigned long ea)
{
- struct mm_struct *mm = spu->mm;
- struct spu_slb slb;
- int psize;
-
- pr_debug("%s\n", __func__);
-
- slb.esid = (ea & ESID_MASK) | SLB_ESID_V;
+ struct copro_slb slb;
+ int ret;
- switch(REGION_ID(ea)) {
- case USER_REGION_ID:
-#ifdef CONFIG_PPC_MM_SLICES
- psize = get_slice_psize(mm, ea);
-#else
- psize = mm->context.user_psize;
-#endif
- slb.vsid = (get_vsid(mm->context.id, ea, MMU_SEGSIZE_256M)
- << SLB_VSID_SHIFT) | SLB_VSID_USER;
- break;
- case VMALLOC_REGION_ID:
- if (ea < VMALLOC_END)
- psize = mmu_vmalloc_psize;
- else
- psize = mmu_io_psize;
- slb.vsid = (get_kernel_vsid(ea, MMU_SEGSIZE_256M)
- << SLB_VSID_SHIFT) | SLB_VSID_KERNEL;
- break;
- case KERNEL_REGION_ID:
- psize = mmu_linear_psize;
- slb.vsid = (get_kernel_vsid(ea, MMU_SEGSIZE_256M)
- << SLB_VSID_SHIFT) | SLB_VSID_KERNEL;
- break;
- default:
- /* Future: support kernel segments so that drivers
- * can use SPUs.
- */
- pr_debug("invalid region access at %016lx\n", ea);
- return 1;
- }
- slb.vsid |= mmu_psize_defs[psize].sllp;
+ ret = copro_calculate_slb(spu->mm, ea, &slb);
+ if (ret)
+ return ret;
spu_load_slb(spu, spu->slb_replace, &slb);
@@ -253,7 +216,7 @@ static int __spu_trap_data_map(struct spu *spu, unsigned long ea, u64 dsisr)
return 0;
}
-static void __spu_kernel_slb(void *addr, struct spu_slb *slb)
+static void __spu_kernel_slb(void *addr, struct copro_slb *slb)
{
unsigned long ea = (unsigned long)addr;
u64 llp;
@@ -272,7 +235,7 @@ static void __spu_kernel_slb(void *addr, struct spu_slb *slb)
* Given an array of @nr_slbs SLB entries, @slbs, return non-zero if the
* address @new_addr is present.
*/
-static inline int __slb_present(struct spu_slb *slbs, int nr_slbs,
+static inline int __slb_present(struct copro_slb *slbs, int nr_slbs,
void *new_addr)
{
unsigned long ea = (unsigned long)new_addr;
@@ -297,7 +260,7 @@ static inline int __slb_present(struct spu_slb *slbs, int nr_slbs,
void spu_setup_kernel_slbs(struct spu *spu, struct spu_lscsa *lscsa,
void *code, int code_size)
{
- struct spu_slb slbs[4];
+ struct copro_slb slbs[4];
int i, nr_slbs = 0;
/* start and end addresses of both mappings */
void *addrs[] = {
--
1.9.1
^ permalink raw reply related
* [PATCH v4 03/16] powerpc/cell: Make spu_flush_all_slbs() generic
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This moves spu_flush_all_slbs() into a generic call copro_flush_all_slbs().
This will be useful when we add cxl which also needs a similar SLB flush call.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/include/asm/copro.h | 6 ++++++
arch/powerpc/mm/copro_fault.c | 9 +++++++++
arch/powerpc/mm/hash_utils_64.c | 10 +++-------
arch/powerpc/mm/slice.c | 10 +++-------
4 files changed, 21 insertions(+), 14 deletions(-)
diff --git a/arch/powerpc/include/asm/copro.h b/arch/powerpc/include/asm/copro.h
index b0e6a18..ce216df 100644
--- a/arch/powerpc/include/asm/copro.h
+++ b/arch/powerpc/include/asm/copro.h
@@ -20,4 +20,10 @@ int copro_handle_mm_fault(struct mm_struct *mm, unsigned long ea,
int copro_calculate_slb(struct mm_struct *mm, u64 ea, struct copro_slb *slb);
+
+#ifdef CONFIG_PPC_COPRO_BASE
+void copro_flush_all_slbs(struct mm_struct *mm);
+#else
+static inline void copro_flush_all_slbs(struct mm_struct *mm) {}
+#endif
#endif /* _ASM_POWERPC_COPRO_H */
diff --git a/arch/powerpc/mm/copro_fault.c b/arch/powerpc/mm/copro_fault.c
index a15a23e..f2aa5a8 100644
--- a/arch/powerpc/mm/copro_fault.c
+++ b/arch/powerpc/mm/copro_fault.c
@@ -25,6 +25,7 @@
#include <linux/export.h>
#include <asm/reg.h>
#include <asm/copro.h>
+#include <asm/spu.h>
/*
* This ought to be kept in sync with the powerpc specific do_page_fault
@@ -136,3 +137,11 @@ int copro_calculate_slb(struct mm_struct *mm, u64 ea, struct copro_slb *slb)
return 0;
}
EXPORT_SYMBOL_GPL(copro_calculate_slb);
+
+void copro_flush_all_slbs(struct mm_struct *mm)
+{
+#ifdef CONFIG_SPU_BASE
+ spu_flush_all_slbs(mm);
+#endif
+}
+EXPORT_SYMBOL_GPL(copro_flush_all_slbs);
diff --git a/arch/powerpc/mm/hash_utils_64.c b/arch/powerpc/mm/hash_utils_64.c
index daee7f4..5c0738d 100644
--- a/arch/powerpc/mm/hash_utils_64.c
+++ b/arch/powerpc/mm/hash_utils_64.c
@@ -51,7 +51,7 @@
#include <asm/cacheflush.h>
#include <asm/cputable.h>
#include <asm/sections.h>
-#include <asm/spu.h>
+#include <asm/copro.h>
#include <asm/udbg.h>
#include <asm/code-patching.h>
#include <asm/fadump.h>
@@ -901,9 +901,7 @@ void demote_segment_4k(struct mm_struct *mm, unsigned long addr)
if (get_slice_psize(mm, addr) == MMU_PAGE_4K)
return;
slice_set_range_psize(mm, addr, 1, MMU_PAGE_4K);
-#ifdef CONFIG_SPU_BASE
- spu_flush_all_slbs(mm);
-#endif
+ copro_flush_all_slbs(mm);
if (get_paca_psize(addr) != MMU_PAGE_4K) {
get_paca()->context = mm->context;
slb_flush_and_rebolt();
@@ -1141,9 +1139,7 @@ int hash_page(unsigned long ea, unsigned long access, unsigned long trap)
"to 4kB pages because of "
"non-cacheable mapping\n");
psize = mmu_vmalloc_psize = MMU_PAGE_4K;
-#ifdef CONFIG_SPU_BASE
- spu_flush_all_slbs(mm);
-#endif
+ copro_flush_all_slbs(mm);
}
}
diff --git a/arch/powerpc/mm/slice.c b/arch/powerpc/mm/slice.c
index b0c75cc..a81791c 100644
--- a/arch/powerpc/mm/slice.c
+++ b/arch/powerpc/mm/slice.c
@@ -32,7 +32,7 @@
#include <linux/export.h>
#include <asm/mman.h>
#include <asm/mmu.h>
-#include <asm/spu.h>
+#include <asm/copro.h>
/* some sanity checks */
#if (PGTABLE_RANGE >> 43) > SLICE_MASK_SIZE
@@ -232,9 +232,7 @@ static void slice_convert(struct mm_struct *mm, struct slice_mask mask, int psiz
spin_unlock_irqrestore(&slice_convert_lock, flags);
-#ifdef CONFIG_SPU_BASE
- spu_flush_all_slbs(mm);
-#endif
+ copro_flush_all_slbs(mm);
}
/*
@@ -671,9 +669,7 @@ void slice_set_psize(struct mm_struct *mm, unsigned long address,
spin_unlock_irqrestore(&slice_convert_lock, flags);
-#ifdef CONFIG_SPU_BASE
- spu_flush_all_slbs(mm);
-#endif
+ copro_flush_all_slbs(mm);
}
void slice_set_range_psize(struct mm_struct *mm, unsigned long start,
--
1.9.1
^ permalink raw reply related
* [PATCH v4 04/16] powerpc/msi: Improve IRQ bitmap allocator
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
Currently msi_bitmap_alloc_hwirqs() will round up any IRQ allocation requests
to the nearest power of 2. eg. ask for 5 IRQs and you'll get 8. This wastes a
lot of IRQs which can be a scarce resource.
For cxl we may require multiple IRQs for every context that is attached to the
accelerator. There may be 1000s of contexts attached, hence we can easily run
out of IRQs, especially if we are needlessly wasting them.
This changes the msi_bitmap_alloc_hwirqs() to allocate only the required number
of IRQs, hence avoiding this wastage. It keeps the natural alignment
requirement though.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/sysdev/msi_bitmap.c | 36 +++++++++++++++++++++++++-----------
1 file changed, 25 insertions(+), 11 deletions(-)
diff --git a/arch/powerpc/sysdev/msi_bitmap.c b/arch/powerpc/sysdev/msi_bitmap.c
index 2ff6302..871d94b 100644
--- a/arch/powerpc/sysdev/msi_bitmap.c
+++ b/arch/powerpc/sysdev/msi_bitmap.c
@@ -20,32 +20,37 @@ int msi_bitmap_alloc_hwirqs(struct msi_bitmap *bmp, int num)
int offset, order = get_count_order(num);
spin_lock_irqsave(&bmp->lock, flags);
- /*
- * This is fast, but stricter than we need. We might want to add
- * a fallback routine which does a linear search with no alignment.
- */
- offset = bitmap_find_free_region(bmp->bitmap, bmp->irq_count, order);
+
+ offset = bitmap_find_next_zero_area(bmp->bitmap, bmp->irq_count, 0,
+ num, (1 << order) - 1);
+ if (offset > bmp->irq_count)
+ goto err;
+
+ bitmap_set(bmp->bitmap, offset, num);
spin_unlock_irqrestore(&bmp->lock, flags);
- pr_debug("msi_bitmap: allocated 0x%x (2^%d) at offset 0x%x\n",
- num, order, offset);
+ pr_debug("msi_bitmap: allocated 0x%x at offset 0x%x\n", num, offset);
return offset;
+err:
+ spin_unlock_irqrestore(&bmp->lock, flags);
+ return -ENOMEM;
}
+EXPORT_SYMBOL(msi_bitmap_alloc_hwirqs);
void msi_bitmap_free_hwirqs(struct msi_bitmap *bmp, unsigned int offset,
unsigned int num)
{
unsigned long flags;
- int order = get_count_order(num);
- pr_debug("msi_bitmap: freeing 0x%x (2^%d) at offset 0x%x\n",
- num, order, offset);
+ pr_debug("msi_bitmap: freeing 0x%x at offset 0x%x\n",
+ num, offset);
spin_lock_irqsave(&bmp->lock, flags);
- bitmap_release_region(bmp->bitmap, offset, order);
+ bitmap_clear(bmp->bitmap, offset, num);
spin_unlock_irqrestore(&bmp->lock, flags);
}
+EXPORT_SYMBOL(msi_bitmap_free_hwirqs);
void msi_bitmap_reserve_hwirq(struct msi_bitmap *bmp, unsigned int hwirq)
{
@@ -180,6 +185,15 @@ void __init test_basics(void)
msi_bitmap_free_hwirqs(&bmp, size / 2, 1);
check(msi_bitmap_alloc_hwirqs(&bmp, 1) == size / 2);
+ /* Check we get a naturally aligned offset */
+ check(msi_bitmap_alloc_hwirqs(&bmp, 2) % 2 == 0);
+ check(msi_bitmap_alloc_hwirqs(&bmp, 4) % 4 == 0);
+ check(msi_bitmap_alloc_hwirqs(&bmp, 8) % 8 == 0);
+ check(msi_bitmap_alloc_hwirqs(&bmp, 9) % 16 == 0);
+ check(msi_bitmap_alloc_hwirqs(&bmp, 3) % 4 == 0);
+ check(msi_bitmap_alloc_hwirqs(&bmp, 7) % 8 == 0);
+ check(msi_bitmap_alloc_hwirqs(&bmp, 121) % 128 == 0);
+
msi_bitmap_free(&bmp);
/* Clients may check bitmap == NULL for "not-allocated" */
--
1.9.1
^ permalink raw reply related
* [PATCH v4 05/16] powerpc/mm: Export mmu_kernel_ssize and mmu_linear_psize
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
Export mmu_kernel_ssize and mmu_linear_psize. These are needed by the cxl
driver which has it's own MMU. To setup the MMU cxl needs access to these.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/mm/hash_utils_64.c | 2 ++
1 file changed, 2 insertions(+)
diff --git a/arch/powerpc/mm/hash_utils_64.c b/arch/powerpc/mm/hash_utils_64.c
index 5c0738d..bbdb054 100644
--- a/arch/powerpc/mm/hash_utils_64.c
+++ b/arch/powerpc/mm/hash_utils_64.c
@@ -98,6 +98,7 @@ unsigned long htab_size_bytes;
unsigned long htab_hash_mask;
EXPORT_SYMBOL_GPL(htab_hash_mask);
int mmu_linear_psize = MMU_PAGE_4K;
+EXPORT_SYMBOL_GPL(mmu_linear_psize);
int mmu_virtual_psize = MMU_PAGE_4K;
int mmu_vmalloc_psize = MMU_PAGE_4K;
#ifdef CONFIG_SPARSEMEM_VMEMMAP
@@ -105,6 +106,7 @@ int mmu_vmemmap_psize = MMU_PAGE_4K;
#endif
int mmu_io_psize = MMU_PAGE_4K;
int mmu_kernel_ssize = MMU_SEGSIZE_256M;
+EXPORT_SYMBOL_GPL(mmu_kernel_ssize);
int mmu_highuser_ssize = MMU_SEGSIZE_256M;
u16 mmu_slb_size = 64;
EXPORT_SYMBOL_GPL(mmu_slb_size);
--
1.9.1
^ permalink raw reply related
* [PATCH v4 06/16] powerpc/powernv: Split out set MSI IRQ chip code
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
Some of the MSI IRQ code in pnv_pci_ioda_msi_setup() is generically useful so
split it out.
This will be used by some of the cxl PCIe code later.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/platforms/powernv/pci-ioda.c | 42 ++++++++++++++++++-------------
1 file changed, 24 insertions(+), 18 deletions(-)
diff --git a/arch/powerpc/platforms/powernv/pci-ioda.c b/arch/powerpc/platforms/powernv/pci-ioda.c
index df241b1..baf3de6 100644
--- a/arch/powerpc/platforms/powernv/pci-ioda.c
+++ b/arch/powerpc/platforms/powernv/pci-ioda.c
@@ -1306,14 +1306,35 @@ static void pnv_ioda2_msi_eoi(struct irq_data *d)
icp_native_eoi(d);
}
+
+static void set_msi_irq_chip(struct pnv_phb *phb, unsigned int virq)
+{
+ struct irq_data *idata;
+ struct irq_chip *ichip;
+
+ if (phb->type != PNV_PHB_IODA2)
+ return;
+
+ if (!phb->ioda.irq_chip_init) {
+ /*
+ * First time we setup an MSI IRQ, we need to setup the
+ * corresponding IRQ chip to route correctly.
+ */
+ idata = irq_get_irq_data(virq);
+ ichip = irq_data_get_irq_chip(idata);
+ phb->ioda.irq_chip_init = 1;
+ phb->ioda.irq_chip = *ichip;
+ phb->ioda.irq_chip.irq_eoi = pnv_ioda2_msi_eoi;
+ }
+ irq_set_chip(virq, &phb->ioda.irq_chip);
+}
+
static int pnv_pci_ioda_msi_setup(struct pnv_phb *phb, struct pci_dev *dev,
unsigned int hwirq, unsigned int virq,
unsigned int is_64, struct msi_msg *msg)
{
struct pnv_ioda_pe *pe = pnv_ioda_get_pe(dev);
struct pci_dn *pdn = pci_get_pdn(dev);
- struct irq_data *idata;
- struct irq_chip *ichip;
unsigned int xive_num = hwirq - phb->msi_base;
__be32 data;
int rc;
@@ -1365,22 +1386,7 @@ static int pnv_pci_ioda_msi_setup(struct pnv_phb *phb, struct pci_dev *dev,
}
msg->data = be32_to_cpu(data);
- /*
- * Change the IRQ chip for the MSI interrupts on PHB3.
- * The corresponding IRQ chip should be populated for
- * the first time.
- */
- if (phb->type == PNV_PHB_IODA2) {
- if (!phb->ioda.irq_chip_init) {
- idata = irq_get_irq_data(virq);
- ichip = irq_data_get_irq_chip(idata);
- phb->ioda.irq_chip_init = 1;
- phb->ioda.irq_chip = *ichip;
- phb->ioda.irq_chip.irq_eoi = pnv_ioda2_msi_eoi;
- }
-
- irq_set_chip(virq, &phb->ioda.irq_chip);
- }
+ set_msi_irq_chip(phb, virq);
pr_devel("%s: %s-bit MSI on hwirq %x (xive #%d),"
" address=%x_%08x data=%x PE# %d\n",
--
1.9.1
^ permalink raw reply related
* [PATCH v4 07/16] cxl: Add new header for call backs and structs
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This new header adds callbacks and structs needed by the rest of the kernel to
hook into the cxl infrastructure.
This adds the cxl_ctx_in_use() function for use in the mm code to see if any
cxl contexts are currently in use. This is used by the tlbie() to determine if
it can do local TLB invalidations or not. This also adds get/put calls for the
cxl driver module to refcount the active cxl contexts.
cxl_ctx_get/put/in_use are static inlined here as they are called in tlbie
which we want to be fast (mpe's suggestion).
Empty functions are provided when CONFIG_CXL_BASE is not enabled.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
include/misc/cxl.h | 48 ++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 48 insertions(+)
create mode 100644 include/misc/cxl.h
diff --git a/include/misc/cxl.h b/include/misc/cxl.h
new file mode 100644
index 0000000..975cc78
--- /dev/null
+++ b/include/misc/cxl.h
@@ -0,0 +1,48 @@
+/*
+ * Copyright 2014 IBM Corp.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License
+ * as published by the Free Software Foundation; either version
+ * 2 of the License, or (at your option) any later version.
+ */
+
+#ifndef _MISC_CXL_H
+#define _MISC_CXL_H
+
+#ifdef CONFIG_CXL_BASE
+
+#define CXL_IRQ_RANGES 4
+
+struct cxl_irq_ranges {
+ irq_hw_number_t offset[CXL_IRQ_RANGES];
+ irq_hw_number_t range[CXL_IRQ_RANGES];
+};
+
+extern atomic_t cxl_use_count;
+
+static inline bool cxl_ctx_in_use(void)
+{
+ return (atomic_read(&cxl_use_count) != 0);
+}
+
+static inline void cxl_ctx_get(void)
+{
+ atomic_inc(&cxl_use_count);
+}
+
+static inline void cxl_ctx_put(void)
+{
+ atomic_dec(&cxl_use_count);
+}
+
+void cxl_slbia(struct mm_struct *mm);
+
+#else /* CONFIG_CXL_BASE */
+
+static inline bool cxl_ctx_in_use(void) { return false; }
+static inline void cxl_slbia(struct mm_struct *mm) {}
+
+#endif /* CONFIG_CXL_BASE */
+
+#endif
--
1.9.1
^ permalink raw reply related
* [PATCH v4 08/16] powerpc/powerpc: Add new PCIe functions for allocating cxl interrupts
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This adds a number of functions for allocating IRQs under powernv PCIe for cxl.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/include/asm/pnv-pci.h | 31 ++++++
arch/powerpc/platforms/powernv/pci-ioda.c | 154 ++++++++++++++++++++++++++++++
2 files changed, 185 insertions(+)
create mode 100644 arch/powerpc/include/asm/pnv-pci.h
diff --git a/arch/powerpc/include/asm/pnv-pci.h b/arch/powerpc/include/asm/pnv-pci.h
new file mode 100644
index 0000000..f09a22f
--- /dev/null
+++ b/arch/powerpc/include/asm/pnv-pci.h
@@ -0,0 +1,31 @@
+/*
+ * Copyright 2014 IBM Corp.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License
+ * as published by the Free Software Foundation; either version
+ * 2 of the License, or (at your option) any later version.
+ */
+
+#ifndef _ASM_PNV_PCI_H
+#define _ASM_PNV_PCI_H
+
+#include <linux/pci.h>
+#include <misc/cxl.h>
+
+int pnv_phb_to_cxl(struct pci_dev *dev);
+int pnv_cxl_ioda_msi_setup(struct pci_dev *dev, unsigned int hwirq,
+ unsigned int virq);
+int pnv_cxl_alloc_hwirqs(struct pci_dev *dev, int num);
+void pnv_cxl_release_hwirqs(struct pci_dev *dev, int hwirq, int num);
+int pnv_cxl_get_irq_count(struct pci_dev *dev);
+struct device_node *pnv_pci_to_phb_node(struct pci_dev *dev);
+
+#ifdef CONFIG_CXL_BASE
+int pnv_cxl_alloc_hwirq_ranges(struct cxl_irq_ranges *irqs,
+ struct pci_dev *dev, int num);
+void pnv_cxl_release_hwirq_ranges(struct cxl_irq_ranges *irqs,
+ struct pci_dev *dev);
+#endif
+
+#endif
diff --git a/arch/powerpc/platforms/powernv/pci-ioda.c b/arch/powerpc/platforms/powernv/pci-ioda.c
index baf3de6..2dfc857 100644
--- a/arch/powerpc/platforms/powernv/pci-ioda.c
+++ b/arch/powerpc/platforms/powernv/pci-ioda.c
@@ -37,6 +37,9 @@
#include <asm/xics.h>
#include <asm/debug.h>
#include <asm/firmware.h>
+#include <asm/pnv-pci.h>
+
+#include <misc/cxl.h>
#include "powernv.h"
#include "pci.h"
@@ -1329,6 +1332,157 @@ static void set_msi_irq_chip(struct pnv_phb *phb, unsigned int virq)
irq_set_chip(virq, &phb->ioda.irq_chip);
}
+#ifdef CONFIG_CXL_BASE
+
+struct device_node *pnv_pci_to_phb_node(struct pci_dev *dev)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+
+ return hose->dn;
+}
+EXPORT_SYMBOL(pnv_pci_to_phb_node);
+
+int pnv_phb_to_cxl(struct pci_dev *dev)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+ struct pnv_phb *phb = hose->private_data;
+ struct pnv_ioda_pe *pe;
+ int rc;
+
+ pe = pnv_ioda_get_pe(dev);
+ if (!pe)
+ return -ENODEV;
+
+ pe_info(pe, "Switching PHB to CXL\n");
+
+ rc = opal_pci_set_phb_cxl_mode(phb->opal_id, 1, pe->pe_number);
+ if (rc)
+ dev_err(&dev->dev, "opal_pci_set_phb_cxl_mode failed: %i\n", rc);
+
+ return rc;
+}
+EXPORT_SYMBOL(pnv_phb_to_cxl);
+
+/* Find PHB for cxl dev and allocate MSI hwirqs?
+ * Returns the absolute hardware IRQ number
+ */
+int pnv_cxl_alloc_hwirqs(struct pci_dev *dev, int num)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+ struct pnv_phb *phb = hose->private_data;
+ int hwirq = msi_bitmap_alloc_hwirqs(&phb->msi_bmp, num);
+
+ if (hwirq < 0) {
+ dev_warn(&dev->dev, "Failed to find a free MSI\n");
+ return -ENOSPC;
+ }
+
+ return phb->msi_base + hwirq;
+}
+EXPORT_SYMBOL(pnv_cxl_alloc_hwirqs);
+
+void pnv_cxl_release_hwirqs(struct pci_dev *dev, int hwirq, int num)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+ struct pnv_phb *phb = hose->private_data;
+
+ msi_bitmap_free_hwirqs(&phb->msi_bmp, hwirq - phb->msi_base, num);
+}
+EXPORT_SYMBOL(pnv_cxl_release_hwirqs);
+
+void pnv_cxl_release_hwirq_ranges(struct cxl_irq_ranges *irqs,
+ struct pci_dev *dev)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+ struct pnv_phb *phb = hose->private_data;
+ int i, hwirq;
+
+ for (i = 1; i < CXL_IRQ_RANGES; i++) {
+ if (!irqs->range[i])
+ continue;
+ pr_devel("cxl release irq range 0x%x: offset: 0x%lx limit: %ld\n",
+ i, irqs->offset[i],
+ irqs->range[i]);
+ hwirq = irqs->offset[i] - phb->msi_base;
+ msi_bitmap_free_hwirqs(&phb->msi_bmp, hwirq,
+ irqs->range[i]);
+ }
+}
+EXPORT_SYMBOL(pnv_cxl_release_hwirq_ranges);
+
+int pnv_cxl_alloc_hwirq_ranges(struct cxl_irq_ranges *irqs,
+ struct pci_dev *dev, int num)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+ struct pnv_phb *phb = hose->private_data;
+ int i, hwirq, try;
+
+ memset(irqs, 0, sizeof(struct cxl_irq_ranges));
+
+ /* 0 is reserved for the multiplexed PSL DSI interrupt */
+ for (i = 1; i < CXL_IRQ_RANGES && num; i++) {
+ try = num;
+ while (try) {
+ hwirq = msi_bitmap_alloc_hwirqs(&phb->msi_bmp, try);
+ if (hwirq >= 0)
+ break;
+ try /= 2;
+ }
+ if (!try)
+ goto fail;
+
+ irqs->offset[i] = phb->msi_base + hwirq;
+ irqs->range[i] = try;
+ pr_devel("cxl alloc irq range 0x%x: offset: 0x%lx limit: %li\n",
+ i, irqs->offset[i], irqs->range[i]);
+ num -= try;
+ }
+ if (num)
+ goto fail;
+
+ return 0;
+fail:
+ pnv_cxl_release_hwirq_ranges(irqs, dev);
+ return -ENOSPC;
+}
+EXPORT_SYMBOL(pnv_cxl_alloc_hwirq_ranges);
+
+int pnv_cxl_get_irq_count(struct pci_dev *dev)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+ struct pnv_phb *phb = hose->private_data;
+
+ return phb->msi_bmp.irq_count;
+}
+EXPORT_SYMBOL(pnv_cxl_get_irq_count);
+
+int pnv_cxl_ioda_msi_setup(struct pci_dev *dev, unsigned int hwirq,
+ unsigned int virq)
+{
+ struct pci_controller *hose = pci_bus_to_host(dev->bus);
+ struct pnv_phb *phb = hose->private_data;
+ unsigned int xive_num = hwirq - phb->msi_base;
+ struct pnv_ioda_pe *pe;
+ int rc;
+
+ if (!(pe = pnv_ioda_get_pe(dev)))
+ return -ENODEV;
+
+ /* Assign XIVE to PE */
+ rc = opal_pci_set_xive_pe(phb->opal_id, pe->pe_number, xive_num);
+ if (rc) {
+ pe_warn(pe, "%s: OPAL error %d setting msi_base 0x%x "
+ "hwirq 0x%x XIVE 0x%x PE\n",
+ pci_name(dev), rc, phb->msi_base, hwirq, xive_num);
+ return -EIO;
+ }
+ set_msi_irq_chip(phb, virq);
+
+ return 0;
+}
+EXPORT_SYMBOL(pnv_cxl_ioda_msi_setup);
+#endif
+
static int pnv_pci_ioda_msi_setup(struct pnv_phb *phb, struct pci_dev *dev,
unsigned int hwirq, unsigned int virq,
unsigned int is_64, struct msi_msg *msg)
--
1.9.1
^ permalink raw reply related
* [PATCH v4 09/16] powerpc/mm: Add new hash_page_mm()
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This adds a new function hash_page_mm() based on the existing hash_page().
This version allows any struct mm to be passed in, rather than assuming
current. This is useful for servicing co-processor faults which are not in the
context of the current running process.
We need to be careful here as the current hash_page() assumes current in a few
places.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/include/asm/mmu-hash64.h | 1 +
arch/powerpc/mm/hash_utils_64.c | 24 +++++++++++++++++-------
2 files changed, 18 insertions(+), 7 deletions(-)
diff --git a/arch/powerpc/include/asm/mmu-hash64.h b/arch/powerpc/include/asm/mmu-hash64.h
index aeabd02..764e141 100644
--- a/arch/powerpc/include/asm/mmu-hash64.h
+++ b/arch/powerpc/include/asm/mmu-hash64.h
@@ -324,6 +324,7 @@ extern int __hash_page_64K(unsigned long ea, unsigned long access,
unsigned int local, int ssize);
struct mm_struct;
unsigned int hash_page_do_lazy_icache(unsigned int pp, pte_t pte, int trap);
+extern int hash_page_mm(struct mm_struct *mm, unsigned long ea, unsigned long access, unsigned long trap);
extern int hash_page(unsigned long ea, unsigned long access, unsigned long trap);
int __hash_page_huge(unsigned long ea, unsigned long access, unsigned long vsid,
pte_t *ptep, unsigned long trap, int local, int ssize,
diff --git a/arch/powerpc/mm/hash_utils_64.c b/arch/powerpc/mm/hash_utils_64.c
index bbdb054..698834d 100644
--- a/arch/powerpc/mm/hash_utils_64.c
+++ b/arch/powerpc/mm/hash_utils_64.c
@@ -904,7 +904,7 @@ void demote_segment_4k(struct mm_struct *mm, unsigned long addr)
return;
slice_set_range_psize(mm, addr, 1, MMU_PAGE_4K);
copro_flush_all_slbs(mm);
- if (get_paca_psize(addr) != MMU_PAGE_4K) {
+ if ((get_paca_psize(addr) != MMU_PAGE_4K) && (current->mm == mm)) {
get_paca()->context = mm->context;
slb_flush_and_rebolt();
}
@@ -989,12 +989,11 @@ static void check_paca_psize(unsigned long ea, struct mm_struct *mm,
* -1 - critical hash insertion error
* -2 - access not permitted by subpage protection mechanism
*/
-int hash_page(unsigned long ea, unsigned long access, unsigned long trap)
+int hash_page_mm(struct mm_struct *mm, unsigned long ea, unsigned long access, unsigned long trap)
{
enum ctx_state prev_state = exception_enter();
pgd_t *pgdir;
unsigned long vsid;
- struct mm_struct *mm;
pte_t *ptep;
unsigned hugeshift;
const struct cpumask *tmp;
@@ -1008,7 +1007,6 @@ int hash_page(unsigned long ea, unsigned long access, unsigned long trap)
switch (REGION_ID(ea)) {
case USER_REGION_ID:
user_region = 1;
- mm = current->mm;
if (! mm) {
DBG_LOW(" user region with no mm !\n");
rc = 1;
@@ -1019,7 +1017,6 @@ int hash_page(unsigned long ea, unsigned long access, unsigned long trap)
vsid = get_vsid(mm->context.id, ea, ssize);
break;
case VMALLOC_REGION_ID:
- mm = &init_mm;
vsid = get_kernel_vsid(ea, mmu_kernel_ssize);
if (ea < VMALLOC_END)
psize = mmu_vmalloc_psize;
@@ -1104,7 +1101,8 @@ int hash_page(unsigned long ea, unsigned long access, unsigned long trap)
WARN_ON(1);
}
#endif
- check_paca_psize(ea, mm, psize, user_region);
+ if (current->mm == mm)
+ check_paca_psize(ea, mm, psize, user_region);
goto bail;
}
@@ -1145,7 +1143,8 @@ int hash_page(unsigned long ea, unsigned long access, unsigned long trap)
}
}
- check_paca_psize(ea, mm, psize, user_region);
+ if (current->mm == mm)
+ check_paca_psize(ea, mm, psize, user_region);
#endif /* CONFIG_PPC_64K_PAGES */
#ifdef CONFIG_PPC_HAS_HASH_64K
@@ -1180,6 +1179,17 @@ bail:
exception_exit(prev_state);
return rc;
}
+EXPORT_SYMBOL_GPL(hash_page_mm);
+
+int hash_page(unsigned long ea, unsigned long access, unsigned long trap)
+{
+ struct mm_struct *mm = current->mm;
+
+ if (REGION_ID(ea) == VMALLOC_REGION_ID)
+ mm = &init_mm;
+
+ return hash_page_mm(mm, ea, access, trap);
+}
EXPORT_SYMBOL_GPL(hash_page);
void hash_preload(struct mm_struct *mm, unsigned long ea,
--
1.9.1
^ permalink raw reply related
* [PATCH v4 10/16] powerpc/opal: Add PHB to cxl mode call
From: Michael Neuling @ 2014-10-08 8:54 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This adds the OPAL call to change a PHB into cxl mode.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/include/asm/opal.h | 2 ++
arch/powerpc/platforms/powernv/opal-wrappers.S | 1 +
2 files changed, 3 insertions(+)
diff --git a/arch/powerpc/include/asm/opal.h b/arch/powerpc/include/asm/opal.h
index 86055e5..84c37c4dbc 100644
--- a/arch/powerpc/include/asm/opal.h
+++ b/arch/powerpc/include/asm/opal.h
@@ -146,6 +146,7 @@ struct opal_sg_list {
#define OPAL_GET_PARAM 89
#define OPAL_SET_PARAM 90
#define OPAL_DUMP_RESEND 91
+#define OPAL_PCI_SET_PHB_CXL_MODE 93
#define OPAL_DUMP_INFO2 94
#define OPAL_PCI_EEH_FREEZE_SET 97
#define OPAL_HANDLE_HMI 98
@@ -924,6 +925,7 @@ int64_t opal_sensor_read(uint32_t sensor_hndl, int token, __be32 *sensor_data);
int64_t opal_handle_hmi(void);
int64_t opal_register_dump_region(uint32_t id, uint64_t start, uint64_t end);
int64_t opal_unregister_dump_region(uint32_t id);
+int64_t opal_pci_set_phb_cxl_mode(uint64_t phb_id, uint64_t mode, uint64_t pe_number);
/* Internal functions */
extern int early_init_dt_scan_opal(unsigned long node, const char *uname,
diff --git a/arch/powerpc/platforms/powernv/opal-wrappers.S b/arch/powerpc/platforms/powernv/opal-wrappers.S
index 2e6ce1b..0fb56dc 100644
--- a/arch/powerpc/platforms/powernv/opal-wrappers.S
+++ b/arch/powerpc/platforms/powernv/opal-wrappers.S
@@ -247,3 +247,4 @@ OPAL_CALL(opal_set_param, OPAL_SET_PARAM);
OPAL_CALL(opal_handle_hmi, OPAL_HANDLE_HMI);
OPAL_CALL(opal_register_dump_region, OPAL_REGISTER_DUMP_REGION);
OPAL_CALL(opal_unregister_dump_region, OPAL_UNREGISTER_DUMP_REGION);
+OPAL_CALL(opal_pci_set_phb_cxl_mode, OPAL_PCI_SET_PHB_CXL_MODE);
--
1.9.1
^ permalink raw reply related
* [PATCH v4 11/16] powerpc/mm: Add hooks for cxl
From: Michael Neuling @ 2014-10-08 8:55 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This adds hooks into the core powerpc mm code for cxl.
The core powerpc code sometimes uses local tlbie. Unfortunately this won't
work with the current cxl driver as it relies on snooping tlbie broadcasts.
The cxl hardware can have TLB entries invalidated via MMIO but this is not
currently supported by the driver. In future we can make local tlbie smarter so
that it invalidates cxl contexts via MMIO when it needs to but for now we have
this workaround.
This workaround checks for any active cxl contexts and if so, disables local
tlbie.
This also adds a hook for when SLBs are invalidated. This ensures any
corresponding SLBs in cxl are also invalidated at the same time. This is
required for segment demotion.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
arch/powerpc/mm/copro_fault.c | 2 ++
arch/powerpc/mm/hash_native_64.c | 6 +++++-
2 files changed, 7 insertions(+), 1 deletion(-)
diff --git a/arch/powerpc/mm/copro_fault.c b/arch/powerpc/mm/copro_fault.c
index f2aa5a8..0f9939e 100644
--- a/arch/powerpc/mm/copro_fault.c
+++ b/arch/powerpc/mm/copro_fault.c
@@ -26,6 +26,7 @@
#include <asm/reg.h>
#include <asm/copro.h>
#include <asm/spu.h>
+#include <misc/cxl.h>
/*
* This ought to be kept in sync with the powerpc specific do_page_fault
@@ -143,5 +144,6 @@ void copro_flush_all_slbs(struct mm_struct *mm)
#ifdef CONFIG_SPU_BASE
spu_flush_all_slbs(mm);
#endif
+ cxl_slbia(mm);
}
EXPORT_SYMBOL_GPL(copro_flush_all_slbs);
diff --git a/arch/powerpc/mm/hash_native_64.c b/arch/powerpc/mm/hash_native_64.c
index afc0a82..ae4962a 100644
--- a/arch/powerpc/mm/hash_native_64.c
+++ b/arch/powerpc/mm/hash_native_64.c
@@ -29,6 +29,8 @@
#include <asm/kexec.h>
#include <asm/ppc-opcode.h>
+#include <misc/cxl.h>
+
#ifdef DEBUG_LOW
#define DBG_LOW(fmt...) udbg_printf(fmt)
#else
@@ -149,9 +151,11 @@ static inline void __tlbiel(unsigned long vpn, int psize, int apsize, int ssize)
static inline void tlbie(unsigned long vpn, int psize, int apsize,
int ssize, int local)
{
- unsigned int use_local = local && mmu_has_feature(MMU_FTR_TLBIEL);
+ unsigned int use_local;
int lock_tlbie = !mmu_has_feature(MMU_FTR_LOCKLESS_TLBIE);
+ use_local = local && mmu_has_feature(MMU_FTR_TLBIEL) && !cxl_ctx_in_use();
+
if (use_local)
use_local = mmu_psize_defs[psize].tlbiel;
if (lock_tlbie && !use_local)
--
1.9.1
^ permalink raw reply related
* [PATCH v4 12/16] cxl: Add base builtin support
From: Michael Neuling @ 2014-10-08 8:55 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This adds the base cxl support that cannot be built as a module. Specifically
it adds the cxl callbacks that are called from the core powerpc mm code which
must always exist irrespective of if the cxl module is loaded or not. This is
similar to how cell works with CONFIG_SPU_BASE.
This adds a cxl_slbia() call (similar to spu_flush_all_slbs()) which checks if
the cxl module is loaded and in use, returning immediately if it is not. If it
is in use it calls into the cxl SLB invalidation code.
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
drivers/misc/Kconfig | 1 +
drivers/misc/Makefile | 1 +
drivers/misc/cxl/Kconfig | 8 +++++
drivers/misc/cxl/Makefile | 1 +
drivers/misc/cxl/base.c | 86 +++++++++++++++++++++++++++++++++++++++++++++++
5 files changed, 97 insertions(+)
create mode 100644 drivers/misc/cxl/Kconfig
create mode 100644 drivers/misc/cxl/Makefile
create mode 100644 drivers/misc/cxl/base.c
diff --git a/drivers/misc/Kconfig b/drivers/misc/Kconfig
index b841180..bbeb451 100644
--- a/drivers/misc/Kconfig
+++ b/drivers/misc/Kconfig
@@ -527,4 +527,5 @@ source "drivers/misc/vmw_vmci/Kconfig"
source "drivers/misc/mic/Kconfig"
source "drivers/misc/genwqe/Kconfig"
source "drivers/misc/echo/Kconfig"
+source "drivers/misc/cxl/Kconfig"
endmenu
diff --git a/drivers/misc/Makefile b/drivers/misc/Makefile
index 5497d02..7d5c4cd 100644
--- a/drivers/misc/Makefile
+++ b/drivers/misc/Makefile
@@ -55,3 +55,4 @@ obj-y += mic/
obj-$(CONFIG_GENWQE) += genwqe/
obj-$(CONFIG_ECHO) += echo/
obj-$(CONFIG_VEXPRESS_SYSCFG) += vexpress-syscfg.o
+obj-$(CONFIG_CXL_BASE) += cxl/
diff --git a/drivers/misc/cxl/Kconfig b/drivers/misc/cxl/Kconfig
new file mode 100644
index 0000000..5cdd319
--- /dev/null
+++ b/drivers/misc/cxl/Kconfig
@@ -0,0 +1,8 @@
+#
+# IBM Coherent Accelerator (CXL) compatible devices
+#
+
+config CXL_BASE
+ bool
+ default n
+ select PPC_COPRO_BASE
diff --git a/drivers/misc/cxl/Makefile b/drivers/misc/cxl/Makefile
new file mode 100644
index 0000000..e30ad0a
--- /dev/null
+++ b/drivers/misc/cxl/Makefile
@@ -0,0 +1 @@
+obj-$(CONFIG_CXL_BASE) += base.o
diff --git a/drivers/misc/cxl/base.c b/drivers/misc/cxl/base.c
new file mode 100644
index 0000000..0654ad8
--- /dev/null
+++ b/drivers/misc/cxl/base.c
@@ -0,0 +1,86 @@
+/*
+ * Copyright 2014 IBM Corp.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License
+ * as published by the Free Software Foundation; either version
+ * 2 of the License, or (at your option) any later version.
+ */
+
+#include <linux/module.h>
+#include <linux/rcupdate.h>
+#include <asm/errno.h>
+#include <misc/cxl.h>
+#include "cxl.h"
+
+/* protected by rcu */
+static struct cxl_calls *cxl_calls;
+
+atomic_t cxl_use_count = ATOMIC_INIT(0);
+EXPORT_SYMBOL(cxl_use_count);
+
+#ifdef CONFIG_CXL_MODULE
+
+static inline struct cxl_calls *cxl_calls_get(void)
+{
+ struct cxl_calls *calls = NULL;
+
+ rcu_read_lock();
+ calls = rcu_dereference(cxl_calls);
+ if (calls && !try_module_get(calls->owner))
+ calls = NULL;
+ rcu_read_unlock();
+
+ return calls;
+}
+
+static inline void cxl_calls_put(struct cxl_calls *calls)
+{
+ BUG_ON(calls != cxl_calls);
+
+ /* we don't need to rcu this, as we hold a reference to the module */
+ module_put(cxl_calls->owner);
+}
+
+#else /* !defined CONFIG_CXL_MODULE */
+
+static inline struct cxl_calls *cxl_calls_get(void)
+{
+ return cxl_calls;
+}
+
+static inline void cxl_calls_put(struct cxl_calls *calls) { }
+
+#endif /* CONFIG_CXL_MODULE */
+
+void cxl_slbia(struct mm_struct *mm)
+{
+ struct cxl_calls *calls;
+
+ calls = cxl_calls_get();
+ if (!calls)
+ return;
+
+ if (cxl_ctx_in_use())
+ calls->cxl_slbia(mm);
+
+ cxl_calls_put(calls);
+}
+
+int register_cxl_calls(struct cxl_calls *calls)
+{
+ if (cxl_calls)
+ return -EBUSY;
+
+ rcu_assign_pointer(cxl_calls, calls);
+ return 0;
+}
+EXPORT_SYMBOL_GPL(register_cxl_calls);
+
+void unregister_cxl_calls(struct cxl_calls *calls)
+{
+ BUG_ON(cxl_calls->owner != calls->owner);
+ RCU_INIT_POINTER(cxl_calls, NULL);
+ synchronize_rcu();
+}
+EXPORT_SYMBOL_GPL(unregister_cxl_calls);
--
1.9.1
^ permalink raw reply related
* [PATCH v4 14/16] cxl: Add userspace header file
From: Michael Neuling @ 2014-10-08 8:55 UTC (permalink / raw)
To: greg, arnd, mpe, benh
Cc: cbe-oss-dev, mikey, Aneesh Kumar K.V, imunsie, linux-kernel,
linuxppc-dev, jk, anton
In-Reply-To: <1412758505-23495-1-git-send-email-mikey@neuling.org>
From: Ian Munsie <imunsie@au1.ibm.com>
This adds a header file for use by userspace programs wanting to interact with
the kernel cxl driver. It defines structs and magic numbers required for
userspace to interact with devices in /dev/cxl/afuM.N.
Further documentation on this interface is added in a subsequent patch in
Documentation/powerpc/cxl.txt.
It also adds this new userspace header file to Kbuild so it's exported when
doing "make headers_installs".
Signed-off-by: Ian Munsie <imunsie@au1.ibm.com>
Signed-off-by: Michael Neuling <mikey@neuling.org>
---
include/uapi/Kbuild | 1 +
include/uapi/misc/Kbuild | 2 ++
include/uapi/misc/cxl.h | 87 ++++++++++++++++++++++++++++++++++++++++++++++++
3 files changed, 90 insertions(+)
create mode 100644 include/uapi/misc/Kbuild
create mode 100644 include/uapi/misc/cxl.h
diff --git a/include/uapi/Kbuild b/include/uapi/Kbuild
index 81d2106..245aa6e 100644
--- a/include/uapi/Kbuild
+++ b/include/uapi/Kbuild
@@ -12,3 +12,4 @@ header-y += video/
header-y += drm/
header-y += xen/
header-y += scsi/
+header-y += misc/
diff --git a/include/uapi/misc/Kbuild b/include/uapi/misc/Kbuild
new file mode 100644
index 0000000..e96cae7
--- /dev/null
+++ b/include/uapi/misc/Kbuild
@@ -0,0 +1,2 @@
+# misc Header export list
+header-y += cxl.h
diff --git a/include/uapi/misc/cxl.h b/include/uapi/misc/cxl.h
new file mode 100644
index 0000000..c232be6
--- /dev/null
+++ b/include/uapi/misc/cxl.h
@@ -0,0 +1,87 @@
+/*
+ * Copyright 2014 IBM Corp.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License
+ * as published by the Free Software Foundation; either version
+ * 2 of the License, or (at your option) any later version.
+ */
+
+#ifndef _UAPI_MISC_CXL_H
+#define _UAPI_MISC_CXL_H
+
+#include <linux/types.h>
+#include <linux/ioctl.h>
+
+/* Structs for IOCTLS for userspace to talk to the kernel */
+struct cxl_ioctl_start_work {
+ __u64 flags;
+ __u64 work_element_descriptor;
+ __u64 amr;
+ __s16 num_interrupts;
+ __s16 reserved1;
+ __s32 reserved2;
+ __u64 reserved3;
+ __u64 reserved4;
+ __u64 reserved5;
+ __u64 reserved6;
+};
+#define CXL_START_WORK_AMR 0x0000000000000001ULL
+#define CXL_START_WORK_NUM_IRQS 0x0000000000000002ULL
+#define CXL_START_WORK_ALL (CXL_START_WORK_AMR |\
+ CXL_START_WORK_NUM_IRQS)
+
+/* IOCTL numbers */
+#define CXL_MAGIC 0xCA
+#define CXL_IOCTL_START_WORK _IOW(CXL_MAGIC, 0x00, struct cxl_ioctl_start_work)
+#define CXL_IOCTL_GET_PROCESS_ELEMENT _IOR(CXL_MAGIC, 0x01, __u32)
+
+/* Events from read() */
+#define CXL_READ_MIN_SIZE 0x1000 /* 4K */
+
+enum cxl_event_type {
+ CXL_EVENT_RESERVED = 0,
+ CXL_EVENT_AFU_INTERRUPT = 1,
+ CXL_EVENT_DATA_STORAGE = 2,
+ CXL_EVENT_AFU_ERROR = 3,
+};
+
+struct cxl_event_header {
+ __u16 type;
+ __u16 size;
+ __u16 process_element;
+ __u16 reserved1;
+};
+
+struct cxl_event_afu_interrupt {
+ __u16 flags;
+ __u16 irq; /* Raised AFU interrupt number */
+ __u32 reserved1;
+};
+
+struct cxl_event_data_storage {
+ __u16 flags;
+ __u16 reserved1;
+ __u32 reserved2;
+ __u64 addr;
+ __u64 dsisr;
+ __u64 reserved3;
+};
+
+struct cxl_event_afu_error {
+ __u16 flags;
+ __u16 reserved1;
+ __u32 reserved2;
+ __u64 error;
+};
+
+struct cxl_event {
+ struct cxl_event_header header;
+ union {
+ struct cxl_event_afu_interrupt irq;
+ struct cxl_event_data_storage fault;
+ struct cxl_event_afu_error afu_error;
+ };
+};
+
+#endif /* _UAPI_MISC_CXL_H */
--
1.9.1
^ permalink raw reply related
page: next (older) | prev (newer) | latest
- recent:[subjects (threaded)|topics (new)|topics (active)]
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox