From: Jacob Shin <jacob.shin@amd.com>
To: Borislav Petkov <bp@alien8.de>,
Doug Thompson <dougthompson@xmission.com>,
<linux-edac@vger.kernel.org>, <linux-kernel@vger.kernel.org>
Subject: Re: [PATCH 2/2] MCE, AMD: MCE decoding support for AMD Family 16h
Date: Tue, 18 Dec 2012 11:30:24 -0600 [thread overview]
Message-ID: <20121218173024.GA18732@jshin-Toonie> (raw)
In-Reply-To: <20121218171915.GE31255@liondog.tnic>
On Tue, Dec 18, 2012 at 06:19:15PM +0100, Borislav Petkov wrote:
> On Mon, Dec 17, 2012 at 01:39:48PM -0600, Jacob Shin wrote:
> > Add MCE decoding logic for AMD Family 16h processors.
> >
> > Signed-off-by: Jacob Shin <jacob.shin@amd.com>
> > ---
> > drivers/edac/mce_amd.c | 120 ++++++++++++++++++++++++++++++++++++++++++++++--
> > drivers/edac/mce_amd.h | 6 +++
> > 2 files changed, 122 insertions(+), 4 deletions(-)
> >
> > diff --git a/drivers/edac/mce_amd.c b/drivers/edac/mce_amd.c
> > index 84320f9..7d2d037 100644
> > --- a/drivers/edac/mce_amd.c
> > +++ b/drivers/edac/mce_amd.c
> > @@ -64,6 +64,10 @@ EXPORT_SYMBOL_GPL(to_msgs);
> > const char * const ii_msgs[] = { "MEM", "RESV", "IO", "GEN" };
> > EXPORT_SYMBOL_GPL(ii_msgs);
> >
> > +/* internal error type */
> > +const char * const uu_msgs[] = { "RESV", "RESV", "HWA", "RESV" };
> > +EXPORT_SYMBOL_GPL(uu_msgs);
>
> Seems like those aren't used anywhere?
>
> > static const char * const f15h_mc1_mce_desc[] = {
> > "UC during a demand linefill from L2",
> > "Parity error during data load from IC",
> > @@ -275,6 +279,23 @@ static bool f15h_mc0_mce(u16 ec, u8 xec)
> > return ret;
> > }
> >
> > +static bool f16h_mc0_mce(u16 ec, u8 xec)
> > +{
> > + u8 r4 = R4(ec);
> > +
> > + if (MEM_ERROR(ec) && TT(ec) == TT_DATA && LL(ec) == LL_L1 &&
> > + (r4 == R4_DRD || r4 == R4_DWR)) {
> > +
> > + pr_cont("%s parity error due to %s.\n",
> > + (xec == 0x0 ? "Data" : "Tag"),
> > + (r4 == R4_DRD ? "load" : "store"));
> > +
> > + return true;
> > + }
> > +
> > + return f14h_mc0_mce(ec, xec);
>
> Looks like this could be merged with f14h_mc0_mce no? You can call the
> function then cat_mc0_mce (for all the *cat cores) and assign it to
> fam_ops->mc0_mce in the f14h and f16h case.
Okay
>
> > +}
> > +
> > static void decode_mc0_mce(struct mce *m)
> > {
> > u16 ec = EC(m->status);
> > @@ -379,6 +400,36 @@ static bool f15h_mc1_mce(u16 ec, u8 xec)
> > return ret;
> > }
> >
> > +static bool f16h_mc1_mce(u16 ec, u8 xec)
> > +{
> > + u8 r4 = R4(ec);
> > + bool ret = true;
> > +
> > + if (MEM_ERROR(ec)) {
> > + if (TT(ec) != TT_INSTR)
> > + ret = false;
> > +
> > + else if (r4 == R4_IRD)
> > + pr_cont("%s array parity error for a tag hit.\n",
> > + (xec == 0x0 ? "Data" : "Tag"));
> > +
> > + else if (r4 == R4_SNOOP)
> > + pr_cont("Tag error during snoop/victimization.\n");
> > +
> > + else if (xec == 0x0)
> > + pr_cont("Tag parity error from victim castout.\n");
> > +
> > + else if (xec == 0x2)
> > + pr_cont("Microcode patch RAM parity error.\n");
>
> Also no need for a family-special function - just rename f14h_mc1_mce
> to cat_mc1_mce() as above and add a special case like this as the last
> else-branch of the if conditional there:
Okay
>
> + if (boot_cpu_data.x86 == 0x16) {
> + if (LL(ec) == LL_LG && xec == 2)
> + pr_cont("Microcode patch RAM parity error.\n");
> + else
> + pr_cont("IC Tag parity error from victim castout.\n");
> + return true;
> + }
>
> > +
> > + else
> > + ret = false;
> > + } else
> > + ret = false;
> > +
> > + return ret;
> > +}
> > +
> > static void decode_mc1_mce(struct mce *m)
> > {
> > u16 ec = EC(m->status);
> > @@ -469,6 +520,48 @@ static bool f15h_mc2_mce(u16 ec, u8 xec)
> > return ret;
> > }
> >
> > +static bool f16h_mc2_mce(u16 ec, u8 xec)
> > +{
> > + u8 r4 = R4(ec);
> > + bool ret = true;
> > +
> > + if (MEM_ERROR(ec) && TT(ec) == TT_GEN && LL(ec) == LL_L2) {
>
> You can exit early here:
>
> if (!MEM_ERROR(ec))
> return false;
>
> Also, no need to test for TT and LL - we're relying on the hardware here
> and if those values are b0rked then we have a more serious problem.
Okay
>
> > + switch (xec) {
> > + case 0x04 ... 0x05:
> > + pr_cont("Parity error in %s.\n",
> > + (r4 == R4_RD ? "IBUFF" : "OBUFF"));
>
> or
> pr_cont("%sBUFF parity error.\n", ((xec == 4) ? "I" : "O"));
>
> > + break;
> > +
> > + case 0x09 ... 0x0b:
> > + case 0x0d ... 0x0f:
> > + pr_cont("ECC error in L2 tag (%s).\n",
> > + (r4 == R4_GEN ? "BankReq" :
> > + (r4 == R4_SNOOP ? "Probe" : "Fill")));
> > + break;
> > +
> > + case 0x10 ... 0x1b:
> > + pr_cont("ECC error in L2 data array (%s).\n",
> > + (r4 == R4_RD ? "Hit" :
> > + (r4 == R4_GEN ? "Attr" :
> > + (r4 == R4_EVICT ? "Vict" : "Fill"))));
> > + break;
> > +
> > + case 0x1c ... 0x1f:
> > + pr_cont("Parity error in L2 attribute bits (%s).\n",
> > + (r4 == R4_RD ? "Hit" :
> > + (r4 == R4_GEN ? "Attr" : "Fill")));
> > + break;
> > +
> > + default:
> > + ret = false;
> > + break;
> > + }
> > + } else
> > + ret = false;
> > +
> > + return ret;
> > +}
> > +
> > static void decode_mc2_mce(struct mce *m)
> > {
> > u16 ec = EC(m->status);
> > @@ -548,7 +641,7 @@ static void decode_mc4_mce(struct mce *m)
> > return;
> >
> > case 0x19:
> > - if (boot_cpu_data.x86 == 0x15)
> > + if (boot_cpu_data.x86 == 0x15 || boot_cpu_data.x86 == 0x16)
> > pr_cont("Compute Unit Data Error.\n");
> > else
> > goto wrong_mc4_mce;
> > @@ -634,6 +727,10 @@ static void decode_mc6_mce(struct mce *m)
> >
> > static inline void amd_decode_err_code(u16 ec)
> > {
> > + if (INT_ERROR(ec)) {
> > + pr_emerg(HW_ERR "internal: %s\n", LL_MSG(ec));
> > + return;
> > + }
>
> Is this correct? I'm just confirming because I don't have the internal
> info anymore.
>
> Uuh, hold on, maybe those otherwise unused uu_msgs above were meant to
> be used here instead of the LL_MSG? IOW,
>
> pr_emerg(HW_ERR "internal: %s\n", UU_MSG(ec));
>
> Right?
Ah, yes thats right, sorry about the typo. It looks like:
Error Code Error Code Type Description
0000 01UU 0000 0000 Internal Unclassified UU = Internal Error Type
And the UU encoding is as is in the mce_amd.h file
>
> >
> > pr_emerg(HW_ERR "cache level: %s", LL_MSG(ec));
> >
> > @@ -738,10 +835,18 @@ int amd_decode_mce(struct notifier_block *nb, unsigned long val, void *data)
> > ((m->status & MCI_STATUS_PCC) ? "PCC" : "-"),
> > ((m->status & MCI_STATUS_ADDRV) ? "AddrV" : "-"));
> >
> > - if (c->x86 == 0x15)
> > - pr_cont("|%s|%s",
> > + if (c->x86 == 0x15 || c->x86 == 0x16) {
> > + char coreid[16];
> > +
> > + if (m->status & MCI_STATUS_COREIDV)
> > + sprintf(coreid, "CoreIdV(Core%d)",
> > + (int)ERR_CORE_ID(m->status));
>
> Uuh, no, this is probably dumping the core which detected the error.. No
> need since we're dumping the core reporting the error anyway above. And
> if that's mismatched for some reason, we're also dumping full MCi_STATUS
> contents so you can decypher CoreId from there if needed.
Okay, will take this out.
Thanks for the feedback! I'll spin a V2 and send it out later today.
-Jacob
>
> Thanks.
>
> --
> Regards/Gruss,
> Boris.
>
> Sent from a fat crate under my desk. Formatting is fine.
> --
>
next prev parent reply other threads:[~2012-12-18 17:31 UTC|newest]
Thread overview: 25+ messages / expand[flat|nested] mbox.gz Atom feed top
2012-12-17 19:39 [PATCH 1/2] MCE, AMD: Make MC2 decoding part of amd_decoder_ops as well Jacob Shin
2012-12-17 19:39 ` [PATCH 2/2] MCE, AMD: MCE decoding support for AMD Family 16h Jacob Shin
2012-12-18 17:19 ` Borislav Petkov
2012-12-18 17:30 ` Jacob Shin [this message]
2012-12-18 18:09 ` Jacob Shin
2012-12-18 18:32 ` Borislav Petkov
2012-12-18 18:24 ` Joe Perches
2012-12-18 18:33 ` Borislav Petkov
2012-12-18 18:37 ` Joe Perches
2012-12-18 19:00 ` Mauro Carvalho Chehab
2012-12-17 19:57 ` [PATCH 1/2] MCE, AMD: Make MC2 decoding part of amd_decoder_ops as well Joe Perches
2012-12-17 20:05 ` Borislav Petkov
2012-12-17 20:16 ` Joe Perches
2012-12-17 20:22 ` Borislav Petkov
2012-12-17 20:34 ` Joe Perches
2012-12-17 20:40 ` Borislav Petkov
2012-12-17 20:43 ` Jacob Shin
2012-12-17 20:59 ` Joe Perches
2012-12-17 21:08 ` Borislav Petkov
2012-12-17 21:11 ` Jacob Shin
2012-12-17 21:09 ` Jacob Shin
-- strict thread matches above, loose matches on Subject: below --
2012-12-18 21:06 [PATCH V2 0/2] MCE, AMD: MCE decoding support for AMD Family 16h Jacob Shin
2012-12-18 21:06 ` [PATCH 2/2] " Jacob Shin
2012-12-18 22:10 ` Joe Perches
2012-12-18 22:56 ` Borislav Petkov
2012-12-20 19:20 ` Borislav Petkov
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20121218173024.GA18732@jshin-Toonie \
--to=jacob.shin@amd.com \
--cc=bp@alien8.de \
--cc=dougthompson@xmission.com \
--cc=linux-edac@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox