[PATCH v3 2/3] ACPI: APEI: GHES: Add NVIDIA Vera decoder
Kai-Heng Feng <[email protected]> Tue, 4 Aug 2026 20:23:16 +0800
| Newsgroups | org.kernel.vger.linux-acpi,org.kernel.vger.linux-hardening,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
Vera is the next-generation NVIDIA server SoC. Its CPER section uses a different GUID and binary layout from Grace, so firmware-reported hardware errors need a dedicated decoder. Follow the NVIDIA event-section CtxSize contract. Require CtxSize to be 16-byte aligned, validate that each context covers its header and data, and advance by CtxSize so alignment padding is skipped. Preserve the legacy format-0 header-only CtxSize used by existing opaque firmware records. Assert all three Vera wire-header sizes and keep the decoded object on the stack. Print at most 32 key/value entries across the entire section, report omitted entries, and retain successfully decoded contexts when a later context is malformed. Keep detailed fatal decode errors unratelimited while rate-limiting those diagnostics for non-fatal records. Signed-off-by: Kai-Heng Feng <[email protected]> --- v3: - Validate and traverse CtxSize according to the firmware contract and libcper implementation; accept aligned padding. - Preserve the legacy format-0 header-only CtxSize encoding. - Add wire-layout assertions and use Vera key/value terminology. - Bound entry output to 32 entries per section and report omissions. - Print partial Vera contexts when decode fails mid-section. - Keep fatal diagnostics unratelimited and non-fatal diagnostics ratelimited. --- drivers/acpi/apei/Kconfig | 4 +- drivers/acpi/apei/ghes-nvidia.c | 665 +++++++++++++++++++++++++++++++- drivers/acpi/apei/ghes-nvidia.h | 69 +++- 3 files changed, 710 insertions(+), 28 deletions(-) diff --git a/drivers/acpi/apei/Kconfig b/drivers/acpi/apei/Kconfig index 428458c623f0..2dcea74c1677 100644 --- a/drivers/acpi/apei/Kconfig +++ b/drivers/acpi/apei/Kconfig @@ -80,8 +80,8 @@ config ACPI_APEI_GHES_NVIDIA help Support for decoding NVIDIA-specific CPER sections delivered via the APEI GHES vendor record notifier chain. Registers a handler - for the NVIDIA section GUID and logs error signatures, severity, - socket, and diagnostic register address-value pairs. + for the NVIDIA Grace and Vera section GUIDs and logs decoded + metadata and per-context key/value entries (bounded). Enable on NVIDIA server platforms (e.g. DGX, HGX) that expose ACPI device NVDA2012 in their firmware tables. diff --git a/drivers/acpi/apei/ghes-nvidia.c b/drivers/acpi/apei/ghes-nvidia.c index e7cc912344fd..adf298dc722b 100644 --- a/drivers/acpi/apei/ghes-nvidia.c +++ b/drivers/acpi/apei/ghes-nvidia.c @@ -6,20 +6,54 @@ */ #include <linux/acpi.h> +#include <linux/align.h> +#include <linux/limits.h> #include <linux/module.h> +#include <linux/overflow.h> #include <linux/platform_device.h> #include <linux/types.h> #include <linux/unaligned.h> #include <linux/uuid.h> +#include <kunit/visibility.h> #include <acpi/ghes.h> -#include <kunit/visibility.h> #include "ghes-nvidia.h" +#define NVIDIA_GHES_VERA_VERSION 1 +#define NVIDIA_GHES_VERA_CPU_INFO_MAJOR 0 +/* Cap section-wide entry log lines so a large dump cannot flood the console. */ +#define NVIDIA_GHES_MAX_PRINT_ENTRIES 32 +#define NVIDIA_GHES_MAX_PRINT_CONTEXTS 16 + +static __printf(3, 4) void nvidia_ghes_decode_err(struct device *dev, + bool fatal, const char *fmt, ...) +{ + struct va_format vaf; + va_list args; + + if (!dev) + return; + + va_start(args, fmt); + vaf.fmt = fmt; + vaf.va = &args; + + if (fatal) + dev_err(dev, "%pV", &vaf); + else + dev_err_ratelimited(dev, "%pV", &vaf); + + va_end(args); +} + static const guid_t nvidia_grace_sec_guid = GUID_INIT(0x6d5244f2, 0x2712, 0x11ec, 0xbe, 0xa7, 0xcb, 0x3f, 0xdb, 0x95, 0xc7, 0x86); +static const guid_t nvidia_vera_sec_guid = + GUID_INIT(0x9068e568, 0x6ca0, 0x11f0, + 0xae, 0xaf, 0x15, 0x93, 0x43, 0x59, 0x1e, 0xac); + /* Grace CPER section wire layout (header without flexible register array). */ struct cper_sec_nvidia { char signature[16]; @@ -35,15 +69,61 @@ struct cper_sec_nvidia { static_assert(sizeof(struct cper_sec_nvidia) == 32); +struct cper_sec_nvidia_vera_event { + u8 version; + u8 event_context_count; + u8 source_device_type; + u8 reserved; + __le16 event_type; + __le16 event_sub_type; + __le64 event_link_id; + char source_module_signature[16]; +} __packed; + +static_assert(sizeof(struct cper_sec_nvidia_vera_event) == 32); + +struct cper_sec_nvidia_vera_cpu_info { + __le16 info_version; + u8 info_size; + u8 socket_number; + __le32 architecture; + u8 chip_serial_number[16]; + __le64 instance_base; +} __packed; + +static_assert(sizeof(struct cper_sec_nvidia_vera_cpu_info) == 32); + +struct cper_sec_nvidia_vera_context { + __le32 context_size; + __le16 context_version; + __le16 reserved; + __le16 data_format_type; + __le16 data_format_version; + __le32 data_size; +} __packed; + +static_assert(sizeof(struct cper_sec_nvidia_vera_context) == 16); + struct nvidia_ghes_private { struct notifier_block nb; struct device *dev; }; VISIBLE_IF_KUNIT -int nvidia_ghes_decode_grace(struct device *dev, const void *buf, - size_t len, - struct nvidia_ghes_decoded *decoded) +enum nvidia_ghes_format nvidia_ghes_format_from_guid(const guid_t *guid) +{ + if (guid_equal(guid, &nvidia_grace_sec_guid)) + return NVIDIA_GHES_FORMAT_GRACE; + if (guid_equal(guid, &nvidia_vera_sec_guid)) + return NVIDIA_GHES_FORMAT_VERA; + return NVIDIA_GHES_FORMAT_UNKNOWN; +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_format_from_guid); + +static int __nvidia_ghes_decode_grace(struct device *dev, const void *buf, + size_t len, + struct nvidia_ghes_decoded *decoded, + bool fatal) { const struct cper_sec_nvidia *nvidia_err = buf; size_t min_size; @@ -52,19 +132,18 @@ int nvidia_ghes_decode_grace(struct device *dev, const void *buf, if (!buf || !decoded) return -EINVAL; if (len < sizeof(*nvidia_err)) { - if (dev) - dev_err_ratelimited(dev, "Section too small (%zu < %zu)\n", - len, sizeof(*nvidia_err)); + nvidia_ghes_decode_err(dev, fatal, + "Section too small (%zu < %zu)\n", + len, sizeof(*nvidia_err)); return -ENODATA; } number_regs = nvidia_err->number_regs; min_size = struct_size(nvidia_err, regs, number_regs); if (len < min_size) { - if (dev) - dev_err_ratelimited(dev, - "Invalid number_regs %u (section size %zu, need %zu)\n", - number_regs, len, min_size); + nvidia_ghes_decode_err(dev, fatal, + "Invalid number_regs %u (section size %zu, need %zu)\n", + number_regs, len, min_size); return -ENODATA; } @@ -83,6 +162,14 @@ int nvidia_ghes_decode_grace(struct device *dev, const void *buf, return 0; } + +VISIBLE_IF_KUNIT +__maybe_unused int nvidia_ghes_decode_grace(struct device *dev, const void *buf, + size_t len, + struct nvidia_ghes_decoded *decoded) +{ + return __nvidia_ghes_decode_grace(dev, buf, len, decoded, false); +} EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_decode_grace); VISIBLE_IF_KUNIT @@ -106,6 +193,361 @@ int nvidia_ghes_grace_reg_pair(const struct nvidia_ghes_decoded *decoded, } EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_grace_reg_pair); +static int nvidia_ghes_vera_validate_context_data(u16 data_format_type, + u32 data_size) +{ + switch (data_format_type) { + case 0: + return 0; + case 1: + return data_size % 16 ? -EINVAL : 0; + case 2: + case 3: + return data_size % 8 ? -EINVAL : 0; + case 4: + return data_size % 4 ? -EINVAL : 0; + default: + return -EOPNOTSUPP; + } +} + +static void +__nvidia_ghes_vera_cursor_init(const struct nvidia_ghes_decoded *decoded, + struct nvidia_ghes_vera_cursor *cursor, u8 count) +{ + *cursor = (struct nvidia_ghes_vera_cursor) { + .pos = decoded->vera_contexts, + .remaining = decoded->vera_contexts_len, + .contexts_left = count, + }; +} + +VISIBLE_IF_KUNIT +void nvidia_ghes_vera_cursor_init(const struct nvidia_ghes_decoded *decoded, + struct nvidia_ghes_vera_cursor *cursor) +{ + if (!cursor) + return; + + memset(cursor, 0, sizeof(*cursor)); + if (!decoded || decoded->format != NVIDIA_GHES_FORMAT_VERA) + return; + + __nvidia_ghes_vera_cursor_init(decoded, cursor, + decoded->valid_context_count); +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_cursor_init); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_cursor_next(struct nvidia_ghes_vera_cursor *cursor, + struct nvidia_ghes_vera_context *ctx) +{ + const struct cper_sec_nvidia_vera_context *wire; + size_t data_end_advance; + size_t advance; + u32 context_size; + u32 data_size; + u16 data_format_type; + int ret; + + if (!cursor || !ctx) + return -EINVAL; + if (!cursor->contexts_left) + return 0; + if (!cursor->pos || cursor->remaining < sizeof(*wire)) + return -ENODATA; + + wire = (const void *)cursor->pos; + context_size = get_unaligned_le32(&wire->context_size); + data_format_type = get_unaligned_le16(&wire->data_format_type); + data_size = get_unaligned_le32(&wire->data_size); + + if (context_size < sizeof(*wire) || !IS_ALIGNED(context_size, 16)) + return -EINVAL; + + ret = nvidia_ghes_vera_validate_context_data(data_format_type, data_size); + if (ret) + return ret; + if (check_add_overflow((size_t)data_size, sizeof(*wire), + &data_end_advance)) + return -EOVERFLOW; + if (data_end_advance > cursor->remaining) + return -ENODATA; + + /* + * CtxSize covers the header, data, and alignment padding. Older opaque + * records used a header-only CtxSize, so advance over their data directly. + */ + if (data_format_type == 0 && context_size == sizeof(*wire)) { + advance = data_end_advance; + } else { + if (data_end_advance > context_size) + return -EINVAL; + if (context_size > cursor->remaining) + return -ENODATA; + advance = context_size; + } + + *ctx = (struct nvidia_ghes_vera_context) { + .context_size = context_size, + .context_version = get_unaligned_le16(&wire->context_version), + .data_format_type = data_format_type, + .data_format_version = get_unaligned_le16(&wire->data_format_version), + .data_size = data_size, + .data = cursor->pos + sizeof(*wire), + }; + cursor->pos += advance; + cursor->remaining -= advance; + cursor->contexts_left--; + + return 1; +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_cursor_next); + +static int __nvidia_ghes_decode_vera(struct device *dev, const void *buf, + size_t len, + struct nvidia_ghes_decoded *decoded, + bool fatal) +{ + const struct cper_sec_nvidia_vera_event *event = buf; + const struct cper_sec_nvidia_vera_cpu_info *cpu_info; + struct nvidia_ghes_vera_cursor cursor; + struct nvidia_ghes_vera_context context; + const u8 *bytes = buf; + size_t offset; + u16 info_version; + int ret; + + if (!buf || !decoded) + return -EINVAL; + if (len < sizeof(*event)) { + nvidia_ghes_decode_err(dev, fatal, + "Vera event header truncated (%zu < %zu)\n", + len, sizeof(*event)); + return -ENODATA; + } + if (event->version != NVIDIA_GHES_VERA_VERSION) + return -EOPNOTSUPP; + if (event->source_device_type != 0) + return -EOPNOTSUPP; + + offset = sizeof(*event); + if (len - offset < sizeof(*cpu_info)) { + nvidia_ghes_decode_err(dev, fatal, + "Vera CPU info truncated (%zu < %zu)\n", + len - offset, sizeof(*cpu_info)); + return -ENODATA; + } + + cpu_info = (const void *)(bytes + offset); + info_version = get_unaligned_le16(&cpu_info->info_version); + if ((info_version >> 8) != NVIDIA_GHES_VERA_CPU_INFO_MAJOR) + return -EOPNOTSUPP; + if (cpu_info->info_size < sizeof(*cpu_info)) { + nvidia_ghes_decode_err(dev, fatal, + "Vera CPU info size %u smaller than header %zu\n", + cpu_info->info_size, sizeof(*cpu_info)); + return -EINVAL; + } + if (len - offset < cpu_info->info_size) { + nvidia_ghes_decode_err(dev, fatal, + "Vera CPU info extends past section (%u > %zu)\n", + cpu_info->info_size, len - offset); + return -ENODATA; + } + + offset += cpu_info->info_size; + + memset(decoded, 0, sizeof(*decoded)); + decoded->format = NVIDIA_GHES_FORMAT_VERA; + memcpy(decoded->signature, event->source_module_signature, + sizeof(event->source_module_signature)); + decoded->signature[sizeof(event->source_module_signature)] = '\0'; + decoded->event_context_count = event->event_context_count; + decoded->source_device_type = event->source_device_type; + decoded->event_type = get_unaligned_le16(&event->event_type); + decoded->event_sub_type = get_unaligned_le16(&event->event_sub_type); + decoded->event_link_id = get_unaligned_le64(&event->event_link_id); + decoded->socket = cpu_info->socket_number; + decoded->architecture = get_unaligned_le32(&cpu_info->architecture); + memcpy(decoded->chip_serial_number, cpu_info->chip_serial_number, + sizeof(cpu_info->chip_serial_number)); + decoded->instance_base = get_unaligned_le64(&cpu_info->instance_base); + decoded->vera_contexts = bytes + offset; + decoded->vera_contexts_len = len - offset; + __nvidia_ghes_vera_cursor_init(decoded, &cursor, + event->event_context_count); + while ((ret = nvidia_ghes_vera_cursor_next(&cursor, &context)) > 0) + decoded->valid_context_count++; + if (ret < 0) { + nvidia_ghes_decode_err(dev, fatal, + "Vera context[%u] is malformed (ret=%d)\n", + decoded->valid_context_count, ret); + return ret; + } + + if (cursor.remaining && dev) + dev_dbg(dev, + "Vera section has %zu trailing byte(s) after %u context(s)\n", + cursor.remaining, decoded->valid_context_count); + + return 0; +} + +VISIBLE_IF_KUNIT +__maybe_unused int nvidia_ghes_decode_vera(struct device *dev, const void *buf, + size_t len, + struct nvidia_ghes_decoded *decoded) +{ + return __nvidia_ghes_decode_vera(dev, buf, len, decoded, false); +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_decode_vera); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_entry_count(const struct nvidia_ghes_vera_context *ctx) +{ + if (!ctx) + return -EINVAL; + if (ctx->data_size && !ctx->data) + return -EINVAL; + if (ctx->data_size > INT_MAX) + return -EOVERFLOW; + + switch (ctx->data_format_type) { + case 0: + return 0; + case 1: + return ctx->data_size / 16; + case 2: + return ctx->data_size / 8; + case 3: + return ctx->data_size / 8; + case 4: + return ctx->data_size / 4; + default: + return -EINVAL; + } +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_context_entry_count); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u64_pair(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u64 *key, u64 *val) +{ + int count; + + if (!ctx || !key || !val || ctx->data_format_type != 1) + return -EINVAL; + + count = nvidia_ghes_vera_context_entry_count(ctx); + if (count < 0) + return count; + if (index >= count) + return -ERANGE; + + *key = get_unaligned_le64(ctx->data + index * 16); + *val = get_unaligned_le64(ctx->data + index * 16 + 8); + + return 0; +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_context_u64_pair); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u32_pair(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u32 *key, u32 *val) +{ + int count; + + if (!ctx || !key || !val || ctx->data_format_type != 2) + return -EINVAL; + + count = nvidia_ghes_vera_context_entry_count(ctx); + if (count < 0) + return count; + if (index >= count) + return -ERANGE; + + *key = get_unaligned_le32(ctx->data + index * 8); + *val = get_unaligned_le32(ctx->data + index * 8 + 4); + + return 0; +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_context_u32_pair); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u64_value(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u64 *val) +{ + int count; + + if (!ctx || !val || ctx->data_format_type != 3) + return -EINVAL; + + count = nvidia_ghes_vera_context_entry_count(ctx); + if (count < 0) + return count; + if (index >= count) + return -ERANGE; + + *val = get_unaligned_le64(ctx->data + index * 8); + + return 0; +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_context_u64_value); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u32_value(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u32 *val) +{ + int count; + + if (!ctx || !val || ctx->data_format_type != 4) + return -EINVAL; + + count = nvidia_ghes_vera_context_entry_count(ctx); + if (count < 0) + return count; + if (index >= count) + return -ERANGE; + + *val = get_unaligned_le32(ctx->data + index * 4); + + return 0; +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_context_u32_value); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_print_count(const struct nvidia_ghes_vera_context *ctx, + unsigned int *remaining) +{ + int entries; + int count; + + if (!remaining) + return -EINVAL; + + entries = nvidia_ghes_vera_context_entry_count(ctx); + if (entries < 0) + return entries; + + count = min_t(unsigned int, entries, *remaining); + *remaining -= count; + + return count; +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_context_print_count); + +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_print_context_count(const struct nvidia_ghes_decoded *decoded) +{ + if (!decoded || decoded->format != NVIDIA_GHES_FORMAT_VERA) + return -EINVAL; + + return min_t(unsigned int, decoded->valid_context_count, + NVIDIA_GHES_MAX_PRINT_CONTEXTS); +} +EXPORT_SYMBOL_IF_KUNIT(nvidia_ghes_vera_print_context_count); + static void nvidia_ghes_print_grace(struct device *dev, const struct nvidia_ghes_decoded *decoded, bool fatal) @@ -113,13 +555,16 @@ static void nvidia_ghes_print_grace(struct device *dev, const char *level = fatal ? KERN_ERR : KERN_INFO; u64 addr, val; - dev_printk(level, dev, "signature: %s\n", decoded->signature); + dev_printk(level, dev, "signature: %*pE\n", + (int)strnlen(decoded->signature, sizeof(decoded->signature)), + decoded->signature); dev_printk(level, dev, "error_type: %u\n", decoded->error_type); dev_printk(level, dev, "error_instance: %u\n", decoded->error_instance); dev_printk(level, dev, "severity: %u\n", decoded->severity); dev_printk(level, dev, "socket: %u\n", decoded->socket); dev_printk(level, dev, "number_regs: %u\n", decoded->number_regs); - dev_printk(level, dev, "instance_base: 0x%016llx\n", decoded->instance_base); + dev_printk(level, dev, "instance_base: 0x%016llx\n", + decoded->instance_base); for (int i = 0; i < decoded->number_regs; i++) { if (nvidia_ghes_grace_reg_pair(decoded, i, &addr, &val)) @@ -129,12 +574,157 @@ static void nvidia_ghes_print_grace(struct device *dev, } } +static void nvidia_ghes_print_vera(struct device *dev, + const struct nvidia_ghes_decoded *decoded, + bool fatal, unsigned long ghes_severity) +{ + struct nvidia_ghes_vera_cursor cursor; + struct nvidia_ghes_vera_context context; + const char *level = fatal ? KERN_ERR : KERN_INFO; + unsigned int remaining = NVIDIA_GHES_MAX_PRINT_ENTRIES; + unsigned int print_contexts; + + dev_printk(level, dev, "signature: %*pE\n", + (int)strnlen(decoded->signature, sizeof(decoded->signature)), + decoded->signature); + dev_printk(level, dev, "event_type: %u\n", decoded->event_type); + dev_printk(level, dev, "event_sub_type: %u\n", decoded->event_sub_type); + dev_printk(level, dev, "ghes_severity: %lu\n", ghes_severity); + dev_printk(level, dev, "event_link_id: 0x%016llx\n", + decoded->event_link_id); + dev_printk(level, dev, "socket: %u\n", decoded->socket); + dev_printk(level, dev, "architecture: 0x%x\n", decoded->architecture); + dev_printk(level, dev, "chip_serial_number: %*phN\n", + (int)sizeof(decoded->chip_serial_number), + decoded->chip_serial_number); + dev_printk(level, dev, "instance_base: 0x%016llx\n", decoded->instance_base); + dev_printk(level, dev, "event_context_count: %u\n", decoded->event_context_count); + if (decoded->valid_context_count != decoded->event_context_count) + dev_printk(level, dev, "valid_context_count: %u\n", + decoded->valid_context_count); + + print_contexts = nvidia_ghes_vera_print_context_count(decoded); + nvidia_ghes_vera_cursor_init(decoded, &cursor); + for (int i = 0; i < print_contexts; i++) { + const struct nvidia_ghes_vera_context *ctx = &context; + int entries; + int print_n; + + if (nvidia_ghes_vera_cursor_next(&cursor, &context) != 1) + break; + entries = nvidia_ghes_vera_context_entry_count(ctx); + dev_printk(level, dev, + "context[%d]: version=%u format=%u format_version=%u context_size=%u data_size=%u\n", + i, ctx->context_version, ctx->data_format_type, + ctx->data_format_version, ctx->context_size, ctx->data_size); + if (ctx->data_format_type == 0 && ctx->data_size > 0) { + int prefix_len = ctx->data_size > 16 ? 16 : ctx->data_size; + + dev_printk(level, dev, "context[%d]_opaque_prefix: %*phN\n", + i, prefix_len, ctx->data); + continue; + } + if (entries < 0) + continue; + + dev_printk(level, dev, "context[%d]_entries: %d\n", i, entries); + print_n = nvidia_ghes_vera_context_print_count(ctx, &remaining); + if (print_n < 0) + continue; + for (int j = 0; j < print_n; j++) { + u64 key64, val64; + u32 key32, val32; + + switch (ctx->data_format_type) { + case 1: + if (nvidia_ghes_vera_context_u64_pair(ctx, j, &key64, &val64)) + break; + dev_printk(level, dev, + "context[%d]_entry[%d]: key=0x%016llx value=0x%016llx\n", + i, j, key64, val64); + break; + case 2: + if (nvidia_ghes_vera_context_u32_pair(ctx, j, &key32, &val32)) + break; + dev_printk(level, dev, + "context[%d]_entry[%d]: key=0x%08x value=0x%08x\n", + i, j, key32, val32); + break; + case 3: + if (nvidia_ghes_vera_context_u64_value(ctx, j, &val64)) + break; + dev_printk(level, dev, + "context[%d]_entry[%d]: value=0x%016llx\n", + i, j, val64); + break; + case 4: + if (nvidia_ghes_vera_context_u32_value(ctx, j, &val32)) + break; + dev_printk(level, dev, + "context[%d]_entry[%d]: value=0x%08x\n", + i, j, val32); + break; + default: + break; + } + } + if (entries > print_n) + dev_printk(level, dev, + "context[%d]_entries_omitted: %d\n", + i, entries - print_n); + } + if (decoded->valid_context_count > print_contexts) + dev_printk(level, dev, "contexts_omitted: %u\n", + decoded->valid_context_count - print_contexts); +} + +static void nvidia_ghes_log_decode_failure(struct device *dev, + enum nvidia_ghes_format format, + bool fatal, u32 len, int ret) +{ + const char *level = fatal ? KERN_ERR : KERN_INFO; + + if (ret == -EOPNOTSUPP && format == NVIDIA_GHES_FORMAT_VERA) { + if (fatal) + dev_printk(level, dev, + "Unsupported NVIDIA Vera CPER section, error_data_length: %u, ret: %d\n", + len, ret); + else + dev_info_ratelimited(dev, + "Unsupported NVIDIA Vera CPER section, error_data_length: %u, ret: %d\n", + len, ret); + return; + } + + if (format == NVIDIA_GHES_FORMAT_GRACE) { + if (fatal) + dev_printk(level, dev, + "Malformed NVIDIA Grace CPER section, error_data_length: %u, ret: %d\n", + len, ret); + else + dev_err_ratelimited(dev, + "Malformed NVIDIA Grace CPER section, error_data_length: %u, ret: %d\n", + len, ret); + return; + } + + if (fatal) + dev_printk(level, dev, + "Malformed NVIDIA Vera CPER section, error_data_length: %u, ret: %d\n", + len, ret); + else + dev_err_ratelimited(dev, + "Malformed NVIDIA Vera CPER section, error_data_length: %u, ret: %d\n", + len, ret); +} + static int nvidia_ghes_notify(struct notifier_block *nb, unsigned long event, void *data) { struct acpi_hest_generic_data *gdata = data; struct nvidia_ghes_decoded decoded = {}; struct nvidia_ghes_private *priv; + enum nvidia_ghes_format format; const void *payload; guid_t sec_guid; u32 len; @@ -142,25 +732,58 @@ static int nvidia_ghes_notify(struct notifier_block *nb, bool fatal; import_guid(&sec_guid, gdata->section_type); - if (!guid_equal(&sec_guid, &nvidia_grace_sec_guid)) + format = nvidia_ghes_format_from_guid(&sec_guid); + if (format == NVIDIA_GHES_FORMAT_UNKNOWN) return NOTIFY_DONE; priv = container_of(nb, struct nvidia_ghes_private, nb); len = acpi_hest_get_error_length(gdata); + payload = acpi_hest_get_payload(gdata); fatal = event >= GHES_SEV_RECOVERABLE; - ret = nvidia_ghes_decode_grace(priv->dev, payload, len, &decoded); + switch (format) { + case NVIDIA_GHES_FORMAT_GRACE: + ret = __nvidia_ghes_decode_grace(priv->dev, payload, len, + &decoded, fatal); + break; + case NVIDIA_GHES_FORMAT_VERA: + ret = __nvidia_ghes_decode_vera(priv->dev, payload, len, + &decoded, fatal); + break; + default: + ret = -EOPNOTSUPP; + break; + } + if (ret) { - dev_err_ratelimited(priv->dev, - "Malformed NVIDIA CPER section, error_data_length: %u, ret: %d\n", - len, ret); + nvidia_ghes_log_decode_failure(priv->dev, format, fatal, len, ret); + /* + * Print any contexts successfully decoded before the failure. + * Raw section bytes remain available via log_non_standard_event(). + */ + if (format == NVIDIA_GHES_FORMAT_VERA && + decoded.format == NVIDIA_GHES_FORMAT_VERA && + decoded.valid_context_count > 0) { + dev_printk(fatal ? KERN_ERR : KERN_INFO, priv->dev, + "NVIDIA Vera CPER section (partial), error_data_length: %u\n", + len); + nvidia_ghes_print_vera(priv->dev, &decoded, fatal, event); + } return NOTIFY_OK; } - dev_printk(fatal ? KERN_ERR : KERN_INFO, priv->dev, - "NVIDIA CPER section, error_data_length: %u\n", len); - nvidia_ghes_print_grace(priv->dev, &decoded, fatal); + if (format == NVIDIA_GHES_FORMAT_GRACE) + dev_printk(fatal ? KERN_ERR : KERN_INFO, priv->dev, + "NVIDIA Grace CPER section, error_data_length: %u\n", len); + else + dev_printk(fatal ? KERN_ERR : KERN_INFO, priv->dev, + "NVIDIA Vera CPER section, error_data_length: %u\n", len); + + if (format == NVIDIA_GHES_FORMAT_VERA) + nvidia_ghes_print_vera(priv->dev, &decoded, fatal, event); + else + nvidia_ghes_print_grace(priv->dev, &decoded, fatal); return NOTIFY_OK; } diff --git a/drivers/acpi/apei/ghes-nvidia.h b/drivers/acpi/apei/ghes-nvidia.h index 965abe3d5c49..b5f0312499fc 100644 --- a/drivers/acpi/apei/ghes-nvidia.h +++ b/drivers/acpi/apei/ghes-nvidia.h @@ -4,13 +4,13 @@ #include <linux/build_bug.h> #include <linux/types.h> +#include <linux/uuid.h> #include <kunit/visibility.h> -struct device; - enum nvidia_ghes_format { NVIDIA_GHES_FORMAT_UNKNOWN, NVIDIA_GHES_FORMAT_GRACE, + NVIDIA_GHES_FORMAT_VERA, }; struct nvidia_ghes_grace_reg { @@ -18,18 +18,52 @@ struct nvidia_ghes_grace_reg { __le64 val; } __packed; +struct nvidia_ghes_vera_context { + u32 context_size; + u16 context_version; + u16 data_format_type; + u16 data_format_version; + u32 data_size; + const u8 *data; +}; + +struct nvidia_ghes_vera_cursor { + const u8 *pos; + size_t remaining; + u8 contexts_left; +}; + struct nvidia_ghes_decoded { enum nvidia_ghes_format format; char signature[17]; u16 error_type; u16 error_instance; + u16 event_type; + u16 event_sub_type; u8 severity; u8 socket; u8 number_regs; + u8 source_device_type; + u8 event_context_count; + u8 valid_context_count; + u32 architecture; + u64 event_link_id; u64 instance_base; + u8 chip_serial_number[16]; const struct nvidia_ghes_grace_reg *grace_regs; + const u8 *vera_contexts; + size_t vera_contexts_len; }; +/* Bound the notifier stack frame. */ +static_assert(sizeof(struct nvidia_ghes_decoded) <= 128); + +struct device; + +VISIBLE_IF_KUNIT enum nvidia_ghes_format nvidia_ghes_format_from_guid(const guid_t *guid); +VISIBLE_IF_KUNIT int nvidia_ghes_decode_grace(struct device *dev, const void *buf, + size_t len, + struct nvidia_ghes_decoded *decoded); /** * nvidia_ghes_grace_reg_pair() - Read one Grace register address/value pair * @decoded: Decoded Grace section; format must be NVIDIA_GHES_FORMAT_GRACE @@ -41,10 +75,35 @@ struct nvidia_ghes_decoded { * Returns -EINVAL for bad arguments / missing grace_regs, -ERANGE for * index >= number_regs, and 0 on success. */ -VISIBLE_IF_KUNIT int nvidia_ghes_decode_grace(struct device *dev, const void *buf, - size_t len, - struct nvidia_ghes_decoded *decoded); VISIBLE_IF_KUNIT int nvidia_ghes_grace_reg_pair(const struct nvidia_ghes_decoded *decoded, unsigned int index, u64 *addr, u64 *val); +VISIBLE_IF_KUNIT int nvidia_ghes_decode_vera(struct device *dev, const void *buf, + size_t len, + struct nvidia_ghes_decoded *decoded); +VISIBLE_IF_KUNIT +void nvidia_ghes_vera_cursor_init(const struct nvidia_ghes_decoded *decoded, + struct nvidia_ghes_vera_cursor *cursor); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_cursor_next(struct nvidia_ghes_vera_cursor *cursor, + struct nvidia_ghes_vera_context *ctx); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_print_context_count(const struct nvidia_ghes_decoded *decoded); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_entry_count(const struct nvidia_ghes_vera_context *ctx); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u64_pair(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u64 *key, u64 *val); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u32_pair(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u32 *key, u32 *val); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u64_value(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u64 *val); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_u32_value(const struct nvidia_ghes_vera_context *ctx, + unsigned int index, u32 *val); +VISIBLE_IF_KUNIT +int nvidia_ghes_vera_context_print_count(const struct nvidia_ghes_vera_context *ctx, + unsigned int *remaining); #endif -- 2.50.1 (Apple Git-155)