From 2c5337a24f7f2d02989dfb733c55d6d8c7e90493 Mon Sep 17 00:00:00 2001 From: Joshua Bakita Date: Sun, 29 Oct 2023 13:07:40 -0400 Subject: Update includes to L4T r32.7.4 and drop nvgpu/gk20a.h dependency Also add instructions for updating `include/`. These files are now only needed to build on Linux 4.9-based Tegra platforms. --- include/nvgpu/acr/nvgpu_acr.h | 6 +- include/nvgpu/bug.h | 20 +- include/nvgpu/enabled.h | 9 +- include/nvgpu/gk20a.h | 8 +- include/nvgpu/hw/gk20a/hw_gr_gk20a.h | 61 ++++++ include/nvgpu/hw/gm20b/hw_gr_gm20b.h | 14 +- include/nvgpu/hw/gp106/hw_gr_gp106.h | 4 + include/nvgpu/hw/gp10b/hw_gr_gp10b.h | 4 + include/nvgpu/hw/gv100/hw_gr_gv100.h | 4 + include/nvgpu/hw/gv11b/hw_gr_gv11b.h | 4 + include/nvgpu/log.h | 3 +- include/nvgpu/nvgpu_err.h | 359 +++++++++++++++++++++++++++++++++++ include/nvgpu/nvlink.h | 2 +- include/nvgpu/pmu.h | 20 +- include/nvgpu/pmuif/gpmuif_pg.h | 14 +- include/nvgpu/tsg.h | 4 +- 16 files changed, 523 insertions(+), 13 deletions(-) create mode 100644 include/nvgpu/nvgpu_err.h (limited to 'include/nvgpu') diff --git a/include/nvgpu/acr/nvgpu_acr.h b/include/nvgpu/acr/nvgpu_acr.h index 7a0143e..cdb7bb8 100644 --- a/include/nvgpu/acr/nvgpu_acr.h +++ b/include/nvgpu/acr/nvgpu_acr.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2016-2018, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2016-2021, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -39,7 +39,11 @@ struct hs_acr; struct nvgpu_acr; #define HSBIN_ACR_BL_UCODE_IMAGE "pmu_bl.bin" +#define GM20B_HSBIN_ACR_PROD_UCODE "nv_acr_ucode_prod.bin" +#define GM20B_HSBIN_ACR_DBG_UCODE "nv_acr_ucode_dbg.bin" #define HSBIN_ACR_UCODE_IMAGE "acr_ucode.bin" +#define HSBIN_ACR_PROD_UCODE "acr_ucode_prod.bin" +#define HSBIN_ACR_DBG_UCODE "acr_ucode_dbg.bin" #define HSBIN_ACR_AHESASC_PROD_UCODE "acr_ahesasc_prod_ucode.bin" #define HSBIN_ACR_ASB_PROD_UCODE "acr_asb_prod_ucode.bin" #define HSBIN_ACR_AHESASC_DBG_UCODE "acr_ahesasc_dbg_ucode.bin" diff --git a/include/nvgpu/bug.h b/include/nvgpu/bug.h index 3d139b7..82d641b 100644 --- a/include/nvgpu/bug.h +++ b/include/nvgpu/bug.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2017-2021, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -24,6 +24,24 @@ #ifdef __KERNEL__ #include +/* + * Define an assert macro that code within nvgpu can use. + * + * The goal of this macro is for debugging but what that means varies from OS + * to OS. On Linux wee don't want to BUG() for general driver misbehaving. BUG() + * is a very heavy handed tool - in fact there's probably no where within the + * nvgpu core code where it makes sense to use a BUG() when running under Linux. + * + * However, on QNX (and POSIX) BUG() will just kill the current process. This + * means we can use it for handling bugs in nvgpu. + * + * As a result this macro varies depending on platform. + */ +#define nvgpu_assert(cond) ((void) WARN_ON(!(cond))) +#define nvgpu_do_assert_print(g, fmt, arg...) \ + do { \ + nvgpu_err(g, fmt, ##arg); \ + } while (false) #elif defined(__NVGPU_POSIX__) #include #else diff --git a/include/nvgpu/enabled.h b/include/nvgpu/enabled.h index ef55dad..51e9358 100644 --- a/include/nvgpu/enabled.h +++ b/include/nvgpu/enabled.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2017-2020, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2017-2022, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -85,7 +85,12 @@ struct gk20a; #define NVGPU_MM_USE_PHYSICAL_SG 27 /* WAR for gm20b chips. */ #define NVGPU_MM_FORCE_128K_PMU_VM 28 - +/* SW ERRATA to disable L3 alloc Bit of the physical address. + * Bit number varies between SOCs. + * E.g. 64GB physical RAM support for gv11b requires this SW errata + * to be enabled. + */ +#define NVGPU_DISABLE_L3_SUPPORT 29 /* * Host flags */ diff --git a/include/nvgpu/gk20a.h b/include/nvgpu/gk20a.h index aa95969..19bfaee 100644 --- a/include/nvgpu/gk20a.h +++ b/include/nvgpu/gk20a.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2011-2020, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2011-2022, NVIDIA CORPORATION. All rights reserved. * * GK20A Graphics * @@ -517,6 +517,7 @@ struct gpu_ops { u32 *priv_addr_table, u32 *priv_addr_table_index); u32 (*fecs_ctxsw_mailbox_size)(void); + u32 (*gpc0_gpccs_ctxsw_mailbox_size)(void); int (*init_sw_bundle64)(struct gk20a *g); int (*alloc_global_ctx_buffers)(struct gk20a *g); int (*map_global_ctx_buffers)(struct gk20a *g, @@ -719,7 +720,7 @@ struct gpu_ops { struct ch_state *ch_state); u32 (*intr_0_error_mask)(struct gk20a *g); int (*is_preempt_pending)(struct gk20a *g, u32 id, - unsigned int id_type); + unsigned int id_type, bool preempt_retries_left); void (*init_pbdma_intr_descs)(struct fifo_gk20a *f); int (*reset_enable_hw)(struct gk20a *g); int (*setup_userd)(struct channel_gk20a *c); @@ -1079,6 +1080,7 @@ struct gpu_ops { u32 (*pmu_pg_supported_engines_list)(struct gk20a *g); u32 (*pmu_pg_engines_feature_list)(struct gk20a *g, u32 pg_engine_id); + int (*pmu_process_pg_event)(struct gk20a *g, void *pmumsg); bool (*pmu_is_lpwr_feature_supported)(struct gk20a *g, u32 feature_id); int (*pmu_lpwr_enable_pg)(struct gk20a *g, bool pstate_lock); @@ -1793,6 +1795,8 @@ bool gk20a_check_poweron(struct gk20a *g); int gk20a_prepare_poweroff(struct gk20a *g); int gk20a_finalize_poweron(struct gk20a *g); +int nvgpu_wait_for_stall_interrupts(struct gk20a *g, u32 timeout); +int nvgpu_wait_for_nonstall_interrupts(struct gk20a *g, u32 timeout); void nvgpu_wait_for_deferred_interrupts(struct gk20a *g); struct gk20a * __must_check gk20a_get(struct gk20a *g); diff --git a/include/nvgpu/hw/gk20a/hw_gr_gk20a.h b/include/nvgpu/hw/gk20a/hw_gr_gk20a.h index 826108f..376cc8f 100644 --- a/include/nvgpu/hw/gk20a/hw_gr_gk20a.h +++ b/include/nvgpu/hw/gk20a/hw_gr_gk20a.h @@ -1380,6 +1380,10 @@ static inline u32 gr_gpc0_gpccs_ctxsw_status_1_r(void) { return 0x00502400U; } +static inline u32 gr_gpc0_gpccs_ctxsw_mailbox__size_1_v(void) +{ + return 0x00000010U; +} static inline u32 gr_fecs_ctxsw_idlestate_r(void) { return 0x00409420U; @@ -3804,4 +3808,61 @@ static inline u32 gr_gpcs_tpcs_sm_dbgr_control0_run_trigger_task_f(void) { return 0x40000000U; } + +static inline u32 gr_gpc0_gpccs_falcon_irqstat_r(void) +{ + return 0x00502008U; +} +static inline u32 gr_gpc0_gpccs_falcon_irqmode_r(void) +{ + return 0x0050200cU; +} +static inline u32 gr_gpc0_gpccs_falcon_irqmask_r(void) +{ + return 0x00502018U; +} +static inline u32 gr_gpc0_gpccs_falcon_irqdest_r(void) +{ + return 0x0050201cU; +} +static inline u32 gr_gpc0_gpccs_falcon_debug1_r(void) +{ + return 0x00502090U; +} +static inline u32 gr_gpc0_gpccs_falcon_debuginfo_r(void) +{ + return 0x00502094U; +} +static inline u32 gr_gpc0_gpccs_falcon_engctl_r(void) +{ + return 0x005020a4U; +} +static inline u32 gr_gpc0_gpccs_falcon_curctx_r(void) +{ + return 0x00502050U; +} +static inline u32 gr_gpc0_gpccs_falcon_nxtctx_r(void) +{ + return 0x00502054U; +} +static inline u32 gr_gpc0_gpccs_ctxsw_mailbox_r(u32 i) +{ + return 0x00502800U + i*4U; +} +static inline u32 gr_gpc0_gpccs_falcon_icd_cmd_r(void) +{ + return 0x00502200U; +} +static inline u32 gr_gpc0_gpccs_falcon_icd_cmd_opc_rreg_f(void) +{ + return 0x8U; +} +static inline u32 gr_gpc0_gpccs_falcon_icd_cmd_idx_f(u32 v) +{ + return (v & 0x1fU) << 8U; +} +static inline u32 gr_gpc_gpccs_falcon_icd_rdata_r(void) +{ + return 0x0050220cU; +} #endif diff --git a/include/nvgpu/hw/gm20b/hw_gr_gm20b.h b/include/nvgpu/hw/gm20b/hw_gr_gm20b.h index 5bbb3b9..79ad326 100644 --- a/include/nvgpu/hw/gm20b/hw_gr_gm20b.h +++ b/include/nvgpu/hw/gm20b/hw_gr_gm20b.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2014-2018, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2014-2023, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -1396,6 +1396,10 @@ static inline u32 gr_gpc0_gpccs_ctxsw_status_1_r(void) { return 0x00502400U; } +static inline u32 gr_gpc0_gpccs_ctxsw_mailbox__size_1_v(void) +{ + return 0x00000010U; +} static inline u32 gr_fecs_ctxsw_idlestate_r(void) { return 0x00409420U; @@ -2344,6 +2348,14 @@ static inline u32 gr_gpcs_tpcs_tex_m_dbg2_su_rd_coalesce_en_m(void) { return 0x1U << 4U; } +static inline u32 gr_gpcs_tpcs_tex_m_dbg2_tex_rd_coalesce_en_f(u32 v) +{ + return (v & 0x1U) << 5U; +} +static inline u32 gr_gpcs_tpcs_tex_m_dbg2_tex_rd_coalesce_en_m(void) +{ + return 0x1U << 5U; +} static inline u32 gr_gpccs_falcon_addr_r(void) { return 0x0041a0acU; diff --git a/include/nvgpu/hw/gp106/hw_gr_gp106.h b/include/nvgpu/hw/gp106/hw_gr_gp106.h index 3ebed7e..ac82901 100644 --- a/include/nvgpu/hw/gp106/hw_gr_gp106.h +++ b/include/nvgpu/hw/gp106/hw_gr_gp106.h @@ -1508,6 +1508,10 @@ static inline u32 gr_gpc0_gpccs_ctxsw_status_1_r(void) { return 0x00502400U; } +static inline u32 gr_gpc0_gpccs_ctxsw_mailbox__size_1_v(void) +{ + return 0x00000010U; +} static inline u32 gr_fecs_ctxsw_idlestate_r(void) { return 0x00409420U; diff --git a/include/nvgpu/hw/gp10b/hw_gr_gp10b.h b/include/nvgpu/hw/gp10b/hw_gr_gp10b.h index f7bc4c2..89c6bba 100644 --- a/include/nvgpu/hw/gp10b/hw_gr_gp10b.h +++ b/include/nvgpu/hw/gp10b/hw_gr_gp10b.h @@ -1584,6 +1584,10 @@ static inline u32 gr_gpc0_gpccs_ctxsw_status_1_r(void) { return 0x00502400U; } +static inline u32 gr_gpc0_gpccs_ctxsw_mailbox__size_1_v(void) +{ + return 0x00000010U; +} static inline u32 gr_fecs_ctxsw_idlestate_r(void) { return 0x00409420U; diff --git a/include/nvgpu/hw/gv100/hw_gr_gv100.h b/include/nvgpu/hw/gv100/hw_gr_gv100.h index 0f83d6b..3955a63 100644 --- a/include/nvgpu/hw/gv100/hw_gr_gv100.h +++ b/include/nvgpu/hw/gv100/hw_gr_gv100.h @@ -1816,6 +1816,10 @@ static inline u32 gr_gpc0_gpccs_ctxsw_status_1_r(void) { return 0x00502400U; } +static inline u32 gr_gpc0_gpccs_ctxsw_mailbox__size_1_v(void) +{ + return 0x00000010U; +} static inline u32 gr_fecs_ctxsw_idlestate_r(void) { return 0x00409420U; diff --git a/include/nvgpu/hw/gv11b/hw_gr_gv11b.h b/include/nvgpu/hw/gv11b/hw_gr_gv11b.h index f7d8089..4a3da79 100644 --- a/include/nvgpu/hw/gv11b/hw_gr_gv11b.h +++ b/include/nvgpu/hw/gv11b/hw_gr_gv11b.h @@ -2420,6 +2420,10 @@ static inline u32 gr_gpc0_gpccs_ctxsw_status_1_r(void) { return 0x00502400U; } +static inline u32 gr_gpc0_gpccs_ctxsw_mailbox__size_1_v(void) +{ + return 0x00000010U; +} static inline u32 gr_fecs_ctxsw_idlestate_r(void) { return 0x00409420U; diff --git a/include/nvgpu/log.h b/include/nvgpu/log.h index 70a1676..2bcca33 100644 --- a/include/nvgpu/log.h +++ b/include/nvgpu/log.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2017-2021, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -80,6 +80,7 @@ void __nvgpu_log_dbg(struct gk20a *g, u64 log_mask, #define gpu_dbg_vidmem BIT(24) /* VIDMEM tracing. */ #define gpu_dbg_nvlink BIT(25) /* nvlink Operation tracing. */ #define gpu_dbg_clk_arb BIT(26) /* Clk arbiter debugging. */ +#define gpu_dbg_ecc BIT(27) /* Print ECC Info Logs. */ #define gpu_dbg_mem BIT(31) /* memory accesses; very verbose. */ /** diff --git a/include/nvgpu/nvgpu_err.h b/include/nvgpu/nvgpu_err.h new file mode 100644 index 0000000..0595faf --- /dev/null +++ b/include/nvgpu/nvgpu_err.h @@ -0,0 +1,359 @@ +/* + * Copyright (c) 2021, NVIDIA CORPORATION. All rights reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a + * copy of this software and associated documentation files (the "Software"), + * to deal in the Software without restriction, including without limitation + * the rights to use, copy, modify, merge, publish, distribute, sublicense, + * and/or sell copies of the Software, and to permit persons to whom the + * Software is furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL + * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING + * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER + * DEALINGS IN THE SOFTWARE. + */ + +#ifndef NVGPU_NVGPU_ERR_H +#define NVGPU_NVGPU_ERR_H + +/** + * @file + * + * Define indices for HW units and errors. Define structures used to carry error + * information. Declare prototype for APIs that are used to report GPU HW errors + * to the Safety_Services framework. + */ + +#include +#include + +struct gk20a; + +/** + * @defgroup INDICES_FOR_GPU_HW_UNITS + * Macros used to assign unique index to GPU HW units. + * @{ + */ +#define NVGPU_ERR_MODULE_SM (0U) +#define NVGPU_ERR_MODULE_FECS (1U) +#define NVGPU_ERR_MODULE_PMU (2U) +/** + * @} + */ + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_SM + * Macros used to assign unique index to errors reported from the SM unit. + * @{ + */ +#define GPU_SM_L1_TAG_ECC_CORRECTED (0U) +#define GPU_SM_L1_TAG_ECC_UNCORRECTED (1U) +#define GPU_SM_CBU_ECC_UNCORRECTED (3U) +#define GPU_SM_LRF_ECC_UNCORRECTED (5U) +#define GPU_SM_L1_DATA_ECC_UNCORRECTED (7U) +#define GPU_SM_ICACHE_L0_DATA_ECC_UNCORRECTED (9U) +#define GPU_SM_ICACHE_L1_DATA_ECC_UNCORRECTED (11U) +#define GPU_SM_ICACHE_L0_PREDECODE_ECC_UNCORRECTED (13U) +#define GPU_SM_L1_TAG_MISS_FIFO_ECC_UNCORRECTED (15U) +#define GPU_SM_L1_TAG_S2R_PIXPRF_ECC_UNCORRECTED (17U) +#define GPU_SM_ICACHE_L1_PREDECODE_ECC_UNCORRECTED (20U) +/** + * @} + */ + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_FECS + * Macros used to assign unique index to errors reported from the FECS unit. + * @{ + */ +#define GPU_FECS_FALCON_IMEM_ECC_CORRECTED (0U) +#define GPU_FECS_FALCON_IMEM_ECC_UNCORRECTED (1U) +#define GPU_FECS_FALCON_DMEM_ECC_UNCORRECTED (3U) +/** + * @} + */ + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_GPCCS + * Macros used to assign unique index to errors reported from the GPCCS unit. + * @{ + */ +#define GPU_GPCCS_FALCON_IMEM_ECC_CORRECTED (0U) +#define GPU_GPCCS_FALCON_IMEM_ECC_UNCORRECTED (1U) +#define GPU_GPCCS_FALCON_DMEM_ECC_UNCORRECTED (3U) +/** + * @} + */ + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_MMU + * Macros used to assign unique index to errors reported from the MMU unit. + * @{ + */ +#define GPU_MMU_L1TLB_SA_DATA_ECC_UNCORRECTED (1U) +#define GPU_MMU_L1TLB_FA_DATA_ECC_UNCORRECTED (3U) +/** + * @} + */ + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_GCC + * Macros used to assign unique index to errors reported from the GCC unit. + * @{ + */ +#define GPU_GCC_L15_ECC_UNCORRECTED (1U) +/** + * @} + */ + + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_PMU + * Macros used to assign unique index to errors reported from the PMU unit. + * @{ + */ +#define GPU_PMU_FALCON_IMEM_ECC_CORRECTED (0U) +#define GPU_PMU_FALCON_IMEM_ECC_UNCORRECTED (1U) +#define GPU_PMU_FALCON_DMEM_ECC_UNCORRECTED (3U) +/** + * @} + */ + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_LTC + * Macros used to assign unique index to errors reported from the LTC unit. + * @{ + */ +#define GPU_LTC_CACHE_DSTG_ECC_CORRECTED (0U) +#define GPU_LTC_CACHE_DSTG_ECC_UNCORRECTED (1U) +#define GPU_LTC_CACHE_TSTG_ECC_UNCORRECTED (3U) +#define GPU_LTC_CACHE_DSTG_BE_ECC_UNCORRECTED (7U) +/** + * @} + */ + +/** + * @defgroup LIST_OF_ERRORS_REPORTED_FROM_HUBMMU + * Macros used to assign unique index to errors reported from the HUBMMU unit. + * @{ + */ +#define GPU_HUBMMU_L2TLB_SA_DATA_ECC_UNCORRECTED (1U) +#define GPU_HUBMMU_TLB_SA_DATA_ECC_UNCORRECTED (3U) +#define GPU_HUBMMU_PTE_DATA_ECC_UNCORRECTED (5U) +#define GPU_HUBMMU_PDE0_DATA_ECC_UNCORRECTED (7U) +#define GPU_HUBMMU_PAGE_FAULT_ERROR (8U) + + +#ifdef CONFIG_NVGPU_SUPPORT_LINUX_ECC_ERROR_REPORTING +/** + * @} + */ + +/** + * nvgpu_err_desc structure holds fields which describe an error along with + * function callback which can be used to inject the error. + */ +struct nvgpu_err_desc { + /** String representation of error. */ + const char *name; + + /** Flag to classify an error as critical or non-critical. */ + bool is_critical; + + /** + * Error Threshold: once this threshold value is reached, then the + * corresponding error counter will be reset to 0 and the error will be + * propagated to Safety_Services. + */ + int err_threshold; + + /** + * Total number of times an error has occurred (since its last reset). + */ + nvgpu_atomic_t err_count; + + /** Error ID. */ + u8 error_id; +}; + +/** + * gpu_err_header structure holds fields which are required to identify the + * version of header, sub-error type, sub-unit id, error address and time stamp. + */ +struct gpu_err_header { + /** Version of GPU error header. */ + struct { + /** Major version number. */ + u16 major; + /** Minor version number. */ + u16 minor; + } version; + + /** Sub error type corresponding to the error that is being reported. */ + u32 sub_err_type; + + /** ID of the sub-unit in a HW unit which encountered an error. */ + u64 sub_unit_id; + + /** Location of the error. */ + u64 address; + + /** Timestamp in nano seconds. */ + u64 timestamp_ns; +}; + +struct gpu_ecc_error_info { + struct gpu_err_header header; + + /** Number of ECC errors. */ + u64 err_cnt; +}; + +/** + * nvgpu_err_hw_module structure holds fields which describe the h/w modules + * error reporting capabilities. + */ +struct nvgpu_err_hw_module { + /** String representation of a given HW unit. */ + const char *name; + + /** HW unit ID. */ + u32 hw_unit; + + /** Total number of errors reported from a given HW unit. */ + u32 num_errs; + + u32 base_ecc_service_id; + + /** Used to get error description from look-up table. */ + struct nvgpu_err_desc *errs; +}; + +struct nvgpu_ecc_reporting_ops { + void (*report_ecc_err)(struct gk20a *g, u32 hw_unit, u32 inst, + u32 err_id, u64 err_addr, u64 err_count); +}; + +struct nvgpu_ecc_reporting { + struct nvgpu_spinlock lock; + /* This flag is protected by the above spinlock */ + bool ecc_reporting_service_enabled; + const struct nvgpu_ecc_reporting_ops *ops; +}; + + /** + * This macro is used to initialize the members of nvgpu_err_desc struct. + */ +#define GPU_ERR(err, critical, id, threshold, ecount) \ +{ \ + .name = (err), \ + .is_critical = (critical), \ + .error_id = (id), \ + .err_threshold = (threshold), \ + .err_count = NVGPU_ATOMIC_INIT(ecount), \ +} + +/** + * This macro is used to initialize critical errors. + */ +#define GPU_CRITERR(err, id, threshold, ecount) \ + GPU_ERR(err, true, id, threshold, ecount) + +/** + * This macro is used to initialize non-critical errors. + */ +#define GPU_NONCRITERR(err, id, threshold, ecount) \ + GPU_ERR(err, false, id, threshold, ecount) + +/** + * @brief GPU HW errors need to be reported to Safety_Services via SDL unit. + * This function provides an interface to report ECC erros to SDL unit. + * + * @param g [in] - The GPU driver struct. + * @param hw_unit [in] - Index of HW unit. + * - List of valid HW unit IDs + * - NVGPU_ERR_MODULE_SM + * - NVGPU_ERR_MODULE_FECS + * - NVGPU_ERR_MODULE_GPCCS + * - NVGPU_ERR_MODULE_MMU + * - NVGPU_ERR_MODULE_GCC + * - NVGPU_ERR_MODULE_PMU + * - NVGPU_ERR_MODULE_LTC + * - NVGPU_ERR_MODULE_HUBMMU + * @param inst [in] - Instance ID. + * - In case of multiple instances of the same HW + * unit (e.g., there are multiple instances of + * SM), it is used to identify the instance + * that encountered a fault. + * @param err_id [in] - Error index. + * - For SM: + * - Min: GPU_SM_L1_TAG_ECC_CORRECTED + * - Max: GPU_SM_ICACHE_L1_PREDECODE_ECC_UNCORRECTED + * - For FECS: + * - Min: GPU_FECS_FALCON_IMEM_ECC_CORRECTED + * - Max: GPU_FECS_INVALID_ERROR + * - For GPCCS: + * - Min: GPU_GPCCS_FALCON_IMEM_ECC_CORRECTED + * - Max: GPU_GPCCS_FALCON_DMEM_ECC_UNCORRECTED + * - For MMU: + * - Min: GPU_MMU_L1TLB_SA_DATA_ECC_UNCORRECTED + * - Max: GPU_MMU_L1TLB_FA_DATA_ECC_UNCORRECTED + * - For GCC: + * - Min: GPU_GCC_L15_ECC_UNCORRECTED + * - Max: GPU_GCC_L15_ECC_UNCORRECTED + * - For PMU: + * - Min: GPU_PMU_FALCON_IMEM_ECC_CORRECTED + * - Max: GPU_PMU_FALCON_DMEM_ECC_UNCORRECTED + * - For LTC: + * - Min: GPU_LTC_CACHE_DSTG_ECC_CORRECTED + * - Max: GPU_LTC_CACHE_DSTG_BE_ECC_UNCORRECTED + * - For HUBMMU: + * - Min: GPU_HUBMMU_L2TLB_SA_DATA_ECC_UNCORRECTED + * - Max: GPU_HUBMMU_PDE0_DATA_ECC_UNCORRECTED + * @param err_addr [in] - Error address. + * - This is the location at which correctable or + * uncorrectable error has occurred. + * @param err_count [in] - Error count. + * + * - Checks whether SDL is supported in the current GPU platform. If SDL is not + * supported, it simply returns. + * - Validates both \a hw_unit and \a err_id indices. In case of a failure, + * invokes #nvgpu_sdl_handle_report_failure() api. + * - Gets the current time of a clock. In case of a failure, invokes + * #nvgpu_sdl_handle_report_failure() api. + * - Gets error description from internal look-up table using \a hw_unit and + * \a err_id indices. + * - Forms error packet using details such as time-stamp, \a hw_unit, \a err_id, + * criticality of the error, \a inst, \a err_addr, \a err_count, error + * description, and size of the error packet. + * - Performs compile-time assert check to ensure that the size of the error + * packet does not exceed the maximum allowable size specified in + * #MAX_ERR_MSG_SIZE. + * + * @return None + */ +void nvgpu_report_ecc_err(struct gk20a *g, u32 hw_unit, u32 inst, + u32 err_id, u64 err_addr, u64 err_count); + +void nvgpu_init_ecc_reporting(struct gk20a *g); +void nvgpu_enable_ecc_reporting(struct gk20a *g); +void nvgpu_disable_ecc_reporting(struct gk20a *g); +void nvgpu_deinit_ecc_reporting(struct gk20a *g); + +#else + +static inline void nvgpu_report_ecc_err(struct gk20a *g, u32 hw_unit, u32 inst, + u32 err_id, u64 err_addr, u64 err_count) { + +} + +#endif /* CONFIG_NVGPU_SUPPORT_LINUX_ECC_ERROR_REPORTING */ + +#endif /* NVGPU_NVGPU_ERR_H */ \ No newline at end of file diff --git a/include/nvgpu/nvlink.h b/include/nvgpu/nvlink.h index 26c83f1..a74111c 100644 --- a/include/nvgpu/nvlink.h +++ b/include/nvgpu/nvlink.h @@ -26,7 +26,7 @@ #include #ifdef __KERNEL__ -//#include +#include #elif defined(__NVGPU_POSIX__) #include #else diff --git a/include/nvgpu/pmu.h b/include/nvgpu/pmu.h index 2b745c7..fb1b016 100644 --- a/include/nvgpu/pmu.h +++ b/include/nvgpu/pmu.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2017-2019, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2017-2022, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -94,6 +94,23 @@ #define PMU_STATE_STARTED 7U /* Fully unitialized */ #define PMU_STATE_EXIT 8U /* Exit PMU state machine */ +/* state transition : + * OFF => [OFF_ON_PENDING optional] => ON_PENDING => ON => OFF + * ON => OFF is always synchronized + */ +/* elpg is off */ +#define PMU_ELPG_STAT_OFF 0U +/* elpg is on */ +#define PMU_ELPG_STAT_ON 1U +/* elpg is off, ALLOW cmd has been sent, wait for ack */ +#define PMU_ELPG_STAT_ON_PENDING 2U +/* elpg is on, DISALLOW cmd has been sent, wait for ack */ +#define PMU_ELPG_STAT_OFF_PENDING 3U +/* elpg is off, caller has requested on, but ALLOW + * cmd hasn't been sent due to ENABLE_ALLOW delay + */ +#define PMU_ELPG_STAT_OFF_ON_PENDING 4U + #define GK20A_PMU_UCODE_NB_MAX_OVERLAY 32U #define GK20A_PMU_UCODE_NB_MAX_DATE_LENGTH 64U @@ -351,6 +368,7 @@ struct nvgpu_pmu { u32 stat_dmem_offset[PMU_PG_ELPG_ENGINE_ID_INVALID_ENGINE]; u32 elpg_stat; + u32 disallow_state; u32 mscg_stat; u32 mscg_transition_state; diff --git a/include/nvgpu/pmuif/gpmuif_pg.h b/include/nvgpu/pmuif/gpmuif_pg.h index 69a7ea4..58311ae 100644 --- a/include/nvgpu/pmuif/gpmuif_pg.h +++ b/include/nvgpu/pmuif/gpmuif_pg.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2017-2022, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -33,6 +33,11 @@ #define PMU_PG_ELPG_ENGINE_ID_INVALID_ENGINE (0x00000005U) #define PMU_PG_ELPG_ENGINE_MAX PMU_PG_ELPG_ENGINE_ID_INVALID_ENGINE +/* Async PG message IDs */ +enum { + PMU_PG_MSG_ASYNC_CMD_DISALLOW, +}; + /* PG message */ enum { PMU_PG_ELPG_MSG_INIT_ACK, @@ -73,12 +78,19 @@ struct pmu_pg_msg_eng_buf_stat { u8 status; }; +struct pmu_pg_msg_async_cmd_resp { + u8 msg_type; + u8 ctrl_id; + u8 msg_id; +}; + struct pmu_pg_msg { union { u8 msg_type; struct pmu_pg_msg_elpg_msg elpg_msg; struct pmu_pg_msg_stat stat; struct pmu_pg_msg_eng_buf_stat eng_buf_stat; + struct pmu_pg_msg_async_cmd_resp async_cmd_resp; /* TBD: other pg messages */ union pmu_ap_msg ap_msg; struct nv_pmu_rppg_msg rppg_msg; diff --git a/include/nvgpu/tsg.h b/include/nvgpu/tsg.h index 7cd97c9..f5391e7 100644 --- a/include/nvgpu/tsg.h +++ b/include/nvgpu/tsg.h @@ -1,5 +1,5 @@ /* - * Copyright (c) 2014-2020, NVIDIA CORPORATION. All rights reserved. + * Copyright (c) 2014-2021, NVIDIA CORPORATION. All rights reserved. * * Permission is hereby granted, free of charge, to any person obtaining a * copy of this software and associated documentation files (the "Software"), @@ -90,7 +90,7 @@ int gk20a_enable_tsg(struct tsg_gk20a *tsg); int gk20a_disable_tsg(struct tsg_gk20a *tsg); int gk20a_tsg_bind_channel(struct tsg_gk20a *tsg, struct channel_gk20a *ch); -int gk20a_tsg_unbind_channel(struct channel_gk20a *ch); +int gk20a_tsg_unbind_channel(struct channel_gk20a *ch, bool force); void gk20a_tsg_event_id_post_event(struct tsg_gk20a *tsg, int event_id); -- cgit v1.2.2