habanalabs/gaudi2: log critical events with no rate limit
When we have a storm of errors of HBM ECC SERR we can reach a situation where driver start hard reset flow without logging the error cause that caused the hard reset due to logs rate limiting. Signed-off-by: farah kassabri <fkassabri@habana.ai> Reviewed-by: Oded Gabbay <ogabbay@kernel.org> Signed-off-by: Oded Gabbay <ogabbay@kernel.org>
This commit is contained in:
parent
d155df4f62
commit
988262ef2f
@ -8200,10 +8200,17 @@ static bool gaudi2_handle_hbm_mc_sei_err(struct hl_device *hdev, u16 event_type,
|
||||
return true;
|
||||
}
|
||||
|
||||
dev_err_ratelimited(hdev->dev,
|
||||
"System Error Interrupt - HBM(%u) MC(%u) MC_CH(%u) MC_PC(%u). Critical(%u). Error cause: %s\n",
|
||||
hbm_id, mc_id, sei_data->hdr.mc_channel, sei_data->hdr.mc_pseudo_channel,
|
||||
sei_data->hdr.is_critical, hbm_mc_sei_cause[cause_idx]);
|
||||
if (sei_data->hdr.is_critical)
|
||||
dev_err(hdev->dev,
|
||||
"System Critical Error Interrupt - HBM(%u) MC(%u) MC_CH(%u) MC_PC(%u). Error cause: %s\n",
|
||||
hbm_id, mc_id, sei_data->hdr.mc_channel, sei_data->hdr.mc_pseudo_channel,
|
||||
hbm_mc_sei_cause[cause_idx]);
|
||||
|
||||
else
|
||||
dev_err_ratelimited(hdev->dev,
|
||||
"System Non-Critical Error Interrupt - HBM(%u) MC(%u) MC_CH(%u) MC_PC(%u). Error cause: %s\n",
|
||||
hbm_id, mc_id, sei_data->hdr.mc_channel, sei_data->hdr.mc_pseudo_channel,
|
||||
hbm_mc_sei_cause[cause_idx]);
|
||||
|
||||
/* Print error-specific info */
|
||||
switch (cause_idx) {
|
||||
|
Loading…
x
Reference in New Issue
Block a user