2018-03-08 09:50:37 +00:00
/*
* SPDX - License - Identifier : MIT
*
* Copyright <EFBFBD> 2008 - 2018 Intel Corporation
*/
# ifndef _I915_GPU_ERROR_H_
# define _I915_GPU_ERROR_H_
# include <linux/kref.h>
# include <linux/ktime.h>
# include <linux/sched.h>
# include <drm/drm_mm.h>
# include "intel_device_info.h"
# include "intel_ringbuffer.h"
# include "intel_uc_fw.h"
# include "i915_gem.h"
# include "i915_gem_gtt.h"
# include "i915_params.h"
2018-04-18 19:40:52 +01:00
# include "i915_scheduler.h"
2018-03-08 09:50:37 +00:00
struct drm_i915_private ;
struct intel_overlay_error_state ;
struct intel_display_error_state ;
struct i915_gpu_state {
struct kref ref ;
ktime_t time ;
ktime_t boottime ;
ktime_t uptime ;
2018-04-30 10:52:59 +03:00
unsigned long capture ;
unsigned long epoch ;
2018-03-08 09:50:37 +00:00
struct drm_i915_private * i915 ;
char error_msg [ 128 ] ;
bool simulated ;
bool awake ;
bool wakelock ;
bool suspended ;
int iommu ;
u32 reset_count ;
u32 suspend_count ;
struct intel_device_info device_info ;
2018-12-31 16:56:41 +02:00
struct intel_runtime_info runtime_info ;
2018-03-08 09:50:37 +00:00
struct intel_driver_caps driver_caps ;
struct i915_params params ;
struct i915_error_uc {
struct intel_uc_fw guc_fw ;
struct intel_uc_fw huc_fw ;
struct drm_i915_error_object * guc_log ;
} uc ;
/* Generic register state */
u32 eir ;
u32 pgtbl_er ;
u32 ier ;
2018-05-10 14:59:55 -07:00
u32 gtier [ 6 ] , ngtier ;
2018-03-08 09:50:37 +00:00
u32 ccid ;
u32 derrmr ;
u32 forcewake ;
u32 error ; /* gen6+ */
u32 err_int ; /* gen7 */
u32 fault_data0 ; /* gen8, gen9 */
u32 fault_data1 ; /* gen8, gen9 */
u32 done_reg ;
u32 gac_eco ;
u32 gam_ecochk ;
u32 gab_ctl ;
u32 gfx_mode ;
u32 nfence ;
u64 fence [ I915_MAX_NUM_FENCES ] ;
struct intel_overlay_error_state * overlay ;
struct intel_display_error_state * display ;
struct drm_i915_error_engine {
int engine_id ;
/* Software tracked state */
bool idle ;
unsigned long hangcheck_timestamp ;
struct i915_address_space * vm ;
int num_requests ;
u32 reset_count ;
/* position of active request inside the ring */
u32 rq_head , rq_post , rq_tail ;
/* our own tracking of ring head and tail */
u32 cpu_ring_head ;
u32 cpu_ring_tail ;
u32 last_seqno ;
/* Register state */
u32 start ;
u32 tail ;
u32 head ;
u32 ctl ;
u32 mode ;
u32 hws ;
u32 ipeir ;
u32 ipehr ;
u32 bbstate ;
u32 instpm ;
u32 instps ;
u32 seqno ;
u64 bbaddr ;
u64 acthd ;
u32 fault_reg ;
u64 faddr ;
u32 rc_psmi ; /* sleep state */
u32 semaphore_mboxes [ I915_NUM_ENGINES - 1 ] ;
struct intel_instdone instdone ;
struct drm_i915_error_context {
char comm [ TASK_COMM_LEN ] ;
pid_t pid ;
u32 handle ;
u32 hw_id ;
int ban_score ;
int active ;
int guilty ;
bool bannable ;
2018-04-18 19:40:52 +01:00
struct i915_sched_attr sched_attr ;
2018-03-08 09:50:37 +00:00
} context ;
struct drm_i915_error_object {
u64 gtt_offset ;
u64 gtt_size ;
2018-10-03 09:24:22 +01:00
int num_pages ;
2018-03-08 09:50:37 +00:00
int page_count ;
int unused ;
u32 * pages [ 0 ] ;
} * ringbuffer , * batchbuffer , * wa_batchbuffer , * ctx , * hws_page ;
struct drm_i915_error_object * * user_bo ;
long user_bo_count ;
struct drm_i915_error_object * wa_ctx ;
struct drm_i915_error_object * default_state ;
struct drm_i915_error_request {
drm/i915: Replace global breadcrumbs with per-context interrupt tracking
A few years ago, see commit 688e6c725816 ("drm/i915: Slaughter the
thundering i915_wait_request herd"), the issue of handling multiple
clients waiting in parallel was brought to our attention. The
requirement was that every client should be woken immediately upon its
request being signaled, without incurring any cpu overhead.
To handle certain fragility of our hw meant that we could not do a
simple check inside the irq handler (some generations required almost
unbounded delays before we could be sure of seqno coherency) and so
request completion checking required delegation.
Before commit 688e6c725816, the solution was simple. Every client
waiting on a request would be woken on every interrupt and each would do
a heavyweight check to see if their request was complete. Commit
688e6c725816 introduced an rbtree so that only the earliest waiter on
the global timeline would woken, and would wake the next and so on.
(Along with various complications to handle requests being reordered
along the global timeline, and also a requirement for kthread to provide
a delegate for fence signaling that had no process context.)
The global rbtree depends on knowing the execution timeline (and global
seqno). Without knowing that order, we must instead check all contexts
queued to the HW to see which may have advanced. We trim that list by
only checking queued contexts that are being waited on, but still we
keep a list of all active contexts and their active signalers that we
inspect from inside the irq handler. By moving the waiters onto the fence
signal list, we can combine the client wakeup with the dma_fence
signaling (a dramatic reduction in complexity, but does require the HW
being coherent, the seqno must be visible from the cpu before the
interrupt is raised - we keep a timer backup just in case).
Having previously fixed all the issues with irq-seqno serialisation (by
inserting delays onto the GPU after each request instead of random delays
on the CPU after each interrupt), we can rely on the seqno state to
perfom direct wakeups from the interrupt handler. This allows us to
preserve our single context switch behaviour of the current routine,
with the only downside that we lose the RT priority sorting of wakeups.
In general, direct wakeup latency of multiple clients is about the same
(about 10% better in most cases) with a reduction in total CPU time spent
in the waiter (about 20-50% depending on gen). Average herd behaviour is
improved, but at the cost of not delegating wakeups on task_prio.
v2: Capture fence signaling state for error state and add comments to
warm even the most cold of hearts.
v3: Check if the request is still active before busywaiting
v4: Reduce the amount of pointer misdirection with list_for_each_safe
and using a local i915_request variable inside the loops
v5: Add a missing pluralisation to a purely informative selftest message.
References: 688e6c725816 ("drm/i915: Slaughter the thundering i915_wait_request herd")
Signed-off-by: Chris Wilson <chris@chris-wilson.co.uk>
Cc: Tvrtko Ursulin <tvrtko.ursulin@intel.com>
Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@intel.com>
Link: https://patchwork.freedesktop.org/patch/msgid/20190129205230.19056-2-chris@chris-wilson.co.uk
2019-01-29 20:52:29 +00:00
unsigned long flags ;
2018-03-08 09:50:37 +00:00
long jiffies ;
pid_t pid ;
u32 context ;
int ban_score ;
u32 seqno ;
2018-05-02 11:41:50 +01:00
u32 start ;
2018-03-08 09:50:37 +00:00
u32 head ;
u32 tail ;
2018-04-18 19:40:52 +01:00
struct i915_sched_attr sched_attr ;
2018-03-08 09:50:37 +00:00
} * requests , execlist [ EXECLIST_MAX_PORTS ] ;
unsigned int num_ports ;
struct {
u32 gfx_mode ;
union {
u64 pdp [ 4 ] ;
u32 pp_dir_base ;
} ;
} vm_info ;
} engine [ I915_NUM_ENGINES ] ;
struct drm_i915_error_buffer {
u32 size ;
u32 name ;
2018-07-06 11:39:46 +01:00
u32 wseqno ;
2018-03-08 09:50:37 +00:00
u64 gtt_offset ;
u32 read_domains ;
u32 write_domain ;
s32 fence_reg : I915_MAX_NUM_FENCE_BITS ;
u32 tiling : 2 ;
u32 dirty : 1 ;
u32 purgeable : 1 ;
u32 userptr : 1 ;
s32 engine : 4 ;
u32 cache_level : 3 ;
} * active_bo [ I915_NUM_ENGINES ] , * pinned_bo ;
u32 active_bo_count [ I915_NUM_ENGINES ] , pinned_bo_count ;
struct i915_address_space * active_vm [ I915_NUM_ENGINES ] ;
2018-11-23 13:23:25 +00:00
struct scatterlist * sgl , * fit ;
2018-03-08 09:50:37 +00:00
} ;
2019-01-25 13:22:28 +00:00
struct i915_gpu_restart ;
2018-03-08 09:50:37 +00:00
struct i915_gpu_error {
/* For hangcheck timer */
# define DRM_I915_HANGCHECK_PERIOD 1500 /* in ms */
# define DRM_I915_HANGCHECK_JIFFIES msecs_to_jiffies(DRM_I915_HANGCHECK_PERIOD)
struct delayed_work hangcheck_work ;
/* For reset and error_state handling. */
spinlock_t lock ;
/* Protected by the above dev->gpu_error.lock. */
struct i915_gpu_state * first_error ;
atomic_t pending_fb_pin ;
/**
* State variable controlling the reset flow and count
*
* This is a counter which gets incremented when reset is triggered ,
*
* Before the reset commences , the I915_RESET_BACKOFF bit is set
* meaning that any waiters holding onto the struct_mutex should
* relinquish the lock immediately in order for the reset to start .
*
* If reset is not completed successfully , the I915_WEDGE bit is
* set meaning that hardware is terminally sour and there is no
* recovery . All waiters on the reset_queue will be woken when
* that happens .
*
* This counter is used by the wait_seqno code to notice that reset
* event happened and it needs to restart the entire ioctl ( since most
* likely the seqno it waited for won ' t ever signal anytime soon ) .
*
* This is important for lock - free wait paths , where no contended lock
* naturally enforces the correct ordering between the bail - out of the
* waiter and the gpu reset work code .
*/
unsigned long reset_count ;
/**
* flags : Control various stages of the GPU reset
*
* # I915_RESET_BACKOFF - When we start a reset , we want to stop any
* other users acquiring the struct_mutex . To do this we set the
* # I915_RESET_BACKOFF bit in the error flags when we detect a reset
* and then check for that bit before acquiring the struct_mutex ( in
* i915_mutex_lock_interruptible ( ) ? ) . I915_RESET_BACKOFF serves a
* secondary role in preventing two concurrent global reset attempts .
*
* # I915_RESET_ENGINE [ num_engines ] - Since the driver doesn ' t need to
* acquire the struct_mutex to reset an engine , we need an explicit
* flag to prevent two concurrent reset attempts in the same engine .
* As the number of engines continues to grow , allocate the flags from
* the most significant bits .
*
* # I915_WEDGED - If reset fails and we can no longer use the GPU ,
* we set the # I915_WEDGED bit . Prior to command submission , e . g .
* i915_request_alloc ( ) , this bit is checked and the sequence
* aborted ( with - EIO reported to userspace ) if set .
*/
unsigned long flags ;
# define I915_RESET_BACKOFF 0
2019-01-25 13:22:28 +00:00
# define I915_RESET_MODESET 1
# define I915_RESET_ENGINE 2
2018-03-08 09:50:37 +00:00
# define I915_WEDGED (BITS_PER_LONG - 1)
/** Number of times an engine has been reset */
u32 reset_engine_count [ I915_NUM_ENGINES ] ;
2019-01-14 21:04:01 +00:00
struct mutex wedge_mutex ; /* serialises wedging/unwedging */
2018-03-08 09:50:37 +00:00
/**
* Waitqueue to signal when a hang is detected . Used to for waiters
* to release the struct_mutex for the reset to procede .
*/
wait_queue_head_t wait_queue ;
/**
* Waitqueue to signal when the reset has completed . Used by clients
* that wait for dev_priv - > mm . wedged to settle .
*/
wait_queue_head_t reset_queue ;
2019-01-25 13:22:28 +00:00
struct i915_gpu_restart * restart ;
2018-03-08 09:50:37 +00:00
} ;
struct drm_i915_error_state_buf {
struct drm_i915_private * i915 ;
2018-11-23 13:23:25 +00:00
struct scatterlist * sgl , * cur , * end ;
char * buf ;
size_t bytes ;
size_t size ;
loff_t iter ;
2018-03-08 09:50:37 +00:00
int err ;
} ;
# if IS_ENABLED(CONFIG_DRM_I915_CAPTURE_ERROR)
__printf ( 2 , 3 )
void i915_error_printf ( struct drm_i915_error_state_buf * e , const char * f , . . . ) ;
struct i915_gpu_state * i915_capture_gpu_state ( struct drm_i915_private * i915 ) ;
void i915_capture_error_state ( struct drm_i915_private * dev_priv ,
2019-01-25 13:22:28 +00:00
unsigned long engine_mask ,
2018-03-08 09:50:37 +00:00
const char * error_msg ) ;
static inline struct i915_gpu_state *
i915_gpu_state_get ( struct i915_gpu_state * gpu )
{
kref_get ( & gpu - > ref ) ;
return gpu ;
}
2018-11-23 13:23:25 +00:00
ssize_t i915_gpu_state_copy_to_buffer ( struct i915_gpu_state * error ,
char * buf , loff_t offset , size_t count ) ;
2018-03-08 09:50:37 +00:00
void __i915_gpu_state_free ( struct kref * kref ) ;
static inline void i915_gpu_state_put ( struct i915_gpu_state * gpu )
{
if ( gpu )
kref_put ( & gpu - > ref , __i915_gpu_state_free ) ;
}
struct i915_gpu_state * i915_first_error_state ( struct drm_i915_private * i915 ) ;
void i915_reset_error_state ( struct drm_i915_private * i915 ) ;
2018-11-02 16:12:12 +00:00
void i915_disable_error_state ( struct drm_i915_private * i915 , int err ) ;
2018-03-08 09:50:37 +00:00
# else
static inline void i915_capture_error_state ( struct drm_i915_private * dev_priv ,
u32 engine_mask ,
const char * error_msg )
{
}
static inline struct i915_gpu_state *
i915_first_error_state ( struct drm_i915_private * i915 )
{
2018-11-02 16:12:12 +00:00
return ERR_PTR ( - ENODEV ) ;
2018-03-08 09:50:37 +00:00
}
static inline void i915_reset_error_state ( struct drm_i915_private * i915 )
{
}
2018-11-02 16:12:12 +00:00
static inline void i915_disable_error_state ( struct drm_i915_private * i915 ,
int err )
{
}
2018-03-08 09:50:37 +00:00
# endif /* IS_ENABLED(CONFIG_DRM_I915_CAPTURE_ERROR) */
# endif /* _I915_GPU_ERROR_H_ */