i915/gt/intel_lrc.c

1.8  riastrad /*	$NetBSD: intel_lrc.c,v 1.8 2021/12/19 12:32:15 riastradh Exp $	*/
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * Copyright  2014 Intel Corporation
1.1  riastrad  *
1.1  riastrad  * Permission is hereby granted, free of charge, to any person obtaining a
1.1  riastrad  * copy of this software and associated documentation files (the "Software"),
1.1  riastrad  * to deal in the Software without restriction, including without limitation
1.1  riastrad  * the rights to use, copy, modify, merge, publish, distribute, sublicense,
1.1  riastrad  * and/or sell copies of the Software, and to permit persons to whom the
1.1  riastrad  * Software is furnished to do so, subject to the following conditions:
1.1  riastrad  *
1.1  riastrad  * The above copyright notice and this permission notice (including the next
1.1  riastrad  * paragraph) shall be included in all copies or substantial portions of the
1.1  riastrad  * Software.
1.1  riastrad  *
1.1  riastrad  * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
1.1  riastrad  * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
1.1  riastrad  * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.  IN NO EVENT SHALL
1.1  riastrad  * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
1.1  riastrad  * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
1.1  riastrad  * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
1.1  riastrad  * IN THE SOFTWARE.
1.1  riastrad  *
1.1  riastrad  * Authors:
1.1  riastrad  *    Ben Widawsky <ben (at) bwidawsk.net>
1.1  riastrad  *    Michel Thierry <michel.thierry (at) intel.com>
1.1  riastrad  *    Thomas Daniel <thomas.daniel (at) intel.com>
1.1  riastrad  *    Oscar Mateo <oscar.mateo (at) intel.com>
1.1  riastrad  *
1.1  riastrad  */
1.1  riastrad
1.1  riastrad /**
1.1  riastrad  * DOC: Logical Rings, Logical Ring Contexts and Execlists
1.1  riastrad  *
1.1  riastrad  * Motivation:
1.1  riastrad  * GEN8 brings an expansion of the HW contexts: "Logical Ring Contexts".
1.1  riastrad  * These expanded contexts enable a number of new abilities, especially
1.1  riastrad  * "Execlists" (also implemented in this file).
1.1  riastrad  *
1.1  riastrad  * One of the main differences with the legacy HW contexts is that logical
1.1  riastrad  * ring contexts incorporate many more things to the context's state, like
1.1  riastrad  * PDPs or ringbuffer control registers:
1.1  riastrad  *
1.1  riastrad  * The reason why PDPs are included in the context is straightforward: as
1.1  riastrad  * PPGTTs (per-process GTTs) are actually per-context, having the PDPs
1.1  riastrad  * contained there mean you don't need to do a ppgtt->switch_mm yourself,
1.1  riastrad  * instead, the GPU will do it for you on the context switch.
1.1  riastrad  *
1.1  riastrad  * But, what about the ringbuffer control registers (head, tail, etc..)?
1.1  riastrad  * shouldn't we just need a set of those per engine command streamer? This is
1.1  riastrad  * where the name "Logical Rings" starts to make sense: by virtualizing the
1.1  riastrad  * rings, the engine cs shifts to a new "ring buffer" with every context
1.1  riastrad  * switch. When you want to submit a workload to the GPU you: A) choose your
1.1  riastrad  * context, B) find its appropriate virtualized ring, C) write commands to it
1.1  riastrad  * and then, finally, D) tell the GPU to switch to that context.
1.1  riastrad  *
1.1  riastrad  * Instead of the legacy MI_SET_CONTEXT, the way you tell the GPU to switch
1.1  riastrad  * to a contexts is via a context execution list, ergo "Execlists".
1.1  riastrad  *
1.1  riastrad  * LRC implementation:
1.1  riastrad  * Regarding the creation of contexts, we have:
1.1  riastrad  *
1.1  riastrad  * - One global default context.
1.1  riastrad  * - One local default context for each opened fd.
1.1  riastrad  * - One local extra context for each context create ioctl call.
1.1  riastrad  *
1.1  riastrad  * Now that ringbuffers belong per-context (and not per-engine, like before)
1.1  riastrad  * and that contexts are uniquely tied to a given engine (and not reusable,
1.1  riastrad  * like before) we need:
1.1  riastrad  *
1.1  riastrad  * - One ringbuffer per-engine inside each context.
1.1  riastrad  * - One backing object per-engine inside each context.
1.1  riastrad  *
1.1  riastrad  * The global default context starts its life with these new objects fully
1.1  riastrad  * allocated and populated. The local default context for each opened fd is
1.1  riastrad  * more complex, because we don't know at creation time which engine is going
1.1  riastrad  * to use them. To handle this, we have implemented a deferred creation of LR
1.1  riastrad  * contexts:
1.1  riastrad  *
1.1  riastrad  * The local context starts its life as a hollow or blank holder, that only
1.1  riastrad  * gets populated for a given engine once we receive an execbuffer. If later
1.1  riastrad  * on we receive another execbuffer ioctl for the same context but a different
1.1  riastrad  * engine, we allocate/populate a new ringbuffer and context backing object and
1.1  riastrad  * so on.
1.1  riastrad  *
1.1  riastrad  * Finally, regarding local contexts created using the ioctl call: as they are
1.1  riastrad  * only allowed with the render ring, we can allocate & populate them right
1.1  riastrad  * away (no need to defer anything, at least for now).
1.1  riastrad  *
1.1  riastrad  * Execlists implementation:
1.1  riastrad  * Execlists are the new method by which, on gen8+ hardware, workloads are
1.1  riastrad  * submitted for execution (as opposed to the legacy, ringbuffer-based, method).
1.1  riastrad  * This method works as follows:
1.1  riastrad  *
1.1  riastrad  * When a request is committed, its commands (the BB start and any leading or
1.1  riastrad  * trailing commands, like the seqno breadcrumbs) are placed in the ringbuffer
1.1  riastrad  * for the appropriate context. The tail pointer in the hardware context is not
1.1  riastrad  * updated at this time, but instead, kept by the driver in the ringbuffer
1.1  riastrad  * structure. A structure representing this request is added to a request queue
1.1  riastrad  * for the appropriate engine: this structure contains a copy of the context's
1.1  riastrad  * tail after the request was written to the ring buffer and a pointer to the
1.1  riastrad  * context itself.
1.1  riastrad  *
1.1  riastrad  * If the engine's request queue was empty before the request was added, the
1.1  riastrad  * queue is processed immediately. Otherwise the queue will be processed during
1.1  riastrad  * a context switch interrupt. In any case, elements on the queue will get sent
1.1  riastrad  * (in pairs) to the GPU's ExecLists Submit Port (ELSP, for short) with a
1.1  riastrad  * globally unique 20-bits submission ID.
1.1  riastrad  *
1.1  riastrad  * When execution of a request completes, the GPU updates the context status
1.1  riastrad  * buffer with a context complete event and generates a context switch interrupt.
1.1  riastrad  * During the interrupt handling, the driver examines the events in the buffer:
1.1  riastrad  * for each context complete event, if the announced ID matches that on the head
1.1  riastrad  * of the request queue, then that request is retired and removed from the queue.
1.1  riastrad  *
1.1  riastrad  * After processing, if any requests were retired and the queue is not empty
1.1  riastrad  * then a new execution list can be submitted. The two requests at the front of
1.1  riastrad  * the queue are next to be submitted but since a context may not occur twice in
1.1  riastrad  * an execution list, if subsequent requests have the same ID as the first then
1.1  riastrad  * the two requests must be combined. This is done simply by discarding requests
1.1  riastrad  * at the head of the queue until either only one requests is left (in which case
1.1  riastrad  * we use a NULL second context) or the first two requests have unique IDs.
1.1  riastrad  *
1.1  riastrad  * By always executing the first two requests in the queue the driver ensures
1.1  riastrad  * that the GPU is kept as busy as possible. In the case where a single context
1.1  riastrad  * completes but a second context is still executing, the request for this second
1.1  riastrad  * context will be at the head of the queue when we remove the first one. This
1.1  riastrad  * request will then be resubmitted along with a new request for a different context,
1.1  riastrad  * which will cause the hardware to continue executing the second request and queue
1.1  riastrad  * the new request (the GPU detects the condition of a context getting preempted
1.1  riastrad  * with the same context and optimizes the context switch flow by not doing
1.1  riastrad  * preemption, but just sampling the new tail pointer).
1.1  riastrad  *
1.1  riastrad  */
1.1  riastrad #include <sys/cdefs.h>
1.8  riastrad __KERNEL_RCSID(0, "$NetBSD: intel_lrc.c,v 1.8 2021/12/19 12:32:15 riastradh Exp $");
1.1  riastrad
1.1  riastrad #include <linux/interrupt.h>
1.1  riastrad
1.1  riastrad #include "i915_drv.h"
1.1  riastrad #include "i915_perf.h"
1.1  riastrad #include "i915_trace.h"
1.1  riastrad #include "i915_vgpu.h"
1.1  riastrad #include "intel_context.h"
1.1  riastrad #include "intel_engine_pm.h"
1.1  riastrad #include "intel_gt.h"
1.1  riastrad #include "intel_gt_pm.h"
1.1  riastrad #include "intel_gt_requests.h"
1.1  riastrad #include "intel_lrc_reg.h"
1.1  riastrad #include "intel_mocs.h"
1.1  riastrad #include "intel_reset.h"
1.1  riastrad #include "intel_ring.h"
1.1  riastrad #include "intel_workarounds.h"
1.1  riastrad
1.5  riastrad #include <linux/nbsd-namespace.h>
1.5  riastrad
1.1  riastrad #define RING_EXECLIST_QFULL		(1 << 0x2)
1.1  riastrad #define RING_EXECLIST1_VALID		(1 << 0x3)
1.1  riastrad #define RING_EXECLIST0_VALID		(1 << 0x4)
1.1  riastrad #define RING_EXECLIST_ACTIVE_STATUS	(3 << 0xE)
1.1  riastrad #define RING_EXECLIST1_ACTIVE		(1 << 0x11)
1.1  riastrad #define RING_EXECLIST0_ACTIVE		(1 << 0x12)
1.1  riastrad
1.1  riastrad #define GEN8_CTX_STATUS_IDLE_ACTIVE	(1 << 0)
1.1  riastrad #define GEN8_CTX_STATUS_PREEMPTED	(1 << 1)
1.1  riastrad #define GEN8_CTX_STATUS_ELEMENT_SWITCH	(1 << 2)
1.1  riastrad #define GEN8_CTX_STATUS_ACTIVE_IDLE	(1 << 3)
1.1  riastrad #define GEN8_CTX_STATUS_COMPLETE	(1 << 4)
1.1  riastrad #define GEN8_CTX_STATUS_LITE_RESTORE	(1 << 15)
1.1  riastrad
1.1  riastrad #define GEN8_CTX_STATUS_COMPLETED_MASK \
1.1  riastrad 	 (GEN8_CTX_STATUS_COMPLETE | GEN8_CTX_STATUS_PREEMPTED)
1.1  riastrad
1.1  riastrad #define CTX_DESC_FORCE_RESTORE BIT_ULL(2)
1.1  riastrad
1.1  riastrad #define GEN12_CTX_STATUS_SWITCHED_TO_NEW_QUEUE	(0x1) /* lower csb dword */
1.1  riastrad #define GEN12_CTX_SWITCH_DETAIL(csb_dw)	((csb_dw) & 0xF) /* upper csb dword */
1.1  riastrad #define GEN12_CSB_SW_CTX_ID_MASK		GENMASK(25, 15)
1.1  riastrad #define GEN12_IDLE_CTX_ID		0x7FF
1.1  riastrad #define GEN12_CSB_CTX_VALID(csb_dw) \
1.1  riastrad 	(FIELD_GET(GEN12_CSB_SW_CTX_ID_MASK, csb_dw) != GEN12_IDLE_CTX_ID)
1.1  riastrad
1.1  riastrad /* Typical size of the average request (2 pipecontrols and a MI_BB) */
1.1  riastrad #define EXECLISTS_REQUEST_SIZE 64 /* bytes */
1.1  riastrad #define WA_TAIL_DWORDS 2
1.1  riastrad #define WA_TAIL_BYTES (sizeof(u32) * WA_TAIL_DWORDS)
1.1  riastrad
1.1  riastrad struct virtual_engine {
1.1  riastrad 	struct intel_engine_cs base;
1.1  riastrad 	struct intel_context context;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We allow only a single request through the virtual engine at a time
1.1  riastrad 	 * (each request in the timeline waits for the completion fence of
1.1  riastrad 	 * the previous before being submitted). By restricting ourselves to
1.1  riastrad 	 * only submitting a single request, each request is placed on to a
1.1  riastrad 	 * physical to maximise load spreading (by virtue of the late greedy
1.1  riastrad 	 * scheduling -- each real engine takes the next available request
1.1  riastrad 	 * upon idling).
1.1  riastrad 	 */
1.1  riastrad 	struct i915_request *request;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We keep a rbtree of available virtual engines inside each physical
1.1  riastrad 	 * engine, sorted by priority. Here we preallocate the nodes we need
1.1  riastrad 	 * for the virtual engine, indexed by physical_engine->id.
1.1  riastrad 	 */
1.1  riastrad 	struct ve_node {
1.1  riastrad 		struct rb_node rb;
1.1  riastrad 		int prio;
1.7  riastrad 		uint64_t order;
1.7  riastrad 		bool inserted;
1.1  riastrad 	} nodes[I915_NUM_ENGINES];
1.7  riastrad 	uint64_t order;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Keep track of bonded pairs -- restrictions upon on our selection
1.1  riastrad 	 * of physical engines any particular request may be submitted to.
1.1  riastrad 	 * If we receive a submit-fence from a master engine, we will only
1.1  riastrad 	 * use one of sibling_mask physical engines.
1.1  riastrad 	 */
1.1  riastrad 	struct ve_bond {
1.1  riastrad 		const struct intel_engine_cs *master;
1.1  riastrad 		intel_engine_mask_t sibling_mask;
1.1  riastrad 	} *bonds;
1.1  riastrad 	unsigned int num_bonds;
1.1  riastrad
1.1  riastrad 	/* And finally, which physical engines this virtual engine maps onto. */
1.1  riastrad 	unsigned int num_siblings;
1.1  riastrad 	struct intel_engine_cs *siblings[0];
1.1  riastrad };
1.1  riastrad
1.7  riastrad #ifdef __NetBSD__
1.7  riastrad static int
1.7  riastrad compare_ve_nodes(void *cookie, const void *va, const void *vb)
1.7  riastrad {
1.7  riastrad 	const struct ve_node *na = va;
1.7  riastrad 	const struct ve_node *nb = vb;
1.7  riastrad
1.7  riastrad 	if (na->prio < nb->prio)
1.7  riastrad 		return -1;
1.7  riastrad 	if (na->prio > nb->prio)
1.7  riastrad 		return +1;
1.7  riastrad 	if (na->order < nb->order)
1.7  riastrad 		return -1;
1.7  riastrad 	if (na->order > nb->order)
1.7  riastrad 		return +1;
1.7  riastrad 	return 0;
1.7  riastrad }
1.7  riastrad
1.7  riastrad static int
1.7  riastrad compare_ve_node_key(void *cookie, const void *vn, const void *vk)
1.7  riastrad {
1.7  riastrad 	const struct ve_node *n = vn;
1.7  riastrad 	const int *k = vk;
1.7  riastrad
1.7  riastrad 	if (n->prio < *k)
1.7  riastrad 		return -1;
1.7  riastrad 	if (n->prio > *k)
1.7  riastrad 		return +1;
1.7  riastrad 	return 0;
1.7  riastrad }
1.7  riastrad
1.7  riastrad static const rb_tree_ops_t ve_tree_ops = {
1.7  riastrad 	.rbto_compare_nodes = compare_ve_nodes,
1.7  riastrad 	.rbto_compare_key = compare_ve_node_key,
1.7  riastrad 	.rbto_node_offset = offsetof(struct ve_node, rb),
1.7  riastrad };
1.7  riastrad #endif
1.7  riastrad
1.1  riastrad static struct virtual_engine *to_virtual_engine(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	GEM_BUG_ON(!intel_engine_is_virtual(engine));
1.1  riastrad 	return container_of(engine, struct virtual_engine, base);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int __execlists_context_alloc(struct intel_context *ce,
1.1  riastrad 				     struct intel_engine_cs *engine);
1.1  riastrad
1.1  riastrad static void execlists_init_reg_state(u32 *reg_state,
1.1  riastrad 				     const struct intel_context *ce,
1.1  riastrad 				     const struct intel_engine_cs *engine,
1.1  riastrad 				     const struct intel_ring *ring,
1.1  riastrad 				     bool close);
1.1  riastrad static void
1.1  riastrad __execlists_update_reg_state(const struct intel_context *ce,
1.1  riastrad 			     const struct intel_engine_cs *engine,
1.1  riastrad 			     u32 head);
1.1  riastrad
1.1  riastrad static void mark_eio(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	if (i915_request_completed(rq))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(i915_request_signaled(rq));
1.1  riastrad
1.1  riastrad 	dma_fence_set_error(&rq->fence, -EIO);
1.1  riastrad 	i915_request_mark_complete(rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static struct i915_request *
1.1  riastrad active_request(const struct intel_timeline * const tl, struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	struct i915_request *active = rq;
1.1  riastrad
1.1  riastrad 	rcu_read_lock();
1.1  riastrad 	list_for_each_entry_continue_reverse(rq, &tl->requests, link) {
1.1  riastrad 		if (i915_request_completed(rq))
1.1  riastrad 			break;
1.1  riastrad
1.1  riastrad 		active = rq;
1.1  riastrad 	}
1.1  riastrad 	rcu_read_unlock();
1.1  riastrad
1.1  riastrad 	return active;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline u32 intel_hws_preempt_address(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	return (i915_ggtt_offset(engine->status_page.vma) +
1.1  riastrad 		I915_GEM_HWS_PREEMPT_ADDR);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void
1.1  riastrad ring_set_paused(const struct intel_engine_cs *engine, int state)
1.1  riastrad {
1.1  riastrad 	/*
1.1  riastrad 	 * We inspect HWS_PREEMPT with a semaphore inside
1.1  riastrad 	 * engine->emit_fini_breadcrumb. If the dword is true,
1.1  riastrad 	 * the ring is paused as the semaphore will busywait
1.1  riastrad 	 * until the dword is false.
1.1  riastrad 	 */
1.1  riastrad 	engine->status_page.addr[I915_GEM_HWS_PREEMPT] = state;
1.1  riastrad 	if (state)
1.1  riastrad 		wmb();
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline struct i915_priolist *to_priolist(struct rb_node *rb)
1.1  riastrad {
1.1  riastrad 	return rb_entry(rb, struct i915_priolist, node);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline int rq_prio(const struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	return rq->sched.attr.priority;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int effective_prio(const struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	int prio = rq_prio(rq);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If this request is special and must not be interrupted at any
1.1  riastrad 	 * cost, so be it. Note we are only checking the most recent request
1.1  riastrad 	 * in the context and so may be masking an earlier vip request. It
1.1  riastrad 	 * is hoped that under the conditions where nopreempt is used, this
1.1  riastrad 	 * will not matter (i.e. all requests to that context will be
1.1  riastrad 	 * nopreempt for as long as desired).
1.1  riastrad 	 */
1.1  riastrad 	if (i915_request_has_nopreempt(rq))
1.1  riastrad 		prio = I915_PRIORITY_UNPREEMPTABLE;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * On unwinding the active request, we give it a priority bump
1.1  riastrad 	 * if it has completed waiting on any semaphore. If we know that
1.1  riastrad 	 * the request has already started, we can prevent an unwanted
1.1  riastrad 	 * preempt-to-idle cycle by taking that into account now.
1.1  riastrad 	 */
1.1  riastrad 	if (__i915_request_has_started(rq))
1.1  riastrad 		prio |= I915_PRIORITY_NOSEMAPHORE;
1.1  riastrad
1.1  riastrad 	/* Restrict mere WAIT boosts from triggering preemption */
1.1  riastrad 	BUILD_BUG_ON(__NO_PREEMPTION & ~I915_PRIORITY_MASK); /* only internal */
1.1  riastrad 	return prio | __NO_PREEMPTION;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int queue_prio(const struct intel_engine_execlists *execlists)
1.1  riastrad {
1.1  riastrad 	struct i915_priolist *p;
1.1  riastrad 	struct rb_node *rb;
1.1  riastrad
1.1  riastrad 	rb = rb_first_cached(&execlists->queue);
1.1  riastrad 	if (!rb)
1.1  riastrad 		return INT_MIN;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * As the priolist[] are inverted, with the highest priority in [0],
1.1  riastrad 	 * we have to flip the index value to become priority.
1.1  riastrad 	 */
1.1  riastrad 	p = to_priolist(rb);
1.1  riastrad 	return ((p->priority + 1) << I915_USER_PRIORITY_SHIFT) - ffs(p->used);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline bool need_preempt(const struct intel_engine_cs *engine,
1.1  riastrad 				const struct i915_request *rq,
1.1  riastrad 				struct rb_node *rb)
1.1  riastrad {
1.1  riastrad 	int last_prio;
1.1  riastrad
1.1  riastrad 	if (!intel_engine_has_semaphores(engine))
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Check if the current priority hint merits a preemption attempt.
1.1  riastrad 	 *
1.1  riastrad 	 * We record the highest value priority we saw during rescheduling
1.1  riastrad 	 * prior to this dequeue, therefore we know that if it is strictly
1.1  riastrad 	 * less than the current tail of ESLP[0], we do not need to force
1.1  riastrad 	 * a preempt-to-idle cycle.
1.1  riastrad 	 *
1.1  riastrad 	 * However, the priority hint is a mere hint that we may need to
1.1  riastrad 	 * preempt. If that hint is stale or we may be trying to preempt
1.1  riastrad 	 * ourselves, ignore the request.
1.1  riastrad 	 *
1.1  riastrad 	 * More naturally we would write
1.1  riastrad 	 *      prio >= max(0, last);
1.1  riastrad 	 * except that we wish to prevent triggering preemption at the same
1.1  riastrad 	 * priority level: the task that is running should remain running
1.1  riastrad 	 * to preserve FIFO ordering of dependencies.
1.1  riastrad 	 */
1.1  riastrad 	last_prio = max(effective_prio(rq), I915_PRIORITY_NORMAL - 1);
1.1  riastrad 	if (engine->execlists.queue_priority_hint <= last_prio)
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Check against the first request in ELSP[1], it will, thanks to the
1.1  riastrad 	 * power of PI, be the highest priority of that context.
1.1  riastrad 	 */
1.1  riastrad 	if (!list_is_last(&rq->sched.link, &engine->active.requests) &&
1.1  riastrad 	    rq_prio(list_next_entry(rq, sched.link)) > last_prio)
1.1  riastrad 		return true;
1.1  riastrad
1.1  riastrad 	if (rb) {
1.1  riastrad 		struct virtual_engine *ve =
1.1  riastrad 			rb_entry(rb, typeof(*ve), nodes[engine->id].rb);
1.1  riastrad 		bool preempt = false;
1.1  riastrad
1.1  riastrad 		if (engine == ve->siblings[0]) { /* only preempt one sibling */
1.1  riastrad 			struct i915_request *next;
1.1  riastrad
1.1  riastrad 			rcu_read_lock();
1.1  riastrad 			next = READ_ONCE(ve->request);
1.1  riastrad 			if (next)
1.1  riastrad 				preempt = rq_prio(next) > last_prio;
1.1  riastrad 			rcu_read_unlock();
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		if (preempt)
1.1  riastrad 			return preempt;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If the inflight context did not trigger the preemption, then maybe
1.1  riastrad 	 * it was the set of queued requests? Pick the highest priority in
1.1  riastrad 	 * the queue (the first active priolist) and see if it deserves to be
1.1  riastrad 	 * running instead of ELSP[0].
1.1  riastrad 	 *
1.1  riastrad 	 * The highest priority request in the queue can not be either
1.1  riastrad 	 * ELSP[0] or ELSP[1] as, thanks again to PI, if it was the same
1.1  riastrad 	 * context, it's priority would not exceed ELSP[0] aka last_prio.
1.1  riastrad 	 */
1.1  riastrad 	return queue_prio(&engine->execlists) > last_prio;
1.1  riastrad }
1.1  riastrad
1.1  riastrad __maybe_unused static inline bool
1.1  riastrad assert_priority_queue(const struct i915_request *prev,
1.1  riastrad 		      const struct i915_request *next)
1.1  riastrad {
1.1  riastrad 	/*
1.1  riastrad 	 * Without preemption, the prev may refer to the still active element
1.1  riastrad 	 * which we refuse to let go.
1.1  riastrad 	 *
1.1  riastrad 	 * Even with preemption, there are times when we think it is better not
1.1  riastrad 	 * to preempt and leave an ostensibly lower priority request in flight.
1.1  riastrad 	 */
1.1  riastrad 	if (i915_request_is_active(prev))
1.1  riastrad 		return true;
1.1  riastrad
1.1  riastrad 	return rq_prio(prev) >= rq_prio(next);
1.1  riastrad }
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * The context descriptor encodes various attributes of a context,
1.1  riastrad  * including its GTT address and some flags. Because it's fairly
1.1  riastrad  * expensive to calculate, we'll just do it once and cache the result,
1.1  riastrad  * which remains valid until the context is unpinned.
1.1  riastrad  *
1.1  riastrad  * This is what a descriptor looks like, from LSB to MSB::
1.1  riastrad  *
1.1  riastrad  *      bits  0-11:    flags, GEN8_CTX_* (cached in ctx->desc_template)
1.1  riastrad  *      bits 12-31:    LRCA, GTT address of (the HWSP of) this context
1.1  riastrad  *      bits 32-52:    ctx ID, a globally unique tag (highest bit used by GuC)
1.1  riastrad  *      bits 53-54:    mbz, reserved for use by hardware
1.1  riastrad  *      bits 55-63:    group ID, currently unused and set to 0
1.1  riastrad  *
1.1  riastrad  * Starting from Gen11, the upper dword of the descriptor has a new format:
1.1  riastrad  *
1.1  riastrad  *      bits 32-36:    reserved
1.1  riastrad  *      bits 37-47:    SW context ID
1.1  riastrad  *      bits 48:53:    engine instance
1.1  riastrad  *      bit 54:        mbz, reserved for use by hardware
1.1  riastrad  *      bits 55-60:    SW counter
1.1  riastrad  *      bits 61-63:    engine class
1.1  riastrad  *
1.1  riastrad  * engine info, SW context ID and SW counter need to form a unique number
1.1  riastrad  * (Context ID) per lrc.
1.1  riastrad  */
1.1  riastrad static u64
1.1  riastrad lrc_descriptor(struct intel_context *ce, struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	u64 desc;
1.1  riastrad
1.1  riastrad 	desc = INTEL_LEGACY_32B_CONTEXT;
1.1  riastrad 	if (i915_vm_is_4lvl(ce->vm))
1.1  riastrad 		desc = INTEL_LEGACY_64B_CONTEXT;
1.1  riastrad 	desc <<= GEN8_CTX_ADDRESSING_MODE_SHIFT;
1.1  riastrad
1.1  riastrad 	desc |= GEN8_CTX_VALID | GEN8_CTX_PRIVILEGE;
1.1  riastrad 	if (IS_GEN(engine->i915, 8))
1.1  riastrad 		desc |= GEN8_CTX_L3LLC_COHERENT;
1.1  riastrad
1.1  riastrad 	desc |= i915_ggtt_offset(ce->state); /* bits 12-31 */
1.1  riastrad 	/*
1.1  riastrad 	 * The following 32bits are copied into the OA reports (dword 2).
1.1  riastrad 	 * Consider updating oa_get_render_ctx_id in i915_perf.c when changing
1.1  riastrad 	 * anything below.
1.1  riastrad 	 */
1.1  riastrad 	if (INTEL_GEN(engine->i915) >= 11) {
1.1  riastrad 		desc |= (u64)engine->instance << GEN11_ENGINE_INSTANCE_SHIFT;
1.1  riastrad 								/* bits 48-53 */
1.1  riastrad
1.1  riastrad 		desc |= (u64)engine->class << GEN11_ENGINE_CLASS_SHIFT;
1.1  riastrad 								/* bits 61-63 */
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return desc;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline unsigned int dword_in_page(void *addr)
1.1  riastrad {
1.1  riastrad 	return offset_in_page(addr) / sizeof(u32);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void set_offsets(u32 *regs,
1.1  riastrad 			const u8 *data,
1.1  riastrad 			const struct intel_engine_cs *engine,
1.1  riastrad 			bool clear)
1.1  riastrad #define NOP(x) (BIT(7) | (x))
1.1  riastrad #define LRI(count, flags) ((flags) << 6 | (count) | BUILD_BUG_ON_ZERO(count >= BIT(6)))
1.1  riastrad #define POSTED BIT(0)
1.1  riastrad #define REG(x) (((x) >> 2) | BUILD_BUG_ON_ZERO(x >= 0x200))
1.1  riastrad #define REG16(x) \
1.1  riastrad 	(((x) >> 9) | BIT(7) | BUILD_BUG_ON_ZERO(x >= 0x10000)), \
1.1  riastrad 	(((x) >> 2) & 0x7f)
1.1  riastrad #define END(x) 0, (x)
1.1  riastrad {
1.1  riastrad 	const u32 base = engine->mmio_base;
1.1  riastrad
1.1  riastrad 	while (*data) {
1.1  riastrad 		u8 count, flags;
1.1  riastrad
1.1  riastrad 		if (*data & BIT(7)) { /* skip */
1.1  riastrad 			count = *data++ & ~BIT(7);
1.1  riastrad 			if (clear)
1.1  riastrad 				memset32(regs, MI_NOOP, count);
1.1  riastrad 			regs += count;
1.1  riastrad 			continue;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		count = *data & 0x3f;
1.1  riastrad 		flags = *data >> 6;
1.1  riastrad 		data++;
1.1  riastrad
1.1  riastrad 		*regs = MI_LOAD_REGISTER_IMM(count);
1.1  riastrad 		if (flags & POSTED)
1.1  riastrad 			*regs |= MI_LRI_FORCE_POSTED;
1.1  riastrad 		if (INTEL_GEN(engine->i915) >= 11)
1.1  riastrad 			*regs |= MI_LRI_CS_MMIO;
1.1  riastrad 		regs++;
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(!count);
1.1  riastrad 		do {
1.1  riastrad 			u32 offset = 0;
1.1  riastrad 			u8 v;
1.1  riastrad
1.1  riastrad 			do {
1.1  riastrad 				v = *data++;
1.1  riastrad 				offset <<= 7;
1.1  riastrad 				offset |= v & ~BIT(7);
1.1  riastrad 			} while (v & BIT(7));
1.1  riastrad
1.1  riastrad 			regs[0] = base + (offset << 2);
1.1  riastrad 			if (clear)
1.1  riastrad 				regs[1] = 0;
1.1  riastrad 			regs += 2;
1.1  riastrad 		} while (--count);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (clear) {
1.1  riastrad 		u8 count = *++data;
1.1  riastrad
1.1  riastrad 		/* Clear past the tail for HW access */
1.1  riastrad 		GEM_BUG_ON(dword_in_page(regs) > count);
1.1  riastrad 		memset32(regs, MI_NOOP, count - dword_in_page(regs));
1.1  riastrad
1.1  riastrad 		/* Close the batch; used mainly by live_lrc_layout() */
1.1  riastrad 		*regs = MI_BATCH_BUFFER_END;
1.1  riastrad 		if (INTEL_GEN(engine->i915) >= 10)
1.1  riastrad 			*regs |= BIT(0);
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static const u8 gen8_xcs_offsets[] = {
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(11, 0),
1.1  riastrad 	REG16(0x244),
1.1  riastrad 	REG(0x034),
1.1  riastrad 	REG(0x030),
1.1  riastrad 	REG(0x038),
1.1  riastrad 	REG(0x03c),
1.1  riastrad 	REG(0x168),
1.1  riastrad 	REG(0x140),
1.1  riastrad 	REG(0x110),
1.1  riastrad 	REG(0x11c),
1.1  riastrad 	REG(0x114),
1.1  riastrad 	REG(0x118),
1.1  riastrad
1.1  riastrad 	NOP(9),
1.1  riastrad 	LRI(9, 0),
1.1  riastrad 	REG16(0x3a8),
1.1  riastrad 	REG16(0x28c),
1.1  riastrad 	REG16(0x288),
1.1  riastrad 	REG16(0x284),
1.1  riastrad 	REG16(0x280),
1.1  riastrad 	REG16(0x27c),
1.1  riastrad 	REG16(0x278),
1.1  riastrad 	REG16(0x274),
1.1  riastrad 	REG16(0x270),
1.1  riastrad
1.1  riastrad 	NOP(13),
1.1  riastrad 	LRI(2, 0),
1.1  riastrad 	REG16(0x200),
1.1  riastrad 	REG(0x028),
1.1  riastrad
1.1  riastrad 	END(80)
1.1  riastrad };
1.1  riastrad
1.1  riastrad static const u8 gen9_xcs_offsets[] = {
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(14, POSTED),
1.1  riastrad 	REG16(0x244),
1.1  riastrad 	REG(0x034),
1.1  riastrad 	REG(0x030),
1.1  riastrad 	REG(0x038),
1.1  riastrad 	REG(0x03c),
1.1  riastrad 	REG(0x168),
1.1  riastrad 	REG(0x140),
1.1  riastrad 	REG(0x110),
1.1  riastrad 	REG(0x11c),
1.1  riastrad 	REG(0x114),
1.1  riastrad 	REG(0x118),
1.1  riastrad 	REG(0x1c0),
1.1  riastrad 	REG(0x1c4),
1.1  riastrad 	REG(0x1c8),
1.1  riastrad
1.1  riastrad 	NOP(3),
1.1  riastrad 	LRI(9, POSTED),
1.1  riastrad 	REG16(0x3a8),
1.1  riastrad 	REG16(0x28c),
1.1  riastrad 	REG16(0x288),
1.1  riastrad 	REG16(0x284),
1.1  riastrad 	REG16(0x280),
1.1  riastrad 	REG16(0x27c),
1.1  riastrad 	REG16(0x278),
1.1  riastrad 	REG16(0x274),
1.1  riastrad 	REG16(0x270),
1.1  riastrad
1.1  riastrad 	NOP(13),
1.1  riastrad 	LRI(1, POSTED),
1.1  riastrad 	REG16(0x200),
1.1  riastrad
1.1  riastrad 	NOP(13),
1.1  riastrad 	LRI(44, POSTED),
1.1  riastrad 	REG(0x028),
1.1  riastrad 	REG(0x09c),
1.1  riastrad 	REG(0x0c0),
1.1  riastrad 	REG(0x178),
1.1  riastrad 	REG(0x17c),
1.1  riastrad 	REG16(0x358),
1.1  riastrad 	REG(0x170),
1.1  riastrad 	REG(0x150),
1.1  riastrad 	REG(0x154),
1.1  riastrad 	REG(0x158),
1.1  riastrad 	REG16(0x41c),
1.1  riastrad 	REG16(0x600),
1.1  riastrad 	REG16(0x604),
1.1  riastrad 	REG16(0x608),
1.1  riastrad 	REG16(0x60c),
1.1  riastrad 	REG16(0x610),
1.1  riastrad 	REG16(0x614),
1.1  riastrad 	REG16(0x618),
1.1  riastrad 	REG16(0x61c),
1.1  riastrad 	REG16(0x620),
1.1  riastrad 	REG16(0x624),
1.1  riastrad 	REG16(0x628),
1.1  riastrad 	REG16(0x62c),
1.1  riastrad 	REG16(0x630),
1.1  riastrad 	REG16(0x634),
1.1  riastrad 	REG16(0x638),
1.1  riastrad 	REG16(0x63c),
1.1  riastrad 	REG16(0x640),
1.1  riastrad 	REG16(0x644),
1.1  riastrad 	REG16(0x648),
1.1  riastrad 	REG16(0x64c),
1.1  riastrad 	REG16(0x650),
1.1  riastrad 	REG16(0x654),
1.1  riastrad 	REG16(0x658),
1.1  riastrad 	REG16(0x65c),
1.1  riastrad 	REG16(0x660),
1.1  riastrad 	REG16(0x664),
1.1  riastrad 	REG16(0x668),
1.1  riastrad 	REG16(0x66c),
1.1  riastrad 	REG16(0x670),
1.1  riastrad 	REG16(0x674),
1.1  riastrad 	REG16(0x678),
1.1  riastrad 	REG16(0x67c),
1.1  riastrad 	REG(0x068),
1.1  riastrad
1.1  riastrad 	END(176)
1.1  riastrad };
1.1  riastrad
1.1  riastrad static const u8 gen12_xcs_offsets[] = {
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(13, POSTED),
1.1  riastrad 	REG16(0x244),
1.1  riastrad 	REG(0x034),
1.1  riastrad 	REG(0x030),
1.1  riastrad 	REG(0x038),
1.1  riastrad 	REG(0x03c),
1.1  riastrad 	REG(0x168),
1.1  riastrad 	REG(0x140),
1.1  riastrad 	REG(0x110),
1.1  riastrad 	REG(0x1c0),
1.1  riastrad 	REG(0x1c4),
1.1  riastrad 	REG(0x1c8),
1.1  riastrad 	REG(0x180),
1.1  riastrad 	REG16(0x2b4),
1.1  riastrad
1.1  riastrad 	NOP(5),
1.1  riastrad 	LRI(9, POSTED),
1.1  riastrad 	REG16(0x3a8),
1.1  riastrad 	REG16(0x28c),
1.1  riastrad 	REG16(0x288),
1.1  riastrad 	REG16(0x284),
1.1  riastrad 	REG16(0x280),
1.1  riastrad 	REG16(0x27c),
1.1  riastrad 	REG16(0x278),
1.1  riastrad 	REG16(0x274),
1.1  riastrad 	REG16(0x270),
1.1  riastrad
1.1  riastrad 	END(80)
1.1  riastrad };
1.1  riastrad
1.1  riastrad static const u8 gen8_rcs_offsets[] = {
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(14, POSTED),
1.1  riastrad 	REG16(0x244),
1.1  riastrad 	REG(0x034),
1.1  riastrad 	REG(0x030),
1.1  riastrad 	REG(0x038),
1.1  riastrad 	REG(0x03c),
1.1  riastrad 	REG(0x168),
1.1  riastrad 	REG(0x140),
1.1  riastrad 	REG(0x110),
1.1  riastrad 	REG(0x11c),
1.1  riastrad 	REG(0x114),
1.1  riastrad 	REG(0x118),
1.1  riastrad 	REG(0x1c0),
1.1  riastrad 	REG(0x1c4),
1.1  riastrad 	REG(0x1c8),
1.1  riastrad
1.1  riastrad 	NOP(3),
1.1  riastrad 	LRI(9, POSTED),
1.1  riastrad 	REG16(0x3a8),
1.1  riastrad 	REG16(0x28c),
1.1  riastrad 	REG16(0x288),
1.1  riastrad 	REG16(0x284),
1.1  riastrad 	REG16(0x280),
1.1  riastrad 	REG16(0x27c),
1.1  riastrad 	REG16(0x278),
1.1  riastrad 	REG16(0x274),
1.1  riastrad 	REG16(0x270),
1.1  riastrad
1.1  riastrad 	NOP(13),
1.1  riastrad 	LRI(1, 0),
1.1  riastrad 	REG(0x0c8),
1.1  riastrad
1.1  riastrad 	END(80)
1.1  riastrad };
1.1  riastrad
1.1  riastrad static const u8 gen9_rcs_offsets[] = {
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(14, POSTED),
1.1  riastrad 	REG16(0x244),
1.1  riastrad 	REG(0x34),
1.1  riastrad 	REG(0x30),
1.1  riastrad 	REG(0x38),
1.1  riastrad 	REG(0x3c),
1.1  riastrad 	REG(0x168),
1.1  riastrad 	REG(0x140),
1.1  riastrad 	REG(0x110),
1.1  riastrad 	REG(0x11c),
1.1  riastrad 	REG(0x114),
1.1  riastrad 	REG(0x118),
1.1  riastrad 	REG(0x1c0),
1.1  riastrad 	REG(0x1c4),
1.1  riastrad 	REG(0x1c8),
1.1  riastrad
1.1  riastrad 	NOP(3),
1.1  riastrad 	LRI(9, POSTED),
1.1  riastrad 	REG16(0x3a8),
1.1  riastrad 	REG16(0x28c),
1.1  riastrad 	REG16(0x288),
1.1  riastrad 	REG16(0x284),
1.1  riastrad 	REG16(0x280),
1.1  riastrad 	REG16(0x27c),
1.1  riastrad 	REG16(0x278),
1.1  riastrad 	REG16(0x274),
1.1  riastrad 	REG16(0x270),
1.1  riastrad
1.1  riastrad 	NOP(13),
1.1  riastrad 	LRI(1, 0),
1.1  riastrad 	REG(0xc8),
1.1  riastrad
1.1  riastrad 	NOP(13),
1.1  riastrad 	LRI(44, POSTED),
1.1  riastrad 	REG(0x28),
1.1  riastrad 	REG(0x9c),
1.1  riastrad 	REG(0xc0),
1.1  riastrad 	REG(0x178),
1.1  riastrad 	REG(0x17c),
1.1  riastrad 	REG16(0x358),
1.1  riastrad 	REG(0x170),
1.1  riastrad 	REG(0x150),
1.1  riastrad 	REG(0x154),
1.1  riastrad 	REG(0x158),
1.1  riastrad 	REG16(0x41c),
1.1  riastrad 	REG16(0x600),
1.1  riastrad 	REG16(0x604),
1.1  riastrad 	REG16(0x608),
1.1  riastrad 	REG16(0x60c),
1.1  riastrad 	REG16(0x610),
1.1  riastrad 	REG16(0x614),
1.1  riastrad 	REG16(0x618),
1.1  riastrad 	REG16(0x61c),
1.1  riastrad 	REG16(0x620),
1.1  riastrad 	REG16(0x624),
1.1  riastrad 	REG16(0x628),
1.1  riastrad 	REG16(0x62c),
1.1  riastrad 	REG16(0x630),
1.1  riastrad 	REG16(0x634),
1.1  riastrad 	REG16(0x638),
1.1  riastrad 	REG16(0x63c),
1.1  riastrad 	REG16(0x640),
1.1  riastrad 	REG16(0x644),
1.1  riastrad 	REG16(0x648),
1.1  riastrad 	REG16(0x64c),
1.1  riastrad 	REG16(0x650),
1.1  riastrad 	REG16(0x654),
1.1  riastrad 	REG16(0x658),
1.1  riastrad 	REG16(0x65c),
1.1  riastrad 	REG16(0x660),
1.1  riastrad 	REG16(0x664),
1.1  riastrad 	REG16(0x668),
1.1  riastrad 	REG16(0x66c),
1.1  riastrad 	REG16(0x670),
1.1  riastrad 	REG16(0x674),
1.1  riastrad 	REG16(0x678),
1.1  riastrad 	REG16(0x67c),
1.1  riastrad 	REG(0x68),
1.1  riastrad
1.1  riastrad 	END(176)
1.1  riastrad };
1.1  riastrad
1.1  riastrad static const u8 gen11_rcs_offsets[] = {
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(15, POSTED),
1.1  riastrad 	REG16(0x244),
1.1  riastrad 	REG(0x034),
1.1  riastrad 	REG(0x030),
1.1  riastrad 	REG(0x038),
1.1  riastrad 	REG(0x03c),
1.1  riastrad 	REG(0x168),
1.1  riastrad 	REG(0x140),
1.1  riastrad 	REG(0x110),
1.1  riastrad 	REG(0x11c),
1.1  riastrad 	REG(0x114),
1.1  riastrad 	REG(0x118),
1.1  riastrad 	REG(0x1c0),
1.1  riastrad 	REG(0x1c4),
1.1  riastrad 	REG(0x1c8),
1.1  riastrad 	REG(0x180),
1.1  riastrad
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(9, POSTED),
1.1  riastrad 	REG16(0x3a8),
1.1  riastrad 	REG16(0x28c),
1.1  riastrad 	REG16(0x288),
1.1  riastrad 	REG16(0x284),
1.1  riastrad 	REG16(0x280),
1.1  riastrad 	REG16(0x27c),
1.1  riastrad 	REG16(0x278),
1.1  riastrad 	REG16(0x274),
1.1  riastrad 	REG16(0x270),
1.1  riastrad
1.1  riastrad 	LRI(1, POSTED),
1.1  riastrad 	REG(0x1b0),
1.1  riastrad
1.1  riastrad 	NOP(10),
1.1  riastrad 	LRI(1, 0),
1.1  riastrad 	REG(0x0c8),
1.1  riastrad
1.1  riastrad 	END(80)
1.1  riastrad };
1.1  riastrad
1.1  riastrad static const u8 gen12_rcs_offsets[] = {
1.1  riastrad 	NOP(1),
1.1  riastrad 	LRI(13, POSTED),
1.1  riastrad 	REG16(0x244),
1.1  riastrad 	REG(0x034),
1.1  riastrad 	REG(0x030),
1.1  riastrad 	REG(0x038),
1.1  riastrad 	REG(0x03c),
1.1  riastrad 	REG(0x168),
1.1  riastrad 	REG(0x140),
1.1  riastrad 	REG(0x110),
1.1  riastrad 	REG(0x1c0),
1.1  riastrad 	REG(0x1c4),
1.1  riastrad 	REG(0x1c8),
1.1  riastrad 	REG(0x180),
1.1  riastrad 	REG16(0x2b4),
1.1  riastrad
1.1  riastrad 	NOP(5),
1.1  riastrad 	LRI(9, POSTED),
1.1  riastrad 	REG16(0x3a8),
1.1  riastrad 	REG16(0x28c),
1.1  riastrad 	REG16(0x288),
1.1  riastrad 	REG16(0x284),
1.1  riastrad 	REG16(0x280),
1.1  riastrad 	REG16(0x27c),
1.1  riastrad 	REG16(0x278),
1.1  riastrad 	REG16(0x274),
1.1  riastrad 	REG16(0x270),
1.1  riastrad
1.1  riastrad 	LRI(3, POSTED),
1.1  riastrad 	REG(0x1b0),
1.1  riastrad 	REG16(0x5a8),
1.1  riastrad 	REG16(0x5ac),
1.1  riastrad
1.1  riastrad 	NOP(6),
1.1  riastrad 	LRI(1, 0),
1.1  riastrad 	REG(0x0c8),
1.1  riastrad
1.1  riastrad 	END(80)
1.1  riastrad };
1.1  riastrad
1.1  riastrad #undef END
1.1  riastrad #undef REG16
1.1  riastrad #undef REG
1.1  riastrad #undef LRI
1.1  riastrad #undef NOP
1.1  riastrad
1.1  riastrad static const u8 *reg_offsets(const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	/*
1.1  riastrad 	 * The gen12+ lists only have the registers we program in the basic
1.1  riastrad 	 * default state. We rely on the context image using relative
1.1  riastrad 	 * addressing to automatic fixup the register state between the
1.1  riastrad 	 * physical engines for virtual engine.
1.1  riastrad 	 */
1.1  riastrad 	GEM_BUG_ON(INTEL_GEN(engine->i915) >= 12 &&
1.1  riastrad 		   !intel_engine_has_relative_mmio(engine));
1.1  riastrad
1.1  riastrad 	if (engine->class == RENDER_CLASS) {
1.1  riastrad 		if (INTEL_GEN(engine->i915) >= 12)
1.1  riastrad 			return gen12_rcs_offsets;
1.1  riastrad 		else if (INTEL_GEN(engine->i915) >= 11)
1.1  riastrad 			return gen11_rcs_offsets;
1.1  riastrad 		else if (INTEL_GEN(engine->i915) >= 9)
1.1  riastrad 			return gen9_rcs_offsets;
1.1  riastrad 		else
1.1  riastrad 			return gen8_rcs_offsets;
1.1  riastrad 	} else {
1.1  riastrad 		if (INTEL_GEN(engine->i915) >= 12)
1.1  riastrad 			return gen12_xcs_offsets;
1.1  riastrad 		else if (INTEL_GEN(engine->i915) >= 9)
1.1  riastrad 			return gen9_xcs_offsets;
1.1  riastrad 		else
1.1  riastrad 			return gen8_xcs_offsets;
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static struct i915_request *
1.1  riastrad __unwind_incomplete_requests(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct i915_request *rq, *rn, *active = NULL;
1.1  riastrad 	struct list_head *uninitialized_var(pl);
1.1  riastrad 	int prio = I915_PRIORITY_INVALID;
1.1  riastrad
1.1  riastrad 	lockdep_assert_held(&engine->active.lock);
1.1  riastrad
1.1  riastrad 	list_for_each_entry_safe_reverse(rq, rn,
1.1  riastrad 					 &engine->active.requests,
1.1  riastrad 					 sched.link) {
1.1  riastrad 		if (i915_request_completed(rq))
1.1  riastrad 			continue; /* XXX */
1.1  riastrad
1.1  riastrad 		__i915_request_unsubmit(rq);
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * Push the request back into the queue for later resubmission.
1.1  riastrad 		 * If this request is not native to this physical engine (i.e.
1.1  riastrad 		 * it came from a virtual source), push it back onto the virtual
1.1  riastrad 		 * engine so that it can be moved across onto another physical
1.1  riastrad 		 * engine as load dictates.
1.1  riastrad 		 */
1.1  riastrad 		if (likely(rq->execution_mask == engine->mask)) {
1.1  riastrad 			GEM_BUG_ON(rq_prio(rq) == I915_PRIORITY_INVALID);
1.1  riastrad 			if (rq_prio(rq) != prio) {
1.1  riastrad 				prio = rq_prio(rq);
1.1  riastrad 				pl = i915_sched_lookup_priolist(engine, prio);
1.1  riastrad 			}
1.1  riastrad 			GEM_BUG_ON(RB_EMPTY_ROOT(&engine->execlists.queue.rb_root));
1.1  riastrad
1.1  riastrad 			list_move(&rq->sched.link, pl);
1.1  riastrad 			set_bit(I915_FENCE_FLAG_PQUEUE, &rq->fence.flags);
1.1  riastrad
1.1  riastrad 			active = rq;
1.1  riastrad 		} else {
1.1  riastrad 			struct intel_engine_cs *owner = rq->context->engine;
1.1  riastrad
1.1  riastrad 			/*
1.1  riastrad 			 * Decouple the virtual breadcrumb before moving it
1.1  riastrad 			 * back to the virtual engine -- we don't want the
1.1  riastrad 			 * request to complete in the background and try
1.1  riastrad 			 * and cancel the breadcrumb on the virtual engine
1.1  riastrad 			 * (instead of the old engine where it is linked)!
1.1  riastrad 			 */
1.1  riastrad 			if (test_bit(DMA_FENCE_FLAG_ENABLE_SIGNAL_BIT,
1.1  riastrad 				     &rq->fence.flags)) {
1.1  riastrad 				spin_lock_nested(&rq->lock,
1.1  riastrad 						 SINGLE_DEPTH_NESTING);
1.1  riastrad 				i915_request_cancel_breadcrumb(rq);
1.1  riastrad 				spin_unlock(&rq->lock);
1.1  riastrad 			}
1.1  riastrad 			rq->engine = owner;
1.1  riastrad 			owner->submit_request(rq);
1.1  riastrad 			active = NULL;
1.1  riastrad 		}
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return active;
1.1  riastrad }
1.1  riastrad
1.1  riastrad struct i915_request *
1.1  riastrad execlists_unwind_incomplete_requests(struct intel_engine_execlists *execlists)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_cs *engine =
1.1  riastrad 		container_of(execlists, typeof(*engine), execlists);
1.1  riastrad
1.1  riastrad 	return __unwind_incomplete_requests(engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void
1.1  riastrad execlists_context_status_change(struct i915_request *rq, unsigned long status)
1.1  riastrad {
1.1  riastrad 	/*
1.1  riastrad 	 * Only used when GVT-g is enabled now. When GVT-g is disabled,
1.1  riastrad 	 * The compiler should eliminate this function as dead-code.
1.1  riastrad 	 */
1.1  riastrad 	if (!IS_ENABLED(CONFIG_DRM_I915_GVT))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	atomic_notifier_call_chain(&rq->engine->context_status_notifier,
1.1  riastrad 				   status, rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void intel_engine_context_in(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	unsigned long flags;
1.1  riastrad
1.1  riastrad 	if (READ_ONCE(engine->stats.enabled) == 0)
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	write_seqlock_irqsave(&engine->stats.lock, flags);
1.1  riastrad
1.1  riastrad 	if (engine->stats.enabled > 0) {
1.1  riastrad 		if (engine->stats.active++ == 0)
1.1  riastrad 			engine->stats.start = ktime_get();
1.1  riastrad 		GEM_BUG_ON(engine->stats.active == 0);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	write_sequnlock_irqrestore(&engine->stats.lock, flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void intel_engine_context_out(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	unsigned long flags;
1.1  riastrad
1.1  riastrad 	if (READ_ONCE(engine->stats.enabled) == 0)
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	write_seqlock_irqsave(&engine->stats.lock, flags);
1.1  riastrad
1.1  riastrad 	if (engine->stats.enabled > 0) {
1.1  riastrad 		ktime_t last;
1.1  riastrad
1.1  riastrad 		if (engine->stats.active && --engine->stats.active == 0) {
1.1  riastrad 			/*
1.1  riastrad 			 * Decrement the active context count and in case GPU
1.1  riastrad 			 * is now idle add up to the running total.
1.1  riastrad 			 */
1.1  riastrad 			last = ktime_sub(ktime_get(), engine->stats.start);
1.1  riastrad
1.1  riastrad 			engine->stats.total = ktime_add(engine->stats.total,
1.1  riastrad 							last);
1.1  riastrad 		} else if (engine->stats.active == 0) {
1.1  riastrad 			/*
1.1  riastrad 			 * After turning on engine stats, context out might be
1.1  riastrad 			 * the first event in which case we account from the
1.1  riastrad 			 * time stats gathering was turned on.
1.1  riastrad 			 */
1.1  riastrad 			last = ktime_sub(ktime_get(), engine->stats.enabled_at);
1.1  riastrad
1.1  riastrad 			engine->stats.total = ktime_add(engine->stats.total,
1.1  riastrad 							last);
1.1  riastrad 		}
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	write_sequnlock_irqrestore(&engine->stats.lock, flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int lrc_ring_mi_mode(const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	if (INTEL_GEN(engine->i915) >= 12)
1.1  riastrad 		return 0x60;
1.1  riastrad 	else if (INTEL_GEN(engine->i915) >= 9)
1.1  riastrad 		return 0x54;
1.1  riastrad 	else if (engine->class == RENDER_CLASS)
1.1  riastrad 		return 0x58;
1.1  riastrad 	else
1.1  riastrad 		return -1;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void
1.1  riastrad execlists_check_context(const struct intel_context *ce,
1.1  riastrad 			const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	const struct intel_ring *ring = ce->ring;
1.1  riastrad 	u32 *regs = ce->lrc_reg_state;
1.1  riastrad 	bool valid = true;
1.1  riastrad 	int x;
1.1  riastrad
1.1  riastrad 	if (regs[CTX_RING_START] != i915_ggtt_offset(ring->vma)) {
1.1  riastrad 		pr_err("%s: context submitted with incorrect RING_START [%08x], expected %08x\n",
1.1  riastrad 		       engine->name,
1.1  riastrad 		       regs[CTX_RING_START],
1.1  riastrad 		       i915_ggtt_offset(ring->vma));
1.1  riastrad 		regs[CTX_RING_START] = i915_ggtt_offset(ring->vma);
1.1  riastrad 		valid = false;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if ((regs[CTX_RING_CTL] & ~(RING_WAIT | RING_WAIT_SEMAPHORE)) !=
1.1  riastrad 	    (RING_CTL_SIZE(ring->size) | RING_VALID)) {
1.1  riastrad 		pr_err("%s: context submitted with incorrect RING_CTL [%08x], expected %08x\n",
1.1  riastrad 		       engine->name,
1.1  riastrad 		       regs[CTX_RING_CTL],
1.1  riastrad 		       (u32)(RING_CTL_SIZE(ring->size) | RING_VALID));
1.1  riastrad 		regs[CTX_RING_CTL] = RING_CTL_SIZE(ring->size) | RING_VALID;
1.1  riastrad 		valid = false;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	x = lrc_ring_mi_mode(engine);
1.1  riastrad 	if (x != -1 && regs[x + 1] & (regs[x + 1] >> 16) & STOP_RING) {
1.1  riastrad 		pr_err("%s: context submitted with STOP_RING [%08x] in RING_MI_MODE\n",
1.1  riastrad 		       engine->name, regs[x + 1]);
1.1  riastrad 		regs[x + 1] &= ~STOP_RING;
1.1  riastrad 		regs[x + 1] |= STOP_RING << 16;
1.1  riastrad 		valid = false;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	WARN_ONCE(!valid, "Invalid lrc state found before submission\n");
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void restore_default_state(struct intel_context *ce,
1.1  riastrad 				  struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	u32 *regs = ce->lrc_reg_state;
1.1  riastrad
1.1  riastrad 	if (engine->pinned_default_state)
1.1  riastrad 		memcpy(regs, /* skip restoring the vanilla PPHWSP */
1.1  riastrad 		       engine->pinned_default_state + LRC_STATE_PN * PAGE_SIZE,
1.1  riastrad 		       engine->context_size - PAGE_SIZE);
1.1  riastrad
1.1  riastrad 	execlists_init_reg_state(regs, ce, engine, ce->ring, false);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void reset_active(struct i915_request *rq,
1.1  riastrad 			 struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_context * const ce = rq->context;
1.1  riastrad 	u32 head;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * The executing context has been cancelled. We want to prevent
1.1  riastrad 	 * further execution along this context and propagate the error on
1.1  riastrad 	 * to anything depending on its results.
1.1  riastrad 	 *
1.1  riastrad 	 * In __i915_request_submit(), we apply the -EIO and remove the
1.1  riastrad 	 * requests' payloads for any banned requests. But first, we must
1.1  riastrad 	 * rewind the context back to the start of the incomplete request so
1.1  riastrad 	 * that we do not jump back into the middle of the batch.
1.1  riastrad 	 *
1.1  riastrad 	 * We preserve the breadcrumbs and semaphores of the incomplete
1.1  riastrad 	 * requests so that inter-timeline dependencies (i.e other timelines)
1.1  riastrad 	 * remain correctly ordered. And we defer to __i915_request_submit()
1.1  riastrad 	 * so that all asynchronous waits are correctly handled.
1.1  riastrad 	 */
1.1  riastrad 	ENGINE_TRACE(engine, "{ rq=%llx:%lld }\n",
1.1  riastrad 		     rq->fence.context, rq->fence.seqno);
1.1  riastrad
1.1  riastrad 	/* On resubmission of the active request, payload will be scrubbed */
1.1  riastrad 	if (i915_request_completed(rq))
1.1  riastrad 		head = rq->tail;
1.1  riastrad 	else
1.1  riastrad 		head = active_request(ce->timeline, rq)->head;
1.1  riastrad 	head = intel_ring_wrap(ce->ring, head);
1.1  riastrad
1.1  riastrad 	/* Scrub the context image to prevent replaying the previous batch */
1.1  riastrad 	restore_default_state(ce, engine);
1.1  riastrad 	__execlists_update_reg_state(ce, engine, head);
1.1  riastrad
1.1  riastrad 	/* We've switched away, so this should be a no-op, but intent matters */
1.1  riastrad 	ce->lrc_desc |= CTX_DESC_FORCE_RESTORE;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline struct intel_engine_cs *
1.1  riastrad __execlists_schedule_in(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_cs * const engine = rq->engine;
1.1  riastrad 	struct intel_context * const ce = rq->context;
1.1  riastrad
1.1  riastrad 	intel_context_get(ce);
1.1  riastrad
1.1  riastrad 	if (unlikely(intel_context_is_banned(ce)))
1.1  riastrad 		reset_active(rq, engine);
1.1  riastrad
1.1  riastrad 	if (IS_ENABLED(CONFIG_DRM_I915_DEBUG_GEM))
1.1  riastrad 		execlists_check_context(ce, engine);
1.1  riastrad
1.1  riastrad 	if (ce->tag) {
1.1  riastrad 		/* Use a fixed tag for OA and friends */
1.1  riastrad 		ce->lrc_desc |= (u64)ce->tag << 32;
1.1  riastrad 	} else {
1.1  riastrad 		/* We don't need a strict matching tag, just different values */
1.1  riastrad 		ce->lrc_desc &= ~GENMASK_ULL(47, 37);
1.1  riastrad 		ce->lrc_desc |=
1.1  riastrad 			(u64)(++engine->context_tag % NUM_CONTEXT_TAG) <<
1.1  riastrad 			GEN11_SW_CTX_ID_SHIFT;
1.1  riastrad 		BUILD_BUG_ON(NUM_CONTEXT_TAG > GEN12_MAX_CONTEXT_HW_ID);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	__intel_gt_pm_get(engine->gt);
1.1  riastrad 	execlists_context_status_change(rq, INTEL_CONTEXT_SCHEDULE_IN);
1.1  riastrad 	intel_engine_context_in(engine);
1.1  riastrad
1.1  riastrad 	return engine;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline struct i915_request *
1.1  riastrad execlists_schedule_in(struct i915_request *rq, int idx)
1.1  riastrad {
1.1  riastrad 	struct intel_context * const ce = rq->context;
1.1  riastrad 	struct intel_engine_cs *old;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!intel_engine_pm_is_awake(rq->engine));
1.1  riastrad 	trace_i915_request_in(rq, idx);
1.1  riastrad
1.1  riastrad 	old = READ_ONCE(ce->inflight);
1.1  riastrad 	do {
1.1  riastrad 		if (!old) {
1.1  riastrad 			WRITE_ONCE(ce->inflight, __execlists_schedule_in(rq));
1.1  riastrad 			break;
1.1  riastrad 		}
1.1  riastrad 	} while (!try_cmpxchg(&ce->inflight, &old, ptr_inc(old)));
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(intel_context_inflight(ce) != rq->engine);
1.1  riastrad 	return i915_request_get(rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void kick_siblings(struct i915_request *rq, struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = container_of(ce, typeof(*ve), context);
1.1  riastrad 	struct i915_request *next = READ_ONCE(ve->request);
1.1  riastrad
1.1  riastrad 	if (next && next->execution_mask & ~rq->execution_mask)
1.1  riastrad 		tasklet_schedule(&ve->base.execlists.tasklet);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void
1.1  riastrad __execlists_schedule_out(struct i915_request *rq,
1.1  riastrad 			 struct intel_engine_cs * const engine)
1.1  riastrad {
1.1  riastrad 	struct intel_context * const ce = rq->context;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * NB process_csb() is not under the engine->active.lock and hence
1.1  riastrad 	 * schedule_out can race with schedule_in meaning that we should
1.1  riastrad 	 * refrain from doing non-trivial work here.
1.1  riastrad 	 */
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If we have just completed this context, the engine may now be
1.1  riastrad 	 * idle and we want to re-enter powersaving.
1.1  riastrad 	 */
1.1  riastrad 	if (list_is_last(&rq->link, &ce->timeline->requests) &&
1.1  riastrad 	    i915_request_completed(rq))
1.1  riastrad 		intel_engine_add_retire(engine, ce->timeline);
1.1  riastrad
1.1  riastrad 	intel_engine_context_out(engine);
1.1  riastrad 	execlists_context_status_change(rq, INTEL_CONTEXT_SCHEDULE_OUT);
1.1  riastrad 	intel_gt_pm_put_async(engine->gt);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If this is part of a virtual engine, its next request may
1.1  riastrad 	 * have been blocked waiting for access to the active context.
1.1  riastrad 	 * We have to kick all the siblings again in case we need to
1.1  riastrad 	 * switch (e.g. the next request is not runnable on this
1.1  riastrad 	 * engine). Hopefully, we will already have submitted the next
1.1  riastrad 	 * request before the tasklet runs and do not need to rebuild
1.1  riastrad 	 * each virtual tree and kick everyone again.
1.1  riastrad 	 */
1.1  riastrad 	if (ce->engine != engine)
1.1  riastrad 		kick_siblings(rq, ce);
1.1  riastrad
1.1  riastrad 	intel_context_put(ce);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void
1.1  riastrad execlists_schedule_out(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	struct intel_context * const ce = rq->context;
1.1  riastrad 	struct intel_engine_cs *cur, *old;
1.1  riastrad
1.1  riastrad 	trace_i915_request_out(rq);
1.1  riastrad
1.1  riastrad 	old = READ_ONCE(ce->inflight);
1.1  riastrad 	do
1.1  riastrad 		cur = ptr_unmask_bits(old, 2) ? ptr_dec(old) : NULL;
1.1  riastrad 	while (!try_cmpxchg(&ce->inflight, &old, cur));
1.1  riastrad 	if (!cur)
1.1  riastrad 		__execlists_schedule_out(rq, old);
1.1  riastrad
1.1  riastrad 	i915_request_put(rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u64 execlists_update_context(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	struct intel_context *ce = rq->context;
1.1  riastrad 	u64 desc = ce->lrc_desc;
1.1  riastrad 	u32 tail, prev;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * WaIdleLiteRestore:bdw,skl
1.1  riastrad 	 *
1.1  riastrad 	 * We should never submit the context with the same RING_TAIL twice
1.1  riastrad 	 * just in case we submit an empty ring, which confuses the HW.
1.1  riastrad 	 *
1.1  riastrad 	 * We append a couple of NOOPs (gen8_emit_wa_tail) after the end of
1.1  riastrad 	 * the normal request to be able to always advance the RING_TAIL on
1.1  riastrad 	 * subsequent resubmissions (for lite restore). Should that fail us,
1.1  riastrad 	 * and we try and submit the same tail again, force the context
1.1  riastrad 	 * reload.
1.1  riastrad 	 *
1.1  riastrad 	 * If we need to return to a preempted context, we need to skip the
1.1  riastrad 	 * lite-restore and force it to reload the RING_TAIL. Otherwise, the
1.1  riastrad 	 * HW has a tendency to ignore us rewinding the TAIL to the end of
1.1  riastrad 	 * an earlier request.
1.1  riastrad 	 */
1.1  riastrad 	tail = intel_ring_set_tail(rq->ring, rq->tail);
1.1  riastrad 	prev = ce->lrc_reg_state[CTX_RING_TAIL];
1.1  riastrad 	if (unlikely(intel_ring_direction(rq->ring, tail, prev) <= 0))
1.1  riastrad 		desc |= CTX_DESC_FORCE_RESTORE;
1.1  riastrad 	ce->lrc_reg_state[CTX_RING_TAIL] = tail;
1.1  riastrad 	rq->tail = rq->wa_tail;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Make sure the context image is complete before we submit it to HW.
1.1  riastrad 	 *
1.1  riastrad 	 * Ostensibly, writes (including the WCB) should be flushed prior to
1.1  riastrad 	 * an uncached write such as our mmio register access, the empirical
1.1  riastrad 	 * evidence (esp. on Braswell) suggests that the WC write into memory
1.1  riastrad 	 * may not be visible to the HW prior to the completion of the UC
1.1  riastrad 	 * register write and that we may begin execution from the context
1.1  riastrad 	 * before its image is complete leading to invalid PD chasing.
1.1  riastrad 	 */
1.1  riastrad 	wmb();
1.1  riastrad
1.1  riastrad 	ce->lrc_desc &= ~CTX_DESC_FORCE_RESTORE;
1.1  riastrad 	return desc;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void write_desc(struct intel_engine_execlists *execlists, u64 desc, u32 port)
1.1  riastrad {
1.4  riastrad #ifdef __NetBSD__
1.4  riastrad 	if (execlists->ctrl_reg) {
1.4  riastrad 		bus_space_write_4(execlists->bst, execlists->bsh, execlists->submit_reg + port * 2, lower_32_bits(desc));
1.4  riastrad 		bus_space_write_4(execlists->bst, execlists->bsh, execlists->submit_reg + port * 2 + 1, upper_32_bits(desc));
1.4  riastrad 	} else {
1.4  riastrad 		bus_space_write_4(execlists->bst, execlists->bsh, execlists->submit_reg, upper_32_bits(desc));
1.4  riastrad 		bus_space_write_4(execlists->bst, execlists->bsh, execlists->submit_reg, lower_32_bits(desc));
1.4  riastrad 	}
1.4  riastrad #else
1.1  riastrad 	if (execlists->ctrl_reg) {
1.1  riastrad 		writel(lower_32_bits(desc), execlists->submit_reg + port * 2);
1.1  riastrad 		writel(upper_32_bits(desc), execlists->submit_reg + port * 2 + 1);
1.1  riastrad 	} else {
1.1  riastrad 		writel(upper_32_bits(desc), execlists->submit_reg);
1.1  riastrad 		writel(lower_32_bits(desc), execlists->submit_reg);
1.1  riastrad 	}
1.4  riastrad #endif
1.1  riastrad }
1.1  riastrad
1.1  riastrad static __maybe_unused void
1.1  riastrad trace_ports(const struct intel_engine_execlists *execlists,
1.1  riastrad 	    const char *msg,
1.1  riastrad 	    struct i915_request * const *ports)
1.1  riastrad {
1.1  riastrad 	const struct intel_engine_cs *engine =
1.5  riastrad 		const_container_of(execlists, typeof(*engine), execlists);
1.1  riastrad
1.1  riastrad 	if (!ports[0])
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	ENGINE_TRACE(engine, "%s { %llx:%lld%s, %llx:%lld }\n", msg,
1.1  riastrad 		     ports[0]->fence.context,
1.1  riastrad 		     ports[0]->fence.seqno,
1.1  riastrad 		     i915_request_completed(ports[0]) ? "!" :
1.1  riastrad 		     i915_request_started(ports[0]) ? "*" :
1.1  riastrad 		     "",
1.1  riastrad 		     ports[1] ? ports[1]->fence.context : 0,
1.1  riastrad 		     ports[1] ? ports[1]->fence.seqno : 0);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static __maybe_unused bool
1.1  riastrad assert_pending_valid(const struct intel_engine_execlists *execlists,
1.1  riastrad 		     const char *msg)
1.1  riastrad {
1.1  riastrad 	struct i915_request * const *port, *rq;
1.1  riastrad 	struct intel_context *ce = NULL;
1.1  riastrad
1.1  riastrad 	trace_ports(execlists, msg, execlists->pending);
1.1  riastrad
1.1  riastrad 	if (!execlists->pending[0]) {
1.1  riastrad 		GEM_TRACE_ERR("Nothing pending for promotion!\n");
1.1  riastrad 		return false;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (execlists->pending[execlists_num_ports(execlists)]) {
1.1  riastrad 		GEM_TRACE_ERR("Excess pending[%d] for promotion!\n",
1.1  riastrad 			      execlists_num_ports(execlists));
1.1  riastrad 		return false;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	for (port = execlists->pending; (rq = *port); port++) {
1.1  riastrad 		unsigned long flags;
1.1  riastrad 		bool ok = true;
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(!kref_read(&rq->fence.refcount));
1.1  riastrad 		GEM_BUG_ON(!i915_request_is_active(rq));
1.1  riastrad
1.1  riastrad 		if (ce == rq->context) {
1.1  riastrad 			GEM_TRACE_ERR("Dup context:%llx in pending[%zd]\n",
1.1  riastrad 				      ce->timeline->fence_context,
1.1  riastrad 				      port - execlists->pending);
1.1  riastrad 			return false;
1.1  riastrad 		}
1.1  riastrad 		ce = rq->context;
1.1  riastrad
1.1  riastrad 		/* Hold tightly onto the lock to prevent concurrent retires! */
1.1  riastrad 		if (!spin_trylock_irqsave(&rq->lock, flags))
1.1  riastrad 			continue;
1.1  riastrad
1.1  riastrad 		if (i915_request_completed(rq))
1.1  riastrad 			goto unlock;
1.1  riastrad
1.1  riastrad 		if (i915_active_is_idle(&ce->active) &&
1.1  riastrad 		    !intel_context_is_barrier(ce)) {
1.1  riastrad 			GEM_TRACE_ERR("Inactive context:%llx in pending[%zd]\n",
1.1  riastrad 				      ce->timeline->fence_context,
1.1  riastrad 				      port - execlists->pending);
1.1  riastrad 			ok = false;
1.1  riastrad 			goto unlock;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		if (!i915_vma_is_pinned(ce->state)) {
1.1  riastrad 			GEM_TRACE_ERR("Unpinned context:%llx in pending[%zd]\n",
1.1  riastrad 				      ce->timeline->fence_context,
1.1  riastrad 				      port - execlists->pending);
1.1  riastrad 			ok = false;
1.1  riastrad 			goto unlock;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		if (!i915_vma_is_pinned(ce->ring->vma)) {
1.1  riastrad 			GEM_TRACE_ERR("Unpinned ring:%llx in pending[%zd]\n",
1.1  riastrad 				      ce->timeline->fence_context,
1.1  riastrad 				      port - execlists->pending);
1.1  riastrad 			ok = false;
1.1  riastrad 			goto unlock;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad unlock:
1.1  riastrad 		spin_unlock_irqrestore(&rq->lock, flags);
1.1  riastrad 		if (!ok)
1.1  riastrad 			return false;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return ce;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_submit_ports(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists *execlists = &engine->execlists;
1.1  riastrad 	unsigned int n;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!assert_pending_valid(execlists, "submit"));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We can skip acquiring intel_runtime_pm_get() here as it was taken
1.1  riastrad 	 * on our behalf by the request (see i915_gem_mark_busy()) and it will
1.1  riastrad 	 * not be relinquished until the device is idle (see
1.1  riastrad 	 * i915_gem_idle_work_handler()). As a precaution, we make sure
1.1  riastrad 	 * that all ELSP are drained i.e. we have processed the CSB,
1.1  riastrad 	 * before allowing ourselves to idle and calling intel_runtime_pm_put().
1.1  riastrad 	 */
1.1  riastrad 	GEM_BUG_ON(!intel_engine_pm_is_awake(engine));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * ELSQ note: the submit queue is not cleared after being submitted
1.1  riastrad 	 * to the HW so we need to make sure we always clean it up. This is
1.1  riastrad 	 * currently ensured by the fact that we always write the same number
1.1  riastrad 	 * of elsq entries, keep this in mind before changing the loop below.
1.1  riastrad 	 */
1.1  riastrad 	for (n = execlists_num_ports(execlists); n--; ) {
1.1  riastrad 		struct i915_request *rq = execlists->pending[n];
1.1  riastrad
1.1  riastrad 		write_desc(execlists,
1.1  riastrad 			   rq ? execlists_update_context(rq) : 0,
1.1  riastrad 			   n);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/* we need to manually load the submit queue */
1.1  riastrad 	if (execlists->ctrl_reg)
1.6  riastrad #ifdef __NetBSD__
1.6  riastrad 		bus_space_write_4(execlists->bst, execlists->bsh, execlists->ctrl_reg, EL_CTRL_LOAD);
1.6  riastrad #else
1.1  riastrad 		writel(EL_CTRL_LOAD, execlists->ctrl_reg);
1.6  riastrad #endif
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool ctx_single_port_submission(const struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	return (IS_ENABLED(CONFIG_DRM_I915_GVT) &&
1.1  riastrad 		intel_context_force_single_submission(ce));
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool can_merge_ctx(const struct intel_context *prev,
1.1  riastrad 			  const struct intel_context *next)
1.1  riastrad {
1.1  riastrad 	if (prev != next)
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	if (ctx_single_port_submission(prev))
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	return true;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool can_merge_rq(const struct i915_request *prev,
1.1  riastrad 			 const struct i915_request *next)
1.1  riastrad {
1.1  riastrad 	GEM_BUG_ON(prev == next);
1.1  riastrad 	GEM_BUG_ON(!assert_priority_queue(prev, next));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We do not submit known completed requests. Therefore if the next
1.1  riastrad 	 * request is already completed, we can pretend to merge it in
1.1  riastrad 	 * with the previous context (and we will skip updating the ELSP
1.1  riastrad 	 * and tracking). Thus hopefully keeping the ELSP full with active
1.1  riastrad 	 * contexts, despite the best efforts of preempt-to-busy to confuse
1.1  riastrad 	 * us.
1.1  riastrad 	 */
1.1  riastrad 	if (i915_request_completed(next))
1.1  riastrad 		return true;
1.1  riastrad
1.1  riastrad 	if (unlikely((prev->fence.flags ^ next->fence.flags) &
1.1  riastrad 		     (BIT(I915_FENCE_FLAG_NOPREEMPT) |
1.1  riastrad 		      BIT(I915_FENCE_FLAG_SENTINEL))))
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	if (!can_merge_ctx(prev->context, next->context))
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	return true;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_update_register_offsets(u32 *regs,
1.1  riastrad 					    struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	set_offsets(regs, reg_offsets(engine), engine, false);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool virtual_matches(const struct virtual_engine *ve,
1.1  riastrad 			    const struct i915_request *rq,
1.1  riastrad 			    const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	const struct intel_engine_cs *inflight;
1.1  riastrad
1.1  riastrad 	if (!(rq->execution_mask & engine->mask)) /* We peeked too soon! */
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We track when the HW has completed saving the context image
1.1  riastrad 	 * (i.e. when we have seen the final CS event switching out of
1.1  riastrad 	 * the context) and must not overwrite the context image before
1.1  riastrad 	 * then. This restricts us to only using the active engine
1.1  riastrad 	 * while the previous virtualized request is inflight (so
1.1  riastrad 	 * we reuse the register offsets). This is a very small
1.1  riastrad 	 * hystersis on the greedy seelction algorithm.
1.1  riastrad 	 */
1.1  riastrad 	inflight = intel_context_inflight(&ve->context);
1.1  riastrad 	if (inflight && inflight != engine)
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	return true;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_xfer_breadcrumbs(struct virtual_engine *ve,
1.1  riastrad 				     struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_cs *old = ve->siblings[0];
1.1  riastrad
1.1  riastrad 	/* All unattached (rq->engine == old) must already be completed */
1.1  riastrad
1.1  riastrad 	spin_lock(&old->breadcrumbs.irq_lock);
1.1  riastrad 	if (!list_empty(&ve->context.signal_link)) {
1.1  riastrad 		list_move_tail(&ve->context.signal_link,
1.1  riastrad 			       &engine->breadcrumbs.signalers);
1.1  riastrad 		intel_engine_signal_breadcrumbs(engine);
1.1  riastrad 	}
1.1  riastrad 	spin_unlock(&old->breadcrumbs.irq_lock);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static struct i915_request *
1.1  riastrad last_active(const struct intel_engine_execlists *execlists)
1.1  riastrad {
1.1  riastrad 	struct i915_request * const *last = READ_ONCE(execlists->active);
1.1  riastrad
1.1  riastrad 	while (*last && i915_request_completed(*last))
1.1  riastrad 		last++;
1.1  riastrad
1.1  riastrad 	return *last;
1.1  riastrad }
1.1  riastrad
1.1  riastrad #define for_each_waiter(p__, rq__) \
1.1  riastrad 	list_for_each_entry_lockless(p__, \
1.1  riastrad 				     &(rq__)->sched.waiters_list, \
1.1  riastrad 				     wait_link)
1.1  riastrad
1.1  riastrad static void defer_request(struct i915_request *rq, struct list_head * const pl)
1.1  riastrad {
1.1  riastrad 	LIST_HEAD(list);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We want to move the interrupted request to the back of
1.1  riastrad 	 * the round-robin list (i.e. its priority level), but
1.1  riastrad 	 * in doing so, we must then move all requests that were in
1.1  riastrad 	 * flight and were waiting for the interrupted request to
1.1  riastrad 	 * be run after it again.
1.1  riastrad 	 */
1.1  riastrad 	do {
1.1  riastrad 		struct i915_dependency *p;
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(i915_request_is_active(rq));
1.1  riastrad 		list_move_tail(&rq->sched.link, pl);
1.1  riastrad
1.1  riastrad 		for_each_waiter(p, rq) {
1.1  riastrad 			struct i915_request *w =
1.1  riastrad 				container_of(p->waiter, typeof(*w), sched);
1.1  riastrad
1.1  riastrad 			/* Leave semaphores spinning on the other engines */
1.1  riastrad 			if (w->engine != rq->engine)
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			/* No waiter should start before its signaler */
1.1  riastrad 			GEM_BUG_ON(i915_request_started(w) &&
1.1  riastrad 				   !i915_request_completed(rq));
1.1  riastrad
1.1  riastrad 			GEM_BUG_ON(i915_request_is_active(w));
1.1  riastrad 			if (!i915_request_is_ready(w))
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			if (rq_prio(w) < rq_prio(rq))
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			GEM_BUG_ON(rq_prio(w) > rq_prio(rq));
1.1  riastrad 			list_move_tail(&w->sched.link, &list);
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		rq = list_first_entry_or_null(&list, typeof(*rq), sched.link);
1.1  riastrad 	} while (rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void defer_active(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct i915_request *rq;
1.1  riastrad
1.1  riastrad 	rq = __unwind_incomplete_requests(engine);
1.1  riastrad 	if (!rq)
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	defer_request(rq, i915_sched_lookup_priolist(engine, rq_prio(rq)));
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool
1.1  riastrad need_timeslice(struct intel_engine_cs *engine, const struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	int hint;
1.1  riastrad
1.1  riastrad 	if (!intel_engine_has_timeslices(engine))
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	if (list_is_last(&rq->sched.link, &engine->active.requests))
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	hint = max(rq_prio(list_next_entry(rq, sched.link)),
1.1  riastrad 		   engine->execlists.queue_priority_hint);
1.1  riastrad
1.1  riastrad 	return hint >= effective_prio(rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int
1.1  riastrad switch_prio(struct intel_engine_cs *engine, const struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	if (list_is_last(&rq->sched.link, &engine->active.requests))
1.1  riastrad 		return INT_MIN;
1.1  riastrad
1.1  riastrad 	return rq_prio(list_next_entry(rq, sched.link));
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline unsigned long
1.1  riastrad timeslice(const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	return READ_ONCE(engine->props.timeslice_duration_ms);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static unsigned long
1.1  riastrad active_timeslice(const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	const struct i915_request *rq = *engine->execlists.active;
1.1  riastrad
1.1  riastrad 	if (!rq || i915_request_completed(rq))
1.1  riastrad 		return 0;
1.1  riastrad
1.1  riastrad 	if (engine->execlists.switch_priority_hint < effective_prio(rq))
1.1  riastrad 		return 0;
1.1  riastrad
1.1  riastrad 	return timeslice(engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void set_timeslice(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	if (!intel_engine_has_timeslices(engine))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	set_timer_ms(&engine->execlists.timer, active_timeslice(engine));
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void record_preemption(struct intel_engine_execlists *execlists)
1.1  riastrad {
1.1  riastrad 	(void)I915_SELFTEST_ONLY(execlists->preempt_hang.count++);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static unsigned long active_preempt_timeout(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct i915_request *rq;
1.1  riastrad
1.1  riastrad 	rq = last_active(&engine->execlists);
1.1  riastrad 	if (!rq)
1.1  riastrad 		return 0;
1.1  riastrad
1.1  riastrad 	/* Force a fast reset for terminated contexts (ignoring sysfs!) */
1.1  riastrad 	if (unlikely(intel_context_is_banned(rq->context)))
1.1  riastrad 		return 1;
1.1  riastrad
1.1  riastrad 	return READ_ONCE(engine->props.preempt_timeout_ms);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void set_preempt_timeout(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	if (!intel_engine_has_preempt_reset(engine))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	set_timer_ms(&engine->execlists.preempt,
1.1  riastrad 		     active_preempt_timeout(engine));
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void clear_ports(struct i915_request **ports, int count)
1.1  riastrad {
1.1  riastrad 	memset_p((void **)ports, NULL, count);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_dequeue(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad 	struct i915_request **port = execlists->pending;
1.1  riastrad 	struct i915_request ** const last_port = port + execlists->port_mask;
1.1  riastrad 	struct i915_request *last;
1.1  riastrad 	struct rb_node *rb;
1.1  riastrad 	bool submit = false;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Hardware submission is through 2 ports. Conceptually each port
1.1  riastrad 	 * has a (RING_START, RING_HEAD, RING_TAIL) tuple. RING_START is
1.1  riastrad 	 * static for a context, and unique to each, so we only execute
1.1  riastrad 	 * requests belonging to a single context from each ring. RING_HEAD
1.1  riastrad 	 * is maintained by the CS in the context image, it marks the place
1.1  riastrad 	 * where it got up to last time, and through RING_TAIL we tell the CS
1.1  riastrad 	 * where we want to execute up to this time.
1.1  riastrad 	 *
1.1  riastrad 	 * In this list the requests are in order of execution. Consecutive
1.1  riastrad 	 * requests from the same context are adjacent in the ringbuffer. We
1.1  riastrad 	 * can combine these requests into a single RING_TAIL update:
1.1  riastrad 	 *
1.1  riastrad 	 *              RING_HEAD...req1...req2
1.1  riastrad 	 *                                    ^- RING_TAIL
1.1  riastrad 	 * since to execute req2 the CS must first execute req1.
1.1  riastrad 	 *
1.1  riastrad 	 * Our goal then is to point each port to the end of a consecutive
1.1  riastrad 	 * sequence of requests as being the most optimal (fewest wake ups
1.1  riastrad 	 * and context switches) submission.
1.1  riastrad 	 */
1.1  riastrad
1.1  riastrad 	for (rb = rb_first_cached(&execlists->virtual); rb; ) {
1.1  riastrad 		struct virtual_engine *ve =
1.1  riastrad 			rb_entry(rb, typeof(*ve), nodes[engine->id].rb);
1.1  riastrad 		struct i915_request *rq = READ_ONCE(ve->request);
1.1  riastrad
1.1  riastrad 		if (!rq) { /* lazily cleanup after another engine handled rq */
1.1  riastrad 			rb_erase_cached(rb, &execlists->virtual);
1.7  riastrad 			container_of(rb, struct ve_node, rb)->inserted =
1.7  riastrad 			    false;
1.1  riastrad 			rb = rb_first_cached(&execlists->virtual);
1.1  riastrad 			continue;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		if (!virtual_matches(ve, rq, engine)) {
1.7  riastrad 			rb = rb_next2(&execlists->virtual.rb_root, rb);
1.1  riastrad 			continue;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		break;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If the queue is higher priority than the last
1.1  riastrad 	 * request in the currently active context, submit afresh.
1.1  riastrad 	 * We will resubmit again afterwards in case we need to split
1.1  riastrad 	 * the active context to interject the preemption request,
1.1  riastrad 	 * i.e. we will retrigger preemption following the ack in case
1.1  riastrad 	 * of trouble.
1.1  riastrad 	 */
1.1  riastrad 	last = last_active(execlists);
1.1  riastrad 	if (last) {
1.1  riastrad 		if (need_preempt(engine, last, rb)) {
1.1  riastrad 			ENGINE_TRACE(engine,
1.1  riastrad 				     "preempting last=%llx:%lld, prio=%d, hint=%d\n",
1.1  riastrad 				     last->fence.context,
1.1  riastrad 				     last->fence.seqno,
1.1  riastrad 				     last->sched.attr.priority,
1.1  riastrad 				     execlists->queue_priority_hint);
1.1  riastrad 			record_preemption(execlists);
1.1  riastrad
1.1  riastrad 			/*
1.1  riastrad 			 * Don't let the RING_HEAD advance past the breadcrumb
1.1  riastrad 			 * as we unwind (and until we resubmit) so that we do
1.1  riastrad 			 * not accidentally tell it to go backwards.
1.1  riastrad 			 */
1.1  riastrad 			ring_set_paused(engine, 1);
1.1  riastrad
1.1  riastrad 			/*
1.1  riastrad 			 * Note that we have not stopped the GPU at this point,
1.1  riastrad 			 * so we are unwinding the incomplete requests as they
1.1  riastrad 			 * remain inflight and so by the time we do complete
1.1  riastrad 			 * the preemption, some of the unwound requests may
1.1  riastrad 			 * complete!
1.1  riastrad 			 */
1.1  riastrad 			__unwind_incomplete_requests(engine);
1.1  riastrad
1.1  riastrad 			last = NULL;
1.1  riastrad 		} else if (need_timeslice(engine, last) &&
1.1  riastrad 			   timer_expired(&engine->execlists.timer)) {
1.1  riastrad 			ENGINE_TRACE(engine,
1.1  riastrad 				     "expired last=%llx:%lld, prio=%d, hint=%d\n",
1.1  riastrad 				     last->fence.context,
1.1  riastrad 				     last->fence.seqno,
1.1  riastrad 				     last->sched.attr.priority,
1.1  riastrad 				     execlists->queue_priority_hint);
1.1  riastrad
1.1  riastrad 			ring_set_paused(engine, 1);
1.1  riastrad 			defer_active(engine);
1.1  riastrad
1.1  riastrad 			/*
1.1  riastrad 			 * Unlike for preemption, if we rewind and continue
1.1  riastrad 			 * executing the same context as previously active,
1.1  riastrad 			 * the order of execution will remain the same and
1.1  riastrad 			 * the tail will only advance. We do not need to
1.1  riastrad 			 * force a full context restore, as a lite-restore
1.1  riastrad 			 * is sufficient to resample the monotonic TAIL.
1.1  riastrad 			 *
1.1  riastrad 			 * If we switch to any other context, similarly we
1.1  riastrad 			 * will not rewind TAIL of current context, and
1.1  riastrad 			 * normal save/restore will preserve state and allow
1.1  riastrad 			 * us to later continue executing the same request.
1.1  riastrad 			 */
1.1  riastrad 			last = NULL;
1.1  riastrad 		} else {
1.1  riastrad 			/*
1.1  riastrad 			 * Otherwise if we already have a request pending
1.1  riastrad 			 * for execution after the current one, we can
1.1  riastrad 			 * just wait until the next CS event before
1.1  riastrad 			 * queuing more. In either case we will force a
1.1  riastrad 			 * lite-restore preemption event, but if we wait
1.1  riastrad 			 * we hopefully coalesce several updates into a single
1.1  riastrad 			 * submission.
1.1  riastrad 			 */
1.1  riastrad 			if (!list_is_last(&last->sched.link,
1.1  riastrad 					  &engine->active.requests)) {
1.1  riastrad 				/*
1.1  riastrad 				 * Even if ELSP[1] is occupied and not worthy
1.1  riastrad 				 * of timeslices, our queue might be.
1.1  riastrad 				 */
1.7  riastrad 				if (!timer_pending(&execlists->timer) &&
1.1  riastrad 				    need_timeslice(engine, last))
1.1  riastrad 					set_timer_ms(&execlists->timer,
1.1  riastrad 						     timeslice(engine));
1.1  riastrad
1.1  riastrad 				return;
1.1  riastrad 			}
1.1  riastrad 		}
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	while (rb) { /* XXX virtual is always taking precedence */
1.1  riastrad 		struct virtual_engine *ve =
1.1  riastrad 			rb_entry(rb, typeof(*ve), nodes[engine->id].rb);
1.1  riastrad 		struct i915_request *rq;
1.1  riastrad
1.1  riastrad 		spin_lock(&ve->base.active.lock);
1.1  riastrad
1.1  riastrad 		rq = ve->request;
1.1  riastrad 		if (unlikely(!rq)) { /* lost the race to a sibling */
1.1  riastrad 			spin_unlock(&ve->base.active.lock);
1.1  riastrad 			rb_erase_cached(rb, &execlists->virtual);
1.7  riastrad 			container_of(rb, struct ve_node, rb)->inserted =
1.7  riastrad 			    false;
1.1  riastrad 			rb = rb_first_cached(&execlists->virtual);
1.1  riastrad 			continue;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(rq != ve->request);
1.1  riastrad 		GEM_BUG_ON(rq->engine != &ve->base);
1.1  riastrad 		GEM_BUG_ON(rq->context != &ve->context);
1.1  riastrad
1.1  riastrad 		if (rq_prio(rq) >= queue_prio(execlists)) {
1.1  riastrad 			if (!virtual_matches(ve, rq, engine)) {
1.1  riastrad 				spin_unlock(&ve->base.active.lock);
1.7  riastrad 				rb = rb_next2(&execlists->virtual.rb_root,
1.7  riastrad 				    rb);
1.1  riastrad 				continue;
1.1  riastrad 			}
1.1  riastrad
1.1  riastrad 			if (last && !can_merge_rq(last, rq)) {
1.1  riastrad 				spin_unlock(&ve->base.active.lock);
1.1  riastrad 				return; /* leave this for another */
1.1  riastrad 			}
1.1  riastrad
1.1  riastrad 			ENGINE_TRACE(engine,
1.1  riastrad 				     "virtual rq=%llx:%lld%s, new engine? %s\n",
1.1  riastrad 				     rq->fence.context,
1.1  riastrad 				     rq->fence.seqno,
1.1  riastrad 				     i915_request_completed(rq) ? "!" :
1.1  riastrad 				     i915_request_started(rq) ? "*" :
1.1  riastrad 				     "",
1.1  riastrad 				     yesno(engine != ve->siblings[0]));
1.1  riastrad
1.1  riastrad 			ve->request = NULL;
1.1  riastrad 			ve->base.execlists.queue_priority_hint = INT_MIN;
1.1  riastrad 			rb_erase_cached(rb, &execlists->virtual);
1.7  riastrad 			container_of(rb, struct ve_node, rb)->inserted =
1.7  riastrad 			    false;
1.1  riastrad
1.1  riastrad 			GEM_BUG_ON(!(rq->execution_mask & engine->mask));
1.1  riastrad 			rq->engine = engine;
1.1  riastrad
1.1  riastrad 			if (engine != ve->siblings[0]) {
1.1  riastrad 				u32 *regs = ve->context.lrc_reg_state;
1.1  riastrad 				unsigned int n;
1.1  riastrad
1.1  riastrad 				GEM_BUG_ON(READ_ONCE(ve->context.inflight));
1.1  riastrad
1.1  riastrad 				if (!intel_engine_has_relative_mmio(engine))
1.1  riastrad 					virtual_update_register_offsets(regs,
1.1  riastrad 									engine);
1.1  riastrad
1.1  riastrad 				if (!list_empty(&ve->context.signals))
1.1  riastrad 					virtual_xfer_breadcrumbs(ve, engine);
1.1  riastrad
1.1  riastrad 				/*
1.1  riastrad 				 * Move the bound engine to the top of the list
1.1  riastrad 				 * for future execution. We then kick this
1.1  riastrad 				 * tasklet first before checking others, so that
1.1  riastrad 				 * we preferentially reuse this set of bound
1.1  riastrad 				 * registers.
1.1  riastrad 				 */
1.1  riastrad 				for (n = 1; n < ve->num_siblings; n++) {
1.1  riastrad 					if (ve->siblings[n] == engine) {
1.1  riastrad 						swap(ve->siblings[n],
1.1  riastrad 						     ve->siblings[0]);
1.1  riastrad 						break;
1.1  riastrad 					}
1.1  riastrad 				}
1.1  riastrad
1.1  riastrad 				GEM_BUG_ON(ve->siblings[0] != engine);
1.1  riastrad 			}
1.1  riastrad
1.1  riastrad 			if (__i915_request_submit(rq)) {
1.1  riastrad 				submit = true;
1.1  riastrad 				last = rq;
1.1  riastrad 			}
1.1  riastrad 			i915_request_put(rq);
1.1  riastrad
1.1  riastrad 			/*
1.1  riastrad 			 * Hmm, we have a bunch of virtual engine requests,
1.1  riastrad 			 * but the first one was already completed (thanks
1.1  riastrad 			 * preempt-to-busy!). Keep looking at the veng queue
1.1  riastrad 			 * until we have no more relevant requests (i.e.
1.1  riastrad 			 * the normal submit queue has higher priority).
1.1  riastrad 			 */
1.1  riastrad 			if (!submit) {
1.1  riastrad 				spin_unlock(&ve->base.active.lock);
1.1  riastrad 				rb = rb_first_cached(&execlists->virtual);
1.1  riastrad 				continue;
1.1  riastrad 			}
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		spin_unlock(&ve->base.active.lock);
1.1  riastrad 		break;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	while ((rb = rb_first_cached(&execlists->queue))) {
1.1  riastrad 		struct i915_priolist *p = to_priolist(rb);
1.1  riastrad 		struct i915_request *rq, *rn;
1.1  riastrad 		int i;
1.1  riastrad
1.1  riastrad 		priolist_for_each_request_consume(rq, rn, p, i) {
1.1  riastrad 			bool merge = true;
1.1  riastrad
1.1  riastrad 			/*
1.1  riastrad 			 * Can we combine this request with the current port?
1.1  riastrad 			 * It has to be the same context/ringbuffer and not
1.1  riastrad 			 * have any exceptions (e.g. GVT saying never to
1.1  riastrad 			 * combine contexts).
1.1  riastrad 			 *
1.1  riastrad 			 * If we can combine the requests, we can execute both
1.1  riastrad 			 * by updating the RING_TAIL to point to the end of the
1.1  riastrad 			 * second request, and so we never need to tell the
1.1  riastrad 			 * hardware about the first.
1.1  riastrad 			 */
1.1  riastrad 			if (last && !can_merge_rq(last, rq)) {
1.1  riastrad 				/*
1.1  riastrad 				 * If we are on the second port and cannot
1.1  riastrad 				 * combine this request with the last, then we
1.1  riastrad 				 * are done.
1.1  riastrad 				 */
1.1  riastrad 				if (port == last_port)
1.1  riastrad 					goto done;
1.1  riastrad
1.1  riastrad 				/*
1.1  riastrad 				 * We must not populate both ELSP[] with the
1.1  riastrad 				 * same LRCA, i.e. we must submit 2 different
1.1  riastrad 				 * contexts if we submit 2 ELSP.
1.1  riastrad 				 */
1.1  riastrad 				if (last->context == rq->context)
1.1  riastrad 					goto done;
1.1  riastrad
1.1  riastrad 				if (i915_request_has_sentinel(last))
1.1  riastrad 					goto done;
1.1  riastrad
1.1  riastrad 				/*
1.1  riastrad 				 * If GVT overrides us we only ever submit
1.1  riastrad 				 * port[0], leaving port[1] empty. Note that we
1.1  riastrad 				 * also have to be careful that we don't queue
1.1  riastrad 				 * the same context (even though a different
1.1  riastrad 				 * request) to the second port.
1.1  riastrad 				 */
1.1  riastrad 				if (ctx_single_port_submission(last->context) ||
1.1  riastrad 				    ctx_single_port_submission(rq->context))
1.1  riastrad 					goto done;
1.1  riastrad
1.1  riastrad 				merge = false;
1.1  riastrad 			}
1.1  riastrad
1.1  riastrad 			if (__i915_request_submit(rq)) {
1.1  riastrad 				if (!merge) {
1.1  riastrad 					*port = execlists_schedule_in(last, port - execlists->pending);
1.1  riastrad 					port++;
1.1  riastrad 					last = NULL;
1.1  riastrad 				}
1.1  riastrad
1.1  riastrad 				GEM_BUG_ON(last &&
1.1  riastrad 					   !can_merge_ctx(last->context,
1.1  riastrad 							  rq->context));
1.1  riastrad
1.1  riastrad 				submit = true;
1.1  riastrad 				last = rq;
1.1  riastrad 			}
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		rb_erase_cached(&p->node, &execlists->queue);
1.1  riastrad 		i915_priolist_free(p);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad done:
1.1  riastrad 	/*
1.1  riastrad 	 * Here be a bit of magic! Or sleight-of-hand, whichever you prefer.
1.1  riastrad 	 *
1.1  riastrad 	 * We choose the priority hint such that if we add a request of greater
1.1  riastrad 	 * priority than this, we kick the submission tasklet to decide on
1.1  riastrad 	 * the right order of submitting the requests to hardware. We must
1.1  riastrad 	 * also be prepared to reorder requests as they are in-flight on the
1.1  riastrad 	 * HW. We derive the priority hint then as the first "hole" in
1.1  riastrad 	 * the HW submission ports and if there are no available slots,
1.1  riastrad 	 * the priority of the lowest executing request, i.e. last.
1.1  riastrad 	 *
1.1  riastrad 	 * When we do receive a higher priority request ready to run from the
1.1  riastrad 	 * user, see queue_request(), the priority hint is bumped to that
1.1  riastrad 	 * request triggering preemption on the next dequeue (or subsequent
1.1  riastrad 	 * interrupt for secondary ports).
1.1  riastrad 	 */
1.1  riastrad 	execlists->queue_priority_hint = queue_prio(execlists);
1.1  riastrad
1.1  riastrad 	if (submit) {
1.1  riastrad 		*port = execlists_schedule_in(last, port - execlists->pending);
1.1  riastrad 		execlists->switch_priority_hint =
1.1  riastrad 			switch_prio(engine, *execlists->pending);
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * Skip if we ended up with exactly the same set of requests,
1.1  riastrad 		 * e.g. trying to timeslice a pair of ordered contexts
1.1  riastrad 		 */
1.1  riastrad 		if (!memcmp(execlists->active, execlists->pending,
1.1  riastrad 			    (port - execlists->pending + 1) * sizeof(*port))) {
1.1  riastrad 			do
1.1  riastrad 				execlists_schedule_out(fetch_and_zero(port));
1.1  riastrad 			while (port-- != execlists->pending);
1.1  riastrad
1.1  riastrad 			goto skip_submit;
1.1  riastrad 		}
1.1  riastrad 		clear_ports(port + 1, last_port - port);
1.1  riastrad
1.1  riastrad 		execlists_submit_ports(engine);
1.1  riastrad 		set_preempt_timeout(engine);
1.1  riastrad 	} else {
1.1  riastrad skip_submit:
1.1  riastrad 		ring_set_paused(engine, 0);
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void
1.1  riastrad cancel_port_requests(struct intel_engine_execlists * const execlists)
1.1  riastrad {
1.1  riastrad 	struct i915_request * const *port;
1.1  riastrad
1.1  riastrad 	for (port = execlists->pending; *port; port++)
1.1  riastrad 		execlists_schedule_out(*port);
1.1  riastrad 	clear_ports(execlists->pending, ARRAY_SIZE(execlists->pending));
1.1  riastrad
1.1  riastrad 	/* Mark the end of active before we overwrite *active */
1.1  riastrad 	for (port = xchg(&execlists->active, execlists->pending); *port; port++)
1.1  riastrad 		execlists_schedule_out(*port);
1.1  riastrad 	clear_ports(execlists->inflight, ARRAY_SIZE(execlists->inflight));
1.1  riastrad
1.1  riastrad 	WRITE_ONCE(execlists->active, execlists->inflight);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void
1.1  riastrad invalidate_csb_entries(const u32 *first, const u32 *last)
1.1  riastrad {
1.7  riastrad 	clflush(__UNCONST(first));
1.7  riastrad 	clflush(__UNCONST(last));
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline bool
1.1  riastrad reset_in_progress(const struct intel_engine_execlists *execlists)
1.1  riastrad {
1.1  riastrad 	return unlikely(!__tasklet_is_enabled(&execlists->tasklet));
1.1  riastrad }
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * Starting with Gen12, the status has a new format:
1.1  riastrad  *
1.1  riastrad  *     bit  0:     switched to new queue
1.1  riastrad  *     bit  1:     reserved
1.1  riastrad  *     bit  2:     semaphore wait mode (poll or signal), only valid when
1.1  riastrad  *                 switch detail is set to "wait on semaphore"
1.1  riastrad  *     bits 3-5:   engine class
1.1  riastrad  *     bits 6-11:  engine instance
1.1  riastrad  *     bits 12-14: reserved
1.1  riastrad  *     bits 15-25: sw context id of the lrc the GT switched to
1.1  riastrad  *     bits 26-31: sw counter of the lrc the GT switched to
1.1  riastrad  *     bits 32-35: context switch detail
1.1  riastrad  *                  - 0: ctx complete
1.1  riastrad  *                  - 1: wait on sync flip
1.1  riastrad  *                  - 2: wait on vblank
1.1  riastrad  *                  - 3: wait on scanline
1.1  riastrad  *                  - 4: wait on semaphore
1.1  riastrad  *                  - 5: context preempted (not on SEMAPHORE_WAIT or
1.1  riastrad  *                       WAIT_FOR_EVENT)
1.1  riastrad  *     bit  36:    reserved
1.1  riastrad  *     bits 37-43: wait detail (for switch detail 1 to 4)
1.1  riastrad  *     bits 44-46: reserved
1.1  riastrad  *     bits 47-57: sw context id of the lrc the GT switched away from
1.1  riastrad  *     bits 58-63: sw counter of the lrc the GT switched away from
1.1  riastrad  */
1.1  riastrad static inline bool
1.1  riastrad gen12_csb_parse(const struct intel_engine_execlists *execlists, const u32 *csb)
1.1  riastrad {
1.1  riastrad 	u32 lower_dw = csb[0];
1.1  riastrad 	u32 upper_dw = csb[1];
1.1  riastrad 	bool ctx_to_valid = GEN12_CSB_CTX_VALID(lower_dw);
1.1  riastrad 	bool ctx_away_valid = GEN12_CSB_CTX_VALID(upper_dw);
1.1  riastrad 	bool new_queue = lower_dw & GEN12_CTX_STATUS_SWITCHED_TO_NEW_QUEUE;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * The context switch detail is not guaranteed to be 5 when a preemption
1.1  riastrad 	 * occurs, so we can't just check for that. The check below works for
1.1  riastrad 	 * all the cases we care about, including preemptions of WAIT
1.1  riastrad 	 * instructions and lite-restore. Preempt-to-idle via the CTRL register
1.1  riastrad 	 * would require some extra handling, but we don't support that.
1.1  riastrad 	 */
1.1  riastrad 	if (!ctx_away_valid || new_queue) {
1.1  riastrad 		GEM_BUG_ON(!ctx_to_valid);
1.1  riastrad 		return true;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * switch detail = 5 is covered by the case above and we do not expect a
1.1  riastrad 	 * context switch on an unsuccessful wait instruction since we always
1.1  riastrad 	 * use polling mode.
1.1  riastrad 	 */
1.1  riastrad 	GEM_BUG_ON(GEN12_CTX_SWITCH_DETAIL(upper_dw));
1.1  riastrad 	return false;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline bool
1.1  riastrad gen8_csb_parse(const struct intel_engine_execlists *execlists, const u32 *csb)
1.1  riastrad {
1.1  riastrad 	return *csb & (GEN8_CTX_STATUS_IDLE_ACTIVE | GEN8_CTX_STATUS_PREEMPTED);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void process_csb(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad 	const u32 * const buf = execlists->csb_status;
1.1  riastrad 	const u8 num_entries = execlists->csb_size;
1.1  riastrad 	u8 head, tail;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * As we modify our execlists state tracking we require exclusive
1.1  riastrad 	 * access. Either we are inside the tasklet, or the tasklet is disabled
1.1  riastrad 	 * and we assume that is only inside the reset paths and so serialised.
1.1  riastrad 	 */
1.1  riastrad 	GEM_BUG_ON(!tasklet_is_locked(&execlists->tasklet) &&
1.1  riastrad 		   !reset_in_progress(execlists));
1.1  riastrad 	GEM_BUG_ON(!intel_engine_in_execlists_submission_mode(engine));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Note that csb_write, csb_status may be either in HWSP or mmio.
1.1  riastrad 	 * When reading from the csb_write mmio register, we have to be
1.1  riastrad 	 * careful to only use the GEN8_CSB_WRITE_PTR portion, which is
1.1  riastrad 	 * the low 4bits. As it happens we know the next 4bits are always
1.1  riastrad 	 * zero and so we can simply masked off the low u8 of the register
1.1  riastrad 	 * and treat it identically to reading from the HWSP (without having
1.1  riastrad 	 * to use explicit shifting and masking, and probably bifurcating
1.1  riastrad 	 * the code to handle the legacy mmio read).
1.1  riastrad 	 */
1.1  riastrad 	head = execlists->csb_head;
1.1  riastrad 	tail = READ_ONCE(*execlists->csb_write);
1.1  riastrad 	ENGINE_TRACE(engine, "cs-irq head=%d, tail=%d\n", head, tail);
1.1  riastrad 	if (unlikely(head == tail))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Hopefully paired with a wmb() in HW!
1.1  riastrad 	 *
1.1  riastrad 	 * We must complete the read of the write pointer before any reads
1.1  riastrad 	 * from the CSB, so that we do not see stale values. Without an rmb
1.1  riastrad 	 * (lfence) the HW may speculatively perform the CSB[] reads *before*
1.1  riastrad 	 * we perform the READ_ONCE(*csb_write).
1.1  riastrad 	 */
1.1  riastrad 	rmb();
1.1  riastrad
1.1  riastrad 	do {
1.1  riastrad 		bool promote;
1.1  riastrad
1.1  riastrad 		if (++head == num_entries)
1.1  riastrad 			head = 0;
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * We are flying near dragons again.
1.1  riastrad 		 *
1.1  riastrad 		 * We hold a reference to the request in execlist_port[]
1.1  riastrad 		 * but no more than that. We are operating in softirq
1.1  riastrad 		 * context and so cannot hold any mutex or sleep. That
1.1  riastrad 		 * prevents us stopping the requests we are processing
1.1  riastrad 		 * in port[] from being retired simultaneously (the
1.1  riastrad 		 * breadcrumb will be complete before we see the
1.1  riastrad 		 * context-switch). As we only hold the reference to the
1.1  riastrad 		 * request, any pointer chasing underneath the request
1.1  riastrad 		 * is subject to a potential use-after-free. Thus we
1.1  riastrad 		 * store all of the bookkeeping within port[] as
1.1  riastrad 		 * required, and avoid using unguarded pointers beneath
1.1  riastrad 		 * request itself. The same applies to the atomic
1.1  riastrad 		 * status notifier.
1.1  riastrad 		 */
1.1  riastrad
1.1  riastrad 		ENGINE_TRACE(engine, "csb[%d]: status=0x%08x:0x%08x\n",
1.1  riastrad 			     head, buf[2 * head + 0], buf[2 * head + 1]);
1.1  riastrad
1.1  riastrad 		if (INTEL_GEN(engine->i915) >= 12)
1.1  riastrad 			promote = gen12_csb_parse(execlists, buf + 2 * head);
1.1  riastrad 		else
1.1  riastrad 			promote = gen8_csb_parse(execlists, buf + 2 * head);
1.1  riastrad 		if (promote) {
1.1  riastrad 			struct i915_request * const *old = execlists->active;
1.1  riastrad
1.1  riastrad 			/* Point active to the new ELSP; prevent overwriting */
1.1  riastrad 			WRITE_ONCE(execlists->active, execlists->pending);
1.1  riastrad
1.1  riastrad 			if (!inject_preempt_hang(execlists))
1.1  riastrad 				ring_set_paused(engine, 0);
1.1  riastrad
1.1  riastrad 			/* cancel old inflight, prepare for switch */
1.1  riastrad 			trace_ports(execlists, "preempted", old);
1.1  riastrad 			while (*old)
1.1  riastrad 				execlists_schedule_out(*old++);
1.1  riastrad
1.1  riastrad 			/* switch pending to inflight */
1.1  riastrad 			GEM_BUG_ON(!assert_pending_valid(execlists, "promote"));
1.1  riastrad 			WRITE_ONCE(execlists->active,
1.1  riastrad 				   memcpy(execlists->inflight,
1.1  riastrad 					  execlists->pending,
1.1  riastrad 					  execlists_num_ports(execlists) *
1.1  riastrad 					  sizeof(*execlists->pending)));
1.1  riastrad
1.1  riastrad 			WRITE_ONCE(execlists->pending[0], NULL);
1.1  riastrad 		} else {
1.1  riastrad 			GEM_BUG_ON(!*execlists->active);
1.1  riastrad
1.1  riastrad 			/* port0 completed, advanced to port1 */
1.1  riastrad 			trace_ports(execlists, "completed", execlists->active);
1.1  riastrad
1.1  riastrad 			/*
1.1  riastrad 			 * We rely on the hardware being strongly
1.1  riastrad 			 * ordered, that the breadcrumb write is
1.1  riastrad 			 * coherent (visible from the CPU) before the
1.1  riastrad 			 * user interrupt and CSB is processed.
1.1  riastrad 			 */
1.1  riastrad 			GEM_BUG_ON(!i915_request_completed(*execlists->active) &&
1.1  riastrad 				   !reset_in_progress(execlists));
1.1  riastrad 			execlists_schedule_out(*execlists->active++);
1.1  riastrad
1.1  riastrad 			GEM_BUG_ON(execlists->active - execlists->inflight >
1.1  riastrad 				   execlists_num_ports(execlists));
1.1  riastrad 		}
1.1  riastrad 	} while (head != tail);
1.1  riastrad
1.1  riastrad 	execlists->csb_head = head;
1.1  riastrad 	set_timeslice(engine);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Gen11 has proven to fail wrt global observation point between
1.1  riastrad 	 * entry and tail update, failing on the ordering and thus
1.1  riastrad 	 * we see an old entry in the context status buffer.
1.1  riastrad 	 *
1.1  riastrad 	 * Forcibly evict out entries for the next gpu csb update,
1.1  riastrad 	 * to increase the odds that we get a fresh entries with non
1.1  riastrad 	 * working hardware. The cost for doing so comes out mostly with
1.1  riastrad 	 * the wash as hardware, working or not, will need to do the
1.1  riastrad 	 * invalidation before.
1.1  riastrad 	 */
1.1  riastrad 	invalidate_csb_entries(&buf[0], &buf[num_entries - 1]);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __execlists_submission_tasklet(struct intel_engine_cs *const engine)
1.1  riastrad {
1.1  riastrad 	lockdep_assert_held(&engine->active.lock);
1.1  riastrad 	if (!engine->execlists.pending[0]) {
1.1  riastrad 		rcu_read_lock(); /* protect peeking at execlists->active */
1.1  riastrad 		execlists_dequeue(engine);
1.1  riastrad 		rcu_read_unlock();
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __execlists_hold(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	LIST_HEAD(list);
1.1  riastrad
1.1  riastrad 	do {
1.1  riastrad 		struct i915_dependency *p;
1.1  riastrad
1.1  riastrad 		if (i915_request_is_active(rq))
1.1  riastrad 			__i915_request_unsubmit(rq);
1.1  riastrad
1.1  riastrad 		RQ_TRACE(rq, "on hold\n");
1.1  riastrad 		clear_bit(I915_FENCE_FLAG_PQUEUE, &rq->fence.flags);
1.1  riastrad 		list_move_tail(&rq->sched.link, &rq->engine->active.hold);
1.1  riastrad 		i915_request_set_hold(rq);
1.1  riastrad
1.1  riastrad 		list_for_each_entry(p, &rq->sched.waiters_list, wait_link) {
1.1  riastrad 			struct i915_request *w =
1.1  riastrad 				container_of(p->waiter, typeof(*w), sched);
1.1  riastrad
1.1  riastrad 			/* Leave semaphores spinning on the other engines */
1.1  riastrad 			if (w->engine != rq->engine)
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			if (!i915_request_is_ready(w))
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			if (i915_request_completed(w))
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			if (i915_request_on_hold(rq))
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			list_move_tail(&w->sched.link, &list);
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		rq = list_first_entry_or_null(&list, typeof(*rq), sched.link);
1.1  riastrad 	} while (rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool execlists_hold(struct intel_engine_cs *engine,
1.1  riastrad 			   struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	spin_lock_irq(&engine->active.lock);
1.1  riastrad
1.1  riastrad 	if (i915_request_completed(rq)) { /* too late! */
1.1  riastrad 		rq = NULL;
1.1  riastrad 		goto unlock;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (rq->engine != engine) { /* preempted virtual engine */
1.1  riastrad 		struct virtual_engine *ve = to_virtual_engine(rq->engine);
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * intel_context_inflight() is only protected by virtue
1.1  riastrad 		 * of process_csb() being called only by the tasklet (or
1.1  riastrad 		 * directly from inside reset while the tasklet is suspended).
1.1  riastrad 		 * Assert that neither of those are allowed to run while we
1.1  riastrad 		 * poke at the request queues.
1.1  riastrad 		 */
1.1  riastrad 		GEM_BUG_ON(!reset_in_progress(&engine->execlists));
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * An unsubmitted request along a virtual engine will
1.1  riastrad 		 * remain on the active (this) engine until we are able
1.1  riastrad 		 * to process the context switch away (and so mark the
1.1  riastrad 		 * context as no longer in flight). That cannot have happened
1.1  riastrad 		 * yet, otherwise we would not be hanging!
1.1  riastrad 		 */
1.1  riastrad 		spin_lock(&ve->base.active.lock);
1.1  riastrad 		GEM_BUG_ON(intel_context_inflight(rq->context) != engine);
1.1  riastrad 		GEM_BUG_ON(ve->request != rq);
1.1  riastrad 		ve->request = NULL;
1.1  riastrad 		spin_unlock(&ve->base.active.lock);
1.1  riastrad 		i915_request_put(rq);
1.1  riastrad
1.1  riastrad 		rq->engine = engine;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Transfer this request onto the hold queue to prevent it
1.1  riastrad 	 * being resumbitted to HW (and potentially completed) before we have
1.1  riastrad 	 * released it. Since we may have already submitted following
1.1  riastrad 	 * requests, we need to remove those as well.
1.1  riastrad 	 */
1.1  riastrad 	GEM_BUG_ON(i915_request_on_hold(rq));
1.1  riastrad 	GEM_BUG_ON(rq->engine != engine);
1.1  riastrad 	__execlists_hold(rq);
1.1  riastrad
1.1  riastrad unlock:
1.1  riastrad 	spin_unlock_irq(&engine->active.lock);
1.1  riastrad 	return rq;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool hold_request(const struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	struct i915_dependency *p;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If one of our ancestors is on hold, we must also be on hold,
1.1  riastrad 	 * otherwise we will bypass it and execute before it.
1.1  riastrad 	 */
1.1  riastrad 	list_for_each_entry(p, &rq->sched.signalers_list, signal_link) {
1.1  riastrad 		const struct i915_request *s =
1.1  riastrad 			container_of(p->signaler, typeof(*s), sched);
1.1  riastrad
1.1  riastrad 		if (s->engine != rq->engine)
1.1  riastrad 			continue;
1.1  riastrad
1.1  riastrad 		if (i915_request_on_hold(s))
1.1  riastrad 			return true;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return false;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __execlists_unhold(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	LIST_HEAD(list);
1.1  riastrad
1.1  riastrad 	do {
1.1  riastrad 		struct i915_dependency *p;
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(!i915_request_on_hold(rq));
1.1  riastrad 		GEM_BUG_ON(!i915_sw_fence_signaled(&rq->submit));
1.1  riastrad
1.1  riastrad 		i915_request_clear_hold(rq);
1.1  riastrad 		list_move_tail(&rq->sched.link,
1.1  riastrad 			       i915_sched_lookup_priolist(rq->engine,
1.1  riastrad 							  rq_prio(rq)));
1.1  riastrad 		set_bit(I915_FENCE_FLAG_PQUEUE, &rq->fence.flags);
1.1  riastrad 		RQ_TRACE(rq, "hold release\n");
1.1  riastrad
1.1  riastrad 		/* Also release any children on this engine that are ready */
1.1  riastrad 		list_for_each_entry(p, &rq->sched.waiters_list, wait_link) {
1.1  riastrad 			struct i915_request *w =
1.1  riastrad 				container_of(p->waiter, typeof(*w), sched);
1.1  riastrad
1.1  riastrad 			if (w->engine != rq->engine)
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			if (!i915_request_on_hold(rq))
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			/* Check that no other parents are also on hold */
1.1  riastrad 			if (hold_request(rq))
1.1  riastrad 				continue;
1.1  riastrad
1.1  riastrad 			list_move_tail(&w->sched.link, &list);
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		rq = list_first_entry_or_null(&list, typeof(*rq), sched.link);
1.1  riastrad 	} while (rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_unhold(struct intel_engine_cs *engine,
1.1  riastrad 			     struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	spin_lock_irq(&engine->active.lock);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Move this request back to the priority queue, and all of its
1.1  riastrad 	 * children and grandchildren that were suspended along with it.
1.1  riastrad 	 */
1.1  riastrad 	__execlists_unhold(rq);
1.1  riastrad
1.1  riastrad 	if (rq_prio(rq) > engine->execlists.queue_priority_hint) {
1.1  riastrad 		engine->execlists.queue_priority_hint = rq_prio(rq);
1.1  riastrad 		tasklet_hi_schedule(&engine->execlists.tasklet);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	spin_unlock_irq(&engine->active.lock);
1.1  riastrad }
1.1  riastrad
1.1  riastrad struct execlists_capture {
1.1  riastrad 	struct work_struct work;
1.1  riastrad 	struct i915_request *rq;
1.1  riastrad 	struct i915_gpu_coredump *error;
1.1  riastrad };
1.1  riastrad
1.1  riastrad static void execlists_capture_work(struct work_struct *work)
1.1  riastrad {
1.1  riastrad 	struct execlists_capture *cap = container_of(work, typeof(*cap), work);
1.1  riastrad 	const gfp_t gfp = GFP_KERNEL | __GFP_RETRY_MAYFAIL | __GFP_NOWARN;
1.1  riastrad 	struct intel_engine_cs *engine = cap->rq->engine;
1.1  riastrad 	struct intel_gt_coredump *gt = cap->error->gt;
1.1  riastrad 	struct intel_engine_capture_vma *vma;
1.1  riastrad
1.1  riastrad 	/* Compress all the objects attached to the request, slow! */
1.1  riastrad 	vma = intel_engine_coredump_add_request(gt->engine, cap->rq, gfp);
1.1  riastrad 	if (vma) {
1.1  riastrad 		struct i915_vma_compress *compress =
1.1  riastrad 			i915_vma_capture_prepare(gt);
1.1  riastrad
1.1  riastrad 		intel_engine_coredump_add_vma(gt->engine, vma, compress);
1.1  riastrad 		i915_vma_capture_finish(gt, compress);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	gt->simulated = gt->engine->simulated;
1.1  riastrad 	cap->error->simulated = gt->simulated;
1.1  riastrad
1.1  riastrad 	/* Publish the error state, and announce it to the world */
1.1  riastrad 	i915_error_state_store(cap->error);
1.1  riastrad 	i915_gpu_coredump_put(cap->error);
1.1  riastrad
1.1  riastrad 	/* Return this request and all that depend upon it for signaling */
1.1  riastrad 	execlists_unhold(engine, cap->rq);
1.1  riastrad 	i915_request_put(cap->rq);
1.1  riastrad
1.1  riastrad 	kfree(cap);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static struct execlists_capture *capture_regs(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	const gfp_t gfp = GFP_ATOMIC | __GFP_NOWARN;
1.1  riastrad 	struct execlists_capture *cap;
1.1  riastrad
1.1  riastrad 	cap = kmalloc(sizeof(*cap), gfp);
1.1  riastrad 	if (!cap)
1.1  riastrad 		return NULL;
1.1  riastrad
1.1  riastrad 	cap->error = i915_gpu_coredump_alloc(engine->i915, gfp);
1.1  riastrad 	if (!cap->error)
1.1  riastrad 		goto err_cap;
1.1  riastrad
1.1  riastrad 	cap->error->gt = intel_gt_coredump_alloc(engine->gt, gfp);
1.1  riastrad 	if (!cap->error->gt)
1.1  riastrad 		goto err_gpu;
1.1  riastrad
1.1  riastrad 	cap->error->gt->engine = intel_engine_coredump_alloc(engine, gfp);
1.1  riastrad 	if (!cap->error->gt->engine)
1.1  riastrad 		goto err_gt;
1.1  riastrad
1.1  riastrad 	return cap;
1.1  riastrad
1.1  riastrad err_gt:
1.1  riastrad 	kfree(cap->error->gt);
1.1  riastrad err_gpu:
1.1  riastrad 	kfree(cap->error);
1.1  riastrad err_cap:
1.1  riastrad 	kfree(cap);
1.1  riastrad 	return NULL;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool execlists_capture(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct execlists_capture *cap;
1.1  riastrad
1.1  riastrad 	if (!IS_ENABLED(CONFIG_DRM_I915_CAPTURE_ERROR))
1.1  riastrad 		return true;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We need to _quickly_ capture the engine state before we reset.
1.1  riastrad 	 * We are inside an atomic section (softirq) here and we are delaying
1.1  riastrad 	 * the forced preemption event.
1.1  riastrad 	 */
1.1  riastrad 	cap = capture_regs(engine);
1.1  riastrad 	if (!cap)
1.1  riastrad 		return true;
1.1  riastrad
1.1  riastrad 	cap->rq = execlists_active(&engine->execlists);
1.1  riastrad 	GEM_BUG_ON(!cap->rq);
1.1  riastrad
1.1  riastrad 	rcu_read_lock();
1.1  riastrad 	cap->rq = active_request(cap->rq->context->timeline, cap->rq);
1.1  riastrad 	cap->rq = i915_request_get_rcu(cap->rq);
1.1  riastrad 	rcu_read_unlock();
1.1  riastrad 	if (!cap->rq)
1.1  riastrad 		goto err_free;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Remove the request from the execlists queue, and take ownership
1.1  riastrad 	 * of the request. We pass it to our worker who will _slowly_ compress
1.1  riastrad 	 * all the pages the _user_ requested for debugging their batch, after
1.1  riastrad 	 * which we return it to the queue for signaling.
1.1  riastrad 	 *
1.1  riastrad 	 * By removing them from the execlists queue, we also remove the
1.1  riastrad 	 * requests from being processed by __unwind_incomplete_requests()
1.1  riastrad 	 * during the intel_engine_reset(), and so they will *not* be replayed
1.1  riastrad 	 * afterwards.
1.1  riastrad 	 *
1.1  riastrad 	 * Note that because we have not yet reset the engine at this point,
1.1  riastrad 	 * it is possible for the request that we have identified as being
1.1  riastrad 	 * guilty, did in fact complete and we will then hit an arbitration
1.1  riastrad 	 * point allowing the outstanding preemption to succeed. The likelihood
1.1  riastrad 	 * of that is very low (as capturing of the engine registers should be
1.1  riastrad 	 * fast enough to run inside an irq-off atomic section!), so we will
1.1  riastrad 	 * simply hold that request accountable for being non-preemptible
1.1  riastrad 	 * long enough to force the reset.
1.1  riastrad 	 */
1.1  riastrad 	if (!execlists_hold(engine, cap->rq))
1.1  riastrad 		goto err_rq;
1.1  riastrad
1.1  riastrad 	INIT_WORK(&cap->work, execlists_capture_work);
1.1  riastrad 	schedule_work(&cap->work);
1.1  riastrad 	return true;
1.1  riastrad
1.1  riastrad err_rq:
1.1  riastrad 	i915_request_put(cap->rq);
1.1  riastrad err_free:
1.1  riastrad 	i915_gpu_coredump_put(cap->error);
1.1  riastrad 	kfree(cap);
1.1  riastrad 	return false;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static noinline void preempt_reset(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	const unsigned int bit = I915_RESET_ENGINE + engine->id;
1.1  riastrad 	unsigned long *lock = &engine->gt->reset.flags;
1.1  riastrad
1.1  riastrad 	if (i915_modparams.reset < 3)
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	if (test_and_set_bit(bit, lock))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	/* Mark this tasklet as disabled to avoid waiting for it to complete */
1.1  riastrad 	tasklet_disable_nosync(&engine->execlists.tasklet);
1.1  riastrad
1.1  riastrad 	ENGINE_TRACE(engine, "preempt timeout %lu+%ums\n",
1.1  riastrad 		     READ_ONCE(engine->props.preempt_timeout_ms),
1.1  riastrad 		     jiffies_to_msecs(jiffies - engine->execlists.preempt.expires));
1.1  riastrad
1.1  riastrad 	ring_set_paused(engine, 1); /* Freeze the current request in place */
1.1  riastrad 	if (execlists_capture(engine))
1.1  riastrad 		intel_engine_reset(engine, "preemption time out");
1.1  riastrad 	else
1.1  riastrad 		ring_set_paused(engine, 0);
1.1  riastrad
1.1  riastrad 	tasklet_enable(&engine->execlists.tasklet);
1.1  riastrad 	clear_and_wake_up_bit(bit, lock);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool preempt_timeout(const struct intel_engine_cs *const engine)
1.1  riastrad {
1.1  riastrad 	const struct timer_list *t = &engine->execlists.preempt;
1.1  riastrad
1.1  riastrad 	if (!CONFIG_DRM_I915_PREEMPT_TIMEOUT)
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	if (!timer_expired(t))
1.1  riastrad 		return false;
1.1  riastrad
1.1  riastrad 	return READ_ONCE(engine->execlists.pending[0]);
1.1  riastrad }
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * Check the unread Context Status Buffers and manage the submission of new
1.1  riastrad  * contexts to the ELSP accordingly.
1.1  riastrad  */
1.1  riastrad static void execlists_submission_tasklet(unsigned long data)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_cs * const engine = (struct intel_engine_cs *)data;
1.1  riastrad 	bool timeout = preempt_timeout(engine);
1.1  riastrad
1.1  riastrad 	process_csb(engine);
1.1  riastrad 	if (!READ_ONCE(engine->execlists.pending[0]) || timeout) {
1.1  riastrad 		unsigned long flags;
1.1  riastrad
1.1  riastrad 		spin_lock_irqsave(&engine->active.lock, flags);
1.1  riastrad 		__execlists_submission_tasklet(engine);
1.1  riastrad 		spin_unlock_irqrestore(&engine->active.lock, flags);
1.1  riastrad
1.1  riastrad 		/* Recheck after serialising with direct-submission */
1.1  riastrad 		if (timeout && preempt_timeout(engine))
1.1  riastrad 			preempt_reset(engine);
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __execlists_kick(struct intel_engine_execlists *execlists)
1.1  riastrad {
1.1  riastrad 	/* Kick the tasklet for some interrupt coalescing and reset handling */
1.1  riastrad 	tasklet_hi_schedule(&execlists->tasklet);
1.1  riastrad }
1.1  riastrad
1.1  riastrad #define execlists_kick(t, member) \
1.1  riastrad 	__execlists_kick(container_of(t, struct intel_engine_execlists, member))
1.1  riastrad
1.1  riastrad static void execlists_timeslice(struct timer_list *timer)
1.1  riastrad {
1.1  riastrad 	execlists_kick(timer, timer);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_preempt(struct timer_list *timer)
1.1  riastrad {
1.1  riastrad 	execlists_kick(timer, preempt);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void queue_request(struct intel_engine_cs *engine,
1.1  riastrad 			  struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	GEM_BUG_ON(!list_empty(&rq->sched.link));
1.1  riastrad 	list_add_tail(&rq->sched.link,
1.1  riastrad 		      i915_sched_lookup_priolist(engine, rq_prio(rq)));
1.1  riastrad 	set_bit(I915_FENCE_FLAG_PQUEUE, &rq->fence.flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __submit_queue_imm(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad
1.1  riastrad 	if (reset_in_progress(execlists))
1.1  riastrad 		return; /* defer until we restart the engine following reset */
1.1  riastrad
1.1  riastrad 	if (execlists->tasklet.func == execlists_submission_tasklet)
1.1  riastrad 		__execlists_submission_tasklet(engine);
1.1  riastrad 	else
1.1  riastrad 		tasklet_hi_schedule(&execlists->tasklet);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void submit_queue(struct intel_engine_cs *engine,
1.1  riastrad 			 const struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists *execlists = &engine->execlists;
1.1  riastrad
1.1  riastrad 	if (rq_prio(rq) <= execlists->queue_priority_hint)
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	execlists->queue_priority_hint = rq_prio(rq);
1.1  riastrad 	__submit_queue_imm(engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool ancestor_on_hold(const struct intel_engine_cs *engine,
1.1  riastrad 			     const struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	GEM_BUG_ON(i915_request_on_hold(rq));
1.1  riastrad 	return !list_empty(&engine->active.hold) && hold_request(rq);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_submit_request(struct i915_request *request)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_cs *engine = request->engine;
1.1  riastrad 	unsigned long flags;
1.1  riastrad
1.1  riastrad 	/* Will be called from irq-context when using foreign fences. */
1.1  riastrad 	spin_lock_irqsave(&engine->active.lock, flags);
1.1  riastrad
1.1  riastrad 	if (unlikely(ancestor_on_hold(engine, request))) {
1.1  riastrad 		list_add_tail(&request->sched.link, &engine->active.hold);
1.1  riastrad 		i915_request_set_hold(request);
1.1  riastrad 	} else {
1.1  riastrad 		queue_request(engine, request);
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(RB_EMPTY_ROOT(&engine->execlists.queue.rb_root));
1.1  riastrad 		GEM_BUG_ON(list_empty(&request->sched.link));
1.1  riastrad
1.1  riastrad 		submit_queue(engine, request);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	spin_unlock_irqrestore(&engine->active.lock, flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __execlists_context_fini(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	intel_ring_put(ce->ring);
1.1  riastrad 	i915_vma_put(ce->state);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_context_destroy(struct kref *kref)
1.1  riastrad {
1.1  riastrad 	struct intel_context *ce = container_of(kref, typeof(*ce), ref);
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!i915_active_is_idle(&ce->active));
1.1  riastrad 	GEM_BUG_ON(intel_context_is_pinned(ce));
1.1  riastrad
1.1  riastrad 	if (ce->state)
1.1  riastrad 		__execlists_context_fini(ce);
1.1  riastrad
1.1  riastrad 	intel_context_fini(ce);
1.1  riastrad 	intel_context_free(ce);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void
1.1  riastrad set_redzone(void *vaddr, const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	if (!IS_ENABLED(CONFIG_DRM_I915_DEBUG_GEM))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	vaddr += engine->context_size;
1.1  riastrad
1.1  riastrad 	memset(vaddr, CONTEXT_REDZONE, I915_GTT_PAGE_SIZE);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void
1.1  riastrad check_redzone(const void *vaddr, const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	if (!IS_ENABLED(CONFIG_DRM_I915_DEBUG_GEM))
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	vaddr += engine->context_size;
1.1  riastrad
1.1  riastrad 	if (memchr_inv(vaddr, CONTEXT_REDZONE, I915_GTT_PAGE_SIZE))
1.1  riastrad 		dev_err_once(engine->i915->drm.dev,
1.1  riastrad 			     "%s context redzone overwritten!\n",
1.1  riastrad 			     engine->name);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_context_unpin(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	check_redzone((void *)ce->lrc_reg_state - LRC_STATE_PN * PAGE_SIZE,
1.1  riastrad 		      ce->engine);
1.1  riastrad
1.1  riastrad 	i915_gem_object_unpin_map(ce->state->obj);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void
1.1  riastrad __execlists_update_reg_state(const struct intel_context *ce,
1.1  riastrad 			     const struct intel_engine_cs *engine,
1.1  riastrad 			     u32 head)
1.1  riastrad {
1.1  riastrad 	struct intel_ring *ring = ce->ring;
1.1  riastrad 	u32 *regs = ce->lrc_reg_state;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!intel_ring_offset_valid(ring, head));
1.1  riastrad 	GEM_BUG_ON(!intel_ring_offset_valid(ring, ring->tail));
1.1  riastrad
1.1  riastrad 	regs[CTX_RING_START] = i915_ggtt_offset(ring->vma);
1.1  riastrad 	regs[CTX_RING_HEAD] = head;
1.1  riastrad 	regs[CTX_RING_TAIL] = ring->tail;
1.1  riastrad
1.1  riastrad 	/* RPCS */
1.1  riastrad 	if (engine->class == RENDER_CLASS) {
1.1  riastrad 		regs[CTX_R_PWR_CLK_STATE] =
1.1  riastrad 			intel_sseu_make_rpcs(engine->i915, &ce->sseu);
1.1  riastrad
1.1  riastrad 		i915_oa_init_reg_state(ce, engine);
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int
1.1  riastrad __execlists_context_pin(struct intel_context *ce,
1.1  riastrad 			struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	void *vaddr;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!ce->state);
1.1  riastrad 	GEM_BUG_ON(!i915_vma_is_pinned(ce->state));
1.1  riastrad
1.1  riastrad 	vaddr = i915_gem_object_pin_map(ce->state->obj,
1.1  riastrad 					i915_coherent_map_type(engine->i915) |
1.1  riastrad 					I915_MAP_OVERRIDE);
1.1  riastrad 	if (IS_ERR(vaddr))
1.1  riastrad 		return PTR_ERR(vaddr);
1.1  riastrad
1.1  riastrad 	ce->lrc_desc = lrc_descriptor(ce, engine) | CTX_DESC_FORCE_RESTORE;
1.1  riastrad 	ce->lrc_reg_state = vaddr + LRC_STATE_PN * PAGE_SIZE;
1.1  riastrad 	__execlists_update_reg_state(ce, engine, ce->ring->tail);
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int execlists_context_pin(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	return __execlists_context_pin(ce, ce->engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int execlists_context_alloc(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	return __execlists_context_alloc(ce, ce->engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_context_reset(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	CE_TRACE(ce, "reset\n");
1.1  riastrad 	GEM_BUG_ON(!intel_context_is_pinned(ce));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Because we emit WA_TAIL_DWORDS there may be a disparity
1.1  riastrad 	 * between our bookkeeping in ce->ring->head and ce->ring->tail and
1.1  riastrad 	 * that stored in context. As we only write new commands from
1.1  riastrad 	 * ce->ring->tail onwards, everything before that is junk. If the GPU
1.1  riastrad 	 * starts reading from its RING_HEAD from the context, it may try to
1.1  riastrad 	 * execute that junk and die.
1.1  riastrad 	 *
1.1  riastrad 	 * The contexts that are stilled pinned on resume belong to the
1.1  riastrad 	 * kernel, and are local to each engine. All other contexts will
1.1  riastrad 	 * have their head/tail sanitized upon pinning before use, so they
1.1  riastrad 	 * will never see garbage,
1.1  riastrad 	 *
1.1  riastrad 	 * So to avoid that we reset the context images upon resume. For
1.1  riastrad 	 * simplicity, we just zero everything out.
1.1  riastrad 	 */
1.1  riastrad 	intel_ring_reset(ce->ring, ce->ring->emit);
1.1  riastrad
1.1  riastrad 	/* Scrub away the garbage */
1.1  riastrad 	execlists_init_reg_state(ce->lrc_reg_state,
1.1  riastrad 				 ce, ce->engine, ce->ring, true);
1.1  riastrad 	__execlists_update_reg_state(ce, ce->engine, ce->ring->tail);
1.1  riastrad
1.1  riastrad 	ce->lrc_desc |= CTX_DESC_FORCE_RESTORE;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static const struct intel_context_ops execlists_context_ops = {
1.1  riastrad 	.alloc = execlists_context_alloc,
1.1  riastrad
1.1  riastrad 	.pin = execlists_context_pin,
1.1  riastrad 	.unpin = execlists_context_unpin,
1.1  riastrad
1.1  riastrad 	.enter = intel_context_enter_engine,
1.1  riastrad 	.exit = intel_context_exit_engine,
1.1  riastrad
1.1  riastrad 	.reset = execlists_context_reset,
1.1  riastrad 	.destroy = execlists_context_destroy,
1.1  riastrad };
1.1  riastrad
1.1  riastrad static int gen8_emit_init_breadcrumb(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	u32 *cs;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!i915_request_timeline(rq)->has_initial_breadcrumb);
1.1  riastrad
1.1  riastrad 	cs = intel_ring_begin(rq, 6);
1.1  riastrad 	if (IS_ERR(cs))
1.1  riastrad 		return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Check if we have been preempted before we even get started.
1.1  riastrad 	 *
1.1  riastrad 	 * After this point i915_request_started() reports true, even if
1.1  riastrad 	 * we get preempted and so are no longer running.
1.1  riastrad 	 */
1.1  riastrad 	*cs++ = MI_ARB_CHECK;
1.1  riastrad 	*cs++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	*cs++ = MI_STORE_DWORD_IMM_GEN4 | MI_USE_GGTT;
1.1  riastrad 	*cs++ = i915_request_timeline(rq)->hwsp_offset;
1.1  riastrad 	*cs++ = 0;
1.1  riastrad 	*cs++ = rq->fence.seqno - 1;
1.1  riastrad
1.1  riastrad 	intel_ring_advance(rq, cs);
1.1  riastrad
1.1  riastrad 	/* Record the updated position of the request's payload */
1.1  riastrad 	rq->infix = intel_ring_offset(rq, cs);
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int execlists_request_alloc(struct i915_request *request)
1.1  riastrad {
1.1  riastrad 	int ret;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!intel_context_is_pinned(request->context));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Flush enough space to reduce the likelihood of waiting after
1.1  riastrad 	 * we start building the request - in which case we will just
1.1  riastrad 	 * have to repeat work.
1.1  riastrad 	 */
1.1  riastrad 	request->reserved_space += EXECLISTS_REQUEST_SIZE;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Note that after this point, we have committed to using
1.1  riastrad 	 * this request as it is being used to both track the
1.1  riastrad 	 * state of engine initialisation and liveness of the
1.1  riastrad 	 * golden renderstate above. Think twice before you try
1.1  riastrad 	 * to cancel/unwind this request now.
1.1  riastrad 	 */
1.1  riastrad
1.1  riastrad 	/* Unconditionally invalidate GPU caches and TLBs. */
1.1  riastrad 	ret = request->engine->emit_flush(request, EMIT_INVALIDATE);
1.1  riastrad 	if (ret)
1.1  riastrad 		return ret;
1.1  riastrad
1.1  riastrad 	request->reserved_space -= EXECLISTS_REQUEST_SIZE;
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * In this WA we need to set GEN8_L3SQCREG4[21:21] and reset it after
1.1  riastrad  * PIPE_CONTROL instruction. This is required for the flush to happen correctly
1.1  riastrad  * but there is a slight complication as this is applied in WA batch where the
1.1  riastrad  * values are only initialized once so we cannot take register value at the
1.1  riastrad  * beginning and reuse it further; hence we save its value to memory, upload a
1.1  riastrad  * constant value with bit21 set and then we restore it back with the saved value.
1.1  riastrad  * To simplify the WA, a constant value is formed by using the default value
1.1  riastrad  * of this register. This shouldn't be a problem because we are only modifying
1.1  riastrad  * it for a short period and this batch in non-premptible. We can ofcourse
1.1  riastrad  * use additional instructions that read the actual value of the register
1.1  riastrad  * at that time and set our bit of interest but it makes the WA complicated.
1.1  riastrad  *
1.1  riastrad  * This WA is also required for Gen9 so extracting as a function avoids
1.1  riastrad  * code duplication.
1.1  riastrad  */
1.1  riastrad static u32 *
1.1  riastrad gen8_emit_flush_coherentl3_wa(struct intel_engine_cs *engine, u32 *batch)
1.1  riastrad {
1.1  riastrad 	/* NB no one else is allowed to scribble over scratch + 256! */
1.1  riastrad 	*batch++ = MI_STORE_REGISTER_MEM_GEN8 | MI_SRM_LRM_GLOBAL_GTT;
1.1  riastrad 	*batch++ = i915_mmio_reg_offset(GEN8_L3SQCREG4);
1.1  riastrad 	*batch++ = intel_gt_scratch_offset(engine->gt,
1.1  riastrad 					   INTEL_GT_SCRATCH_FIELD_COHERENTL3_WA);
1.1  riastrad 	*batch++ = 0;
1.1  riastrad
1.1  riastrad 	*batch++ = MI_LOAD_REGISTER_IMM(1);
1.1  riastrad 	*batch++ = i915_mmio_reg_offset(GEN8_L3SQCREG4);
1.1  riastrad 	*batch++ = 0x40400000 | GEN8_LQSC_FLUSH_COHERENT_LINES;
1.1  riastrad
1.1  riastrad 	batch = gen8_emit_pipe_control(batch,
1.1  riastrad 				       PIPE_CONTROL_CS_STALL |
1.1  riastrad 				       PIPE_CONTROL_DC_FLUSH_ENABLE,
1.1  riastrad 				       0);
1.1  riastrad
1.1  riastrad 	*batch++ = MI_LOAD_REGISTER_MEM_GEN8 | MI_SRM_LRM_GLOBAL_GTT;
1.1  riastrad 	*batch++ = i915_mmio_reg_offset(GEN8_L3SQCREG4);
1.1  riastrad 	*batch++ = intel_gt_scratch_offset(engine->gt,
1.1  riastrad 					   INTEL_GT_SCRATCH_FIELD_COHERENTL3_WA);
1.1  riastrad 	*batch++ = 0;
1.1  riastrad
1.1  riastrad 	return batch;
1.1  riastrad }
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * Typically we only have one indirect_ctx and per_ctx batch buffer which are
1.1  riastrad  * initialized at the beginning and shared across all contexts but this field
1.1  riastrad  * helps us to have multiple batches at different offsets and select them based
1.1  riastrad  * on a criteria. At the moment this batch always start at the beginning of the page
1.1  riastrad  * and at this point we don't have multiple wa_ctx batch buffers.
1.1  riastrad  *
1.1  riastrad  * The number of WA applied are not known at the beginning; we use this field
1.1  riastrad  * to return the no of DWORDS written.
1.1  riastrad  *
1.1  riastrad  * It is to be noted that this batch does not contain MI_BATCH_BUFFER_END
1.1  riastrad  * so it adds NOOPs as padding to make it cacheline aligned.
1.1  riastrad  * MI_BATCH_BUFFER_END will be added to perctx batch and both of them together
1.1  riastrad  * makes a complete batch buffer.
1.1  riastrad  */
1.1  riastrad static u32 *gen8_init_indirectctx_bb(struct intel_engine_cs *engine, u32 *batch)
1.1  riastrad {
1.1  riastrad 	/* WaDisableCtxRestoreArbitration:bdw,chv */
1.1  riastrad 	*batch++ = MI_ARB_ON_OFF | MI_ARB_DISABLE;
1.1  riastrad
1.1  riastrad 	/* WaFlushCoherentL3CacheLinesAtContextSwitch:bdw */
1.1  riastrad 	if (IS_BROADWELL(engine->i915))
1.1  riastrad 		batch = gen8_emit_flush_coherentl3_wa(engine, batch);
1.1  riastrad
1.1  riastrad 	/* WaClearSlmSpaceAtContextSwitch:bdw,chv */
1.1  riastrad 	/* Actual scratch location is at 128 bytes offset */
1.1  riastrad 	batch = gen8_emit_pipe_control(batch,
1.1  riastrad 				       PIPE_CONTROL_FLUSH_L3 |
1.1  riastrad 				       PIPE_CONTROL_STORE_DATA_INDEX |
1.1  riastrad 				       PIPE_CONTROL_CS_STALL |
1.1  riastrad 				       PIPE_CONTROL_QW_WRITE,
1.1  riastrad 				       LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad
1.1  riastrad 	*batch++ = MI_ARB_ON_OFF | MI_ARB_ENABLE;
1.1  riastrad
1.1  riastrad 	/* Pad to end of cacheline */
1.1  riastrad 	while ((unsigned long)batch % CACHELINE_BYTES)
1.1  riastrad 		*batch++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * MI_BATCH_BUFFER_END is not required in Indirect ctx BB because
1.1  riastrad 	 * execution depends on the length specified in terms of cache lines
1.1  riastrad 	 * in the register CTX_RCS_INDIRECT_CTX
1.1  riastrad 	 */
1.1  riastrad
1.1  riastrad 	return batch;
1.1  riastrad }
1.1  riastrad
1.1  riastrad struct lri {
1.1  riastrad 	i915_reg_t reg;
1.1  riastrad 	u32 value;
1.1  riastrad };
1.1  riastrad
1.1  riastrad static u32 *emit_lri(u32 *batch, const struct lri *lri, unsigned int count)
1.1  riastrad {
1.1  riastrad 	GEM_BUG_ON(!count || count > 63);
1.1  riastrad
1.1  riastrad 	*batch++ = MI_LOAD_REGISTER_IMM(count);
1.1  riastrad 	do {
1.1  riastrad 		*batch++ = i915_mmio_reg_offset(lri->reg);
1.1  riastrad 		*batch++ = lri->value;
1.1  riastrad 	} while (lri++, --count);
1.1  riastrad 	*batch++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	return batch;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *gen9_init_indirectctx_bb(struct intel_engine_cs *engine, u32 *batch)
1.1  riastrad {
1.1  riastrad 	static const struct lri lri[] = {
1.1  riastrad 		/* WaDisableGatherAtSetShaderCommonSlice:skl,bxt,kbl,glk */
1.1  riastrad 		{
1.1  riastrad 			COMMON_SLICE_CHICKEN2,
1.1  riastrad 			__MASKED_FIELD(GEN9_DISABLE_GATHER_AT_SET_SHADER_COMMON_SLICE,
1.1  riastrad 				       0),
1.1  riastrad 		},
1.1  riastrad
1.1  riastrad 		/* BSpec: 11391 */
1.1  riastrad 		{
1.1  riastrad 			FF_SLICE_CHICKEN,
1.1  riastrad 			__MASKED_FIELD(FF_SLICE_CHICKEN_CL_PROVOKING_VERTEX_FIX,
1.1  riastrad 				       FF_SLICE_CHICKEN_CL_PROVOKING_VERTEX_FIX),
1.1  riastrad 		},
1.1  riastrad
1.1  riastrad 		/* BSpec: 11299 */
1.1  riastrad 		{
1.1  riastrad 			_3D_CHICKEN3,
1.1  riastrad 			__MASKED_FIELD(_3D_CHICKEN_SF_PROVOKING_VERTEX_FIX,
1.1  riastrad 				       _3D_CHICKEN_SF_PROVOKING_VERTEX_FIX),
1.1  riastrad 		}
1.1  riastrad 	};
1.1  riastrad
1.1  riastrad 	*batch++ = MI_ARB_ON_OFF | MI_ARB_DISABLE;
1.1  riastrad
1.1  riastrad 	/* WaFlushCoherentL3CacheLinesAtContextSwitch:skl,bxt,glk */
1.1  riastrad 	batch = gen8_emit_flush_coherentl3_wa(engine, batch);
1.1  riastrad
1.1  riastrad 	/* WaClearSlmSpaceAtContextSwitch:skl,bxt,kbl,glk,cfl */
1.1  riastrad 	batch = gen8_emit_pipe_control(batch,
1.1  riastrad 				       PIPE_CONTROL_FLUSH_L3 |
1.1  riastrad 				       PIPE_CONTROL_STORE_DATA_INDEX |
1.1  riastrad 				       PIPE_CONTROL_CS_STALL |
1.1  riastrad 				       PIPE_CONTROL_QW_WRITE,
1.1  riastrad 				       LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad
1.1  riastrad 	batch = emit_lri(batch, lri, ARRAY_SIZE(lri));
1.1  riastrad
1.1  riastrad 	/* WaMediaPoolStateCmdInWABB:bxt,glk */
1.1  riastrad 	if (HAS_POOLED_EU(engine->i915)) {
1.1  riastrad 		/*
1.1  riastrad 		 * EU pool configuration is setup along with golden context
1.1  riastrad 		 * during context initialization. This value depends on
1.1  riastrad 		 * device type (2x6 or 3x6) and needs to be updated based
1.1  riastrad 		 * on which subslice is disabled especially for 2x6
1.1  riastrad 		 * devices, however it is safe to load default
1.1  riastrad 		 * configuration of 3x6 device instead of masking off
1.1  riastrad 		 * corresponding bits because HW ignores bits of a disabled
1.1  riastrad 		 * subslice and drops down to appropriate config. Please
1.1  riastrad 		 * see render_state_setup() in i915_gem_render_state.c for
1.1  riastrad 		 * possible configurations, to avoid duplication they are
1.1  riastrad 		 * not shown here again.
1.1  riastrad 		 */
1.1  riastrad 		*batch++ = GEN9_MEDIA_POOL_STATE;
1.1  riastrad 		*batch++ = GEN9_MEDIA_POOL_ENABLE;
1.1  riastrad 		*batch++ = 0x00777000;
1.1  riastrad 		*batch++ = 0;
1.1  riastrad 		*batch++ = 0;
1.1  riastrad 		*batch++ = 0;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	*batch++ = MI_ARB_ON_OFF | MI_ARB_ENABLE;
1.1  riastrad
1.1  riastrad 	/* Pad to end of cacheline */
1.1  riastrad 	while ((unsigned long)batch % CACHELINE_BYTES)
1.1  riastrad 		*batch++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	return batch;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *
1.1  riastrad gen10_init_indirectctx_bb(struct intel_engine_cs *engine, u32 *batch)
1.1  riastrad {
1.1  riastrad 	int i;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * WaPipeControlBefore3DStateSamplePattern: cnl
1.1  riastrad 	 *
1.1  riastrad 	 * Ensure the engine is idle prior to programming a
1.1  riastrad 	 * 3DSTATE_SAMPLE_PATTERN during a context restore.
1.1  riastrad 	 */
1.1  riastrad 	batch = gen8_emit_pipe_control(batch,
1.1  riastrad 				       PIPE_CONTROL_CS_STALL,
1.1  riastrad 				       0);
1.1  riastrad 	/*
1.1  riastrad 	 * WaPipeControlBefore3DStateSamplePattern says we need 4 dwords for
1.1  riastrad 	 * the PIPE_CONTROL followed by 12 dwords of 0x0, so 16 dwords in
1.1  riastrad 	 * total. However, a PIPE_CONTROL is 6 dwords long, not 4, which is
1.1  riastrad 	 * confusing. Since gen8_emit_pipe_control() already advances the
1.1  riastrad 	 * batch by 6 dwords, we advance the other 10 here, completing a
1.1  riastrad 	 * cacheline. It's not clear if the workaround requires this padding
1.1  riastrad 	 * before other commands, or if it's just the regular padding we would
1.1  riastrad 	 * already have for the workaround bb, so leave it here for now.
1.1  riastrad 	 */
1.1  riastrad 	for (i = 0; i < 10; i++)
1.1  riastrad 		*batch++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	/* Pad to end of cacheline */
1.1  riastrad 	while ((unsigned long)batch % CACHELINE_BYTES)
1.1  riastrad 		*batch++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	return batch;
1.1  riastrad }
1.1  riastrad
1.1  riastrad #define CTX_WA_BB_OBJ_SIZE (PAGE_SIZE)
1.1  riastrad
1.1  riastrad static int lrc_setup_wa_ctx(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct drm_i915_gem_object *obj;
1.1  riastrad 	struct i915_vma *vma;
1.1  riastrad 	int err;
1.1  riastrad
1.1  riastrad 	obj = i915_gem_object_create_shmem(engine->i915, CTX_WA_BB_OBJ_SIZE);
1.1  riastrad 	if (IS_ERR(obj))
1.1  riastrad 		return PTR_ERR(obj);
1.1  riastrad
1.1  riastrad 	vma = i915_vma_instance(obj, &engine->gt->ggtt->vm, NULL);
1.1  riastrad 	if (IS_ERR(vma)) {
1.1  riastrad 		err = PTR_ERR(vma);
1.1  riastrad 		goto err;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	err = i915_vma_pin(vma, 0, 0, PIN_GLOBAL | PIN_HIGH);
1.1  riastrad 	if (err)
1.1  riastrad 		goto err;
1.1  riastrad
1.1  riastrad 	engine->wa_ctx.vma = vma;
1.1  riastrad 	return 0;
1.1  riastrad
1.1  riastrad err:
1.1  riastrad 	i915_gem_object_put(obj);
1.1  riastrad 	return err;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void lrc_destroy_wa_ctx(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	i915_vma_unpin_and_release(&engine->wa_ctx.vma, 0);
1.1  riastrad }
1.1  riastrad
1.1  riastrad typedef u32 *(*wa_bb_func_t)(struct intel_engine_cs *engine, u32 *batch);
1.1  riastrad
1.1  riastrad static int intel_init_workaround_bb(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct i915_ctx_workarounds *wa_ctx = &engine->wa_ctx;
1.1  riastrad 	struct i915_wa_ctx_bb *wa_bb[2] = { &wa_ctx->indirect_ctx,
1.1  riastrad 					    &wa_ctx->per_ctx };
1.1  riastrad 	wa_bb_func_t wa_bb_fn[2];
1.1  riastrad 	struct page *page;
1.1  riastrad 	void *batch, *batch_ptr;
1.1  riastrad 	unsigned int i;
1.1  riastrad 	int ret;
1.1  riastrad
1.1  riastrad 	if (engine->class != RENDER_CLASS)
1.1  riastrad 		return 0;
1.1  riastrad
1.1  riastrad 	switch (INTEL_GEN(engine->i915)) {
1.1  riastrad 	case 12:
1.1  riastrad 	case 11:
1.1  riastrad 		return 0;
1.1  riastrad 	case 10:
1.1  riastrad 		wa_bb_fn[0] = gen10_init_indirectctx_bb;
1.1  riastrad 		wa_bb_fn[1] = NULL;
1.1  riastrad 		break;
1.1  riastrad 	case 9:
1.1  riastrad 		wa_bb_fn[0] = gen9_init_indirectctx_bb;
1.1  riastrad 		wa_bb_fn[1] = NULL;
1.1  riastrad 		break;
1.1  riastrad 	case 8:
1.1  riastrad 		wa_bb_fn[0] = gen8_init_indirectctx_bb;
1.1  riastrad 		wa_bb_fn[1] = NULL;
1.1  riastrad 		break;
1.1  riastrad 	default:
1.1  riastrad 		MISSING_CASE(INTEL_GEN(engine->i915));
1.1  riastrad 		return 0;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	ret = lrc_setup_wa_ctx(engine);
1.1  riastrad 	if (ret) {
1.1  riastrad 		DRM_DEBUG_DRIVER("Failed to setup context WA page: %d\n", ret);
1.1  riastrad 		return ret;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	page = i915_gem_object_get_dirty_page(wa_ctx->vma->obj, 0);
1.1  riastrad 	batch = batch_ptr = kmap_atomic(page);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Emit the two workaround batch buffers, recording the offset from the
1.1  riastrad 	 * start of the workaround batch buffer object for each and their
1.1  riastrad 	 * respective sizes.
1.1  riastrad 	 */
1.1  riastrad 	for (i = 0; i < ARRAY_SIZE(wa_bb_fn); i++) {
1.1  riastrad 		wa_bb[i]->offset = batch_ptr - batch;
1.1  riastrad 		if (GEM_DEBUG_WARN_ON(!IS_ALIGNED(wa_bb[i]->offset,
1.1  riastrad 						  CACHELINE_BYTES))) {
1.1  riastrad 			ret = -EINVAL;
1.1  riastrad 			break;
1.1  riastrad 		}
1.1  riastrad 		if (wa_bb_fn[i])
1.1  riastrad 			batch_ptr = wa_bb_fn[i](engine, batch_ptr);
1.1  riastrad 		wa_bb[i]->size = batch_ptr - (batch + wa_bb[i]->offset);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	BUG_ON(batch_ptr - batch > CTX_WA_BB_OBJ_SIZE);
1.1  riastrad
1.1  riastrad 	kunmap_atomic(batch);
1.1  riastrad 	if (ret)
1.1  riastrad 		lrc_destroy_wa_ctx(engine);
1.1  riastrad
1.1  riastrad 	return ret;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void enable_execlists(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	u32 mode;
1.1  riastrad
1.1  riastrad 	assert_forcewakes_active(engine->uncore, FORCEWAKE_ALL);
1.1  riastrad
1.1  riastrad 	intel_engine_set_hwsp_writemask(engine, ~0u); /* HWSTAM */
1.1  riastrad
1.1  riastrad 	if (INTEL_GEN(engine->i915) >= 11)
1.1  riastrad 		mode = _MASKED_BIT_ENABLE(GEN11_GFX_DISABLE_LEGACY_MODE);
1.1  riastrad 	else
1.1  riastrad 		mode = _MASKED_BIT_ENABLE(GFX_RUN_LIST_ENABLE);
1.1  riastrad 	ENGINE_WRITE_FW(engine, RING_MODE_GEN7, mode);
1.1  riastrad
1.1  riastrad 	ENGINE_WRITE_FW(engine, RING_MI_MODE, _MASKED_BIT_DISABLE(STOP_RING));
1.1  riastrad
1.1  riastrad 	ENGINE_WRITE_FW(engine,
1.1  riastrad 			RING_HWS_PGA,
1.1  riastrad 			i915_ggtt_offset(engine->status_page.vma));
1.1  riastrad 	ENGINE_POSTING_READ(engine, RING_HWS_PGA);
1.1  riastrad
1.1  riastrad 	engine->context_tag = 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static bool unexpected_starting_state(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	bool unexpected = false;
1.1  riastrad
1.1  riastrad 	if (ENGINE_READ_FW(engine, RING_MI_MODE) & STOP_RING) {
1.1  riastrad 		DRM_DEBUG_DRIVER("STOP_RING still set in RING_MI_MODE\n");
1.1  riastrad 		unexpected = true;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return unexpected;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int execlists_resume(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	intel_engine_apply_workarounds(engine);
1.1  riastrad 	intel_engine_apply_whitelist(engine);
1.1  riastrad
1.1  riastrad 	intel_mocs_init_engine(engine);
1.1  riastrad
1.1  riastrad 	intel_engine_reset_breadcrumbs(engine);
1.1  riastrad
1.1  riastrad 	if (GEM_SHOW_DEBUG() && unexpected_starting_state(engine)) {
1.1  riastrad 		struct drm_printer p = drm_debug_printer(__func__);
1.1  riastrad
1.1  riastrad 		intel_engine_dump(engine, &p, NULL);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	enable_execlists(engine);
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_reset_prepare(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad 	unsigned long flags;
1.1  riastrad
1.1  riastrad 	ENGINE_TRACE(engine, "depth<-%d\n",
1.1  riastrad 		     atomic_read(&execlists->tasklet.count));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Prevent request submission to the hardware until we have
1.1  riastrad 	 * completed the reset in i915_gem_reset_finish(). If a request
1.1  riastrad 	 * is completed by one engine, it may then queue a request
1.1  riastrad 	 * to a second via its execlists->tasklet *just* as we are
1.1  riastrad 	 * calling engine->resume() and also writing the ELSP.
1.1  riastrad 	 * Turning off the execlists->tasklet until the reset is over
1.1  riastrad 	 * prevents the race.
1.1  riastrad 	 */
1.1  riastrad 	__tasklet_disable_sync_once(&execlists->tasklet);
1.1  riastrad 	GEM_BUG_ON(!reset_in_progress(execlists));
1.1  riastrad
1.1  riastrad 	/* And flush any current direct submission. */
1.1  riastrad 	spin_lock_irqsave(&engine->active.lock, flags);
1.1  riastrad 	spin_unlock_irqrestore(&engine->active.lock, flags);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We stop engines, otherwise we might get failed reset and a
1.1  riastrad 	 * dead gpu (on elk). Also as modern gpu as kbl can suffer
1.1  riastrad 	 * from system hang if batchbuffer is progressing when
1.1  riastrad 	 * the reset is issued, regardless of READY_TO_RESET ack.
1.1  riastrad 	 * Thus assume it is best to stop engines on all gens
1.1  riastrad 	 * where we have a gpu reset.
1.1  riastrad 	 *
1.1  riastrad 	 * WaKBLVECSSemaphoreWaitPoll:kbl (on ALL_ENGINES)
1.1  riastrad 	 *
1.1  riastrad 	 * FIXME: Wa for more modern gens needs to be validated
1.1  riastrad 	 */
1.1  riastrad 	intel_engine_stop_cs(engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void reset_csb_pointers(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad 	const unsigned int reset_value = execlists->csb_size - 1;
1.1  riastrad
1.1  riastrad 	ring_set_paused(engine, 0);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * After a reset, the HW starts writing into CSB entry [0]. We
1.1  riastrad 	 * therefore have to set our HEAD pointer back one entry so that
1.1  riastrad 	 * the *first* entry we check is entry 0. To complicate this further,
1.1  riastrad 	 * as we don't wait for the first interrupt after reset, we have to
1.1  riastrad 	 * fake the HW write to point back to the last entry so that our
1.1  riastrad 	 * inline comparison of our cached head position against the last HW
1.1  riastrad 	 * write works even before the first interrupt.
1.1  riastrad 	 */
1.1  riastrad 	execlists->csb_head = reset_value;
1.1  riastrad 	WRITE_ONCE(*execlists->csb_write, reset_value);
1.1  riastrad 	wmb(); /* Make sure this is visible to HW (paranoia?) */
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Sometimes Icelake forgets to reset its pointers on a GPU reset.
1.1  riastrad 	 * Bludgeon them with a mmio update to be sure.
1.1  riastrad 	 */
1.1  riastrad 	ENGINE_WRITE(engine, RING_CONTEXT_STATUS_PTR,
1.1  riastrad 		     reset_value << 8 | reset_value);
1.1  riastrad 	ENGINE_POSTING_READ(engine, RING_CONTEXT_STATUS_PTR);
1.1  riastrad
1.1  riastrad 	invalidate_csb_entries(&execlists->csb_status[0],
1.1  riastrad 			       &execlists->csb_status[reset_value]);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __reset_stop_ring(u32 *regs, const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	int x;
1.1  riastrad
1.1  riastrad 	x = lrc_ring_mi_mode(engine);
1.1  riastrad 	if (x != -1) {
1.1  riastrad 		regs[x + 1] &= ~STOP_RING;
1.1  riastrad 		regs[x + 1] |= STOP_RING << 16;
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __execlists_reset_reg_state(const struct intel_context *ce,
1.1  riastrad 					const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	u32 *regs = ce->lrc_reg_state;
1.1  riastrad
1.1  riastrad 	__reset_stop_ring(regs, engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void __execlists_reset(struct intel_engine_cs *engine, bool stalled)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad 	struct intel_context *ce;
1.1  riastrad 	struct i915_request *rq;
1.1  riastrad 	u32 head;
1.1  riastrad
1.1  riastrad 	mb(); /* paranoia: read the CSB pointers from after the reset */
1.1  riastrad 	clflush(execlists->csb_write);
1.1  riastrad 	mb();
1.1  riastrad
1.1  riastrad 	process_csb(engine); /* drain preemption events */
1.1  riastrad
1.1  riastrad 	/* Following the reset, we need to reload the CSB read/write pointers */
1.1  riastrad 	reset_csb_pointers(engine);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Save the currently executing context, even if we completed
1.1  riastrad 	 * its request, it was still running at the time of the
1.1  riastrad 	 * reset and will have been clobbered.
1.1  riastrad 	 */
1.1  riastrad 	rq = execlists_active(execlists);
1.1  riastrad 	if (!rq)
1.1  riastrad 		goto unwind;
1.1  riastrad
1.1  riastrad 	/* We still have requests in-flight; the engine should be active */
1.1  riastrad 	GEM_BUG_ON(!intel_engine_pm_is_awake(engine));
1.1  riastrad
1.1  riastrad 	ce = rq->context;
1.1  riastrad 	GEM_BUG_ON(!i915_vma_is_pinned(ce->state));
1.1  riastrad
1.1  riastrad 	if (i915_request_completed(rq)) {
1.1  riastrad 		/* Idle context; tidy up the ring so we can restart afresh */
1.1  riastrad 		head = intel_ring_wrap(ce->ring, rq->tail);
1.1  riastrad 		goto out_replay;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/* Context has requests still in-flight; it should not be idle! */
1.1  riastrad 	GEM_BUG_ON(i915_active_is_idle(&ce->active));
1.1  riastrad 	rq = active_request(ce->timeline, rq);
1.1  riastrad 	head = intel_ring_wrap(ce->ring, rq->head);
1.1  riastrad 	GEM_BUG_ON(head == ce->ring->tail);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If this request hasn't started yet, e.g. it is waiting on a
1.1  riastrad 	 * semaphore, we need to avoid skipping the request or else we
1.1  riastrad 	 * break the signaling chain. However, if the context is corrupt
1.1  riastrad 	 * the request will not restart and we will be stuck with a wedged
1.1  riastrad 	 * device. It is quite often the case that if we issue a reset
1.1  riastrad 	 * while the GPU is loading the context image, that the context
1.1  riastrad 	 * image becomes corrupt.
1.1  riastrad 	 *
1.1  riastrad 	 * Otherwise, if we have not started yet, the request should replay
1.1  riastrad 	 * perfectly and we do not need to flag the result as being erroneous.
1.1  riastrad 	 */
1.1  riastrad 	if (!i915_request_started(rq))
1.1  riastrad 		goto out_replay;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * If the request was innocent, we leave the request in the ELSP
1.1  riastrad 	 * and will try to replay it on restarting. The context image may
1.1  riastrad 	 * have been corrupted by the reset, in which case we may have
1.1  riastrad 	 * to service a new GPU hang, but more likely we can continue on
1.1  riastrad 	 * without impact.
1.1  riastrad 	 *
1.1  riastrad 	 * If the request was guilty, we presume the context is corrupt
1.1  riastrad 	 * and have to at least restore the RING register in the context
1.1  riastrad 	 * image back to the expected values to skip over the guilty request.
1.1  riastrad 	 */
1.1  riastrad 	__i915_request_reset(rq, stalled);
1.1  riastrad 	if (!stalled)
1.1  riastrad 		goto out_replay;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We want a simple context + ring to execute the breadcrumb update.
1.1  riastrad 	 * We cannot rely on the context being intact across the GPU hang,
1.1  riastrad 	 * so clear it and rebuild just what we need for the breadcrumb.
1.1  riastrad 	 * All pending requests for this context will be zapped, and any
1.1  riastrad 	 * future request will be after userspace has had the opportunity
1.1  riastrad 	 * to recreate its own state.
1.1  riastrad 	 */
1.1  riastrad 	GEM_BUG_ON(!intel_context_is_pinned(ce));
1.1  riastrad 	restore_default_state(ce, engine);
1.1  riastrad
1.1  riastrad out_replay:
1.1  riastrad 	ENGINE_TRACE(engine, "replay {head:%04x, tail:%04x}\n",
1.1  riastrad 		     head, ce->ring->tail);
1.1  riastrad 	__execlists_reset_reg_state(ce, engine);
1.1  riastrad 	__execlists_update_reg_state(ce, engine, head);
1.1  riastrad 	ce->lrc_desc |= CTX_DESC_FORCE_RESTORE; /* paranoid: GPU was reset! */
1.1  riastrad
1.1  riastrad unwind:
1.1  riastrad 	/* Push back any incomplete requests for replay after the reset. */
1.1  riastrad 	cancel_port_requests(execlists);
1.1  riastrad 	__unwind_incomplete_requests(engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_reset_rewind(struct intel_engine_cs *engine, bool stalled)
1.1  riastrad {
1.1  riastrad 	unsigned long flags;
1.1  riastrad
1.1  riastrad 	ENGINE_TRACE(engine, "\n");
1.1  riastrad
1.1  riastrad 	spin_lock_irqsave(&engine->active.lock, flags);
1.1  riastrad
1.1  riastrad 	__execlists_reset(engine, stalled);
1.1  riastrad
1.1  riastrad 	spin_unlock_irqrestore(&engine->active.lock, flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void nop_submission_tasklet(unsigned long data)
1.1  riastrad {
1.1  riastrad 	/* The driver is wedged; don't process any more events. */
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_reset_cancel(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad 	struct i915_request *rq, *rn;
1.1  riastrad 	struct rb_node *rb;
1.1  riastrad 	unsigned long flags;
1.1  riastrad
1.1  riastrad 	ENGINE_TRACE(engine, "\n");
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Before we call engine->cancel_requests(), we should have exclusive
1.1  riastrad 	 * access to the submission state. This is arranged for us by the
1.1  riastrad 	 * caller disabling the interrupt generation, the tasklet and other
1.1  riastrad 	 * threads that may then access the same state, giving us a free hand
1.1  riastrad 	 * to reset state. However, we still need to let lockdep be aware that
1.1  riastrad 	 * we know this state may be accessed in hardirq context, so we
1.1  riastrad 	 * disable the irq around this manipulation and we want to keep
1.1  riastrad 	 * the spinlock focused on its duties and not accidentally conflate
1.1  riastrad 	 * coverage to the submission's irq state. (Similarly, although we
1.1  riastrad 	 * shouldn't need to disable irq around the manipulation of the
1.1  riastrad 	 * submission's irq state, we also wish to remind ourselves that
1.1  riastrad 	 * it is irq state.)
1.1  riastrad 	 */
1.1  riastrad 	spin_lock_irqsave(&engine->active.lock, flags);
1.1  riastrad
1.1  riastrad 	__execlists_reset(engine, true);
1.1  riastrad
1.1  riastrad 	/* Mark all executing requests as skipped. */
1.1  riastrad 	list_for_each_entry(rq, &engine->active.requests, sched.link)
1.1  riastrad 		mark_eio(rq);
1.1  riastrad
1.1  riastrad 	/* Flush the queued requests to the timeline list (for retiring). */
1.1  riastrad 	while ((rb = rb_first_cached(&execlists->queue))) {
1.1  riastrad 		struct i915_priolist *p = to_priolist(rb);
1.1  riastrad 		int i;
1.1  riastrad
1.1  riastrad 		priolist_for_each_request_consume(rq, rn, p, i) {
1.1  riastrad 			mark_eio(rq);
1.1  riastrad 			__i915_request_submit(rq);
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		rb_erase_cached(&p->node, &execlists->queue);
1.1  riastrad 		i915_priolist_free(p);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/* On-hold requests will be flushed to timeline upon their release */
1.1  riastrad 	list_for_each_entry(rq, &engine->active.hold, sched.link)
1.1  riastrad 		mark_eio(rq);
1.1  riastrad
1.1  riastrad 	/* Cancel all attached virtual engines */
1.1  riastrad 	while ((rb = rb_first_cached(&execlists->virtual))) {
1.1  riastrad 		struct virtual_engine *ve =
1.1  riastrad 			rb_entry(rb, typeof(*ve), nodes[engine->id].rb);
1.1  riastrad
1.1  riastrad 		rb_erase_cached(rb, &execlists->virtual);
1.7  riastrad 		container_of(rb, struct ve_node, rb)->inserted = false;
1.1  riastrad
1.1  riastrad 		spin_lock(&ve->base.active.lock);
1.1  riastrad 		rq = fetch_and_zero(&ve->request);
1.1  riastrad 		if (rq) {
1.1  riastrad 			mark_eio(rq);
1.1  riastrad
1.1  riastrad 			rq->engine = engine;
1.1  riastrad 			__i915_request_submit(rq);
1.1  riastrad 			i915_request_put(rq);
1.1  riastrad
1.1  riastrad 			ve->base.execlists.queue_priority_hint = INT_MIN;
1.1  riastrad 		}
1.1  riastrad 		spin_unlock(&ve->base.active.lock);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/* Remaining _unready_ requests will be nop'ed when submitted */
1.1  riastrad
1.1  riastrad 	execlists->queue_priority_hint = INT_MIN;
1.7  riastrad #ifdef __NetBSD__
1.7  riastrad 	i915_sched_init(execlists);
1.7  riastrad 	rb_tree_init(&execlists->virtual.rb_root.rbr_tree, &ve_tree_ops);
1.7  riastrad #else
1.1  riastrad 	execlists->queue = RB_ROOT_CACHED;
1.7  riastrad #endif
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(__tasklet_is_enabled(&execlists->tasklet));
1.1  riastrad 	execlists->tasklet.func = nop_submission_tasklet;
1.1  riastrad
1.1  riastrad 	spin_unlock_irqrestore(&engine->active.lock, flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_reset_finish(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * After a GPU reset, we may have requests to replay. Do so now while
1.1  riastrad 	 * we still have the forcewake to be sure that the GPU is not allowed
1.1  riastrad 	 * to sleep before we restart and reload a context.
1.1  riastrad 	 */
1.1  riastrad 	GEM_BUG_ON(!reset_in_progress(execlists));
1.1  riastrad 	if (!RB_EMPTY_ROOT(&execlists->queue.rb_root))
1.1  riastrad 		execlists->tasklet.func(execlists->tasklet.data);
1.1  riastrad
1.1  riastrad 	if (__tasklet_enable(&execlists->tasklet))
1.1  riastrad 		/* And kick in case we missed a new request submission. */
1.1  riastrad 		tasklet_hi_schedule(&execlists->tasklet);
1.1  riastrad 	ENGINE_TRACE(engine, "depth->%d\n",
1.1  riastrad 		     atomic_read(&execlists->tasklet.count));
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int gen8_emit_bb_start_noarb(struct i915_request *rq,
1.1  riastrad 				    u64 offset, u32 len,
1.1  riastrad 				    const unsigned int flags)
1.1  riastrad {
1.1  riastrad 	u32 *cs;
1.1  riastrad
1.1  riastrad 	cs = intel_ring_begin(rq, 4);
1.1  riastrad 	if (IS_ERR(cs))
1.1  riastrad 		return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * WaDisableCtxRestoreArbitration:bdw,chv
1.1  riastrad 	 *
1.1  riastrad 	 * We don't need to perform MI_ARB_ENABLE as often as we do (in
1.1  riastrad 	 * particular all the gen that do not need the w/a at all!), if we
1.1  riastrad 	 * took care to make sure that on every switch into this context
1.1  riastrad 	 * (both ordinary and for preemption) that arbitrartion was enabled
1.1  riastrad 	 * we would be fine.  However, for gen8 there is another w/a that
1.1  riastrad 	 * requires us to not preempt inside GPGPU execution, so we keep
1.1  riastrad 	 * arbitration disabled for gen8 batches. Arbitration will be
1.1  riastrad 	 * re-enabled before we close the request
1.1  riastrad 	 * (engine->emit_fini_breadcrumb).
1.1  riastrad 	 */
1.1  riastrad 	*cs++ = MI_ARB_ON_OFF | MI_ARB_DISABLE;
1.1  riastrad
1.1  riastrad 	/* FIXME(BDW+): Address space and security selectors. */
1.1  riastrad 	*cs++ = MI_BATCH_BUFFER_START_GEN8 |
1.1  riastrad 		(flags & I915_DISPATCH_SECURE ? 0 : BIT(8));
1.1  riastrad 	*cs++ = lower_32_bits(offset);
1.1  riastrad 	*cs++ = upper_32_bits(offset);
1.1  riastrad
1.1  riastrad 	intel_ring_advance(rq, cs);
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int gen8_emit_bb_start(struct i915_request *rq,
1.1  riastrad 			      u64 offset, u32 len,
1.1  riastrad 			      const unsigned int flags)
1.1  riastrad {
1.1  riastrad 	u32 *cs;
1.1  riastrad
1.1  riastrad 	cs = intel_ring_begin(rq, 6);
1.1  riastrad 	if (IS_ERR(cs))
1.1  riastrad 		return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 	*cs++ = MI_ARB_ON_OFF | MI_ARB_ENABLE;
1.1  riastrad
1.1  riastrad 	*cs++ = MI_BATCH_BUFFER_START_GEN8 |
1.1  riastrad 		(flags & I915_DISPATCH_SECURE ? 0 : BIT(8));
1.1  riastrad 	*cs++ = lower_32_bits(offset);
1.1  riastrad 	*cs++ = upper_32_bits(offset);
1.1  riastrad
1.1  riastrad 	*cs++ = MI_ARB_ON_OFF | MI_ARB_DISABLE;
1.1  riastrad 	*cs++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	intel_ring_advance(rq, cs);
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void gen8_logical_ring_enable_irq(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	ENGINE_WRITE(engine, RING_IMR,
1.1  riastrad 		     ~(engine->irq_enable_mask | engine->irq_keep_mask));
1.1  riastrad 	ENGINE_POSTING_READ(engine, RING_IMR);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void gen8_logical_ring_disable_irq(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	ENGINE_WRITE(engine, RING_IMR, ~engine->irq_keep_mask);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int gen8_emit_flush(struct i915_request *request, u32 mode)
1.1  riastrad {
1.1  riastrad 	u32 cmd, *cs;
1.1  riastrad
1.1  riastrad 	cs = intel_ring_begin(request, 4);
1.1  riastrad 	if (IS_ERR(cs))
1.1  riastrad 		return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 	cmd = MI_FLUSH_DW + 1;
1.1  riastrad
1.1  riastrad 	/* We always require a command barrier so that subsequent
1.1  riastrad 	 * commands, such as breadcrumb interrupts, are strictly ordered
1.1  riastrad 	 * wrt the contents of the write cache being flushed to memory
1.1  riastrad 	 * (and thus being coherent from the CPU).
1.1  riastrad 	 */
1.1  riastrad 	cmd |= MI_FLUSH_DW_STORE_INDEX | MI_FLUSH_DW_OP_STOREDW;
1.1  riastrad
1.1  riastrad 	if (mode & EMIT_INVALIDATE) {
1.1  riastrad 		cmd |= MI_INVALIDATE_TLB;
1.1  riastrad 		if (request->engine->class == VIDEO_DECODE_CLASS)
1.1  riastrad 			cmd |= MI_INVALIDATE_BSD;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	*cs++ = cmd;
1.1  riastrad 	*cs++ = LRC_PPHWSP_SCRATCH_ADDR;
1.1  riastrad 	*cs++ = 0; /* upper addr */
1.1  riastrad 	*cs++ = 0; /* value */
1.1  riastrad 	intel_ring_advance(request, cs);
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int gen8_emit_flush_render(struct i915_request *request,
1.1  riastrad 				  u32 mode)
1.1  riastrad {
1.1  riastrad 	bool vf_flush_wa = false, dc_flush_wa = false;
1.1  riastrad 	u32 *cs, flags = 0;
1.1  riastrad 	int len;
1.1  riastrad
1.1  riastrad 	flags |= PIPE_CONTROL_CS_STALL;
1.1  riastrad
1.1  riastrad 	if (mode & EMIT_FLUSH) {
1.1  riastrad 		flags |= PIPE_CONTROL_RENDER_TARGET_CACHE_FLUSH;
1.1  riastrad 		flags |= PIPE_CONTROL_DEPTH_CACHE_FLUSH;
1.1  riastrad 		flags |= PIPE_CONTROL_DC_FLUSH_ENABLE;
1.1  riastrad 		flags |= PIPE_CONTROL_FLUSH_ENABLE;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (mode & EMIT_INVALIDATE) {
1.1  riastrad 		flags |= PIPE_CONTROL_TLB_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_INSTRUCTION_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_TEXTURE_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_VF_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_CONST_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_STATE_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_QW_WRITE;
1.1  riastrad 		flags |= PIPE_CONTROL_STORE_DATA_INDEX;
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * On GEN9: before VF_CACHE_INVALIDATE we need to emit a NULL
1.1  riastrad 		 * pipe control.
1.1  riastrad 		 */
1.1  riastrad 		if (IS_GEN(request->i915, 9))
1.1  riastrad 			vf_flush_wa = true;
1.1  riastrad
1.1  riastrad 		/* WaForGAMHang:kbl */
1.1  riastrad 		if (IS_KBL_REVID(request->i915, 0, KBL_REVID_B0))
1.1  riastrad 			dc_flush_wa = true;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	len = 6;
1.1  riastrad
1.1  riastrad 	if (vf_flush_wa)
1.1  riastrad 		len += 6;
1.1  riastrad
1.1  riastrad 	if (dc_flush_wa)
1.1  riastrad 		len += 12;
1.1  riastrad
1.1  riastrad 	cs = intel_ring_begin(request, len);
1.1  riastrad 	if (IS_ERR(cs))
1.1  riastrad 		return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 	if (vf_flush_wa)
1.1  riastrad 		cs = gen8_emit_pipe_control(cs, 0, 0);
1.1  riastrad
1.1  riastrad 	if (dc_flush_wa)
1.1  riastrad 		cs = gen8_emit_pipe_control(cs, PIPE_CONTROL_DC_FLUSH_ENABLE,
1.1  riastrad 					    0);
1.1  riastrad
1.1  riastrad 	cs = gen8_emit_pipe_control(cs, flags, LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad
1.1  riastrad 	if (dc_flush_wa)
1.1  riastrad 		cs = gen8_emit_pipe_control(cs, PIPE_CONTROL_CS_STALL, 0);
1.1  riastrad
1.1  riastrad 	intel_ring_advance(request, cs);
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int gen11_emit_flush_render(struct i915_request *request,
1.1  riastrad 				   u32 mode)
1.1  riastrad {
1.1  riastrad 	if (mode & EMIT_FLUSH) {
1.1  riastrad 		u32 *cs;
1.1  riastrad 		u32 flags = 0;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_CS_STALL;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_TILE_CACHE_FLUSH;
1.1  riastrad 		flags |= PIPE_CONTROL_RENDER_TARGET_CACHE_FLUSH;
1.1  riastrad 		flags |= PIPE_CONTROL_DEPTH_CACHE_FLUSH;
1.1  riastrad 		flags |= PIPE_CONTROL_DC_FLUSH_ENABLE;
1.1  riastrad 		flags |= PIPE_CONTROL_FLUSH_ENABLE;
1.1  riastrad 		flags |= PIPE_CONTROL_QW_WRITE;
1.1  riastrad 		flags |= PIPE_CONTROL_STORE_DATA_INDEX;
1.1  riastrad
1.1  riastrad 		cs = intel_ring_begin(request, 6);
1.1  riastrad 		if (IS_ERR(cs))
1.1  riastrad 			return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 		cs = gen8_emit_pipe_control(cs, flags, LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad 		intel_ring_advance(request, cs);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (mode & EMIT_INVALIDATE) {
1.1  riastrad 		u32 *cs;
1.1  riastrad 		u32 flags = 0;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_CS_STALL;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_COMMAND_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_TLB_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_INSTRUCTION_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_TEXTURE_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_VF_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_CONST_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_STATE_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_QW_WRITE;
1.1  riastrad 		flags |= PIPE_CONTROL_STORE_DATA_INDEX;
1.1  riastrad
1.1  riastrad 		cs = intel_ring_begin(request, 6);
1.1  riastrad 		if (IS_ERR(cs))
1.1  riastrad 			return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 		cs = gen8_emit_pipe_control(cs, flags, LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad 		intel_ring_advance(request, cs);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 preparser_disable(bool state)
1.1  riastrad {
1.1  riastrad 	return MI_ARB_CHECK | 1 << 8 | state;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int gen12_emit_flush_render(struct i915_request *request,
1.1  riastrad 				   u32 mode)
1.1  riastrad {
1.1  riastrad 	if (mode & EMIT_FLUSH) {
1.1  riastrad 		u32 flags = 0;
1.1  riastrad 		u32 *cs;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_TILE_CACHE_FLUSH;
1.1  riastrad 		flags |= PIPE_CONTROL_RENDER_TARGET_CACHE_FLUSH;
1.1  riastrad 		flags |= PIPE_CONTROL_DEPTH_CACHE_FLUSH;
1.1  riastrad 		/* Wa_1409600907:tgl */
1.1  riastrad 		flags |= PIPE_CONTROL_DEPTH_STALL;
1.1  riastrad 		flags |= PIPE_CONTROL_DC_FLUSH_ENABLE;
1.1  riastrad 		flags |= PIPE_CONTROL_FLUSH_ENABLE;
1.1  riastrad 		flags |= PIPE_CONTROL_HDC_PIPELINE_FLUSH;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_STORE_DATA_INDEX;
1.1  riastrad 		flags |= PIPE_CONTROL_QW_WRITE;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_CS_STALL;
1.1  riastrad
1.1  riastrad 		cs = intel_ring_begin(request, 6);
1.1  riastrad 		if (IS_ERR(cs))
1.1  riastrad 			return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 		cs = gen8_emit_pipe_control(cs, flags, LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad 		intel_ring_advance(request, cs);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (mode & EMIT_INVALIDATE) {
1.1  riastrad 		u32 flags = 0;
1.1  riastrad 		u32 *cs;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_COMMAND_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_TLB_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_INSTRUCTION_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_TEXTURE_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_VF_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_CONST_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_STATE_CACHE_INVALIDATE;
1.1  riastrad 		flags |= PIPE_CONTROL_L3_RO_CACHE_INVALIDATE;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_STORE_DATA_INDEX;
1.1  riastrad 		flags |= PIPE_CONTROL_QW_WRITE;
1.1  riastrad
1.1  riastrad 		flags |= PIPE_CONTROL_CS_STALL;
1.1  riastrad
1.1  riastrad 		cs = intel_ring_begin(request, 8);
1.1  riastrad 		if (IS_ERR(cs))
1.1  riastrad 			return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * Prevent the pre-parser from skipping past the TLB
1.1  riastrad 		 * invalidate and loading a stale page for the batch
1.1  riastrad 		 * buffer / request payload.
1.1  riastrad 		 */
1.1  riastrad 		*cs++ = preparser_disable(true);
1.1  riastrad
1.1  riastrad 		cs = gen8_emit_pipe_control(cs, flags, LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad
1.1  riastrad 		*cs++ = preparser_disable(false);
1.1  riastrad 		intel_ring_advance(request, cs);
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * Wa_1604544889:tgl
1.1  riastrad 		 */
1.1  riastrad 		if (IS_TGL_REVID(request->i915, TGL_REVID_A0, TGL_REVID_A0)) {
1.1  riastrad 			flags = 0;
1.1  riastrad 			flags |= PIPE_CONTROL_CS_STALL;
1.1  riastrad 			flags |= PIPE_CONTROL_HDC_PIPELINE_FLUSH;
1.1  riastrad
1.1  riastrad 			flags |= PIPE_CONTROL_STORE_DATA_INDEX;
1.1  riastrad 			flags |= PIPE_CONTROL_QW_WRITE;
1.1  riastrad
1.1  riastrad 			cs = intel_ring_begin(request, 6);
1.1  riastrad 			if (IS_ERR(cs))
1.1  riastrad 				return PTR_ERR(cs);
1.1  riastrad
1.1  riastrad 			cs = gen8_emit_pipe_control(cs, flags,
1.1  riastrad 						    LRC_PPHWSP_SCRATCH_ADDR);
1.1  riastrad 			intel_ring_advance(request, cs);
1.1  riastrad 		}
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * Reserve space for 2 NOOPs at the end of each request to be
1.1  riastrad  * used as a workaround for not being allowed to do lite
1.1  riastrad  * restore with HEAD==TAIL (WaIdleLiteRestore).
1.1  riastrad  */
1.1  riastrad static u32 *gen8_emit_wa_tail(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	/* Ensure there's always at least one preemption point per-request. */
1.1  riastrad 	*cs++ = MI_ARB_CHECK;
1.1  riastrad 	*cs++ = MI_NOOP;
1.1  riastrad 	request->wa_tail = intel_ring_offset(request, cs);
1.1  riastrad
1.1  riastrad 	return cs;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *emit_preempt_busywait(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	*cs++ = MI_SEMAPHORE_WAIT |
1.1  riastrad 		MI_SEMAPHORE_GLOBAL_GTT |
1.1  riastrad 		MI_SEMAPHORE_POLL |
1.1  riastrad 		MI_SEMAPHORE_SAD_EQ_SDD;
1.1  riastrad 	*cs++ = 0;
1.1  riastrad 	*cs++ = intel_hws_preempt_address(request->engine);
1.1  riastrad 	*cs++ = 0;
1.1  riastrad
1.1  riastrad 	return cs;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static __always_inline u32*
1.1  riastrad gen8_emit_fini_breadcrumb_footer(struct i915_request *request,
1.1  riastrad 				 u32 *cs)
1.1  riastrad {
1.1  riastrad 	*cs++ = MI_USER_INTERRUPT;
1.1  riastrad
1.1  riastrad 	*cs++ = MI_ARB_ON_OFF | MI_ARB_ENABLE;
1.1  riastrad 	if (intel_engine_has_semaphores(request->engine))
1.1  riastrad 		cs = emit_preempt_busywait(request, cs);
1.1  riastrad
1.1  riastrad 	request->tail = intel_ring_offset(request, cs);
1.1  riastrad 	assert_ring_tail_valid(request->ring, request->tail);
1.1  riastrad
1.1  riastrad 	return gen8_emit_wa_tail(request, cs);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *gen8_emit_fini_breadcrumb(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	cs = gen8_emit_ggtt_write(cs,
1.1  riastrad 				  request->fence.seqno,
1.1  riastrad 				  i915_request_active_timeline(request)->hwsp_offset,
1.1  riastrad 				  0);
1.1  riastrad
1.1  riastrad 	return gen8_emit_fini_breadcrumb_footer(request, cs);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *gen8_emit_fini_breadcrumb_rcs(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	cs = gen8_emit_pipe_control(cs,
1.1  riastrad 				    PIPE_CONTROL_RENDER_TARGET_CACHE_FLUSH |
1.1  riastrad 				    PIPE_CONTROL_DEPTH_CACHE_FLUSH |
1.1  riastrad 				    PIPE_CONTROL_DC_FLUSH_ENABLE,
1.1  riastrad 				    0);
1.1  riastrad
1.1  riastrad 	/* XXX flush+write+CS_STALL all in one upsets gem_concurrent_blt:kbl */
1.1  riastrad 	cs = gen8_emit_ggtt_write_rcs(cs,
1.1  riastrad 				      request->fence.seqno,
1.1  riastrad 				      i915_request_active_timeline(request)->hwsp_offset,
1.1  riastrad 				      PIPE_CONTROL_FLUSH_ENABLE |
1.1  riastrad 				      PIPE_CONTROL_CS_STALL);
1.1  riastrad
1.1  riastrad 	return gen8_emit_fini_breadcrumb_footer(request, cs);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *
1.1  riastrad gen11_emit_fini_breadcrumb_rcs(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	cs = gen8_emit_ggtt_write_rcs(cs,
1.1  riastrad 				      request->fence.seqno,
1.1  riastrad 				      i915_request_active_timeline(request)->hwsp_offset,
1.1  riastrad 				      PIPE_CONTROL_CS_STALL |
1.1  riastrad 				      PIPE_CONTROL_TILE_CACHE_FLUSH |
1.1  riastrad 				      PIPE_CONTROL_RENDER_TARGET_CACHE_FLUSH |
1.1  riastrad 				      PIPE_CONTROL_DEPTH_CACHE_FLUSH |
1.1  riastrad 				      PIPE_CONTROL_DC_FLUSH_ENABLE |
1.1  riastrad 				      PIPE_CONTROL_FLUSH_ENABLE);
1.1  riastrad
1.1  riastrad 	return gen8_emit_fini_breadcrumb_footer(request, cs);
1.1  riastrad }
1.1  riastrad
1.1  riastrad /*
1.1  riastrad  * Note that the CS instruction pre-parser will not stall on the breadcrumb
1.1  riastrad  * flush and will continue pre-fetching the instructions after it before the
1.1  riastrad  * memory sync is completed. On pre-gen12 HW, the pre-parser will stop at
1.1  riastrad  * BB_START/END instructions, so, even though we might pre-fetch the pre-amble
1.1  riastrad  * of the next request before the memory has been flushed, we're guaranteed that
1.1  riastrad  * we won't access the batch itself too early.
1.1  riastrad  * However, on gen12+ the parser can pre-fetch across the BB_START/END commands,
1.1  riastrad  * so, if the current request is modifying an instruction in the next request on
1.1  riastrad  * the same intel_context, we might pre-fetch and then execute the pre-update
1.1  riastrad  * instruction. To avoid this, the users of self-modifying code should either
1.1  riastrad  * disable the parser around the code emitting the memory writes, via a new flag
1.1  riastrad  * added to MI_ARB_CHECK, or emit the writes from a different intel_context. For
1.1  riastrad  * the in-kernel use-cases we've opted to use a separate context, see
1.1  riastrad  * reloc_gpu() as an example.
1.1  riastrad  * All the above applies only to the instructions themselves. Non-inline data
1.1  riastrad  * used by the instructions is not pre-fetched.
1.1  riastrad  */
1.1  riastrad
1.1  riastrad static u32 *gen12_emit_preempt_busywait(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	*cs++ = MI_SEMAPHORE_WAIT_TOKEN |
1.1  riastrad 		MI_SEMAPHORE_GLOBAL_GTT |
1.1  riastrad 		MI_SEMAPHORE_POLL |
1.1  riastrad 		MI_SEMAPHORE_SAD_EQ_SDD;
1.1  riastrad 	*cs++ = 0;
1.1  riastrad 	*cs++ = intel_hws_preempt_address(request->engine);
1.1  riastrad 	*cs++ = 0;
1.1  riastrad 	*cs++ = 0;
1.1  riastrad 	*cs++ = MI_NOOP;
1.1  riastrad
1.1  riastrad 	return cs;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static __always_inline u32*
1.1  riastrad gen12_emit_fini_breadcrumb_footer(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	*cs++ = MI_USER_INTERRUPT;
1.1  riastrad
1.1  riastrad 	*cs++ = MI_ARB_ON_OFF | MI_ARB_ENABLE;
1.1  riastrad 	if (intel_engine_has_semaphores(request->engine))
1.1  riastrad 		cs = gen12_emit_preempt_busywait(request, cs);
1.1  riastrad
1.1  riastrad 	request->tail = intel_ring_offset(request, cs);
1.1  riastrad 	assert_ring_tail_valid(request->ring, request->tail);
1.1  riastrad
1.1  riastrad 	return gen8_emit_wa_tail(request, cs);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *gen12_emit_fini_breadcrumb(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	cs = gen8_emit_ggtt_write(cs,
1.1  riastrad 				  request->fence.seqno,
1.1  riastrad 				  i915_request_active_timeline(request)->hwsp_offset,
1.1  riastrad 				  0);
1.1  riastrad
1.1  riastrad 	return gen12_emit_fini_breadcrumb_footer(request, cs);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 *
1.1  riastrad gen12_emit_fini_breadcrumb_rcs(struct i915_request *request, u32 *cs)
1.1  riastrad {
1.1  riastrad 	cs = gen8_emit_ggtt_write_rcs(cs,
1.1  riastrad 				      request->fence.seqno,
1.1  riastrad 				      i915_request_active_timeline(request)->hwsp_offset,
1.1  riastrad 				      PIPE_CONTROL_CS_STALL |
1.1  riastrad 				      PIPE_CONTROL_TILE_CACHE_FLUSH |
1.1  riastrad 				      PIPE_CONTROL_RENDER_TARGET_CACHE_FLUSH |
1.1  riastrad 				      PIPE_CONTROL_DEPTH_CACHE_FLUSH |
1.1  riastrad 				      /* Wa_1409600907:tgl */
1.1  riastrad 				      PIPE_CONTROL_DEPTH_STALL |
1.1  riastrad 				      PIPE_CONTROL_DC_FLUSH_ENABLE |
1.1  riastrad 				      PIPE_CONTROL_FLUSH_ENABLE |
1.1  riastrad 				      PIPE_CONTROL_HDC_PIPELINE_FLUSH);
1.1  riastrad
1.1  riastrad 	return gen12_emit_fini_breadcrumb_footer(request, cs);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_park(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	cancel_timer(&engine->execlists.timer);
1.1  riastrad 	cancel_timer(&engine->execlists.preempt);
1.1  riastrad }
1.1  riastrad
1.1  riastrad void intel_execlists_set_default_submission(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	engine->submit_request = execlists_submit_request;
1.1  riastrad 	engine->schedule = i915_schedule;
1.1  riastrad 	engine->execlists.tasklet.func = execlists_submission_tasklet;
1.1  riastrad
1.1  riastrad 	engine->reset.prepare = execlists_reset_prepare;
1.1  riastrad 	engine->reset.rewind = execlists_reset_rewind;
1.1  riastrad 	engine->reset.cancel = execlists_reset_cancel;
1.1  riastrad 	engine->reset.finish = execlists_reset_finish;
1.1  riastrad
1.1  riastrad 	engine->park = execlists_park;
1.1  riastrad 	engine->unpark = NULL;
1.1  riastrad
1.1  riastrad 	engine->flags |= I915_ENGINE_SUPPORTS_STATS;
1.1  riastrad 	if (!intel_vgpu_active(engine->i915)) {
1.1  riastrad 		engine->flags |= I915_ENGINE_HAS_SEMAPHORES;
1.1  riastrad 		if (HAS_LOGICAL_RING_PREEMPTION(engine->i915))
1.1  riastrad 			engine->flags |= I915_ENGINE_HAS_PREEMPTION;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (INTEL_GEN(engine->i915) >= 12)
1.1  riastrad 		engine->flags |= I915_ENGINE_HAS_RELATIVE_MMIO;
1.1  riastrad
1.1  riastrad 	if (intel_engine_has_preemption(engine))
1.1  riastrad 		engine->emit_bb_start = gen8_emit_bb_start;
1.1  riastrad 	else
1.1  riastrad 		engine->emit_bb_start = gen8_emit_bb_start_noarb;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_shutdown(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	/* Synchronise with residual timers and any softirq they raise */
1.1  riastrad 	del_timer_sync(&engine->execlists.timer);
1.1  riastrad 	del_timer_sync(&engine->execlists.preempt);
1.1  riastrad 	tasklet_kill(&engine->execlists.tasklet);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_release(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	execlists_shutdown(engine);
1.1  riastrad
1.1  riastrad 	intel_engine_cleanup_common(engine);
1.1  riastrad 	lrc_destroy_wa_ctx(engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void
1.1  riastrad logical_ring_default_vfuncs(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	/* Default vfuncs which can be overriden by each engine. */
1.1  riastrad
1.1  riastrad 	engine->resume = execlists_resume;
1.1  riastrad
1.1  riastrad 	engine->cops = &execlists_context_ops;
1.1  riastrad 	engine->request_alloc = execlists_request_alloc;
1.1  riastrad
1.1  riastrad 	engine->emit_flush = gen8_emit_flush;
1.1  riastrad 	engine->emit_init_breadcrumb = gen8_emit_init_breadcrumb;
1.1  riastrad 	engine->emit_fini_breadcrumb = gen8_emit_fini_breadcrumb;
1.1  riastrad 	if (INTEL_GEN(engine->i915) >= 12)
1.1  riastrad 		engine->emit_fini_breadcrumb = gen12_emit_fini_breadcrumb;
1.1  riastrad
1.1  riastrad 	engine->set_default_submission = intel_execlists_set_default_submission;
1.1  riastrad
1.1  riastrad 	if (INTEL_GEN(engine->i915) < 11) {
1.1  riastrad 		engine->irq_enable = gen8_logical_ring_enable_irq;
1.1  riastrad 		engine->irq_disable = gen8_logical_ring_disable_irq;
1.1  riastrad 	} else {
1.1  riastrad 		/*
1.1  riastrad 		 * TODO: On Gen11 interrupt masks need to be clear
1.1  riastrad 		 * to allow C6 entry. Keep interrupts enabled at
1.1  riastrad 		 * and take the hit of generating extra interrupts
1.1  riastrad 		 * until a more refined solution exists.
1.1  riastrad 		 */
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static inline void
1.1  riastrad logical_ring_default_irqs(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	unsigned int shift = 0;
1.1  riastrad
1.1  riastrad 	if (INTEL_GEN(engine->i915) < 11) {
1.1  riastrad 		const u8 irq_shifts[] = {
1.1  riastrad 			[RCS0]  = GEN8_RCS_IRQ_SHIFT,
1.1  riastrad 			[BCS0]  = GEN8_BCS_IRQ_SHIFT,
1.1  riastrad 			[VCS0]  = GEN8_VCS0_IRQ_SHIFT,
1.1  riastrad 			[VCS1]  = GEN8_VCS1_IRQ_SHIFT,
1.1  riastrad 			[VECS0] = GEN8_VECS_IRQ_SHIFT,
1.1  riastrad 		};
1.1  riastrad
1.1  riastrad 		shift = irq_shifts[engine->id];
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	engine->irq_enable_mask = GT_RENDER_USER_INTERRUPT << shift;
1.1  riastrad 	engine->irq_keep_mask = GT_CONTEXT_SWITCH_INTERRUPT << shift;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void rcs_submission_override(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	switch (INTEL_GEN(engine->i915)) {
1.1  riastrad 	case 12:
1.1  riastrad 		engine->emit_flush = gen12_emit_flush_render;
1.1  riastrad 		engine->emit_fini_breadcrumb = gen12_emit_fini_breadcrumb_rcs;
1.1  riastrad 		break;
1.1  riastrad 	case 11:
1.1  riastrad 		engine->emit_flush = gen11_emit_flush_render;
1.1  riastrad 		engine->emit_fini_breadcrumb = gen11_emit_fini_breadcrumb_rcs;
1.1  riastrad 		break;
1.1  riastrad 	default:
1.1  riastrad 		engine->emit_flush = gen8_emit_flush_render;
1.1  riastrad 		engine->emit_fini_breadcrumb = gen8_emit_fini_breadcrumb_rcs;
1.1  riastrad 		break;
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad int intel_execlists_submission_setup(struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct intel_engine_execlists * const execlists = &engine->execlists;
1.1  riastrad 	struct drm_i915_private *i915 = engine->i915;
1.1  riastrad 	struct intel_uncore *uncore = engine->uncore;
1.1  riastrad 	u32 base = engine->mmio_base;
1.1  riastrad
1.3  riastrad 	i915_sched_init(&engine->execlists);
1.3  riastrad
1.1  riastrad 	tasklet_init(&engine->execlists.tasklet,
1.1  riastrad 		     execlists_submission_tasklet, (unsigned long)engine);
1.1  riastrad 	timer_setup(&engine->execlists.timer, execlists_timeslice, 0);
1.1  riastrad 	timer_setup(&engine->execlists.preempt, execlists_preempt, 0);
1.1  riastrad
1.1  riastrad 	logical_ring_default_vfuncs(engine);
1.1  riastrad 	logical_ring_default_irqs(engine);
1.1  riastrad
1.1  riastrad 	if (engine->class == RENDER_CLASS)
1.1  riastrad 		rcs_submission_override(engine);
1.1  riastrad
1.1  riastrad 	if (intel_init_workaround_bb(engine))
1.1  riastrad 		/*
1.1  riastrad 		 * We continue even if we fail to initialize WA batch
1.1  riastrad 		 * because we only expect rare glitches but nothing
1.1  riastrad 		 * critical to prevent us from using GPU
1.1  riastrad 		 */
1.1  riastrad 		DRM_ERROR("WA batch buffer initialization failed\n");
1.1  riastrad
1.1  riastrad 	if (HAS_LOGICAL_RING_ELSQ(i915)) {
1.4  riastrad #ifdef __NetBSD__
1.4  riastrad 		execlists->submit_reg = i915_mmio_reg_offset(RING_EXECLIST_SQ_CONTENTS(base));
1.4  riastrad 		execlists->ctrl_reg = i915_mmio_reg_offset(RING_EXECLIST_CONTROL(base));
1.4  riastrad 		execlists->bsh = uncore->regs_bsh;
1.4  riastrad 		execlists->bst = uncore->regs_bst;
1.4  riastrad #else
1.1  riastrad 		execlists->submit_reg = uncore->regs +
1.1  riastrad 			i915_mmio_reg_offset(RING_EXECLIST_SQ_CONTENTS(base));
1.1  riastrad 		execlists->ctrl_reg = uncore->regs +
1.1  riastrad 			i915_mmio_reg_offset(RING_EXECLIST_CONTROL(base));
1.4  riastrad #endif
1.1  riastrad 	} else {
1.4  riastrad #ifdef __NetBSD__
1.4  riastrad 		execlists->submit_reg = i915_mmio_reg_offset(RING_ELSP(base));
1.4  riastrad 		execlists->bsh = uncore->regs_bsh;
1.4  riastrad 		execlists->bst = uncore->regs_bst;
1.4  riastrad #else
1.1  riastrad 		execlists->submit_reg = uncore->regs +
1.1  riastrad 			i915_mmio_reg_offset(RING_ELSP(base));
1.4  riastrad #endif
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	execlists->csb_status =
1.1  riastrad 		&engine->status_page.addr[I915_HWS_CSB_BUF0_INDEX];
1.1  riastrad
1.1  riastrad 	execlists->csb_write =
1.1  riastrad 		&engine->status_page.addr[intel_hws_csb_write_index(i915)];
1.1  riastrad
1.1  riastrad 	if (INTEL_GEN(i915) < 11)
1.1  riastrad 		execlists->csb_size = GEN8_CSB_ENTRIES;
1.1  riastrad 	else
1.1  riastrad 		execlists->csb_size = GEN11_CSB_ENTRIES;
1.1  riastrad
1.1  riastrad 	reset_csb_pointers(engine);
1.1  riastrad
1.1  riastrad 	/* Finally, take ownership and responsibility for cleanup! */
1.1  riastrad 	engine->release = execlists_release;
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static u32 intel_lr_indirect_ctx_offset(const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	u32 indirect_ctx_offset;
1.1  riastrad
1.1  riastrad 	switch (INTEL_GEN(engine->i915)) {
1.1  riastrad 	default:
1.1  riastrad 		MISSING_CASE(INTEL_GEN(engine->i915));
1.1  riastrad 		/* fall through */
1.1  riastrad 	case 12:
1.1  riastrad 		indirect_ctx_offset =
1.1  riastrad 			GEN12_CTX_RCS_INDIRECT_CTX_OFFSET_DEFAULT;
1.1  riastrad 		break;
1.1  riastrad 	case 11:
1.1  riastrad 		indirect_ctx_offset =
1.1  riastrad 			GEN11_CTX_RCS_INDIRECT_CTX_OFFSET_DEFAULT;
1.1  riastrad 		break;
1.1  riastrad 	case 10:
1.1  riastrad 		indirect_ctx_offset =
1.1  riastrad 			GEN10_CTX_RCS_INDIRECT_CTX_OFFSET_DEFAULT;
1.1  riastrad 		break;
1.1  riastrad 	case 9:
1.1  riastrad 		indirect_ctx_offset =
1.1  riastrad 			GEN9_CTX_RCS_INDIRECT_CTX_OFFSET_DEFAULT;
1.1  riastrad 		break;
1.1  riastrad 	case 8:
1.1  riastrad 		indirect_ctx_offset =
1.1  riastrad 			GEN8_CTX_RCS_INDIRECT_CTX_OFFSET_DEFAULT;
1.1  riastrad 		break;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return indirect_ctx_offset;
1.1  riastrad }
1.1  riastrad
1.1  riastrad
1.1  riastrad static void init_common_reg_state(u32 * const regs,
1.1  riastrad 				  const struct intel_engine_cs *engine,
1.1  riastrad 				  const struct intel_ring *ring,
1.1  riastrad 				  bool inhibit)
1.1  riastrad {
1.1  riastrad 	u32 ctl;
1.1  riastrad
1.1  riastrad 	ctl = _MASKED_BIT_ENABLE(CTX_CTRL_INHIBIT_SYN_CTX_SWITCH);
1.1  riastrad 	ctl |= _MASKED_BIT_DISABLE(CTX_CTRL_ENGINE_CTX_RESTORE_INHIBIT);
1.1  riastrad 	if (inhibit)
1.1  riastrad 		ctl |= CTX_CTRL_ENGINE_CTX_RESTORE_INHIBIT;
1.1  riastrad 	if (INTEL_GEN(engine->i915) < 11)
1.1  riastrad 		ctl |= _MASKED_BIT_DISABLE(CTX_CTRL_ENGINE_CTX_SAVE_INHIBIT |
1.1  riastrad 					   CTX_CTRL_RS_CTX_ENABLE);
1.1  riastrad 	regs[CTX_CONTEXT_CONTROL] = ctl;
1.1  riastrad
1.1  riastrad 	regs[CTX_RING_CTL] = RING_CTL_SIZE(ring->size) | RING_VALID;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void init_wa_bb_reg_state(u32 * const regs,
1.1  riastrad 				 const struct intel_engine_cs *engine,
1.1  riastrad 				 u32 pos_bb_per_ctx)
1.1  riastrad {
1.1  riastrad 	const struct i915_ctx_workarounds * const wa_ctx = &engine->wa_ctx;
1.1  riastrad
1.1  riastrad 	if (wa_ctx->per_ctx.size) {
1.1  riastrad 		const u32 ggtt_offset = i915_ggtt_offset(wa_ctx->vma);
1.1  riastrad
1.1  riastrad 		regs[pos_bb_per_ctx] =
1.1  riastrad 			(ggtt_offset + wa_ctx->per_ctx.offset) | 0x01;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (wa_ctx->indirect_ctx.size) {
1.1  riastrad 		const u32 ggtt_offset = i915_ggtt_offset(wa_ctx->vma);
1.1  riastrad
1.1  riastrad 		regs[pos_bb_per_ctx + 2] =
1.1  riastrad 			(ggtt_offset + wa_ctx->indirect_ctx.offset) |
1.1  riastrad 			(wa_ctx->indirect_ctx.size / CACHELINE_BYTES);
1.1  riastrad
1.1  riastrad 		regs[pos_bb_per_ctx + 4] =
1.1  riastrad 			intel_lr_indirect_ctx_offset(engine) << 6;
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void init_ppgtt_reg_state(u32 *regs, const struct i915_ppgtt *ppgtt)
1.1  riastrad {
1.1  riastrad 	if (i915_vm_is_4lvl(&ppgtt->vm)) {
1.1  riastrad 		/* 64b PPGTT (48bit canonical)
1.1  riastrad 		 * PDP0_DESCRIPTOR contains the base address to PML4 and
1.1  riastrad 		 * other PDP Descriptors are ignored.
1.1  riastrad 		 */
1.1  riastrad 		ASSIGN_CTX_PML4(ppgtt, regs);
1.1  riastrad 	} else {
1.1  riastrad 		ASSIGN_CTX_PDP(ppgtt, regs, 3);
1.1  riastrad 		ASSIGN_CTX_PDP(ppgtt, regs, 2);
1.1  riastrad 		ASSIGN_CTX_PDP(ppgtt, regs, 1);
1.1  riastrad 		ASSIGN_CTX_PDP(ppgtt, regs, 0);
1.1  riastrad 	}
1.1  riastrad }
1.1  riastrad
1.1  riastrad static struct i915_ppgtt *vm_alias(struct i915_address_space *vm)
1.1  riastrad {
1.1  riastrad 	if (i915_is_ggtt(vm))
1.1  riastrad 		return i915_vm_to_ggtt(vm)->alias;
1.1  riastrad 	else
1.1  riastrad 		return i915_vm_to_ppgtt(vm);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void execlists_init_reg_state(u32 *regs,
1.1  riastrad 				     const struct intel_context *ce,
1.1  riastrad 				     const struct intel_engine_cs *engine,
1.1  riastrad 				     const struct intel_ring *ring,
1.1  riastrad 				     bool inhibit)
1.1  riastrad {
1.1  riastrad 	/*
1.1  riastrad 	 * A context is actually a big batch buffer with several
1.1  riastrad 	 * MI_LOAD_REGISTER_IMM commands followed by (reg, value) pairs. The
1.1  riastrad 	 * values we are setting here are only for the first context restore:
1.1  riastrad 	 * on a subsequent save, the GPU will recreate this batchbuffer with new
1.1  riastrad 	 * values (including all the missing MI_LOAD_REGISTER_IMM commands that
1.1  riastrad 	 * we are not initializing here).
1.1  riastrad 	 *
1.1  riastrad 	 * Must keep consistent with virtual_update_register_offsets().
1.1  riastrad 	 */
1.1  riastrad 	set_offsets(regs, reg_offsets(engine), engine, inhibit);
1.1  riastrad
1.1  riastrad 	init_common_reg_state(regs, engine, ring, inhibit);
1.1  riastrad 	init_ppgtt_reg_state(regs, vm_alias(ce->vm));
1.1  riastrad
1.1  riastrad 	init_wa_bb_reg_state(regs, engine,
1.1  riastrad 			     INTEL_GEN(engine->i915) >= 12 ?
1.1  riastrad 			     GEN12_CTX_BB_PER_CTX_PTR :
1.1  riastrad 			     CTX_BB_PER_CTX_PTR);
1.1  riastrad
1.1  riastrad 	__reset_stop_ring(regs, engine);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int
1.1  riastrad populate_lr_context(struct intel_context *ce,
1.1  riastrad 		    struct drm_i915_gem_object *ctx_obj,
1.1  riastrad 		    struct intel_engine_cs *engine,
1.1  riastrad 		    struct intel_ring *ring)
1.1  riastrad {
1.1  riastrad 	bool inhibit = true;
1.1  riastrad 	void *vaddr;
1.1  riastrad 	int ret;
1.1  riastrad
1.1  riastrad 	vaddr = i915_gem_object_pin_map(ctx_obj, I915_MAP_WB);
1.1  riastrad 	if (IS_ERR(vaddr)) {
1.1  riastrad 		ret = PTR_ERR(vaddr);
1.1  riastrad 		DRM_DEBUG_DRIVER("Could not map object pages! (%d)\n", ret);
1.1  riastrad 		return ret;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	set_redzone(vaddr, engine);
1.1  riastrad
1.1  riastrad 	if (engine->default_state) {
1.1  riastrad 		void *defaults;
1.1  riastrad
1.1  riastrad 		defaults = i915_gem_object_pin_map(engine->default_state,
1.1  riastrad 						   I915_MAP_WB);
1.1  riastrad 		if (IS_ERR(defaults)) {
1.1  riastrad 			ret = PTR_ERR(defaults);
1.1  riastrad 			goto err_unpin_ctx;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		memcpy(vaddr, defaults, engine->context_size);
1.1  riastrad 		i915_gem_object_unpin_map(engine->default_state);
1.1  riastrad 		__set_bit(CONTEXT_VALID_BIT, &ce->flags);
1.1  riastrad 		inhibit = false;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	/* The second page of the context object contains some fields which must
1.1  riastrad 	 * be set up prior to the first execution. */
1.1  riastrad 	execlists_init_reg_state(vaddr + LRC_STATE_PN * PAGE_SIZE,
1.1  riastrad 				 ce, engine, ring, inhibit);
1.1  riastrad
1.1  riastrad 	ret = 0;
1.1  riastrad err_unpin_ctx:
1.1  riastrad 	__i915_gem_object_flush_map(ctx_obj, 0, engine->context_size);
1.1  riastrad 	i915_gem_object_unpin_map(ctx_obj);
1.1  riastrad 	return ret;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int __execlists_context_alloc(struct intel_context *ce,
1.1  riastrad 				     struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	struct drm_i915_gem_object *ctx_obj;
1.1  riastrad 	struct intel_ring *ring;
1.1  riastrad 	struct i915_vma *vma;
1.1  riastrad 	u32 context_size;
1.1  riastrad 	int ret;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(ce->state);
1.1  riastrad 	context_size = round_up(engine->context_size, I915_GTT_PAGE_SIZE);
1.1  riastrad
1.1  riastrad 	if (IS_ENABLED(CONFIG_DRM_I915_DEBUG_GEM))
1.1  riastrad 		context_size += I915_GTT_PAGE_SIZE; /* for redzone */
1.1  riastrad
1.1  riastrad 	ctx_obj = i915_gem_object_create_shmem(engine->i915, context_size);
1.1  riastrad 	if (IS_ERR(ctx_obj))
1.1  riastrad 		return PTR_ERR(ctx_obj);
1.1  riastrad
1.1  riastrad 	vma = i915_vma_instance(ctx_obj, &engine->gt->ggtt->vm, NULL);
1.1  riastrad 	if (IS_ERR(vma)) {
1.1  riastrad 		ret = PTR_ERR(vma);
1.1  riastrad 		goto error_deref_obj;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (!ce->timeline) {
1.1  riastrad 		struct intel_timeline *tl;
1.1  riastrad
1.1  riastrad 		tl = intel_timeline_create(engine->gt, NULL);
1.1  riastrad 		if (IS_ERR(tl)) {
1.1  riastrad 			ret = PTR_ERR(tl);
1.1  riastrad 			goto error_deref_obj;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		ce->timeline = tl;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	ring = intel_engine_create_ring(engine, (unsigned long)ce->ring);
1.1  riastrad 	if (IS_ERR(ring)) {
1.1  riastrad 		ret = PTR_ERR(ring);
1.1  riastrad 		goto error_deref_obj;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	ret = populate_lr_context(ce, ctx_obj, engine, ring);
1.1  riastrad 	if (ret) {
1.1  riastrad 		DRM_DEBUG_DRIVER("Failed to populate LRC: %d\n", ret);
1.1  riastrad 		goto error_ring_free;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	ce->ring = ring;
1.1  riastrad 	ce->state = vma;
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad
1.1  riastrad error_ring_free:
1.1  riastrad 	intel_ring_put(ring);
1.1  riastrad error_deref_obj:
1.1  riastrad 	i915_gem_object_put(ctx_obj);
1.1  riastrad 	return ret;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static struct list_head *virtual_queue(struct virtual_engine *ve)
1.1  riastrad {
1.1  riastrad 	return &ve->base.execlists.default_priolist.requests[0];
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_context_destroy(struct kref *kref)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve =
1.1  riastrad 		container_of(kref, typeof(*ve), context.ref);
1.1  riastrad 	unsigned int n;
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(!list_empty(virtual_queue(ve)));
1.1  riastrad 	GEM_BUG_ON(ve->request);
1.1  riastrad 	GEM_BUG_ON(ve->context.inflight);
1.1  riastrad
1.1  riastrad 	for (n = 0; n < ve->num_siblings; n++) {
1.1  riastrad 		struct intel_engine_cs *sibling = ve->siblings[n];
1.1  riastrad 		struct rb_node *node = &ve->nodes[sibling->id].rb;
1.1  riastrad 		unsigned long flags;
1.1  riastrad
1.7  riastrad 		if (!ve->nodes[sibling->id].inserted)
1.1  riastrad 			continue;
1.1  riastrad
1.1  riastrad 		spin_lock_irqsave(&sibling->active.lock, flags);
1.1  riastrad
1.1  riastrad 		/* Detachment is lazily performed in the execlists tasklet */
1.7  riastrad 		if (ve->nodes[sibling->id].inserted) {
1.1  riastrad 			rb_erase_cached(node, &sibling->execlists.virtual);
1.7  riastrad 			ve->nodes[sibling->id].inserted = false;
1.7  riastrad 		}
1.1  riastrad
1.1  riastrad 		spin_unlock_irqrestore(&sibling->active.lock, flags);
1.1  riastrad 	}
1.1  riastrad 	GEM_BUG_ON(__tasklet_is_scheduled(&ve->base.execlists.tasklet));
1.1  riastrad
1.1  riastrad 	if (ve->context.state)
1.1  riastrad 		__execlists_context_fini(&ve->context);
1.1  riastrad 	intel_context_fini(&ve->context);
1.1  riastrad
1.8  riastrad 	intel_engine_fini_breadcrumbs(&ve->base);
1.8  riastrad 	spin_lock_destroy(&ve->base.active.lock);
1.8  riastrad
1.1  riastrad 	kfree(ve->bonds);
1.1  riastrad 	kfree(ve);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_engine_initial_hint(struct virtual_engine *ve)
1.1  riastrad {
1.1  riastrad 	int swp;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * Pick a random sibling on starting to help spread the load around.
1.1  riastrad 	 *
1.1  riastrad 	 * New contexts are typically created with exactly the same order
1.1  riastrad 	 * of siblings, and often started in batches. Due to the way we iterate
1.1  riastrad 	 * the array of sibling when submitting requests, sibling[0] is
1.1  riastrad 	 * prioritised for dequeuing. If we make sure that sibling[0] is fairly
1.1  riastrad 	 * randomised across the system, we also help spread the load by the
1.1  riastrad 	 * first engine we inspect being different each time.
1.1  riastrad 	 *
1.1  riastrad 	 * NB This does not force us to execute on this engine, it will just
1.1  riastrad 	 * typically be the first we inspect for submission.
1.1  riastrad 	 */
1.1  riastrad 	swp = prandom_u32_max(ve->num_siblings);
1.1  riastrad 	if (!swp)
1.1  riastrad 		return;
1.1  riastrad
1.1  riastrad 	swap(ve->siblings[swp], ve->siblings[0]);
1.1  riastrad 	if (!intel_engine_has_relative_mmio(ve->siblings[0]))
1.1  riastrad 		virtual_update_register_offsets(ve->context.lrc_reg_state,
1.1  riastrad 						ve->siblings[0]);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int virtual_context_alloc(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = container_of(ce, typeof(*ve), context);
1.1  riastrad
1.1  riastrad 	return __execlists_context_alloc(ce, ve->siblings[0]);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static int virtual_context_pin(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = container_of(ce, typeof(*ve), context);
1.1  riastrad 	int err;
1.1  riastrad
1.1  riastrad 	/* Note: we must use a real engine class for setting up reg state */
1.1  riastrad 	err = __execlists_context_pin(ce, ve->siblings[0]);
1.1  riastrad 	if (err)
1.1  riastrad 		return err;
1.1  riastrad
1.1  riastrad 	virtual_engine_initial_hint(ve);
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_context_enter(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = container_of(ce, typeof(*ve), context);
1.1  riastrad 	unsigned int n;
1.1  riastrad
1.1  riastrad 	for (n = 0; n < ve->num_siblings; n++)
1.1  riastrad 		intel_engine_pm_get(ve->siblings[n]);
1.1  riastrad
1.1  riastrad 	intel_timeline_enter(ce->timeline);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_context_exit(struct intel_context *ce)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = container_of(ce, typeof(*ve), context);
1.1  riastrad 	unsigned int n;
1.1  riastrad
1.1  riastrad 	intel_timeline_exit(ce->timeline);
1.1  riastrad
1.1  riastrad 	for (n = 0; n < ve->num_siblings; n++)
1.1  riastrad 		intel_engine_pm_put(ve->siblings[n]);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static const struct intel_context_ops virtual_context_ops = {
1.1  riastrad 	.alloc = virtual_context_alloc,
1.1  riastrad
1.1  riastrad 	.pin = virtual_context_pin,
1.1  riastrad 	.unpin = execlists_context_unpin,
1.1  riastrad
1.1  riastrad 	.enter = virtual_context_enter,
1.1  riastrad 	.exit = virtual_context_exit,
1.1  riastrad
1.1  riastrad 	.destroy = virtual_context_destroy,
1.1  riastrad };
1.1  riastrad
1.1  riastrad static intel_engine_mask_t virtual_submission_mask(struct virtual_engine *ve)
1.1  riastrad {
1.1  riastrad 	struct i915_request *rq;
1.1  riastrad 	intel_engine_mask_t mask;
1.1  riastrad
1.1  riastrad 	rq = READ_ONCE(ve->request);
1.1  riastrad 	if (!rq)
1.1  riastrad 		return 0;
1.1  riastrad
1.1  riastrad 	/* The rq is ready for submission; rq->execution_mask is now stable. */
1.1  riastrad 	mask = rq->execution_mask;
1.1  riastrad 	if (unlikely(!mask)) {
1.1  riastrad 		/* Invalid selection, submit to a random engine in error */
1.1  riastrad 		i915_request_skip(rq, -ENODEV);
1.1  riastrad 		mask = ve->siblings[0]->mask;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	ENGINE_TRACE(&ve->base, "rq=%llx:%lld, mask=%x, prio=%d\n",
1.1  riastrad 		     rq->fence.context, rq->fence.seqno,
1.1  riastrad 		     mask, ve->base.execlists.queue_priority_hint);
1.1  riastrad
1.1  riastrad 	return mask;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_submission_tasklet(unsigned long data)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine * const ve = (struct virtual_engine *)data;
1.1  riastrad 	const int prio = ve->base.execlists.queue_priority_hint;
1.1  riastrad 	intel_engine_mask_t mask;
1.1  riastrad 	unsigned int n;
1.1  riastrad
1.1  riastrad 	rcu_read_lock();
1.1  riastrad 	mask = virtual_submission_mask(ve);
1.1  riastrad 	rcu_read_unlock();
1.1  riastrad 	if (unlikely(!mask))
1.1  riastrad 		return;
1.1  riastrad
1.7  riastrad #ifdef __NetBSD__
1.7  riastrad 	int s = splsoftserial(); /* block tasklets=softints */
1.7  riastrad #else
1.1  riastrad 	local_irq_disable();
1.7  riastrad #endif
1.1  riastrad 	for (n = 0; READ_ONCE(ve->request) && n < ve->num_siblings; n++) {
1.1  riastrad 		struct intel_engine_cs *sibling = ve->siblings[n];
1.1  riastrad 		struct ve_node * const node = &ve->nodes[sibling->id];
1.1  riastrad 		struct rb_node **parent, *rb;
1.1  riastrad 		bool first;
1.1  riastrad
1.1  riastrad 		if (unlikely(!(mask & sibling->mask))) {
1.7  riastrad 			if (node->inserted) {
1.1  riastrad 				spin_lock(&sibling->active.lock);
1.1  riastrad 				rb_erase_cached(&node->rb,
1.1  riastrad 						&sibling->execlists.virtual);
1.7  riastrad 				node->inserted = false;
1.1  riastrad 				spin_unlock(&sibling->active.lock);
1.1  riastrad 			}
1.1  riastrad 			continue;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		spin_lock(&sibling->active.lock);
1.1  riastrad
1.7  riastrad 		if (node->inserted) {
1.1  riastrad 			/*
1.1  riastrad 			 * Cheat and avoid rebalancing the tree if we can
1.1  riastrad 			 * reuse this node in situ.
1.1  riastrad 			 */
1.1  riastrad 			first = rb_first_cached(&sibling->execlists.virtual) ==
1.1  riastrad 				&node->rb;
1.1  riastrad 			if (prio == node->prio || (prio > node->prio && first))
1.1  riastrad 				goto submit_engine;
1.1  riastrad
1.1  riastrad 			rb_erase_cached(&node->rb, &sibling->execlists.virtual);
1.7  riastrad 			node->inserted = false;
1.1  riastrad 		}
1.1  riastrad
1.7  riastrad #ifdef __NetBSD__
1.7  riastrad 		__USE(parent);
1.7  riastrad 		__USE(rb);
1.7  riastrad 		struct ve_node *collision __diagused;
1.7  riastrad 		/* XXX kludge to get insertion order */
1.7  riastrad 		node->order = ve->order++;
1.7  riastrad 		collision = rb_tree_insert_node(
1.7  riastrad 			&sibling->execlists.virtual.rb_root.rbr_tree,
1.7  riastrad 			node);
1.7  riastrad 		KASSERT(collision == node);
1.7  riastrad 		node->inserted = true;
1.7  riastrad 		first = rb_tree_find_node_geq(
1.7  riastrad 			&sibling->execlists.virtual.rb_root.rbr_tree,
1.7  riastrad 			&node->prio) == node;
1.7  riastrad #else
1.1  riastrad 		rb = NULL;
1.1  riastrad 		first = true;
1.1  riastrad 		parent = &sibling->execlists.virtual.rb_root.rb_node;
1.1  riastrad 		while (*parent) {
1.1  riastrad 			struct ve_node *other;
1.1  riastrad
1.1  riastrad 			rb = *parent;
1.1  riastrad 			other = rb_entry(rb, typeof(*other), rb);
1.1  riastrad 			if (prio > other->prio) {
1.1  riastrad 				parent = &rb->rb_left;
1.1  riastrad 			} else {
1.1  riastrad 				parent = &rb->rb_right;
1.1  riastrad 				first = false;
1.1  riastrad 			}
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		rb_link_node(&node->rb, rb, parent);
1.1  riastrad 		rb_insert_color_cached(&node->rb,
1.1  riastrad 				       &sibling->execlists.virtual,
1.1  riastrad 				       first);
1.7  riastrad #endif
1.1  riastrad
1.1  riastrad submit_engine:
1.7  riastrad 		GEM_BUG_ON(!node->inserted);
1.1  riastrad 		node->prio = prio;
1.1  riastrad 		if (first && prio > sibling->execlists.queue_priority_hint) {
1.1  riastrad 			sibling->execlists.queue_priority_hint = prio;
1.1  riastrad 			tasklet_hi_schedule(&sibling->execlists.tasklet);
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		spin_unlock(&sibling->active.lock);
1.1  riastrad 	}
1.7  riastrad #ifdef __NetBSD__
1.7  riastrad 	splx(s);
1.7  riastrad #else
1.1  riastrad 	local_irq_enable();
1.7  riastrad #endif
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void virtual_submit_request(struct i915_request *rq)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = to_virtual_engine(rq->engine);
1.1  riastrad 	struct i915_request *old;
1.1  riastrad 	unsigned long flags;
1.1  riastrad
1.1  riastrad 	ENGINE_TRACE(&ve->base, "rq=%llx:%lld\n",
1.1  riastrad 		     rq->fence.context,
1.1  riastrad 		     rq->fence.seqno);
1.1  riastrad
1.1  riastrad 	GEM_BUG_ON(ve->base.submit_request != virtual_submit_request);
1.1  riastrad
1.1  riastrad 	spin_lock_irqsave(&ve->base.active.lock, flags);
1.1  riastrad
1.1  riastrad 	old = ve->request;
1.1  riastrad 	if (old) { /* background completion event from preempt-to-busy */
1.1  riastrad 		GEM_BUG_ON(!i915_request_completed(old));
1.1  riastrad 		__i915_request_submit(old);
1.1  riastrad 		i915_request_put(old);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	if (i915_request_completed(rq)) {
1.1  riastrad 		__i915_request_submit(rq);
1.1  riastrad
1.1  riastrad 		ve->base.execlists.queue_priority_hint = INT_MIN;
1.1  riastrad 		ve->request = NULL;
1.1  riastrad 	} else {
1.1  riastrad 		ve->base.execlists.queue_priority_hint = rq_prio(rq);
1.1  riastrad 		ve->request = i915_request_get(rq);
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(!list_empty(virtual_queue(ve)));
1.1  riastrad 		list_move_tail(&rq->sched.link, virtual_queue(ve));
1.1  riastrad
1.1  riastrad 		tasklet_schedule(&ve->base.execlists.tasklet);
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	spin_unlock_irqrestore(&ve->base.active.lock, flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad static struct ve_bond *
1.1  riastrad virtual_find_bond(struct virtual_engine *ve,
1.1  riastrad 		  const struct intel_engine_cs *master)
1.1  riastrad {
1.1  riastrad 	int i;
1.1  riastrad
1.1  riastrad 	for (i = 0; i < ve->num_bonds; i++) {
1.1  riastrad 		if (ve->bonds[i].master == master)
1.1  riastrad 			return &ve->bonds[i];
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return NULL;
1.1  riastrad }
1.1  riastrad
1.1  riastrad static void
1.1  riastrad virtual_bond_execute(struct i915_request *rq, struct dma_fence *signal)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = to_virtual_engine(rq->engine);
1.1  riastrad 	intel_engine_mask_t allowed, exec;
1.1  riastrad 	struct ve_bond *bond;
1.1  riastrad
1.1  riastrad 	allowed = ~to_request(signal)->engine->mask;
1.1  riastrad
1.1  riastrad 	bond = virtual_find_bond(ve, to_request(signal)->engine);
1.1  riastrad 	if (bond)
1.1  riastrad 		allowed &= bond->sibling_mask;
1.1  riastrad
1.1  riastrad 	/* Restrict the bonded request to run on only the available engines */
1.1  riastrad 	exec = READ_ONCE(rq->execution_mask);
1.1  riastrad 	while (!try_cmpxchg(&rq->execution_mask, &exec, exec & allowed))
1.1  riastrad 		;
1.1  riastrad
1.1  riastrad 	/* Prevent the master from being re-run on the bonded engines */
1.1  riastrad 	to_request(signal)->execution_mask &= ~allowed;
1.1  riastrad }
1.1  riastrad
1.1  riastrad struct intel_context *
1.1  riastrad intel_execlists_create_virtual(struct intel_engine_cs **siblings,
1.1  riastrad 			       unsigned int count)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve;
1.1  riastrad 	unsigned int n;
1.1  riastrad 	int err;
1.1  riastrad
1.1  riastrad 	if (count == 0)
1.1  riastrad 		return ERR_PTR(-EINVAL);
1.1  riastrad
1.1  riastrad 	if (count == 1)
1.1  riastrad 		return intel_context_create(siblings[0]);
1.1  riastrad
1.1  riastrad 	ve = kzalloc(struct_size(ve, siblings, count), GFP_KERNEL);
1.1  riastrad 	if (!ve)
1.1  riastrad 		return ERR_PTR(-ENOMEM);
1.1  riastrad
1.1  riastrad 	ve->base.i915 = siblings[0]->i915;
1.1  riastrad 	ve->base.gt = siblings[0]->gt;
1.1  riastrad 	ve->base.uncore = siblings[0]->uncore;
1.1  riastrad 	ve->base.id = -1;
1.1  riastrad
1.1  riastrad 	ve->base.class = OTHER_CLASS;
1.1  riastrad 	ve->base.uabi_class = I915_ENGINE_CLASS_INVALID;
1.1  riastrad 	ve->base.instance = I915_ENGINE_CLASS_INVALID_VIRTUAL;
1.1  riastrad 	ve->base.uabi_instance = I915_ENGINE_CLASS_INVALID_VIRTUAL;
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * The decision on whether to submit a request using semaphores
1.1  riastrad 	 * depends on the saturated state of the engine. We only compute
1.1  riastrad 	 * this during HW submission of the request, and we need for this
1.1  riastrad 	 * state to be globally applied to all requests being submitted
1.1  riastrad 	 * to this engine. Virtual engines encompass more than one physical
1.1  riastrad 	 * engine and so we cannot accurately tell in advance if one of those
1.1  riastrad 	 * engines is already saturated and so cannot afford to use a semaphore
1.1  riastrad 	 * and be pessimized in priority for doing so -- if we are the only
1.1  riastrad 	 * context using semaphores after all other clients have stopped, we
1.1  riastrad 	 * will be starved on the saturated system. Such a global switch for
1.1  riastrad 	 * semaphores is less than ideal, but alas is the current compromise.
1.1  riastrad 	 */
1.1  riastrad 	ve->base.saturated = ALL_ENGINES;
1.1  riastrad
1.1  riastrad 	snprintf(ve->base.name, sizeof(ve->base.name), "virtual");
1.1  riastrad
1.1  riastrad 	intel_engine_init_active(&ve->base, ENGINE_VIRTUAL);
1.1  riastrad 	intel_engine_init_breadcrumbs(&ve->base);
1.1  riastrad 	intel_engine_init_execlists(&ve->base);
1.1  riastrad
1.1  riastrad 	ve->base.cops = &virtual_context_ops;
1.1  riastrad 	ve->base.request_alloc = execlists_request_alloc;
1.1  riastrad
1.1  riastrad 	ve->base.schedule = i915_schedule;
1.1  riastrad 	ve->base.submit_request = virtual_submit_request;
1.1  riastrad 	ve->base.bond_execute = virtual_bond_execute;
1.1  riastrad
1.1  riastrad 	INIT_LIST_HEAD(virtual_queue(ve));
1.1  riastrad 	ve->base.execlists.queue_priority_hint = INT_MIN;
1.1  riastrad 	tasklet_init(&ve->base.execlists.tasklet,
1.1  riastrad 		     virtual_submission_tasklet,
1.1  riastrad 		     (unsigned long)ve);
1.1  riastrad
1.1  riastrad 	intel_context_init(&ve->context, &ve->base);
1.1  riastrad
1.1  riastrad 	for (n = 0; n < count; n++) {
1.1  riastrad 		struct intel_engine_cs *sibling = siblings[n];
1.1  riastrad
1.1  riastrad 		GEM_BUG_ON(!is_power_of_2(sibling->mask));
1.1  riastrad 		if (sibling->mask & ve->base.mask) {
1.1  riastrad 			DRM_DEBUG("duplicate %s entry in load balancer\n",
1.1  riastrad 				  sibling->name);
1.1  riastrad 			err = -EINVAL;
1.1  riastrad 			goto err_put;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * The virtual engine implementation is tightly coupled to
1.1  riastrad 		 * the execlists backend -- we push out request directly
1.1  riastrad 		 * into a tree inside each physical engine. We could support
1.1  riastrad 		 * layering if we handle cloning of the requests and
1.1  riastrad 		 * submitting a copy into each backend.
1.1  riastrad 		 */
1.1  riastrad 		if (sibling->execlists.tasklet.func !=
1.1  riastrad 		    execlists_submission_tasklet) {
1.1  riastrad 			err = -ENODEV;
1.1  riastrad 			goto err_put;
1.1  riastrad 		}
1.1  riastrad
1.7  riastrad 		GEM_BUG_ON(!ve->nodes[sibling->id].inserted);
1.7  riastrad 		ve->nodes[sibling->id].inserted = false;
1.1  riastrad
1.1  riastrad 		ve->siblings[ve->num_siblings++] = sibling;
1.1  riastrad 		ve->base.mask |= sibling->mask;
1.1  riastrad
1.1  riastrad 		/*
1.1  riastrad 		 * All physical engines must be compatible for their emission
1.1  riastrad 		 * functions (as we build the instructions during request
1.1  riastrad 		 * construction and do not alter them before submission
1.1  riastrad 		 * on the physical engine). We use the engine class as a guide
1.1  riastrad 		 * here, although that could be refined.
1.1  riastrad 		 */
1.1  riastrad 		if (ve->base.class != OTHER_CLASS) {
1.1  riastrad 			if (ve->base.class != sibling->class) {
1.1  riastrad 				DRM_DEBUG("invalid mixing of engine class, sibling %d, already %d\n",
1.1  riastrad 					  sibling->class, ve->base.class);
1.1  riastrad 				err = -EINVAL;
1.1  riastrad 				goto err_put;
1.1  riastrad 			}
1.1  riastrad 			continue;
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		ve->base.class = sibling->class;
1.1  riastrad 		ve->base.uabi_class = sibling->uabi_class;
1.1  riastrad 		snprintf(ve->base.name, sizeof(ve->base.name),
1.1  riastrad 			 "v%dx%d", ve->base.class, count);
1.1  riastrad 		ve->base.context_size = sibling->context_size;
1.1  riastrad
1.1  riastrad 		ve->base.emit_bb_start = sibling->emit_bb_start;
1.1  riastrad 		ve->base.emit_flush = sibling->emit_flush;
1.1  riastrad 		ve->base.emit_init_breadcrumb = sibling->emit_init_breadcrumb;
1.1  riastrad 		ve->base.emit_fini_breadcrumb = sibling->emit_fini_breadcrumb;
1.1  riastrad 		ve->base.emit_fini_breadcrumb_dw =
1.1  riastrad 			sibling->emit_fini_breadcrumb_dw;
1.1  riastrad
1.1  riastrad 		ve->base.flags = sibling->flags;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	ve->base.flags |= I915_ENGINE_IS_VIRTUAL;
1.1  riastrad
1.1  riastrad 	return &ve->context;
1.1  riastrad
1.1  riastrad err_put:
1.1  riastrad 	intel_context_put(&ve->context);
1.1  riastrad 	return ERR_PTR(err);
1.1  riastrad }
1.1  riastrad
1.1  riastrad struct intel_context *
1.1  riastrad intel_execlists_clone_virtual(struct intel_engine_cs *src)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *se = to_virtual_engine(src);
1.1  riastrad 	struct intel_context *dst;
1.1  riastrad
1.1  riastrad 	dst = intel_execlists_create_virtual(se->siblings,
1.1  riastrad 					     se->num_siblings);
1.1  riastrad 	if (IS_ERR(dst))
1.1  riastrad 		return dst;
1.1  riastrad
1.1  riastrad 	if (se->num_bonds) {
1.1  riastrad 		struct virtual_engine *de = to_virtual_engine(dst->engine);
1.1  riastrad
1.1  riastrad 		de->bonds = kmemdup(se->bonds,
1.1  riastrad 				    sizeof(*se->bonds) * se->num_bonds,
1.1  riastrad 				    GFP_KERNEL);
1.1  riastrad 		if (!de->bonds) {
1.1  riastrad 			intel_context_put(dst);
1.1  riastrad 			return ERR_PTR(-ENOMEM);
1.1  riastrad 		}
1.1  riastrad
1.1  riastrad 		de->num_bonds = se->num_bonds;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	return dst;
1.1  riastrad }
1.1  riastrad
1.1  riastrad int intel_virtual_engine_attach_bond(struct intel_engine_cs *engine,
1.1  riastrad 				     const struct intel_engine_cs *master,
1.1  riastrad 				     const struct intel_engine_cs *sibling)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = to_virtual_engine(engine);
1.1  riastrad 	struct ve_bond *bond;
1.1  riastrad 	int n;
1.1  riastrad
1.1  riastrad 	/* Sanity check the sibling is part of the virtual engine */
1.1  riastrad 	for (n = 0; n < ve->num_siblings; n++)
1.1  riastrad 		if (sibling == ve->siblings[n])
1.1  riastrad 			break;
1.1  riastrad 	if (n == ve->num_siblings)
1.1  riastrad 		return -EINVAL;
1.1  riastrad
1.1  riastrad 	bond = virtual_find_bond(ve, master);
1.1  riastrad 	if (bond) {
1.1  riastrad 		bond->sibling_mask |= sibling->mask;
1.1  riastrad 		return 0;
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	bond = krealloc(ve->bonds,
1.1  riastrad 			sizeof(*bond) * (ve->num_bonds + 1),
1.1  riastrad 			GFP_KERNEL);
1.1  riastrad 	if (!bond)
1.1  riastrad 		return -ENOMEM;
1.1  riastrad
1.1  riastrad 	bond[ve->num_bonds].master = master;
1.1  riastrad 	bond[ve->num_bonds].sibling_mask = sibling->mask;
1.1  riastrad
1.1  riastrad 	ve->bonds = bond;
1.1  riastrad 	ve->num_bonds++;
1.1  riastrad
1.1  riastrad 	return 0;
1.1  riastrad }
1.1  riastrad
1.1  riastrad struct intel_engine_cs *
1.1  riastrad intel_virtual_engine_get_sibling(struct intel_engine_cs *engine,
1.1  riastrad 				 unsigned int sibling)
1.1  riastrad {
1.1  riastrad 	struct virtual_engine *ve = to_virtual_engine(engine);
1.1  riastrad
1.1  riastrad 	if (sibling >= ve->num_siblings)
1.1  riastrad 		return NULL;
1.1  riastrad
1.1  riastrad 	return ve->siblings[sibling];
1.1  riastrad }
1.1  riastrad
1.1  riastrad void intel_execlists_show_requests(struct intel_engine_cs *engine,
1.1  riastrad 				   struct drm_printer *m,
1.1  riastrad 				   void (*show_request)(struct drm_printer *m,
1.1  riastrad 							struct i915_request *rq,
1.1  riastrad 							const char *prefix),
1.1  riastrad 				   unsigned int max)
1.1  riastrad {
1.1  riastrad 	const struct intel_engine_execlists *execlists = &engine->execlists;
1.1  riastrad 	struct i915_request *rq, *last;
1.1  riastrad 	unsigned long flags;
1.1  riastrad 	unsigned int count;
1.1  riastrad 	struct rb_node *rb;
1.1  riastrad
1.1  riastrad 	spin_lock_irqsave(&engine->active.lock, flags);
1.1  riastrad
1.1  riastrad 	last = NULL;
1.1  riastrad 	count = 0;
1.1  riastrad 	list_for_each_entry(rq, &engine->active.requests, sched.link) {
1.1  riastrad 		if (count++ < max - 1)
1.1  riastrad 			show_request(m, rq, "\t\tE ");
1.1  riastrad 		else
1.1  riastrad 			last = rq;
1.1  riastrad 	}
1.1  riastrad 	if (last) {
1.1  riastrad 		if (count > max) {
1.1  riastrad 			drm_printf(m,
1.1  riastrad 				   "\t\t...skipping %d executing requests...\n",
1.1  riastrad 				   count - max);
1.1  riastrad 		}
1.1  riastrad 		show_request(m, last, "\t\tE ");
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	last = NULL;
1.1  riastrad 	count = 0;
1.1  riastrad 	if (execlists->queue_priority_hint != INT_MIN)
1.1  riastrad 		drm_printf(m, "\t\tQueue priority hint: %d\n",
1.1  riastrad 			   execlists->queue_priority_hint);
1.7  riastrad 	for (rb = rb_first_cached(&execlists->queue);
1.7  riastrad 	     rb;
1.7  riastrad 	     rb = rb_next2(&execlists->queue.rb_root, rb)) {
1.1  riastrad 		struct i915_priolist *p = rb_entry(rb, typeof(*p), node);
1.1  riastrad 		int i;
1.1  riastrad
1.1  riastrad 		priolist_for_each_request(rq, p, i) {
1.1  riastrad 			if (count++ < max - 1)
1.1  riastrad 				show_request(m, rq, "\t\tQ ");
1.1  riastrad 			else
1.1  riastrad 				last = rq;
1.1  riastrad 		}
1.1  riastrad 	}
1.1  riastrad 	if (last) {
1.1  riastrad 		if (count > max) {
1.1  riastrad 			drm_printf(m,
1.1  riastrad 				   "\t\t...skipping %d queued requests...\n",
1.1  riastrad 				   count - max);
1.1  riastrad 		}
1.1  riastrad 		show_request(m, last, "\t\tQ ");
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	last = NULL;
1.1  riastrad 	count = 0;
1.7  riastrad 	for (rb = rb_first_cached(&execlists->virtual);
1.7  riastrad 	     rb;
1.7  riastrad 	     rb = rb_next2(&execlists->virtual.rb_root, rb)) {
1.1  riastrad 		struct virtual_engine *ve =
1.1  riastrad 			rb_entry(rb, typeof(*ve), nodes[engine->id].rb);
1.1  riastrad 		struct i915_request *rq = READ_ONCE(ve->request);
1.1  riastrad
1.1  riastrad 		if (rq) {
1.1  riastrad 			if (count++ < max - 1)
1.1  riastrad 				show_request(m, rq, "\t\tV ");
1.1  riastrad 			else
1.1  riastrad 				last = rq;
1.1  riastrad 		}
1.1  riastrad 	}
1.1  riastrad 	if (last) {
1.1  riastrad 		if (count > max) {
1.1  riastrad 			drm_printf(m,
1.1  riastrad 				   "\t\t...skipping %d virtual requests...\n",
1.1  riastrad 				   count - max);
1.1  riastrad 		}
1.1  riastrad 		show_request(m, last, "\t\tV ");
1.1  riastrad 	}
1.1  riastrad
1.1  riastrad 	spin_unlock_irqrestore(&engine->active.lock, flags);
1.1  riastrad }
1.1  riastrad
1.1  riastrad void intel_lr_context_reset(struct intel_engine_cs *engine,
1.1  riastrad 			    struct intel_context *ce,
1.1  riastrad 			    u32 head,
1.1  riastrad 			    bool scrub)
1.1  riastrad {
1.1  riastrad 	GEM_BUG_ON(!intel_context_is_pinned(ce));
1.1  riastrad
1.1  riastrad 	/*
1.1  riastrad 	 * We want a simple context + ring to execute the breadcrumb update.
1.1  riastrad 	 * We cannot rely on the context being intact across the GPU hang,
1.1  riastrad 	 * so clear it and rebuild just what we need for the breadcrumb.
1.1  riastrad 	 * All pending requests for this context will be zapped, and any
1.1  riastrad 	 * future request will be after userspace has had the opportunity
1.1  riastrad 	 * to recreate its own state.
1.1  riastrad 	 */
1.1  riastrad 	if (scrub)
1.1  riastrad 		restore_default_state(ce, engine);
1.1  riastrad
1.1  riastrad 	/* Rerun the request; its payload has been neutered (if guilty). */
1.1  riastrad 	__execlists_update_reg_state(ce, engine, head);
1.1  riastrad }
1.1  riastrad
1.1  riastrad bool
1.1  riastrad intel_engine_in_execlists_submission_mode(const struct intel_engine_cs *engine)
1.1  riastrad {
1.1  riastrad 	return engine->set_default_submission ==
1.1  riastrad 	       intel_execlists_set_default_submission;
1.1  riastrad }
1.1  riastrad
1.1  riastrad #if IS_ENABLED(CONFIG_DRM_I915_SELFTEST)
1.1  riastrad #include "selftest_lrc.c"
1.1  riastrad #endif