2564fe708b
Asking the GPU to busywait on a memory address, perhaps not unexpectedly in hindsight for a shared system, leads to bus contention that affects CPU programs trying to concurrently access memory. This can manifest as a drop in transcode throughput on highly over-saturated workloads. The only clue offered by perf, is that the bus-cycles (perf stat -e bus-cycles) jumped by 50% when enabling semaphores. This corresponds with extra CPU active cycles being attributed to intel_idle's mwait. This patch introduces a heuristic to try and detect when more than one client is submitting to the GPU pushing it into an oversaturated state. As we already keep track of when the semaphores are signaled, we can inspect their state on submitting the busywait batch and if we planned to use a semaphore but were too late, conclude that the GPU is overloaded and not try to use semaphores in future requests. In practice, this means we optimistically try to use semaphores for the first frame of a transcode job split over multiple engines, and fail if there are multiple clients active and continue not to use semaphores for the subsequent frames in the sequence. Periodically, we try to optimistically switch semaphores back on whenever the client waits to catch up with the transcode results. With 1 client, on Broxton J3455, with the relative fps normalized by %cpu: x no semaphores + drm-tip * patched +------------------------------------------------------------------------+ | * | | *+ | | **+ | | **+ x | | x * +**+ x | | x x * * +***x xx | | x x * * *+***x *x | | x x* + * * *****x *x x | | + x xx+x* + *** * ********* x * | | + x xx+x* * *** +** ********* xx * | | * + ++++* + x*x****+*+* ***+*************+x* * | |*+ +** *+ + +* + *++****** *xxx**********x***+*****************+*++ *| | |__________A_____M_____| | | |_______________A____M_________| | | |____________A___M________| | +------------------------------------------------------------------------+ N Min Max Median Avg Stddev x 120 2.60475 3.50941 3.31123 3.2143953 0.21117399 + 120 2.3826 3.57077 3.25101 3.1414161 0.28146407 Difference at 95.0% confidence -0.0729792 +/- 0.0629585 -2.27039% +/- 1.95864% (Student's t, pooled s = 0.248814) * 120 2.35536 3.66713 3.2849 3.2059917 0.24618565 No difference proven at 95.0% confidence With 10 clients over-saturating the pipeline: x no semaphores + drm-tip * patched +------------------------------------------------------------------------+ | ++ ** | | ++ ** | | ++ ** | | ++ ** | | ++ xx *** | | ++ xx *** | | ++ xxx*** | | ++ xxx*** | | +++ xxx*** | | +++ xx**** | | +++ xx**** | | +++ xx**** | | +++ xx**** | | ++++ xx**** | | +++++ xx**** | | +++++ x x****** | | ++++++ xxx******* | | ++++++ xxx******* | | ++++++ xxx******* | | ++++++ xx******** | | ++++++ xxxx******** | | ++++++ xxxx******** | | ++++++++ xxxxx********* | |+ + + + ++++++++ xxx*xx**********x* *| | |__A__| | | |__AM__| | | |__A_| | +------------------------------------------------------------------------+ N Min Max Median Avg Stddev x 120 2.47855 2.8972 2.72376 2.7193402 0.074604933 + 120 1.17367 1.77459 1.71977 1.6966782 0.085850697 Difference at 95.0% confidence -1.02266 +/- 0.0203502 -37.607% +/- 0.748352% (Student's t, pooled s = 0.0804246) * 120 2.57868 3.00821 2.80142 2.7923878 0.058646477 Difference at 95.0% confidence 0.0730476 +/- 0.0169791 2.68622% +/- 0.624383% (Student's t, pooled s = 0.0671018) Indicating that we've recovered the regression from enabling semaphores on this saturated setup, with a hint towards an overall improvement. Very similar, but of smaller magnitude, results are observed on both Skylake(gt2) and Kabylake(gt4). This may be due to the reduced impact of bus-cycles, where we see a 50% hit on Broxton, it is only 10% on the big core, in this particular test. One observation to make here is that for a greedy client trying to maximise its own throughput, using semaphores is the right choice. It is only the holistic system-wide view that semaphores of one client impacts another and reduces the overall throughput where we would choose to disable semaphores. The most noticeable negactive impact this has is on the no-op microbenchmarks, which are also very notable for having no cpu bus load. In particular, this increases the runtime and energy consumption of gem_exec_whisper. Fixes:e886196469
("drm/i915: Use HW semaphores for inter-engine synchronisation on gen8+") Signed-off-by: Chris Wilson <chris@chris-wilson.co.uk> Cc: Tvrtko Ursulin <tvrtko.ursulin@intel.com> Cc: Dmitry Rogozhkin <dmitry.v.rogozhkin@intel.com> Cc: Dmitry Ermilov <dmitry.ermilov@intel.com> Cc: Joonas Lahtinen <joonas.lahtinen@linux.intel.com> Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@intel.com> Link: https://patchwork.freedesktop.org/patch/msgid/20190504070707.30902-1-chris@chris-wilson.co.uk (cherry picked from commitca6e56f654
) Signed-off-by: Joonas Lahtinen <joonas.lahtinen@linux.intel.com>
271 lines
5.5 KiB
C
271 lines
5.5 KiB
C
/*
|
|
* SPDX-License-Identifier: MIT
|
|
*
|
|
* Copyright © 2019 Intel Corporation
|
|
*/
|
|
|
|
#include "i915_drv.h"
|
|
#include "i915_gem_context.h"
|
|
#include "i915_globals.h"
|
|
#include "intel_context.h"
|
|
#include "intel_ringbuffer.h"
|
|
|
|
static struct i915_global_context {
|
|
struct i915_global base;
|
|
struct kmem_cache *slab_ce;
|
|
} global;
|
|
|
|
struct intel_context *intel_context_alloc(void)
|
|
{
|
|
return kmem_cache_zalloc(global.slab_ce, GFP_KERNEL);
|
|
}
|
|
|
|
void intel_context_free(struct intel_context *ce)
|
|
{
|
|
kmem_cache_free(global.slab_ce, ce);
|
|
}
|
|
|
|
struct intel_context *
|
|
intel_context_lookup(struct i915_gem_context *ctx,
|
|
struct intel_engine_cs *engine)
|
|
{
|
|
struct intel_context *ce = NULL;
|
|
struct rb_node *p;
|
|
|
|
spin_lock(&ctx->hw_contexts_lock);
|
|
p = ctx->hw_contexts.rb_node;
|
|
while (p) {
|
|
struct intel_context *this =
|
|
rb_entry(p, struct intel_context, node);
|
|
|
|
if (this->engine == engine) {
|
|
GEM_BUG_ON(this->gem_context != ctx);
|
|
ce = this;
|
|
break;
|
|
}
|
|
|
|
if (this->engine < engine)
|
|
p = p->rb_right;
|
|
else
|
|
p = p->rb_left;
|
|
}
|
|
spin_unlock(&ctx->hw_contexts_lock);
|
|
|
|
return ce;
|
|
}
|
|
|
|
struct intel_context *
|
|
__intel_context_insert(struct i915_gem_context *ctx,
|
|
struct intel_engine_cs *engine,
|
|
struct intel_context *ce)
|
|
{
|
|
struct rb_node **p, *parent;
|
|
int err = 0;
|
|
|
|
spin_lock(&ctx->hw_contexts_lock);
|
|
|
|
parent = NULL;
|
|
p = &ctx->hw_contexts.rb_node;
|
|
while (*p) {
|
|
struct intel_context *this;
|
|
|
|
parent = *p;
|
|
this = rb_entry(parent, struct intel_context, node);
|
|
|
|
if (this->engine == engine) {
|
|
err = -EEXIST;
|
|
ce = this;
|
|
break;
|
|
}
|
|
|
|
if (this->engine < engine)
|
|
p = &parent->rb_right;
|
|
else
|
|
p = &parent->rb_left;
|
|
}
|
|
if (!err) {
|
|
rb_link_node(&ce->node, parent, p);
|
|
rb_insert_color(&ce->node, &ctx->hw_contexts);
|
|
}
|
|
|
|
spin_unlock(&ctx->hw_contexts_lock);
|
|
|
|
return ce;
|
|
}
|
|
|
|
void __intel_context_remove(struct intel_context *ce)
|
|
{
|
|
struct i915_gem_context *ctx = ce->gem_context;
|
|
|
|
spin_lock(&ctx->hw_contexts_lock);
|
|
rb_erase(&ce->node, &ctx->hw_contexts);
|
|
spin_unlock(&ctx->hw_contexts_lock);
|
|
}
|
|
|
|
static struct intel_context *
|
|
intel_context_instance(struct i915_gem_context *ctx,
|
|
struct intel_engine_cs *engine)
|
|
{
|
|
struct intel_context *ce, *pos;
|
|
|
|
ce = intel_context_lookup(ctx, engine);
|
|
if (likely(ce))
|
|
return ce;
|
|
|
|
ce = intel_context_alloc();
|
|
if (!ce)
|
|
return ERR_PTR(-ENOMEM);
|
|
|
|
intel_context_init(ce, ctx, engine);
|
|
|
|
pos = __intel_context_insert(ctx, engine, ce);
|
|
if (unlikely(pos != ce)) /* Beaten! Use their HW context instead */
|
|
intel_context_free(ce);
|
|
|
|
GEM_BUG_ON(intel_context_lookup(ctx, engine) != pos);
|
|
return pos;
|
|
}
|
|
|
|
struct intel_context *
|
|
intel_context_pin_lock(struct i915_gem_context *ctx,
|
|
struct intel_engine_cs *engine)
|
|
__acquires(ce->pin_mutex)
|
|
{
|
|
struct intel_context *ce;
|
|
|
|
ce = intel_context_instance(ctx, engine);
|
|
if (IS_ERR(ce))
|
|
return ce;
|
|
|
|
if (mutex_lock_interruptible(&ce->pin_mutex))
|
|
return ERR_PTR(-EINTR);
|
|
|
|
return ce;
|
|
}
|
|
|
|
struct intel_context *
|
|
intel_context_pin(struct i915_gem_context *ctx,
|
|
struct intel_engine_cs *engine)
|
|
{
|
|
struct intel_context *ce;
|
|
int err;
|
|
|
|
ce = intel_context_instance(ctx, engine);
|
|
if (IS_ERR(ce))
|
|
return ce;
|
|
|
|
if (likely(atomic_inc_not_zero(&ce->pin_count)))
|
|
return ce;
|
|
|
|
if (mutex_lock_interruptible(&ce->pin_mutex))
|
|
return ERR_PTR(-EINTR);
|
|
|
|
if (likely(!atomic_read(&ce->pin_count))) {
|
|
err = ce->ops->pin(ce);
|
|
if (err)
|
|
goto err;
|
|
|
|
i915_gem_context_get(ctx);
|
|
GEM_BUG_ON(ce->gem_context != ctx);
|
|
|
|
mutex_lock(&ctx->mutex);
|
|
list_add(&ce->active_link, &ctx->active_engines);
|
|
mutex_unlock(&ctx->mutex);
|
|
|
|
intel_context_get(ce);
|
|
smp_mb__before_atomic(); /* flush pin before it is visible */
|
|
}
|
|
|
|
atomic_inc(&ce->pin_count);
|
|
GEM_BUG_ON(!intel_context_is_pinned(ce)); /* no overflow! */
|
|
|
|
mutex_unlock(&ce->pin_mutex);
|
|
return ce;
|
|
|
|
err:
|
|
mutex_unlock(&ce->pin_mutex);
|
|
return ERR_PTR(err);
|
|
}
|
|
|
|
void intel_context_unpin(struct intel_context *ce)
|
|
{
|
|
if (likely(atomic_add_unless(&ce->pin_count, -1, 1)))
|
|
return;
|
|
|
|
/* We may be called from inside intel_context_pin() to evict another */
|
|
intel_context_get(ce);
|
|
mutex_lock_nested(&ce->pin_mutex, SINGLE_DEPTH_NESTING);
|
|
|
|
if (likely(atomic_dec_and_test(&ce->pin_count))) {
|
|
ce->ops->unpin(ce);
|
|
|
|
mutex_lock(&ce->gem_context->mutex);
|
|
list_del(&ce->active_link);
|
|
mutex_unlock(&ce->gem_context->mutex);
|
|
|
|
i915_gem_context_put(ce->gem_context);
|
|
intel_context_put(ce);
|
|
}
|
|
|
|
mutex_unlock(&ce->pin_mutex);
|
|
intel_context_put(ce);
|
|
}
|
|
|
|
static void intel_context_retire(struct i915_active_request *active,
|
|
struct i915_request *rq)
|
|
{
|
|
struct intel_context *ce =
|
|
container_of(active, typeof(*ce), active_tracker);
|
|
|
|
intel_context_unpin(ce);
|
|
}
|
|
|
|
void
|
|
intel_context_init(struct intel_context *ce,
|
|
struct i915_gem_context *ctx,
|
|
struct intel_engine_cs *engine)
|
|
{
|
|
kref_init(&ce->ref);
|
|
|
|
ce->gem_context = ctx;
|
|
ce->engine = engine;
|
|
ce->ops = engine->cops;
|
|
ce->saturated = 0;
|
|
|
|
INIT_LIST_HEAD(&ce->signal_link);
|
|
INIT_LIST_HEAD(&ce->signals);
|
|
|
|
mutex_init(&ce->pin_mutex);
|
|
|
|
/* Use the whole device by default */
|
|
ce->sseu = intel_device_default_sseu(ctx->i915);
|
|
|
|
i915_active_request_init(&ce->active_tracker,
|
|
NULL, intel_context_retire);
|
|
}
|
|
|
|
static void i915_global_context_shrink(void)
|
|
{
|
|
kmem_cache_shrink(global.slab_ce);
|
|
}
|
|
|
|
static void i915_global_context_exit(void)
|
|
{
|
|
kmem_cache_destroy(global.slab_ce);
|
|
}
|
|
|
|
static struct i915_global_context global = { {
|
|
.shrink = i915_global_context_shrink,
|
|
.exit = i915_global_context_exit,
|
|
} };
|
|
|
|
int __init i915_global_context_init(void)
|
|
{
|
|
global.slab_ce = KMEM_CACHE(intel_context, SLAB_HWCACHE_ALIGN);
|
|
if (!global.slab_ce)
|
|
return -ENOMEM;
|
|
|
|
i915_global_register(&global.base);
|
|
return 0;
|
|
}
|