perf(capture): overlap GPU readbacks with monotonic publish gate (slice 17)

Measure take ty-1824 on the slice-16 build: the downscale fix worked (conv
~47ms, ring allocs 0) but desktop band was still 88% frozen at 6.8 updates/s.
Telemetry isolated the real wall — CreateCopyFromSurfaceAsync readback ~45ms
of each conversion, serialized one-in-flight => ~17/s capture cap. Docs fact:
pool-sized surfaces CLIP, not scale (Microsoft Learn), so readback stays
native; the lever is concurrency.

- MaxConcurrentConversions=3 with pool 2->5 buffers (in-flight frames fit)
- new MonotonicGate (Interlocked compare-exchange): stale OLDER completions
  are dropped, never overwrite a newer LatestFrame (mirror of 1742 tear)
- FrameRingBuffer.Rent/ConsumeAllocations now lock; downscale row scratch is
  per-conversion locals
- Good Dog test PublishGate_TryPublish_OnlyStrictlyNewerWins; 296/296 green,
  0 warnings; docs cited Microsoft screen-capture page + libyuv fixed-point.

Local only, no push.
This commit is contained in:
2026-09-14 18:40:20 -07:00
parent 71932b9756
commit b37b8a30f9
7 changed files with 238 additions and 114 deletions
+32 -22
View File
@@ -25,6 +25,7 @@ internal sealed class FrameRingBuffer
private readonly byte[]?[] _slots;
private readonly long[] _lastHandout;
private readonly int _redLine;
private readonly object _lock = new();
private int _next;
private long _seq;
private long _allocations;
@@ -38,43 +39,52 @@ internal sealed class FrameRingBuffer
}
/// <summary>Buffers freshly allocated after the last <see cref="ConsumeAllocations"/>.</summary>
public long Allocations => _allocations;
public long Allocations
{
get { lock (_lock) return _allocations; }
}
/// <summary>Returns the allocation count since the last call and resets it.</summary>
public long ConsumeAllocations()
{
var count = _allocations;
_allocations = 0;
return count;
lock (_lock)
{
var count = _allocations;
_allocations = 0;
return count;
}
}
/// <summary>Hands out a scratch buffer of <paramref name="size"/> bytes.</summary>
public byte[] Rent(int size)
{
var seq = ++_seq;
for (var tries = 0; tries < _slots.Length; tries++)
lock (_lock)
{
var idx = (_next + tries) % _slots.Length;
// Red line: rewriting this slot could hit a frame a consumer still reads.
if (seq - _lastHandout[idx] < _redLine) continue;
_next = (idx + 1) % _slots.Length;
if (_slots[idx] is { Length: var len } buf && len == size)
var seq = ++_seq;
for (var tries = 0; tries < _slots.Length; tries++)
{
var idx = (_next + tries) % _slots.Length;
// Red line: rewriting this slot could hit a frame a consumer still reads.
if (seq - _lastHandout[idx] < _redLine) continue;
_next = (idx + 1) % _slots.Length;
if (_slots[idx] is { Length: var len } buf && len == size)
{
_lastHandout[idx] = seq;
return buf;
}
// Length mismatch (or never allocated): a fresh array, never an in-place
// overwrite — the previous loan's bytes stay valid for whoever holds it.
_allocations++;
var fresh = new byte[size];
_slots[idx] = fresh;
_lastHandout[idx] = seq;
return buf;
return fresh;
}
// Length mismatch (or never allocated): a fresh array, never an in-place
// overwrite — the previous loan's bytes stay valid for whoever holds it.
// Defensive: the whole ring is inside its red line — do not lap a loaned slot.
_allocations++;
var fresh = new byte[size];
_slots[idx] = fresh;
_lastHandout[idx] = seq;
return fresh;
return new byte[size];
}
// Defensive: the whole ring is inside its red line — do not lap a loaned slot.
_allocations++;
return new byte[size];
}
}