perf(capture): overlap GPU readbacks with monotonic publish gate (slice 17)
Measure take ty-1824 on the slice-16 build: the downscale fix worked (conv ~47ms, ring allocs 0) but desktop band was still 88% frozen at 6.8 updates/s. Telemetry isolated the real wall — CreateCopyFromSurfaceAsync readback ~45ms of each conversion, serialized one-in-flight => ~17/s capture cap. Docs fact: pool-sized surfaces CLIP, not scale (Microsoft Learn), so readback stays native; the lever is concurrency. - MaxConcurrentConversions=3 with pool 2->5 buffers (in-flight frames fit) - new MonotonicGate (Interlocked compare-exchange): stale OLDER completions are dropped, never overwrite a newer LatestFrame (mirror of 1742 tear) - FrameRingBuffer.Rent/ConsumeAllocations now lock; downscale row scratch is per-conversion locals - Good Dog test PublishGate_TryPublish_OnlyStrictlyNewerWins; 296/296 green, 0 warnings; docs cited Microsoft screen-capture page + libyuv fixed-point. Local only, no push.
This commit is contained in:
+32
-22
@@ -25,6 +25,7 @@ internal sealed class FrameRingBuffer
|
||||
private readonly byte[]?[] _slots;
|
||||
private readonly long[] _lastHandout;
|
||||
private readonly int _redLine;
|
||||
private readonly object _lock = new();
|
||||
private int _next;
|
||||
private long _seq;
|
||||
private long _allocations;
|
||||
@@ -38,43 +39,52 @@ internal sealed class FrameRingBuffer
|
||||
}
|
||||
|
||||
/// <summary>Buffers freshly allocated after the last <see cref="ConsumeAllocations"/>.</summary>
|
||||
public long Allocations => _allocations;
|
||||
public long Allocations
|
||||
{
|
||||
get { lock (_lock) return _allocations; }
|
||||
}
|
||||
|
||||
/// <summary>Returns the allocation count since the last call and resets it.</summary>
|
||||
public long ConsumeAllocations()
|
||||
{
|
||||
var count = _allocations;
|
||||
_allocations = 0;
|
||||
return count;
|
||||
lock (_lock)
|
||||
{
|
||||
var count = _allocations;
|
||||
_allocations = 0;
|
||||
return count;
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>Hands out a scratch buffer of <paramref name="size"/> bytes.</summary>
|
||||
public byte[] Rent(int size)
|
||||
{
|
||||
var seq = ++_seq;
|
||||
for (var tries = 0; tries < _slots.Length; tries++)
|
||||
lock (_lock)
|
||||
{
|
||||
var idx = (_next + tries) % _slots.Length;
|
||||
// Red line: rewriting this slot could hit a frame a consumer still reads.
|
||||
if (seq - _lastHandout[idx] < _redLine) continue;
|
||||
_next = (idx + 1) % _slots.Length;
|
||||
if (_slots[idx] is { Length: var len } buf && len == size)
|
||||
var seq = ++_seq;
|
||||
for (var tries = 0; tries < _slots.Length; tries++)
|
||||
{
|
||||
var idx = (_next + tries) % _slots.Length;
|
||||
// Red line: rewriting this slot could hit a frame a consumer still reads.
|
||||
if (seq - _lastHandout[idx] < _redLine) continue;
|
||||
_next = (idx + 1) % _slots.Length;
|
||||
if (_slots[idx] is { Length: var len } buf && len == size)
|
||||
{
|
||||
_lastHandout[idx] = seq;
|
||||
return buf;
|
||||
}
|
||||
|
||||
// Length mismatch (or never allocated): a fresh array, never an in-place
|
||||
// overwrite — the previous loan's bytes stay valid for whoever holds it.
|
||||
_allocations++;
|
||||
var fresh = new byte[size];
|
||||
_slots[idx] = fresh;
|
||||
_lastHandout[idx] = seq;
|
||||
return buf;
|
||||
return fresh;
|
||||
}
|
||||
|
||||
// Length mismatch (or never allocated): a fresh array, never an in-place
|
||||
// overwrite — the previous loan's bytes stay valid for whoever holds it.
|
||||
// Defensive: the whole ring is inside its red line — do not lap a loaned slot.
|
||||
_allocations++;
|
||||
var fresh = new byte[size];
|
||||
_slots[idx] = fresh;
|
||||
_lastHandout[idx] = seq;
|
||||
return fresh;
|
||||
return new byte[size];
|
||||
}
|
||||
|
||||
// Defensive: the whole ring is inside its red line — do not lap a loaned slot.
|
||||
_allocations++;
|
||||
return new byte[size];
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
using System;
|
||||
using System.Threading;
|
||||
|
||||
namespace ytLive.Services;
|
||||
|
||||
/// <summary>
|
||||
/// An atomic "publish only if strictly newer" gate for the capture's latest-wins
|
||||
/// hand-out. Screen capture conversions may now overlap (slice 17), so two
|
||||
/// conversions can finish briefly out of order; without a monotonic gate the slower
|
||||
/// (older) completion would overwrite the faster (newer) one's LatestFrame and the
|
||||
/// compositor would render STALE content — a time hole in the other direction.
|
||||
/// Sequence must be a strictly increasing per-source counter
|
||||
/// (<c>Interlocked.Increment</c> in <c>ScreenCaptureFrameSource</c>).
|
||||
/// </summary>
|
||||
internal sealed class MonotonicGate
|
||||
{
|
||||
private long _last;
|
||||
|
||||
/// <summary>Returns true iff <paramref name="seq"/> is strictly newer than every
|
||||
/// seq accepted so far (and records it).</summary>
|
||||
public bool TryPublish(long seq)
|
||||
{
|
||||
while (true)
|
||||
{
|
||||
var current = Interlocked.Read(ref _last);
|
||||
if (seq <= current) return false;
|
||||
if (Interlocked.CompareExchange(ref _last, seq, current) == current) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,7 @@ using System;
|
||||
using System.Diagnostics;
|
||||
using System.Runtime.InteropServices;
|
||||
using System.Runtime.InteropServices.WindowsRuntime;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using Windows.Graphics;
|
||||
using Windows.Graphics.Capture;
|
||||
@@ -28,7 +29,7 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
private GraphicsCaptureSession? _session;
|
||||
private SizeInt32 _poolSize;
|
||||
private bool _started;
|
||||
private bool _framePending;
|
||||
private int _convertInFlight;
|
||||
private DateTime _lastErrorLog = DateTime.MinValue;
|
||||
|
||||
// Buffer recycling (take-11 spikes, 2026-09-04; slice 16, 2026-09-14): a fresh
|
||||
@@ -43,8 +44,7 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
// consumers (the compositor's paste cache) cannot false-hit a recycled array.
|
||||
private readonly FrameRingBuffer _ring = new(8, redLine: 4);
|
||||
private long _epoch;
|
||||
private byte[]? _row0;
|
||||
private byte[]? _row1;
|
||||
private readonly MonotonicGate _publishGate = new();
|
||||
|
||||
// The composition master frame (see ai.md "Resolution tiers"): the background
|
||||
// is an input layer, so we never hold a CPU frame bigger than the master.
|
||||
@@ -54,14 +54,19 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
// A failing conversion must not re-flood the log at frame rate.
|
||||
private static readonly TimeSpan ErrorLogThrottle = TimeSpan.FromSeconds(5);
|
||||
|
||||
// Conversion cadence (slice 16, 2026-09-14): the monitor delivers at the 240Hz DWM
|
||||
// cadence (~4.2ms) while slots are 16.6ms. A 10ms floor between conversion starts
|
||||
// keeps the open edge above the ~60 conversions/s the pump can actually use, so the
|
||||
// 240Hz tail stops chewing a conversion thread that the 1742 take measured at
|
||||
// ~150ms/frame (a 90%-frozen desktop, ~6 fresh frames/s). Drop counters and the
|
||||
// rolling conversion stats feed the 2-second startup.log telemetry line.
|
||||
// Conversion cadence (slice 16, 2026-09-14; slice 17, 2026-09-14): the monitor
|
||||
// delivers at the 240Hz DWM cadence (~4.2ms) while slots are 16.6ms. A 10ms floor
|
||||
// between conversion starts keeps the open edge above the ~60 conversions/s the pump
|
||||
// can use. What actually bounded the desktop feed on takes 1742/1824 was the SERIAL
|
||||
// GPU→CPU readback (`CreateCopyFromSurfaceAsync` ≈ 40-50ms of the ~47ms conversion
|
||||
// on a 240Hz-HDR box shared with the encoder), capping captures at ~17-20/s.
|
||||
// Slice 17 overlaps up to MaxConcurrentConversions readbacks (pool sized to
|
||||
// accommodate in-flight frames) and publishes only monotonically newer frames
|
||||
// (MonotonicGate — a slow older completion must never overwrite a newer LatestFrame).
|
||||
private static readonly TimeSpan MinConvertInterval = TimeSpan.FromMilliseconds(10);
|
||||
private static readonly TimeSpan TelemetryInterval = TimeSpan.FromSeconds(2);
|
||||
private const int MaxConcurrentConversions = 3;
|
||||
private const int PoolBufferCount = 5;
|
||||
private DateTime _lastConvertAt = DateTime.MinValue;
|
||||
private DateTime _telemetryFrom = DateTime.UtcNow;
|
||||
private DateTime _lastTelemetry = DateTime.UtcNow;
|
||||
@@ -107,7 +112,7 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
var item = _item;
|
||||
var device = Direct3D11Helper.CreateDevice();
|
||||
var framePool = Direct3D11CaptureFramePool.CreateFreeThreaded(
|
||||
device, DirectXPixelFormat.B8G8R8A8UIntNormalized, 2, item.Size);
|
||||
device, DirectXPixelFormat.B8G8R8A8UIntNormalized, PoolBufferCount, item.Size);
|
||||
_poolSize = item.Size;
|
||||
var session = framePool.CreateCaptureSession(item);
|
||||
framePool.FrameArrived += OnFrameArrived;
|
||||
@@ -124,7 +129,7 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
lock (_gate)
|
||||
{
|
||||
_started = false;
|
||||
_framePending = false;
|
||||
_convertInFlight = 0;
|
||||
if (_framePool != null)
|
||||
_framePool.FrameArrived -= OnFrameArrived;
|
||||
_session?.Dispose();
|
||||
@@ -151,14 +156,15 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
if (frame.ContentSize.Width != _poolSize.Width || frame.ContentSize.Height != _poolSize.Height)
|
||||
{
|
||||
sender.Recreate(Direct3D11Helper.CreateDevice(),
|
||||
DirectXPixelFormat.B8G8R8A8UIntNormalized, 2, frame.ContentSize);
|
||||
DirectXPixelFormat.B8G8R8A8UIntNormalized, PoolBufferCount, frame.ContentSize);
|
||||
_poolSize = frame.ContentSize;
|
||||
}
|
||||
|
||||
// One conversion at a time, spaced by MinConvertInterval (slice 16): the
|
||||
// 240Hz delivery otherwise queued a conversion every ~4.2ms and the
|
||||
// 1742 take's ~150ms conversion pinned the desktop layer ~90% frozen.
|
||||
if (_framePending)
|
||||
// Up to MaxConcurrentConversions readbacks in flight (slice 17), spaced by
|
||||
// MinConvertInterval (slice 16): the 240Hz delivery otherwise queued one
|
||||
// conversion every ~4.2ms, and the serial ~47ms readback pinned the desktop
|
||||
// layer to ~17 updates/s on the 1824 take.
|
||||
if (_convertInFlight >= MaxConcurrentConversions)
|
||||
{
|
||||
_skippedBusy++;
|
||||
frame.Dispose();
|
||||
@@ -172,7 +178,7 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
return;
|
||||
}
|
||||
_lastConvertAt = now;
|
||||
_framePending = true;
|
||||
_convertInFlight++;
|
||||
}
|
||||
_ = ProcessFrameAsync(frame);
|
||||
}
|
||||
@@ -186,7 +192,14 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
using (var softwareBitmap = await SoftwareBitmap.CreateCopyFromSurfaceAsync(
|
||||
frame.Surface, BitmapAlphaMode.Ignore))
|
||||
{
|
||||
FrameAvailable?.Invoke(CopyToVideoFrame(softwareBitmap));
|
||||
// Monotonic sequencing across the overlapping conversions (slice 17): a
|
||||
// completed readback is published only if strictly newer than the last
|
||||
// one published — a slow older completion must never overwrite a newer
|
||||
// LatestFrame (that would be a time hole in the other direction).
|
||||
var epoch = Interlocked.Increment(ref _epoch);
|
||||
var videoFrame = CopyToVideoFrame(softwareBitmap, epoch);
|
||||
if (_publishGate.TryPublish(epoch))
|
||||
FrameAvailable?.Invoke(videoFrame);
|
||||
}
|
||||
}
|
||||
catch (Exception ex)
|
||||
@@ -203,8 +216,8 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
lock (_gate)
|
||||
{
|
||||
EmitTelemetry(sw.ElapsedMilliseconds);
|
||||
_convertInFlight--;
|
||||
}
|
||||
_framePending = false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -234,7 +247,7 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
_skippedCadence = 0;
|
||||
}
|
||||
|
||||
private VideoFrame CopyToVideoFrame(SoftwareBitmap bitmap)
|
||||
private VideoFrame CopyToVideoFrame(SoftwareBitmap bitmap, long epoch)
|
||||
{
|
||||
var sw = bitmap.PixelWidth;
|
||||
var sh = bitmap.PixelHeight;
|
||||
@@ -257,12 +270,12 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
// DWM delivers an opaque surface (alpha 255); bilinear keeps it 255.
|
||||
var scaled = _ring.Rent(dw * dh * 4);
|
||||
return new VideoFrame(dw, dh, DownscaleBgra(data, sw, sh, srcStride, dw, dh, scaled))
|
||||
{ IsOpaque = true, Epoch = ++_epoch };
|
||||
{ IsOpaque = true, Epoch = epoch };
|
||||
}
|
||||
|
||||
var pixels = _ring.Rent(count);
|
||||
Marshal.Copy(data, pixels, 0, pixels.Length);
|
||||
return new VideoFrame(sw, sh, pixels) { IsOpaque = true, Epoch = ++_epoch };
|
||||
return new VideoFrame(sw, sh, pixels) { IsOpaque = true, Epoch = epoch };
|
||||
}
|
||||
|
||||
// Integer 8.8 fixed-point bilinear downscale to the master frame (slice 16,
|
||||
@@ -275,9 +288,11 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
// row pair through Marshal.Copy (no unsafe), writing tightly packed BGRA output.
|
||||
private byte[] DownscaleBgra(IntPtr src, int sw, int sh, int srcStride, int dw, int dh, byte[] dst)
|
||||
{
|
||||
// Row scratch is per-capture-thread and reused across frames (same churn lesson).
|
||||
var row0 = _row0 != null && _row0.Length >= srcStride ? _row0 : (_row0 = new byte[srcStride]);
|
||||
var row1 = _row1 != null && _row1.Length >= srcStride ? _row1 : (_row1 = new byte[srcStride]);
|
||||
// Row scratch is per-conversion (overlapping conversions since slice 17 each
|
||||
// bring their own — two 10KB arrays, no shared state). Size: a 2560-wide row
|
||||
// pair, the largest the monitor path delivers before the downscale.
|
||||
var row0 = new byte[srcStride];
|
||||
var row1 = new byte[srcStride];
|
||||
|
||||
for (var y = 0; y < dh; y++)
|
||||
{
|
||||
|
||||
Reference in New Issue
Block a user