perf(capture): slice 8 — buffer ring + paste-cache Epoch; gen2 visibility (take-11 spikes)

Take 11 (c10ce06c) validated the off-UI architecture: typical frames land
work ~10ms + wait ~6.8ms = 16.7 exactly on the deadline; 212/300 best yet.
The ENTIRE remaining gap is periodic 35-65ms render spikes that WORSENED
across the take (189 -> 147) — the signature of gen2 GC pauses. Biggest
churner is structural: the screen capture minted a fresh ~8.3MB byte[] per
DWM frame (~500MB/s of LOH), a producer OBS never does (it owns fixed
surface pools).

- ScreenCaptureFrameSource: 4-deep buffer ring with size-matched slots (a
  <=17ms consumer cannot be lapped at 60Hz) + reused downscale row scratch.
- VideoFrame.Epoch: monotonic per producer frame. The paste cache keys on
  array IDENTITY, so recycled arrays MUST be distinguished — epoch joins the
  PasteKey. Producers handing fresh arrays leave it 0 (key unchanged effect).
- Stats print 'gen2 +N' per 5s window: next take acquits or convicts GC
  without another guess (rule: prove the stage).
- Test (the ONE): PasteCache_RecycledArrayWithNewEpoch_ReRasterizes_NotStaleHits
  — same array, new content, bumped epoch; fails on the old key by
  construction. 37/37 compositor/pump, clean build.
- Next suspect if gen2 stays hot: the 10Hz WebView2 capture (full-canvas PNG
  decode + fresh arrays on the UI thread) — recorded, untouched.

Creator audio ask queued in the same working session (+40% post-mix master
gain before the -1dBFS limiter) lands as its own commit next.
This commit is contained in:
2026-09-04 12:36:14 -07:00
parent eb4c379b91
commit fbc8562cf4
7 changed files with 128 additions and 20 deletions
+41 -8
View File
@@ -30,6 +30,18 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
private bool _framePending;
private DateTime _lastErrorLog = DateTime.MinValue;
// Buffer recycling (take-11 spikes, 2026-09-04): a fresh ~8.3MB byte[] per DWM
// frame ≈ 500MB/s of LOH churn — the gen2 pauses it forces surfaced as the
// "worst render 35-65ms" spikes that capped fps at ~42 long after the compositor
// itself was fast. A 4-deep ring rotated round-robin is never lapped by a
// ≤17ms consumer at 60Hz; each hand-out carries an Epoch so identity-keyed
// consumers (the compositor's paste cache) cannot false-hit a recycled array.
private readonly byte[]?[] _frameRing = new byte[4][];
private int _ringNext;
private long _epoch;
private byte[]? _row0;
private byte[]? _row1;
// The composition master frame (see ai.md "Resolution tiers"): the background
// is an input layer, so we never hold a CPU frame bigger than the master.
private const int MaxBackgroundWidth = 1920;
@@ -157,7 +169,26 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
}
}
private static VideoFrame CopyToVideoFrame(SoftwareBitmap bitmap)
private byte[] RentRingBuffer(int size)
{
for (var tries = 0; tries < _frameRing.Length; tries++)
{
var idx = (_ringNext + tries) % _frameRing.Length;
var buf = _frameRing[idx];
if (buf is { Length: var len } && len == size)
{
_ringNext = (idx + 1) % _frameRing.Length;
return buf;
}
}
var slot = _ringNext;
_ringNext = (slot + 1) % _frameRing.Length;
var fresh = new byte[size];
_frameRing[slot] = fresh;
return fresh;
}
private VideoFrame CopyToVideoFrame(SoftwareBitmap bitmap)
{
var sw = bitmap.PixelWidth;
var sh = bitmap.PixelHeight;
@@ -178,21 +209,23 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
var dw = Math.Max(1, (int)(sw * scale));
var dh = Math.Max(1, (int)(sh * scale));
// DWM delivers an opaque surface (alpha 255); bilinear keeps it 255.
return new VideoFrame(dw, dh, DownscaleBgra(data, sw, sh, srcStride, dw, dh)) { IsOpaque = true };
var scaled = RentRingBuffer(dw * dh * 4);
return new VideoFrame(dw, dh, DownscaleBgra(data, sw, sh, srcStride, dw, dh, scaled))
{ IsOpaque = true, Epoch = ++_epoch };
}
var pixels = new byte[count];
var pixels = RentRingBuffer(count);
Marshal.Copy(data, pixels, 0, pixels.Length);
return new VideoFrame(sw, sh, pixels) { IsOpaque = true };
return new VideoFrame(sw, sh, pixels) { IsOpaque = true, Epoch = ++_epoch };
}
// Bilinear downscale to the master frame. Reads each source row pair through
// Marshal.Copy (no unsafe), writing tightly packed BGRA output.
private static byte[] DownscaleBgra(IntPtr src, int sw, int sh, int srcStride, int dw, int dh)
private byte[] DownscaleBgra(IntPtr src, int sw, int sh, int srcStride, int dw, int dh, byte[] dst)
{
var row0 = new byte[srcStride];
var row1 = new byte[srcStride];
var dst = new byte[dw * dh * 4];
// Row scratch is per-capture-thread and reused across frames (same churn lesson).
var row0 = _row0 != null && _row0.Length >= srcStride ? _row0 : (_row0 = new byte[srcStride]);
var row1 = _row1 != null && _row1.Length >= srcStride ? _row1 : (_row1 = new byte[srcStride]);
var xs = sw / (double)dw;
var ys = sh / (double)dh;