perf(capture): slice 8 — buffer ring + paste-cache Epoch; gen2 visibility (take-11 spikes)
Take 11 (c10ce06c) validated the off-UI architecture: typical frames land work ~10ms + wait ~6.8ms = 16.7 exactly on the deadline; 212/300 best yet. The ENTIRE remaining gap is periodic 35-65ms render spikes that WORSENED across the take (189 -> 147) — the signature of gen2 GC pauses. Biggest churner is structural: the screen capture minted a fresh ~8.3MB byte[] per DWM frame (~500MB/s of LOH), a producer OBS never does (it owns fixed surface pools). - ScreenCaptureFrameSource: 4-deep buffer ring with size-matched slots (a <=17ms consumer cannot be lapped at 60Hz) + reused downscale row scratch. - VideoFrame.Epoch: monotonic per producer frame. The paste cache keys on array IDENTITY, so recycled arrays MUST be distinguished — epoch joins the PasteKey. Producers handing fresh arrays leave it 0 (key unchanged effect). - Stats print 'gen2 +N' per 5s window: next take acquits or convicts GC without another guess (rule: prove the stage). - Test (the ONE): PasteCache_RecycledArrayWithNewEpoch_ReRasterizes_NotStaleHits — same array, new content, bumped epoch; fails on the old key by construction. 37/37 compositor/pump, clean build. - Next suspect if gen2 stays hot: the 10Hz WebView2 capture (full-canvas PNG decode + fresh arrays on the UI thread) — recorded, untouched. Creator audio ask queued in the same working session (+40% post-mix master gain before the -1dBFS limiter) lands as its own commit next.
This commit is contained in:
@@ -30,6 +30,18 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
private bool _framePending;
|
||||
private DateTime _lastErrorLog = DateTime.MinValue;
|
||||
|
||||
// Buffer recycling (take-11 spikes, 2026-09-04): a fresh ~8.3MB byte[] per DWM
|
||||
// frame ≈ 500MB/s of LOH churn — the gen2 pauses it forces surfaced as the
|
||||
// "worst render 35-65ms" spikes that capped fps at ~42 long after the compositor
|
||||
// itself was fast. A 4-deep ring rotated round-robin is never lapped by a
|
||||
// ≤17ms consumer at 60Hz; each hand-out carries an Epoch so identity-keyed
|
||||
// consumers (the compositor's paste cache) cannot false-hit a recycled array.
|
||||
private readonly byte[]?[] _frameRing = new byte[4][];
|
||||
private int _ringNext;
|
||||
private long _epoch;
|
||||
private byte[]? _row0;
|
||||
private byte[]? _row1;
|
||||
|
||||
// The composition master frame (see ai.md "Resolution tiers"): the background
|
||||
// is an input layer, so we never hold a CPU frame bigger than the master.
|
||||
private const int MaxBackgroundWidth = 1920;
|
||||
@@ -157,7 +169,26 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
}
|
||||
}
|
||||
|
||||
private static VideoFrame CopyToVideoFrame(SoftwareBitmap bitmap)
|
||||
private byte[] RentRingBuffer(int size)
|
||||
{
|
||||
for (var tries = 0; tries < _frameRing.Length; tries++)
|
||||
{
|
||||
var idx = (_ringNext + tries) % _frameRing.Length;
|
||||
var buf = _frameRing[idx];
|
||||
if (buf is { Length: var len } && len == size)
|
||||
{
|
||||
_ringNext = (idx + 1) % _frameRing.Length;
|
||||
return buf;
|
||||
}
|
||||
}
|
||||
var slot = _ringNext;
|
||||
_ringNext = (slot + 1) % _frameRing.Length;
|
||||
var fresh = new byte[size];
|
||||
_frameRing[slot] = fresh;
|
||||
return fresh;
|
||||
}
|
||||
|
||||
private VideoFrame CopyToVideoFrame(SoftwareBitmap bitmap)
|
||||
{
|
||||
var sw = bitmap.PixelWidth;
|
||||
var sh = bitmap.PixelHeight;
|
||||
@@ -178,21 +209,23 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
|
||||
var dw = Math.Max(1, (int)(sw * scale));
|
||||
var dh = Math.Max(1, (int)(sh * scale));
|
||||
// DWM delivers an opaque surface (alpha 255); bilinear keeps it 255.
|
||||
return new VideoFrame(dw, dh, DownscaleBgra(data, sw, sh, srcStride, dw, dh)) { IsOpaque = true };
|
||||
var scaled = RentRingBuffer(dw * dh * 4);
|
||||
return new VideoFrame(dw, dh, DownscaleBgra(data, sw, sh, srcStride, dw, dh, scaled))
|
||||
{ IsOpaque = true, Epoch = ++_epoch };
|
||||
}
|
||||
|
||||
var pixels = new byte[count];
|
||||
var pixels = RentRingBuffer(count);
|
||||
Marshal.Copy(data, pixels, 0, pixels.Length);
|
||||
return new VideoFrame(sw, sh, pixels) { IsOpaque = true };
|
||||
return new VideoFrame(sw, sh, pixels) { IsOpaque = true, Epoch = ++_epoch };
|
||||
}
|
||||
|
||||
// Bilinear downscale to the master frame. Reads each source row pair through
|
||||
// Marshal.Copy (no unsafe), writing tightly packed BGRA output.
|
||||
private static byte[] DownscaleBgra(IntPtr src, int sw, int sh, int srcStride, int dw, int dh)
|
||||
private byte[] DownscaleBgra(IntPtr src, int sw, int sh, int srcStride, int dw, int dh, byte[] dst)
|
||||
{
|
||||
var row0 = new byte[srcStride];
|
||||
var row1 = new byte[srcStride];
|
||||
var dst = new byte[dw * dh * 4];
|
||||
// Row scratch is per-capture-thread and reused across frames (same churn lesson).
|
||||
var row0 = _row0 != null && _row0.Length >= srcStride ? _row0 : (_row0 = new byte[srcStride]);
|
||||
var row1 = _row1 != null && _row1.Length >= srcStride ? _row1 : (_row1 = new byte[srcStride]);
|
||||
var xs = sw / (double)dw;
|
||||
var ys = sh / (double)dh;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user