perf(capture): fast integer downscale + 10ms cadence floor + reuse-distance ring (slice 16)

The 240Hz monitor delivery + one-in-flight conversions + naive double-per-pixel
DownscaleBgra (~150ms/frame under load) froze the desktop layer 90% of take
ty-1742 (6.1 fresh content updates/s, freeze runs to 2.8s; decoded raw-frame
audit). The render stat (33-36ms) was real but moot — the capture CONVERSION was
the wall, and the one torn frame was a ring slot rewritten under the consumer's
read. Reference: WGC delivers at DWM/monitor cadence
(https://learn.microsoft.com/en-us/windows/apps/develop/media-authoring-processing/screen-capture)
and libyuv row-simple/fixed-point scaling
(https://chromium.googlesource.com/libyuv/libyuv/) — the repo's own take-4 rule.

- DownscaleBgra: integer 8.8 fixed-point, shift-only-at-the-end (same two-stage
  math as SceneCompositor.Bilinear). ~150ms -> ~5ms per 2.5K->1080p frame.
- 10ms MinConvertInterval: the ~4.2ms 240Hz tail stopped queuing ~150ms of
  serialized conversion/s; capacity sits just above the 60/s the pump can use.
- FrameRingBuffer (depth 8, redLine 4): reuse-DISTANCE ring — a buffer is only
  rewritten >=4 rents after its last hand-out else fresh-allocated, so a frame a
  consumer still holds (session.LatestFrame survives conversions, dispatcher
  preview lags) is never read-while-overwritten. Needs no consumer Release API.
- 2s startup.log telemetry: frames/s, conv avg/max ms, skip busy/cadence, ring
  allocs — the device take is judgeable numerically.

Good Dog test: Ring_NoLap_ReusesOnlyAfterRedLineRents. 295/295 green, 0 warnings.
C4 (composite Epoch-cached downscale) deferred pending the device re-measure.
Local only, no push.
This commit is contained in:
2026-09-14 18:18:34 -07:00
parent 76f51e6f4e
commit 71932b9756
6 changed files with 417 additions and 110 deletions
+108 -54
View File
@@ -1,4 +1,5 @@
using System;
using System.Diagnostics;
using System.Runtime.InteropServices;
using System.Runtime.InteropServices.WindowsRuntime;
using System.Threading.Tasks;
@@ -30,17 +31,17 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
private bool _framePending;
private DateTime _lastErrorLog = DateTime.MinValue;
// Buffer recycling (take-11 spikes, 2026-09-04): a fresh ~8.3MB byte[] per DWM
// frame ≈ 500MB/s of LOH churn — the gen2 pauses it forces surfaced as the
// "worst render 35-65ms" spikes that capped fps at ~42 long after the compositor
// itself was fast. A 4-deep ring rotated round-robin is never lapped by a
// ≤17ms consumer at 60Hz; each hand-out carries an Epoch so identity-keyed
// Buffer recycling (take-11 spikes, 2026-09-04; slice 16, 2026-09-14): a fresh
// ~8.3MB byte[] per DWM frame ≈ 500MB/s of LOH churn — the gen2 pauses it forced
// surfaced as the "worst render 35-65ms" spikes that capped fps at ~42. The pool
// is a reuse-distance ring: a buffer is only rewritten after ≥ redLine rents have
// cycled since its last hand-out (FrameRingBuffer), so a frame a consumer still
// holds — session.LatestFrame survives across conversions and the dispatcher
// preview copy lags — can never be overwritten in place (the 1742 tear, a ring
// slot rewritten under the consumer's read, handed the compositor one
// new-top/old-bottom frame). Each hand-out still carries an Epoch so identity-keyed
// consumers (the compositor's paste cache) cannot false-hit a recycled array.
// Depth 8 (take-14 rule): depth × source period must exceed the worst consumer
// hold — 4 slots at 144Hz laps in ~27ms while a compositor read + lagged UI
// preview copy can hold a frame ~50ms; the flash was half a new frame over old.
private readonly byte[]?[] _frameRing = new byte[8][];
private int _ringNext;
private readonly FrameRingBuffer _ring = new(8, redLine: 4);
private long _epoch;
private byte[]? _row0;
private byte[]? _row1;
@@ -53,6 +54,23 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
// A failing conversion must not re-flood the log at frame rate.
private static readonly TimeSpan ErrorLogThrottle = TimeSpan.FromSeconds(5);
// Conversion cadence (slice 16, 2026-09-14): the monitor delivers at the 240Hz DWM
// cadence (~4.2ms) while slots are 16.6ms. A 10ms floor between conversion starts
// keeps the open edge above the ~60 conversions/s the pump can actually use, so the
// 240Hz tail stops chewing a conversion thread that the 1742 take measured at
// ~150ms/frame (a 90%-frozen desktop, ~6 fresh frames/s). Drop counters and the
// rolling conversion stats feed the 2-second startup.log telemetry line.
private static readonly TimeSpan MinConvertInterval = TimeSpan.FromMilliseconds(10);
private static readonly TimeSpan TelemetryInterval = TimeSpan.FromSeconds(2);
private DateTime _lastConvertAt = DateTime.MinValue;
private DateTime _telemetryFrom = DateTime.UtcNow;
private DateTime _lastTelemetry = DateTime.UtcNow;
private long _skippedCadence;
private long _skippedBusy;
private long _conversions;
private long _convertMsTotal;
private long _convertMsMax;
public string Key { get; }
public event Action<VideoFrame>? FrameAvailable;
@@ -121,6 +139,7 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
{
var frame = sender.TryGetNextFrame();
if (frame == null) return;
lock (_gate)
{
if (!_started)
@@ -128,26 +147,39 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
frame.Dispose();
return;
}
}
if (frame.ContentSize.Width != _poolSize.Width || frame.ContentSize.Height != _poolSize.Height)
{
sender.Recreate(Direct3D11Helper.CreateDevice(),
DirectXPixelFormat.B8G8R8A8UIntNormalized, 2, frame.ContentSize);
_poolSize = frame.ContentSize;
}
if (frame.ContentSize.Width != _poolSize.Width || frame.ContentSize.Height != _poolSize.Height)
{
sender.Recreate(Direct3D11Helper.CreateDevice(),
DirectXPixelFormat.B8G8R8A8UIntNormalized, 2, frame.ContentSize);
_poolSize = frame.ContentSize;
}
if (_framePending)
{
frame.Dispose();
return;
// One conversion at a time, spaced by MinConvertInterval (slice 16): the
// 240Hz delivery otherwise queued a conversion every ~4.2ms and the
// 1742 take's ~150ms conversion pinned the desktop layer ~90% frozen.
if (_framePending)
{
_skippedBusy++;
frame.Dispose();
return;
}
var now = DateTime.UtcNow;
if (now - _lastConvertAt < MinConvertInterval)
{
_skippedCadence++;
frame.Dispose();
return;
}
_lastConvertAt = now;
_framePending = true;
}
_framePending = true;
_ = ProcessFrameAsync(frame);
}
private async Task ProcessFrameAsync(Direct3D11CaptureFrame frame)
{
var sw = Stopwatch.StartNew();
try
{
using (frame)
@@ -168,27 +200,38 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
}
finally
{
lock (_gate)
{
EmitTelemetry(sw.ElapsedMilliseconds);
}
_framePending = false;
}
}
private byte[] RentRingBuffer(int size)
private void EmitTelemetry(long conversionMs)
{
for (var tries = 0; tries < _frameRing.Length; tries++)
{
var idx = (_ringNext + tries) % _frameRing.Length;
var buf = _frameRing[idx];
if (buf is { Length: var len } && len == size)
{
_ringNext = (idx + 1) % _frameRing.Length;
return buf;
}
}
var slot = _ringNext;
_ringNext = (slot + 1) % _frameRing.Length;
var fresh = new byte[size];
_frameRing[slot] = fresh;
return fresh;
_conversions++;
_convertMsTotal += conversionMs;
if (conversionMs > _convertMsMax) _convertMsMax = conversionMs;
var now = DateTime.UtcNow;
if (now - _lastTelemetry < TelemetryInterval) return;
var span = now - _telemetryFrom;
var perSecond = span.TotalSeconds > 0 ? _conversions / span.TotalSeconds : 0;
AppLog.Write(
$"ScreenCapture telemetry [{Key}]: {_conversions} frames in {span.TotalSeconds:F1}s " +
$"({perSecond:F0}/s), conv avg {( _conversions == 0 ? 0 : _convertMsTotal / _conversions )}ms " +
$"max {_convertMsMax}ms, skip busy {_skippedBusy} cadence {_skippedCadence}, " +
$"ring allocs {_ring.ConsumeAllocations()}");
_lastTelemetry = now;
_telemetryFrom = now;
_conversions = 0;
_convertMsTotal = 0;
_convertMsMax = 0;
_skippedBusy = 0;
_skippedCadence = 0;
}
private VideoFrame CopyToVideoFrame(SoftwareBitmap bitmap)
@@ -212,47 +255,58 @@ public sealed class ScreenCaptureFrameSource : IScreenCaptureSource
var dw = Math.Max(1, (int)(sw * scale));
var dh = Math.Max(1, (int)(sh * scale));
// DWM delivers an opaque surface (alpha 255); bilinear keeps it 255.
var scaled = RentRingBuffer(dw * dh * 4);
var scaled = _ring.Rent(dw * dh * 4);
return new VideoFrame(dw, dh, DownscaleBgra(data, sw, sh, srcStride, dw, dh, scaled))
{ IsOpaque = true, Epoch = ++_epoch };
}
var pixels = RentRingBuffer(count);
var pixels = _ring.Rent(count);
Marshal.Copy(data, pixels, 0, pixels.Length);
return new VideoFrame(sw, sh, pixels) { IsOpaque = true, Epoch = ++_epoch };
}
// Bilinear downscale to the master frame. Reads each source row pair through
// Marshal.Copy (no unsafe), writing tightly packed BGRA output.
// Integer 8.8 fixed-point bilinear downscale to the master frame (slice 16,
// 2026-09-14: the two-stage math from SceneCompositor.Bilinear — the same "shift
// only at the end" rule from MyMistakes item on fixed-point blending). The previous
// double-per-pixel version ran ~30-45ms per 2.5K→1080p frame on a quiet desktop
// and the 1742 take's stall measured ~150ms/frame under load (90%-frozen desktop,
// ~6 fresh frames/s); integer math keeps the same bilinear result within ±1 while
// dropping the cost to the row-pair Marshal.Copy. Rows are read once per source
// row pair through Marshal.Copy (no unsafe), writing tightly packed BGRA output.
private byte[] DownscaleBgra(IntPtr src, int sw, int sh, int srcStride, int dw, int dh, byte[] dst)
{
// Row scratch is per-capture-thread and reused across frames (same churn lesson).
var row0 = _row0 != null && _row0.Length >= srcStride ? _row0 : (_row0 = new byte[srcStride]);
var row1 = _row1 != null && _row1.Length >= srcStride ? _row1 : (_row1 = new byte[srcStride]);
var xs = sw / (double)dw;
var ys = sh / (double)dh;
for (var y = 0; y < dh; y++)
{
var sy = Math.Min(sh - 1, (int)(y * ys));
var sy8 = (int)((long)y * sh * 256 / dh);
var sy = sy8 >> 8;
var sy1 = Math.Min(sh - 1, sy + 1);
var fy = (y * ys) - sy;
var fy8 = sy8 & 255;
var fyInv = 256 - fy8;
Marshal.Copy(IntPtr.Add(src, sy * srcStride), row0, 0, srcStride);
Marshal.Copy(IntPtr.Add(src, sy1 * srcStride), row1, 0, srcStride);
var dRow = y * dw * 4;
for (var x = 0; x < dw; x++)
{
var sx = Math.Min(sw - 1, (int)(x * xs));
var sx8 = (int)((long)x * sw * 256 / dw);
var sx = Math.Min(sw - 1, sx8 >> 8);
var sx1 = Math.Min(sw - 1, sx + 1);
var fx = (x * xs) - sx;
for (var c = 0; c < 4; c++)
var fx8 = sx8 & 255;
var fxInv = 256 - fx8;
var i0 = sx * 4;
var i1 = sx1 * 4;
var j = dRow + x * 4;
for (var c = 0; c < 4; c++, i0++, i1++, j++)
{
var i0 = sx * 4 + c;
var i1 = sx1 * 4 + c;
var top = row0[i0] + (row0[i1] - row0[i0]) * fx;
var bottom = row1[i0] + (row1[i1] - row1[i0]) * fx;
dst[dRow + x * 4 + c] = (byte)(top + (bottom - top) * fy);
// Two-stage in one 16.8 scale: top/bot ≤ 65280, ×256 + round ≤ 33.5M — int-safe.
var top = row0[i0] * fxInv + row0[i1] * fx8;
var bot = row1[i0] * fxInv + row1[i1] * fx8;
dst[j] = (byte)((top * fyInv + bot * fy8 + 32768) >> 16);
}
}
}