Files
rdplib/plugin/rdpgfx/avc.go
T

2329 lines
87 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package rdpgfx
// AVC420 / AVC444 bitmap stream parsing (MS-RDPEGFX 2.2.4.6 / 2.2.4.7).
import (
"encoding/binary"
"fmt"
"log/slog"
"runtime"
"sync"
"time"
)
type avcRect struct {
left, top, right, bottom uint16
}
type avc420Stream struct {
regions []avcRect
h264Data []byte
}
// fillAVC420Stream parses data into out in-place, reusing out.regions if its
// capacity is sufficient. This avoids a heap allocation for the regions slice
// on every AVC frame when called with a pre-allocated GfxHandler field.
func fillAVC420Stream(data []byte, out *avc420Stream) error {
if len(data) < 4 {
return fmt.Errorf("avc420 stream too short (%d bytes)", len(data))
}
numRegions := binary.LittleEndian.Uint32(data[:4])
if numRegions > 65536 {
return fmt.Errorf("avc420: too many regions: %d", numRegions)
}
// 4 bytes header + 10 bytes per region (8-byte rect + 2-byte quant/quality)
metaSize := 4 + int(numRegions)*10
if metaSize > len(data) {
return fmt.Errorf("avc420: metadata truncated (need %d, have %d)", metaSize, len(data))
}
if cap(out.regions) >= int(numRegions) {
out.regions = out.regions[:numRegions]
} else {
out.regions = make([]avcRect, numRegions)
}
off := 4
for i := range numRegions {
out.regions[i] = avcRect{
left: binary.LittleEndian.Uint16(data[off:]),
top: binary.LittleEndian.Uint16(data[off+2:]),
right: binary.LittleEndian.Uint16(data[off+4:]),
bottom: binary.LittleEndian.Uint16(data[off+6:]),
}
off += 8
}
out.h264Data = data[metaSize:]
return nil
}
// parseAVC420Stream parses RDPGFX_AVC420_BITMAP_STREAM into a new struct.
// Callers that run on the decode goroutine should prefer fillAVC420Stream with
// a pre-allocated GfxHandler field to avoid per-frame heap allocations.
func parseAVC420Stream(data []byte) (*avc420Stream, error) {
var out avc420Stream
if err := fillAVC420Stream(data, &out); err != nil {
return nil, err
}
return &out, nil
}
// parseAVC444Stream parses RDPGFX_AVC444_BITMAP_STREAM.
// Returns the main AVC420 stream, the auxiliary AVC420 stream, and the LC
// (luma-chroma) field.
//
// LC=0: both streams present; stream1 = main (YUV420), stream2 = chroma upgrade.
// LC=1: main stream only; stream2 is nil.
// LC=2: auxiliary only (chroma upgrade); stream1 is nil.
func parseAVC444Stream(data []byte) (stream1, stream2 *avc420Stream, lc uint8, err error) {
if len(data) < 4 {
return nil, nil, 0, fmt.Errorf("avc444 stream too short")
}
cbField := binary.LittleEndian.Uint32(data[:4])
lc = uint8((cbField >> 30) & 0x03)
cbStream1 := int(cbField & 0x3FFFFFFF)
rest := data[4:]
switch lc {
case 0: // Both streams present
if cbStream1 > len(rest) {
return nil, nil, lc, fmt.Errorf("avc444: stream1 size %d exceeds data %d", cbStream1, len(rest))
}
stream1, err = parseAVC420Stream(rest[:cbStream1])
if err != nil {
return nil, nil, lc, err
}
if cbStream1 < len(rest) {
stream2, err = parseAVC420Stream(rest[cbStream1:])
if err != nil {
slog.Debug("RDPGFX: AVC444 stream2 parse error (LC=0)", "err", err)
stream2 = nil
err = nil
}
}
return stream1, stream2, lc, nil
case 1: // Main stream only
streamData := rest
if cbStream1 > 0 && cbStream1 <= len(rest) {
streamData = rest[:cbStream1]
}
stream1, err = parseAVC420Stream(streamData)
return stream1, nil, lc, err
case 2: // Auxiliary only (chroma upgrade)
streamData := rest
if cbStream1 > 0 && cbStream1 <= len(rest) {
streamData = rest[:cbStream1]
}
stream2, err = parseAVC420Stream(streamData)
return nil, stream2, lc, err
default:
return nil, nil, lc, fmt.Errorf("avc444: invalid LC=%d", lc)
}
}
// fillAVC444Stream parses data into g.avcStream1 and g.avcStream2, reusing
// their regions slices to avoid per-frame heap allocations.
// Safe: always called on the single decode goroutine.
func (g *GfxHandler) fillAVC444Stream(data []byte) (stream1, stream2 *avc420Stream, lc uint8, err error) {
if len(data) < 4 {
return nil, nil, 0, fmt.Errorf("avc444 stream too short")
}
cbField := binary.LittleEndian.Uint32(data[:4])
lc = uint8((cbField >> 30) & 0x03)
cbStream1 := int(cbField & 0x3FFFFFFF)
rest := data[4:]
switch lc {
case 0: // Both streams present
if cbStream1 > len(rest) {
return nil, nil, lc, fmt.Errorf("avc444: stream1 size %d exceeds data %d", cbStream1, len(rest))
}
if err = fillAVC420Stream(rest[:cbStream1], &g.avcStream1); err != nil {
return nil, nil, lc, err
}
stream1 = &g.avcStream1
if cbStream1 < len(rest) {
if err2 := fillAVC420Stream(rest[cbStream1:], &g.avcStream2); err2 != nil {
slog.Debug("RDPGFX: AVC444 stream2 parse error (LC=0)", "err", err2)
} else {
stream2 = &g.avcStream2
}
}
return stream1, stream2, lc, nil
case 1: // Main stream only
streamData := rest
if cbStream1 > 0 && cbStream1 <= len(rest) {
streamData = rest[:cbStream1]
}
if err = fillAVC420Stream(streamData, &g.avcStream1); err != nil {
return nil, nil, lc, err
}
return &g.avcStream1, nil, lc, nil
case 2: // Auxiliary only (chroma upgrade)
streamData := rest
if cbStream1 > 0 && cbStream1 <= len(rest) {
streamData = rest[:cbStream1]
}
if err = fillAVC420Stream(streamData, &g.avcStream2); err != nil {
return nil, nil, lc, err
}
return nil, &g.avcStream2, lc, nil
default:
return nil, nil, lc, fmt.Errorf("avc444: invalid LC=%d", lc)
}
}
// avc444YPlane caches the tightly-packed luma plane (stride = Width) from the
// most recently decoded AVC444 main stream. It is used to combine with the
// auxiliary chroma stream when LC=2 frames arrive.
type avc444YPlane struct {
data []byte // luma Y, tight-packed, stride = w
u []byte // Cb (U) plane from stream1, half-res, stride = (w+1)/2
v []byte // Cr (V) plane from stream1, half-res, stride = (w+1)/2
stride int // = w
uvStride int // = (w+1)/2
w, h int
fullRange bool
updatedAt time.Time // last time the cache was refreshed from a live main-stream decode
}
// avc444YStaleness is the maximum age of the Y-plane cache before LC=2
// combines are suppressed. When the main decoder (h264dec) stalls, the
// Y-plane is frozen while incoming LC=2 frames carry fresh chroma — combining
// stale luma with fresh chroma produces wrong colours. 500 ms is well above
// the inter-frame interval at typical RDP frame rates (≥2 fps) yet much lower
// than the 7-second hard stall threshold, so normal operation is unaffected.
const avc444YStaleness = 500 * time.Millisecond
// avcHWStallQueueDepthHint is the queueDepth value reported in
// FRAME_ACKNOWLEDGE PDUs while the HW decoder is stalling (Y cache stale).
// Reporting a depth of 10 signals to the Windows RDP server that the client's
// decode backlog is growing, prompting it to reduce encoding quality and
// bitrate. This reduces the stream of LC=2 frames that accumulate during a
// VideoToolbox null-frame period and gives VT more headroom to flush its
// pipeline. The hint is cleared when the Y cache is refreshed (stall over).
const avcHWStallQueueDepthHint uint32 = 10
// isH264Keyframe returns true when data contains an IDR NAL unit (type 5),
// which marks the start of a new GOP (key frame). The scan handles both
// 3-byte (00 00 01) and 4-byte (00 00 00 01) Annex-B start codes.
func isH264Keyframe(data []byte) bool {
for i := 0; i+4 <= len(data); i++ {
// Look for Annex-B start code: 00 00 01 or 00 00 00 01.
if data[i] == 0x00 && data[i+1] == 0x00 {
var nalByte byte
if data[i+2] == 0x01 && i+3 < len(data) {
nalByte = data[i+3]
i += 2
} else if data[i+2] == 0x00 && i+3 < len(data) && data[i+3] == 0x01 && i+4 < len(data) {
nalByte = data[i+4]
i += 3
} else {
continue
}
nalType := nalByte & 0x1F
if nalType == 5 { // IDR slice
return true
}
}
}
return false
}
// firstNALType returns the NAL unit type byte of the first Annex-B NAL in
// data, or 0xFF if none found. Useful for diagnosing decoder "buffering".
func firstNALType(data []byte) byte {
for i := 0; i+4 <= len(data); i++ {
if data[i] == 0x00 && data[i+1] == 0x00 {
if data[i+2] == 0x01 && i+3 < len(data) {
return data[i+3] & 0x1F
} else if data[i+2] == 0x00 && i+3 < len(data) && data[i+3] == 0x01 && i+4 < len(data) {
return data[i+4] & 0x1F
}
}
}
return 0xFF
}
// decoded frame plus the dirty rectangle list reported in the AVC420 stream
// header (in decoded-frame coordinates). When regions is non-empty callers
// can blit only those regions instead of the whole frame, which dramatically
// reduces per-frame copying for typical desktop video where most of the
// frame is unchanged from the previous frame.
// The pooled return value is true when the returned slice was acquired from
// bitmapBufPool; the caller must then call releaseBitmapBuf on it.
func (g *GfxHandler) decodeAVC420(data []byte, destX, destY, destW, destH int) ([]byte, []avcRect, bool) {
// Parse the stream header once and reuse for both the raw-NAL callback and
// the actual decode path, avoiding a redundant walk of the metadata.
parseErr := fillAVC420Stream(data, &g.avcStream1)
stream := &g.avcStream1
// Compute isKF once — shared between the onH264Raw forwarding path and the
// maybeCacheStream1IDR call below, avoiding two linear scans of the NAL data.
var isKF bool
if parseErr == nil && len(stream.h264Data) > 0 {
isKF = isH264Keyframe(stream.h264Data)
}
if g.onH264Raw != nil && parseErr == nil && len(stream.h264Data) > 0 {
nalData := make([]byte, len(stream.h264Data))
copy(nalData, stream.h264Data)
// AVC420(WTS1)语义:H264 流按整桌面尺寸编码,帧内"有效像素"
// 由元数据 regions 子矩形逐个刻画(实测子矩形之外散布色度清零
// 的绿像素,连 PDU 大矩形内部也不例外——大矩形只是外框)。
// 元数据缺失时回退 PDU 矩形。帧原点 (0,0)。
var regions []int32
if len(stream.regions) > 0 {
for _, r := range stream.regions {
regions = append(regions, int32(r.left), int32(r.top), int32(r.right), int32(r.bottom))
}
} else {
regions = []int32{int32(destX), int32(destY), int32(destX + destW), int32(destY + destH)}
}
g.onH264Raw(0, 0, destX+destW, destY+destH, isKF, nalData, regions)
}
if g.h264dec == nil {
return nil, nil, false
}
if parseErr != nil {
slog.Warn("RDPGFX: AVC420 parse error", "err", parseErr)
return nil, nil, false
}
if len(stream.h264Data) == 0 {
return nil, nil, false
}
if isKF {
g.maybeCacheStream1IDR(stream.h264Data)
}
// For frames where only a small dirty area changed, pass region hints so
// the decoder can skip converting pixels outside those rectangles. This
// is safe here because decodeAVC420 uses blitAndEmitAVCRegions (which only
// reads dirty pixels) when shouldUseAVCRegions returns true.
if rh, ok := g.h264dec.(RegionHinter); ok &&
len(stream.regions) > 0 && shouldUseAVCRegions(stream.regions, destW, destH) {
if cap(g.regionHintBuf) >= len(stream.regions) {
g.regionHintBuf = g.regionHintBuf[:len(stream.regions)]
} else {
g.regionHintBuf = make([][4]uint16, len(stream.regions))
}
for i, r := range stream.regions {
g.regionHintBuf[i] = [4]uint16{r.left, r.top, r.right, r.bottom}
}
rh.SetRegionHint(g.regionHintBuf)
}
frame, err := g.h264dec.Decode(stream.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error", "err", err)
return nil, nil, false
}
if frame == nil {
g.maybeRequestKeyframe()
g.maybeNotifyDecoderBroken()
slog.Debug("RDPGFX: H.264 decode returned nil frame (buffering?)")
return nil, nil, false
}
if frame.Dropped {
slog.Debug("RDPGFX: AVC420 frame intentionally dropped (zero-fill)")
g.trackSWFallbackDroppedFrame()
g.maybeRequestKeyframe()
return nil, nil, false
}
if slog.Default().Enabled(nil, slog.LevelDebug) {
slog.Debug("RDPGFX: AVC420 decoded", "frameW", frame.Width, "frameH", frame.Height, "destW", destW, "destH", destH, "regions", len(stream.regions), "h264Len", len(stream.h264Data))
}
g.noteSuccessfulDecode()
decoded, pooled := cropBGRA(frame.Data, frame.Width, frame.Height, destW, destH)
return decoded, stream.regions, pooled
}
// decodeAVC444 decodes AVC444 bitmap data to BGRA pixels.
// LC=0 and LC=1 decode the main YUV420 stream and cache the luma plane for
// potential LC=2 combine. LC=2 combines the cached luma with the auxiliary
// chroma stream decoded by the secondary decoder.
// The pooled return value is true when the returned slice was acquired from
// bitmapBufPool; the caller must then call releaseBitmapBuf on it.
func (g *GfxHandler) decodeAVC444(data []byte, destX, destY, destW, destH int) ([]byte, []avcRect, bool) {
// Parse the stream header once and reuse for both the raw-NAL callback and
// the actual decode path, avoiding a redundant walk of the metadata.
stream1, stream2, lc, parseErr := g.fillAVC444Stream(data)
// Compute isKF once — shared between the onH264Raw forwarding path and the
// maybeCacheStream1IDR call below, avoiding two linear scans of the NAL data.
var isKF bool
if parseErr == nil && stream1 != nil && len(stream1.h264Data) > 0 {
isKF = isH264Keyframe(stream1.h264Data)
}
if g.onH264Raw != nil && parseErr == nil && stream1 != nil && len(stream1.h264Data) > 0 {
nalData := make([]byte, len(stream1.h264Data))
copy(nalData, stream1.h264Data)
// 同 AVC420:元数据子矩形优先(权威有效范围),缺失回退整表面矩形
var regions []int32
if len(stream1.regions) > 0 {
for _, r := range stream1.regions {
regions = append(regions, int32(r.left), int32(r.top), int32(r.right), int32(r.bottom))
}
} else {
regions = []int32{int32(destX), int32(destY), int32(destX + destW), int32(destY + destH)}
}
g.onH264Raw(0, 0, destW, destH, isKF, nalData, regions)
}
if g.h264dec == nil {
return nil, nil, false
}
if parseErr != nil {
slog.Warn("RDPGFX: AVC444 parse error", "err", parseErr)
return nil, nil, false
}
if lc == 2 {
return g.decodeAVC444LC2(stream2, destW, destH)
}
if stream1 == nil || len(stream1.h264Data) == 0 {
return nil, nil, false
}
// Pass region hints so the decoder skips converting pixels outside the
// dirty rectangles. Safe here because decodeAVC444 also uses
// blitAndEmitAVCRegions when shouldUseAVCRegions returns true.
if rh, ok := g.h264dec.(RegionHinter); ok &&
len(stream1.regions) > 0 && shouldUseAVCRegions(stream1.regions, destW, destH) {
if cap(g.regionHintBuf) >= len(stream1.regions) {
g.regionHintBuf = g.regionHintBuf[:len(stream1.regions)]
} else {
g.regionHintBuf = make([][4]uint16, len(stream1.regions))
}
for i, r := range stream1.regions {
g.regionHintBuf[i] = [4]uint16{r.left, r.top, r.right, r.bottom}
}
rh.SetRegionHint(g.regionHintBuf)
}
var frame *H264Frame
var i420out *H264FrameI420
var err error
isKeyFrame := isKF
if isKeyFrame {
// Cache IDR NAL data so the SW fallback decoder can be primed
// immediately after a VideoToolbox stall without waiting for VBox.
g.maybeCacheStream1IDR(stream1.h264Data)
}
isIDR := g.h264dec2 != nil && isKeyFrame
if isIDR {
// Reset per-GOP diagnostic flags so the LC=0 IDR and LC=2 combine
// after this IDR are sampled again for colour diagnostics.
g.lc2SampleLogged = false
g.lc2PFrameSampleLogged = false
g.lc0SampleLogged = false
}
if g.h264dec2 != nil {
// Cache luma for future LC=2 combine.
if i420dec, ok := g.h264dec.(I420Decoder); ok {
frame, i420out, err = i420dec.DecodeWithI420(stream1.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error (AVC444)", "err", err)
return nil, nil, false
}
if i420out != nil {
g.updateAVC444YCache(i420out)
if isIDR {
// Snapshot the IDR luma separately. When a standalone
// LC=2 packet carries a stream2 IDR, the chroma data
// belongs to this GOP's first frame, so we must combine
// it with the IDR luma — not with a later P-frame's luma
// that has since overwritten avc444YPlane.
g.copyAVC444YToIDRCache()
}
}
} else {
frame, err = g.h264dec.Decode(stream1.h264Data)
}
} else {
frame, err = g.h264dec.Decode(stream1.h264Data)
}
if err != nil {
slog.Warn("RDPGFX: H.264 decode error (AVC444)", "err", err)
return nil, nil, false
}
// Prime the aux decoder before any nil/drop checks so the stream2 IDR is
// never lost. On macOS, VideoToolbox returns nil frames for 1–3 s during
// initial warm-up; without this early call the stream2 IDR carried by the
// first LC=0 packet would be discarded (h264dec2 never created) and the
// renegotiation timer would degrade LC=2 to LC=0-only after retrying.
if lc == 0 && stream2 != nil && len(stream2.h264Data) > 0 {
g.primeAuxDecoder(stream2.h264Data)
}
if frame == nil {
if i420out == nil {
g.maybeRequestKeyframe()
g.maybeNotifyDecoderBroken()
return nil, nil, false
}
// I420 fast path: HW decoder returned planar I420 instead of BGRA.
// Convert to BGRA using BT.709 (AVC444 standard encoding) so the
// BGRA rendering path can continue normally.
bgra, _ := i420ToBGRA(i420out)
if bgra == nil {
return nil, nil, false
}
frame = &H264Frame{Data: bgra, Width: i420out.Width, Height: i420out.Height}
}
if frame.Dropped {
slog.Debug("RDPGFX: AVC444 frame intentionally dropped (zero-fill)")
g.trackSWFallbackDroppedFrame()
g.maybeRequestKeyframe()
// Touch Y cache timestamp so LC=2 can still combine with last valid luma
// while the server delivers a recovery IDR.
if g.avc444YPlane.w > 0 && !g.avc444YPlane.updatedAt.IsZero() {
g.avc444YPlane.updatedAt = time.Now()
}
return nil, nil, false
}
if !g.lc0SampleLogged && isIDR {
g.lc0SampleLogged = true
bgraData := frame.Data
w, h := frame.Width, frame.Height
for _, p := range [][2]int{{960, 400}, {480, 400}, {1440, 400}, {960, 600}, {100, 100}} {
px, py := p[0], p[1]
if px >= w || py >= h {
continue
}
off := (py*w + px) * 4
if off+3 < len(bgraData) {
var rawY, rawU, rawV byte
if i420out != nil && py < i420out.Height && px < i420out.Width {
rawY = i420out.Y[py*i420out.YStride+px]
rawU = i420out.U[(py/2)*i420out.UStride+(px/2)]
rawV = i420out.V[(py/2)*i420out.VStride+(px/2)]
}
slog.Debug("H.264: pixel sample (LC=0 IDR frame)",
"x", px, "y", py,
"rawY", rawY, "rawU", rawU, "rawV", rawV,
"fullRange", i420out != nil && i420out.FullRange,
"B", bgraData[off], "G", bgraData[off+1], "R", bgraData[off+2])
}
}
}
if slog.Default().Enabled(nil, slog.LevelDebug) {
slog.Debug("RDPGFX: AVC444 decoded", "frameW", frame.Width, "frameH", frame.Height,
"destW", destW, "destH", destH, "h264Len", len(stream1.h264Data))
}
g.noteSuccessfulDecode()
decoded, pooled := cropBGRA(frame.Data, frame.Width, frame.Height, destW, destH)
return decoded, stream1.regions, pooled
}
// decodeAVC420WithI420 decodes AVC420 bitmap data, returning BGRA pixels for
// the surface backing store and, when the underlying decoder supports I420
// output, an optional H264FrameI420 for GPU-accelerated IYUV texture upload.
// i420 is nil when I420 extraction is unsupported or the frame dimensions are
// smaller than destW×destH. Callers must fall back to BGRA rendering when
// i420 is nil.
func (g *GfxHandler) decodeAVC420WithI420(data []byte, destX, destY, destW, destH int) (decoded []byte, i420 *H264FrameI420, regions []avcRect, pooled bool) {
if err := fillAVC420Stream(data, &g.avcStream1); err != nil {
slog.Warn("RDPGFX: AVC420 parse error", "err", err)
return
}
stream := &g.avcStream1
if g.onH264Raw != nil && len(stream.h264Data) > 0 {
isKF := isH264Keyframe(stream.h264Data)
nalData := make([]byte, len(stream.h264Data))
copy(nalData, stream.h264Data)
g.onH264Raw(destX, destY, destW, destH, isKF, nalData, nil)
}
if g.h264dec == nil || len(stream.h264Data) == 0 {
return
}
if isH264Keyframe(stream.h264Data) {
g.maybeCacheStream1IDR(stream.h264Data)
}
var frame *H264Frame
var err error
i420dec, hasI420 := g.h264dec.(I420Decoder)
if hasI420 {
var i420out *H264FrameI420
frame, i420out, err = i420dec.DecodeWithI420(stream.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error", "err", err)
return
}
if i420out != nil && i420out.Width >= destW && i420out.Height >= destH {
i420 = i420out
}
} else {
frame, err = g.h264dec.Decode(stream.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error", "err", err)
return
}
}
// I420 fast path: frame is nil but i420 is non-nil — decoder produced output
// via the direct NV12/YUV420P copy path. Still counts as a successful decode.
if frame == nil && i420 == nil {
g.maybeRequestKeyframe()
g.maybeNotifyDecoderBroken()
slog.Debug("RDPGFX: H.264 decode returned nil frame (buffering?)")
return
}
if frame != nil && frame.Dropped {
slog.Debug("RDPGFX: AVC420 (WithI420) frame intentionally dropped (zero-fill)")
g.trackSWFallbackDroppedFrame()
g.maybeRequestKeyframe()
return
}
g.noteSuccessfulDecode()
if frame != nil {
if slog.Default().Enabled(nil, slog.LevelDebug) {
slog.Debug("RDPGFX: AVC420 decoded (WithI420)", "frameW", frame.Width, "frameH", frame.Height,
"destW", destW, "destH", destH, "hasI420", i420 != nil,
"regions", len(stream.regions), "h264Len", len(stream.h264Data))
}
decoded, pooled = cropBGRA(frame.Data, frame.Width, frame.Height, destW, destH)
}
regions = stream.regions
return
}
// decodeAVC420WithNV12 decodes AVC420 bitmap data, returning native NV12
// planes when the underlying decoder produces NV12 (typically VideoToolbox).
// If NV12 is unavailable, decoded may contain a BGRA fallback frame.
func (g *GfxHandler) decodeAVC420WithNV12(data []byte, destX, destY, destW, destH int) (decoded []byte, nv12 *H264FrameNV12, regions []avcRect, pooled bool) {
if err := fillAVC420Stream(data, &g.avcStream1); err != nil {
slog.Warn("RDPGFX: AVC420 parse error", "err", err)
return
}
stream := &g.avcStream1
// Compute isKF once — shared between the onH264Raw forwarding path and the
// maybeCacheStream1IDR call below, avoiding two linear scans of the NAL data.
var isKF bool
if len(stream.h264Data) > 0 {
isKF = isH264Keyframe(stream.h264Data)
}
if g.onH264Raw != nil && len(stream.h264Data) > 0 {
nalData := make([]byte, len(stream.h264Data))
copy(nalData, stream.h264Data)
g.onH264Raw(destX, destY, destW, destH, isKF, nalData, nil)
}
if g.h264dec == nil || len(stream.h264Data) == 0 {
return
}
if isKF {
g.maybeCacheStream1IDR(stream.h264Data)
}
var frame *H264Frame
var err error
nv12dec, hasNV12 := g.h264dec.(NV12Decoder)
if hasNV12 {
var nv12out *H264FrameNV12
frame, nv12out, err = nv12dec.DecodeWithNV12(stream.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error", "err", err)
return
}
if nv12out != nil && nv12out.Width >= destW && nv12out.Height >= destH {
nv12 = nv12out
}
} else {
frame, err = g.h264dec.Decode(stream.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error", "err", err)
return
}
}
if frame == nil && nv12 == nil {
g.maybeRequestKeyframe()
g.maybeNotifyDecoderBroken()
slog.Debug("RDPGFX: H.264 decode returned nil frame (buffering?)")
return
}
if frame != nil && frame.Dropped {
slog.Debug("RDPGFX: AVC420 (WithNV12) frame intentionally dropped (zero-fill)")
g.trackSWFallbackDroppedFrame()
g.maybeRequestKeyframe()
return
}
g.noteSuccessfulDecode()
if frame != nil {
if slog.Default().Enabled(nil, slog.LevelDebug) {
slog.Debug("RDPGFX: AVC420 decoded (WithNV12)", "frameW", frame.Width, "frameH", frame.Height,
"destW", destW, "destH", destH, "hasNV12", nv12 != nil,
"regions", len(stream.regions), "h264Len", len(stream.h264Data))
}
decoded, pooled = cropBGRA(frame.Data, frame.Width, frame.Height, destW, destH)
}
regions = stream.regions
return
}
// decodeAVC444WithI420 decodes AVC444 bitmap data, returning BGRA pixels and
// an optional I420 frame. LC=0 and LC=1 decode the main stream and cache the
// luma plane. LC=2 decodes the auxiliary chroma stream and combines it with
// the cached luma to produce BGRA; i420 is nil for LC=2 frames (GPU path falls
// back to BGRA).
func (g *GfxHandler) decodeAVC444WithI420(data []byte, destX, destY, destW, destH int) (decoded []byte, i420 *H264FrameI420, regions []avcRect, pooled bool) {
stream1, stream2, lc, err := g.fillAVC444Stream(data)
// Compute isKF once — shared between the onH264Raw forwarding path and the
// maybeCacheStream1IDR call below, avoiding two linear scans of the NAL data.
var isKF bool
if stream1 != nil && len(stream1.h264Data) > 0 {
isKF = isH264Keyframe(stream1.h264Data)
}
if g.onH264Raw != nil && stream1 != nil && len(stream1.h264Data) > 0 {
nalData := make([]byte, len(stream1.h264Data))
copy(nalData, stream1.h264Data)
g.onH264Raw(destX, destY, destW, destH, isKF, nalData, nil)
}
if err != nil {
slog.Warn("RDPGFX: AVC444 parse error", "err", err)
return
}
if lc == 2 {
decoded, regions, pooled = g.decodeAVC444LC2(stream2, destW, destH)
return
}
if g.h264dec == nil || stream1 == nil || len(stream1.h264Data) == 0 {
return
}
if isKF {
g.maybeCacheStream1IDR(stream1.h264Data)
}
var frame *H264Frame
i420dec, hasI420 := g.h264dec.(I420Decoder)
if hasI420 {
var i420out *H264FrameI420
frame, i420out, err = i420dec.DecodeWithI420(stream1.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error (AVC444)", "err", err)
return
}
if i420out != nil {
if g.h264dec2 != nil {
g.updateAVC444YCache(i420out)
}
if i420out.Width >= destW && i420out.Height >= destH {
i420 = i420out
}
}
} else {
frame, err = g.h264dec.Decode(stream1.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error (AVC444)", "err", err)
return
}
}
// Prime the aux decoder before checking frame.Dropped: stream2 IDR data
// must not be lost when the main frame is discarded due to zero-fill.
if lc == 0 && stream2 != nil && len(stream2.h264Data) > 0 {
g.primeAuxDecoder(stream2.h264Data)
}
// I420 fast path: frame is nil but i420 is non-nil — decoder produced output
// via the direct NV12/YUV420P copy path. Still counts as a successful decode.
if frame == nil && i420 == nil {
g.maybeRequestKeyframe()
g.maybeNotifyDecoderBroken()
return
}
if frame != nil && frame.Dropped {
slog.Debug("RDPGFX: AVC444 (WithI420) frame intentionally dropped (zero-fill)")
g.trackSWFallbackDroppedFrame()
g.maybeRequestKeyframe()
if g.avc444YPlane.w > 0 && !g.avc444YPlane.updatedAt.IsZero() {
g.avc444YPlane.updatedAt = time.Now()
}
return
}
g.noteSuccessfulDecode()
if frame != nil {
if slog.Default().Enabled(nil, slog.LevelDebug) {
slog.Debug("RDPGFX: AVC444 decoded (WithI420)", "frameW", frame.Width, "frameH", frame.Height,
"destW", destW, "destH", destH, "hasI420", i420 != nil, "h264Len", len(stream1.h264Data))
}
decoded, pooled = cropBGRA(frame.Data, frame.Width, frame.Height, destW, destH)
regions = stream1.regions
}
return
}
// decodeAVC444WithNV12 decodes AVC444 bitmap data, returning native NV12 planes
// for LC=0/LC=1 frames and BGRA for LC=2 chroma-combination frames.
// nv12 is non-nil only when the hardware decoder (VideoToolbox) produced NV12
// output for a LC=0/LC=1 stream1 packet. LC=2 frames always return decoded
// BGRA with nv12==nil because the chroma supplement requires a CPU combine step.
func (g *GfxHandler) decodeAVC444WithNV12(data []byte, destX, destY, destW, destH int) (decoded []byte, nv12 *H264FrameNV12, regions []avcRect, pooled bool) {
stream1, stream2, lc, err := g.fillAVC444Stream(data)
var isKF bool
if stream1 != nil && len(stream1.h264Data) > 0 {
isKF = isH264Keyframe(stream1.h264Data)
}
if g.onH264Raw != nil && stream1 != nil && len(stream1.h264Data) > 0 {
nalData := make([]byte, len(stream1.h264Data))
copy(nalData, stream1.h264Data)
g.onH264Raw(destX, destY, destW, destH, isKF, nalData, nil)
}
if err != nil {
slog.Warn("RDPGFX: AVC444 parse error", "err", err)
return
}
if lc == 2 {
decoded, regions, pooled = g.decodeAVC444LC2(stream2, destW, destH)
return
}
if g.h264dec == nil || stream1 == nil || len(stream1.h264Data) == 0 {
return
}
if isKF {
g.maybeCacheStream1IDR(stream1.h264Data)
}
isIDR := g.h264dec2 != nil && isKF
if isIDR {
g.lc2SampleLogged = false
g.lc2PFrameSampleLogged = false
g.lc0SampleLogged = false
}
var frame *H264Frame
nv12dec, hasNV12 := g.h264dec.(NV12Decoder)
if hasNV12 {
var nv12out *H264FrameNV12
frame, nv12out, err = nv12dec.DecodeWithNV12(stream1.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error (AVC444 WithNV12)", "err", err)
return
}
if nv12out != nil && nv12out.Width >= destW && nv12out.Height >= destH {
nv12 = nv12out
if g.h264dec2 != nil {
g.updateAVC444YCacheFromNV12(nv12out)
if isIDR {
g.copyAVC444YToIDRCache()
}
}
}
} else {
frame, err = g.h264dec.Decode(stream1.h264Data)
if err != nil {
slog.Warn("RDPGFX: H.264 decode error (AVC444 WithNV12 fallback)", "err", err)
return
}
}
// When the NV12 decoder returned a BGRA frame but no NV12 planes (e.g. the
// software decoder produced YUV420P instead of NV12), recover I420 from the
// decoder's side channel for Y cache update so LC=2 frames are not stalled.
if hasNV12 && nv12 == nil && frame != nil && g.h264dec2 != nil {
type i420LastDecoder interface {
LastI420() *H264FrameI420
}
if p, ok := g.h264dec.(i420LastDecoder); ok {
if i420out := p.LastI420(); i420out != nil {
g.updateAVC444YCache(i420out)
if isIDR {
g.copyAVC444YToIDRCache()
}
}
}
}
// Prime aux decoder before nil/dropped checks so stream2 IDR data is never lost.
if lc == 0 && stream2 != nil && len(stream2.h264Data) > 0 {
g.primeAuxDecoder(stream2.h264Data)
}
if frame == nil && nv12 == nil {
g.maybeRequestKeyframe()
g.maybeNotifyDecoderBroken()
slog.Debug("RDPGFX: H.264 decode returned nil frame/nv12 (buffering?)")
return
}
if frame != nil && frame.Dropped {
slog.Debug("RDPGFX: AVC444 (WithNV12) frame intentionally dropped (zero-fill)")
g.trackSWFallbackDroppedFrame()
g.maybeRequestKeyframe()
if g.avc444YPlane.w > 0 && !g.avc444YPlane.updatedAt.IsZero() {
g.avc444YPlane.updatedAt = time.Now()
}
return
}
g.noteSuccessfulDecode()
if frame != nil {
if slog.Default().Enabled(nil, slog.LevelDebug) {
slog.Debug("RDPGFX: AVC444 decoded (WithNV12)", "frameW", frame.Width, "frameH", frame.Height,
"destW", destW, "destH", destH, "hasNV12", nv12 != nil, "h264Len", len(stream1.h264Data))
}
decoded, pooled = cropBGRA(frame.Data, frame.Width, frame.Height, destW, destH)
}
regions = stream1.regions
return
}
// updateAVC444YCache copies the Y, U, and V planes from stream1's i420 into
// g.avc444YPlane for use when combining with an LC=2 auxiliary chroma frame.
// The U/V planes are stored half-res (stride = (w+1)/2) and provide the B2/B3
// chroma values (even column, even row positions) that stream2 does not cover.
func (g *GfxHandler) updateAVC444YCache(i420 *H264FrameI420) {
w, h := i420.Width, i420.Height
uvStride := (w + 1) / 2
uvH := (h + 1) / 2
neededY := w * h
neededUV := uvStride * uvH
if cap(g.avc444YPlane.data) < neededY {
g.avc444YPlane.data = make([]byte, neededY)
} else {
g.avc444YPlane.data = g.avc444YPlane.data[:neededY]
}
if cap(g.avc444YPlane.u) < neededUV {
g.avc444YPlane.u = make([]byte, neededUV)
} else {
g.avc444YPlane.u = g.avc444YPlane.u[:neededUV]
}
if cap(g.avc444YPlane.v) < neededUV {
g.avc444YPlane.v = make([]byte, neededUV)
} else {
g.avc444YPlane.v = g.avc444YPlane.v[:neededUV]
}
// i420 planes are already tight-packed (strides == width/height from extractI420fromSrc).
// Run Y, U, V copies in parallel for large frames: each slice is an independent
// allocation so there is no aliasing between the goroutines' writes.
totalBytes := neededY + neededUV*2
if totalBytes >= parallelConvertMinPixels*4 {
var wg sync.WaitGroup
wg.Add(3)
go func() { defer wg.Done(); copy(g.avc444YPlane.data, i420.Y) }()
go func() { defer wg.Done(); copy(g.avc444YPlane.u, i420.U) }()
go func() { defer wg.Done(); copy(g.avc444YPlane.v, i420.V) }()
wg.Wait()
} else {
copy(g.avc444YPlane.data, i420.Y)
copy(g.avc444YPlane.u, i420.U)
copy(g.avc444YPlane.v, i420.V)
}
g.avc444YPlane.stride = w
g.avc444YPlane.uvStride = uvStride
g.avc444YPlane.w = w
g.avc444YPlane.h = h
g.avc444YPlane.fullRange = i420.FullRange
g.avc444YPlane.updatedAt = time.Now()
// HW decoder is producing real frames again — clear any stall throttle so
// the server resumes its normal quality/bitrate.
g.SetQueueDepthHint(0)
}
// updateAVC444YCacheFromNV12 updates the AVC444 Y/UV cache from a native NV12
// frame produced by the hardware decoder (VideoToolbox on macOS). The NV12
// interleaved UV plane is de-interleaved into separate U/V planes so that the
// cache layout matches what combineAVC444v2BGRA expects.
//
// fullRange is forced to false regardless of nv12.FullRange. VideoToolbox
// expands the H.264 limited-range chroma to full range when it outputs
// kCVPixelFormatType_420YpCbCr8BiPlanarFullRange, but the SDL2 Metal renderer
// applies limited-range BT.709 coefficients to the NV12 texture (SDL2 auto-
// selects BT.709 for HD resolutions). The LC=2 auxiliary-chroma stream is
// decoded by the SW decoder and keeps limited-range values. By always using
// fullRange=false here, combineAVC444v2BGRA applies the same limited-range
// BT.709 formula that SDL2 uses for the NV12 texture, eliminating the colour
// shift that was visible when the display alternated between LC=0 NV12 frames
// (SDL2 limited BT.709) and LC=2 BGRA overlay frames (previously BT.709 full-
// range).
func (g *GfxHandler) updateAVC444YCacheFromNV12(nv12 *H264FrameNV12) {
w, h := nv12.Width, nv12.Height
uvStride := (w + 1) / 2
uvH := (h + 1) / 2
neededY := w * h
neededUV := uvStride * uvH
if cap(g.avc444YPlane.data) < neededY {
g.avc444YPlane.data = make([]byte, neededY)
} else {
g.avc444YPlane.data = g.avc444YPlane.data[:neededY]
}
if cap(g.avc444YPlane.u) < neededUV {
g.avc444YPlane.u = make([]byte, neededUV)
} else {
g.avc444YPlane.u = g.avc444YPlane.u[:neededUV]
}
if cap(g.avc444YPlane.v) < neededUV {
g.avc444YPlane.v = make([]byte, neededUV)
} else {
g.avc444YPlane.v = g.avc444YPlane.v[:neededUV]
}
// Copy Y rows, respecting the source stride.
srcYStride := nv12.YStride
if srcYStride <= 0 {
srcYStride = w
}
for row := range h {
copy(g.avc444YPlane.data[row*w:row*w+w], nv12.Y[row*srcYStride:row*srcYStride+w])
}
// De-interleave NV12 UV (interleaved UVUVUV…) into separate U/V planes.
srcUVStride := nv12.UVStride
if srcUVStride <= 0 {
srcUVStride = w
}
for row := range uvH {
srcRow := nv12.UV[row*srcUVStride : row*srcUVStride+w]
dstU := g.avc444YPlane.u[row*uvStride : row*uvStride+uvStride]
dstV := g.avc444YPlane.v[row*uvStride : row*uvStride+uvStride]
for col := range uvStride {
dstU[col] = srcRow[col*2]
dstV[col] = srcRow[col*2+1]
}
}
g.avc444YPlane.stride = w
g.avc444YPlane.uvStride = uvStride
g.avc444YPlane.w = w
g.avc444YPlane.h = h
// Force limited-range so combineAVC444v2BGRA uses the same BT.709 limited-
// range coefficients as the SDL2 Metal renderer (see comment above).
g.avc444YPlane.fullRange = false
g.avc444YPlane.updatedAt = time.Now()
g.SetQueueDepthHint(0)
}
// copyAVC444YToIDRCache copies the current avc444YPlane content into
// avc444IDRYPlane. Called immediately after updating avc444YPlane from a
// stream1 IDR decode, so the IDR luma snapshot stays separate from any
// subsequent P-frame luma updates.
func (g *GfxHandler) copyAVC444YToIDRCache() {
src := &g.avc444YPlane
dst := &g.avc444IDRYPlane
if cap(dst.data) < len(src.data) {
dst.data = make([]byte, len(src.data))
} else {
dst.data = dst.data[:len(src.data)]
}
if cap(dst.u) < len(src.u) {
dst.u = make([]byte, len(src.u))
} else {
dst.u = dst.u[:len(src.u)]
}
if cap(dst.v) < len(src.v) {
dst.v = make([]byte, len(src.v))
} else {
dst.v = dst.v[:len(src.v)]
}
// Parallel copy for large frames: each slice is a separate allocation.
totalBytes := len(src.data) + len(src.u) + len(src.v)
if totalBytes >= parallelConvertMinPixels*4 {
var wg sync.WaitGroup
wg.Add(3)
go func() { defer wg.Done(); copy(dst.data, src.data) }()
go func() { defer wg.Done(); copy(dst.u, src.u) }()
go func() { defer wg.Done(); copy(dst.v, src.v) }()
wg.Wait()
} else {
copy(dst.data, src.data)
copy(dst.u, src.u)
copy(dst.v, src.v)
}
dst.stride = src.stride
dst.uvStride = src.uvStride
dst.w = src.w
dst.h = src.h
dst.fullRange = src.fullRange
dst.updatedAt = src.updatedAt
}
// maybeCacheStream1IDR stores h264Data as the latest stream1 IDR NAL data for
// later use when priming the SW fallback decoder after a VideoToolbox stall.
// The data is copied so the caller's buffer may be reused freely.
// Only call when isH264Keyframe(h264Data) is true.
func (g *GfxHandler) maybeCacheStream1IDR(h264Data []byte) {
g.lastStream1IDR = append(g.lastStream1IDR[:0], h264Data...)
g.lastStream1IDRTime = time.Now()
g.lastStream1IDRFrame = g.framesDecoded.Load()
if g.usingSWFallback {
// A natural IDR from the server arrived while in SW fallback mode.
// From this frame onwards the SW decoder has a fresh reference point and
// error concealment (block noise) should stop. This is the genuine
// resync point after a stale-IDR prime, so disarm the stale-prime
// corruption tracking: the primed frames were only suspect until a real
// IDR healed the picture.
g.swFallbackPrimed = false
g.swFallbackDroppedCount = 0
slog.Debug("H.264: natural IDR received during SW fallback — block noise should stop",
"idrLen", len(h264Data),
"framesDecoded", g.lastStream1IDRFrame)
} else {
slog.Debug("H.264: stream1 IDR cached for SW fallback priming",
"idrLen", len(h264Data),
"framesDecoded", g.lastStream1IDRFrame)
}
}
// isPlaneRegionBlank samples a 3×3 grid inside a rectangular region of a
// single plane and returns true when a majority of the samples are either
// near-zero (< loThreshold) or near-saturated (>= hiThreshold). It is used
// to detect the uninitialised/corrupt chroma states that produce green or
// pink overlays in AVC444v2 reconstruction.
func isPlaneRegionBlank(data []byte, stride, x0, y0, w, h int) bool {
if len(data) == 0 || w <= 0 || h <= 0 {
return false
}
const (
loThreshold = 72 // below this: abnormally low (green monochrome)
hiThreshold = 235 // at or above this: near-saturation (pink overlay)
)
nearZero, nearSat, total := 0, 0, 0
for i := range 3 {
row := y0 + (i+1)*h/4
if row < y0 || row >= y0+h {
continue
}
for j := range 3 {
col := x0 + (j+1)*w/4
if col < x0 || col >= x0+w {
continue
}
total++
v := data[row*stride+col]
if v < loThreshold {
nearZero++
} else if v >= hiThreshold {
nearSat++
}
}
}
if total == 0 {
return false
}
return nearZero*2 > total || nearSat*2 > total
}
// isAuxChromaBlank returns true when any of the chroma-carrying planes in the
// stream2 auxiliary frame looks uninitialised or corrupt. In AVC444v2 the Y
// plane carries Cb (left half) and Cr (right half) for odd columns, while the
// U and V planes carry the remaining chroma positions for even columns on odd
// rows. Near-zero values in any of these planes produce a bright green frame;
// near-saturated values produce a pink/magenta overlay. Detecting this early
// lets decodeAVC444LC2 skip the combine and wait for real data.
//
// Two failure modes are detected:
// - Near-zero (< 20): codec not yet initialised; Windows Server initialises
// stream2 IDR with Cb≈0, Cr≈0 and sometimes emits sparse artefact pixels
// (Cb≈9–12) that a threshold of 8 would pass. Raising to 20 keeps those
// from triggering a combine that produces bright green blocks.
// - Near-saturation (≥ 235): indicates DPB mismatch or corruption in the aux
// decoder (h264dec2); a P-frame decoded against the wrong reference can
// produce near-maximal values, which encode Cb≈255/Cr≈255 and result in a
// pink/magenta overlay when combined with any luma.
func isAuxChromaBlank(f *H264FrameI420) bool {
if f == nil || f.Width < 16 || f.Height < 4 || len(f.Y) == 0 || len(f.U) == 0 || len(f.V) == 0 {
return false
}
w, h := f.Width, f.Height
halfW := w / 2
uvW := halfW / 2
uvH := (h + 1) / 2
// Y plane left half: Cb for odd columns.
if isPlaneRegionBlank(f.Y, f.YStride, 0, 0, halfW, h) {
return true
}
// Y plane right half: Cr for odd columns.
if isPlaneRegionBlank(f.Y, f.YStride, halfW, 0, halfW, h) {
return true
}
// U plane: Cb/Cr for even columns on odd rows.
if isPlaneRegionBlank(f.U, f.UStride, 0, 0, uvW, uvH) {
return true
}
// V plane: Cb/Cr for even columns on odd rows.
if isPlaneRegionBlank(f.V, f.VStride, 0, 0, uvW, uvH) {
return true
}
return false
}
// isAVC444YPlaneChromaBlank returns true when the cached stream1 chroma (U/V)
// looks corrupt. Green monochrome requires both Cb and Cr to collapse to
// near-zero, so the grid check requires both U and V at a sample point to be
// low. Near-saturation in both planes produces a pink/magenta overlay.
//
// The threshold (72) matches the low-chroma guard in the ffmpeg decoder plugin
// so a frame that poisoned the cache would also have been dropped there.
func isAVC444YPlaneChromaBlank(yp *avc444YPlane) bool {
if yp == nil || yp.w < 16 || yp.h < 4 || len(yp.u) == 0 || len(yp.v) == 0 {
return false
}
uvH := (yp.h + 1) / 2
const (
loThreshold = 72 // below this: near-zero (green monochrome)
hiThreshold = 235 // at or above this: near-saturation (pink overlay)
)
nearZero, nearSat, total := 0, 0, 0
for i := range 3 {
row := (i + 1) * uvH / 4
if row >= uvH {
continue
}
for j := range 3 {
col := (j + 1) * yp.uvStride / 4
if col >= yp.uvStride {
continue
}
total++
u := yp.u[row*yp.uvStride+col]
v := yp.v[row*yp.uvStride+col]
if u < loThreshold && v < loThreshold {
nearZero++
} else if u >= hiThreshold && v >= hiThreshold {
nearSat++
}
}
}
if total == 0 {
return false
}
return nearZero*2 > total || nearSat*2 > total
}
// combineAVC444v2BGRA implements the AVC444v2 chroma reconstruction defined in
// [MS-RDPEGFX 3.3.8.3.3] ("YUV420p Stream Combination for YUV444v2 mode").
//
// Stream2 encodes the missing chroma positions that stream1's 4:2:0 quantiser
// discards, split across three "Bx areas" of the auxiliary I420 frame:
//
// B4/B5 — stream2 Y plane, each row:
// bytes [0, w/2) = Cb at all odd-x columns (U444[2k+1, y] for k=0..w/2-1)
// bytes [w/2, w) = Cr at all odd-x columns (V444[2k+1, y] for k=0..w/2-1)
//
// B6/B7 — stream2 U plane, each half-height row j:
// bytes [0, w/4) = Cb at even-x multiples of 4 (U444[4k, 2j+1])
// bytes [w/4, w/2) = Cr at even-x multiples of 4 (V444[4k, 2j+1])
//
// B8/B9 — stream2 V plane, each half-height row j:
// bytes [0, w/4) = Cb at even-x offset-2 cols (U444[4k+2, 2j+1])
// bytes [w/4, w/2) = Cr at even-x offset-2 cols (V444[4k+2, 2j+1])
//
// Positions not covered by stream2 (even-x, even-y) use stream1's half-res
// B2/B3 chroma values from the cached cachedU/cachedV planes.
//
// Parameters:
//
// yPlane/yStride – luma Y from stream1, tight-packed (stride=w)
// cachedU/cachedV – Cb/Cr from stream1, half-res (stride=uvStride=(w+1)/2)
// i420aux – I420 output from decoding stream2
// fullRange – true for PC-range [0-255], false for video [16-235]
// maxConvertWorkers caps the number of goroutines used to parallelise the
// per-row YCbCr→BGRA conversions. Beyond ~8 workers the conversion is limited
// by memory bandwidth rather than CPU, so additional workers only add
// scheduling overhead without speeding up the conversion.
const maxConvertWorkers = 8
// parallelConvertMinPixels is the frame-area threshold below which conversion
// runs serially: for small frames the goroutine spawn/join overhead exceeds the
// work saved by splitting the rows across cores.
const parallelConvertMinPixels = 256 * 256
// parallelRows splits the row range [0,h) into up to maxConvertWorkers
// contiguous chunks and runs fn(y0,y1) for each chunk concurrently, returning
// only once every chunk has finished. Each chunk writes a disjoint set of
// output rows and reads the shared input planes read-only, so the chunks are
// data-race free and the combined result is identical to a serial run.
//
// For small frames (area < parallelConvertMinPixels) fn is invoked once over
// the full range on the calling goroutine, avoiding goroutine overhead.
func parallelRows(w, h int, fn func(y0, y1 int)) {
workers := runtime.GOMAXPROCS(0)
if workers > maxConvertWorkers {
workers = maxConvertWorkers
}
if workers <= 1 || h < 2 || w*h < parallelConvertMinPixels {
fn(0, h)
return
}
if workers > h {
workers = h
}
chunk := (h + workers - 1) / workers
var wg sync.WaitGroup
for y0 := 0; y0 < h; y0 += chunk {
y1 := y0 + chunk
if y1 > h {
y1 = h
}
wg.Add(1)
go func(a, b int) {
defer wg.Done()
fn(a, b)
}(y0, y1)
}
wg.Wait()
}
// combineAVC444v2BGRA combines luma from stream1 with per-pixel chroma from
// stream1 and stream2 to produce a BGRA frame. The fullRange branch is hoisted
// outside the inner loop; row-level offsets are computed once per row. The row
// range is split across cores via parallelRows for large frames.
//
// When dirtyRegions is non-nil, only rows covered by at least one region are
// converted; all other rows in the output buffer are left as pool garbage.
// Callers must only pass non-nil dirtyRegions when shouldUseAVCRegions is true,
// because in that case blitAndEmitAVCRegions reads only within the dirty rects
// and never accesses the uninitialised rows.
func combineAVC444v2BGRA(
yPlane []byte, yStride int,
cachedU, cachedV []byte, uvStride int,
i420aux *H264FrameI420,
fullRange bool,
w, h int,
dirtyRegions []avcRect,
) (out []byte, pooled bool) {
if len(yPlane) == 0 || len(cachedU) == 0 || len(cachedV) == 0 || w <= 0 || h <= 0 {
return nil, false
}
if i420aux == nil || len(i420aux.Y) == 0 || len(i420aux.U) == 0 || len(i420aux.V) == 0 {
return nil, false
}
out = acquireBitmapBuf(w * h * 4)
halfW := w / 2
quarterW := w / 4
auxYStride := i420aux.YStride
auxUStride := i420aux.UStride
auxVStride := i420aux.VStride
// Build per-row dirty mask when only partial conversion is needed. Rows not
// covered by any dirty region are skipped inside rowFn; the corresponding
// output bytes remain as pool-buffer data that blitAndEmitAVCRegions never
// reads (it only accesses pixels within the dirty rectangles).
var rowDirty []bool
if len(dirtyRegions) > 0 {
rowDirty = make([]bool, h)
for _, r := range dirtyRegions {
r0 := max(0, int(r.top))
r1 := min(h, int(r.bottom))
for y := r0; y < r1; y++ {
rowDirty[y] = true
}
}
}
// Split on fullRange once so the inner loop body is branch-free for the
// YCbCr→BGRA conversion coefficients. Each output row is independent, so
// parallelRows splits the rows across cores for large frames.
var rowFn func(y0, y1 int)
if fullRange {
rowFn = func(y0, y1 int) {
for row := y0; row < y1; row++ {
if rowDirty != nil && !rowDirty[row] {
continue
}
yRowOff := row * yStride
uvRow := row >> 1
uvRowOff := uvRow * uvStride
auxYRowOff := row * auxYStride
auxURowOff := uvRow * auxUStride
auxVRowOff := uvRow * auxVStride
outIdx := row * w * 4
for col := range w {
Y := yPlane[yRowOff+col]
var Cb, Cr byte
if col&1 == 1 {
k := col >> 1
Cb = i420aux.Y[auxYRowOff+k]
Cr = i420aux.Y[auxYRowOff+halfW+k]
} else if row&1 == 0 {
k := col >> 1
Cb = cachedU[uvRowOff+k]
Cr = cachedV[uvRowOff+k]
} else {
k := col >> 2
if col&2 == 0 {
Cb = i420aux.U[auxURowOff+k]
Cr = i420aux.U[auxURowOff+quarterW+k]
} else {
Cb = i420aux.V[auxVRowOff+k]
Cr = i420aux.V[auxVRowOff+quarterW+k]
}
}
y := int(Y)
u := int(Cb) - 128
v := int(Cr) - 128
out[outIdx] = clampByte((256*y + 475*u + 128) >> 8)
out[outIdx+1] = clampByte((256*y - 48*u - 120*v + 128) >> 8)
out[outIdx+2] = clampByte((256*y + 403*v + 128) >> 8)
out[outIdx+3] = 255
outIdx += 4
}
}
}
} else {
rowFn = func(y0, y1 int) {
for row := y0; row < y1; row++ {
if rowDirty != nil && !rowDirty[row] {
continue
}
yRowOff := row * yStride
uvRow := row >> 1
uvRowOff := uvRow * uvStride
auxYRowOff := row * auxYStride
auxURowOff := uvRow * auxUStride
auxVRowOff := uvRow * auxVStride
outIdx := row * w * 4
for col := range w {
Y := yPlane[yRowOff+col]
var Cb, Cr byte
if col&1 == 1 {
k := col >> 1
Cb = i420aux.Y[auxYRowOff+k]
Cr = i420aux.Y[auxYRowOff+halfW+k]
} else if row&1 == 0 {
k := col >> 1
Cb = cachedU[uvRowOff+k]
Cr = cachedV[uvRowOff+k]
} else {
k := col >> 2
if col&2 == 0 {
Cb = i420aux.U[auxURowOff+k]
Cr = i420aux.U[auxURowOff+quarterW+k]
} else {
Cb = i420aux.V[auxVRowOff+k]
Cr = i420aux.V[auxVRowOff+quarterW+k]
}
}
c := int(Y) - 16
u := int(Cb) - 128
v := int(Cr) - 128
out[outIdx] = clampByte((298*c + 541*u + 128) >> 8)
out[outIdx+1] = clampByte((298*c - 55*u - 136*v + 128) >> 8)
out[outIdx+2] = clampByte((298*c + 459*v + 128) >> 8)
out[outIdx+3] = 255
outIdx += 4
}
}
}
}
parallelRows(w, h, rowFn)
return out, true
}
// i420ToBGRA converts a planar I420 frame to a packed BGRA buffer using BT.709
// coefficients (matching AVC444 content encoding). Used when the I420 fast path
// is active and a BGRA output is required by the rendering path.
//
// Optimised: the fullRange branch is hoisted outside both loops so the inner
// loop body is branch-free, row offsets are computed once per row, and outIdx
// advances by 4 instead of recomputing col*4 per pixel.
func i420ToBGRA(src *H264FrameI420) ([]byte, bool) {
if src == nil || src.Width <= 0 || src.Height <= 0 {
return nil, false
}
w, h := src.Width, src.Height
out := acquireBitmapBuf(w * h * 4)
var rowFn func(y0, y1 int)
if src.FullRange {
rowFn = func(yStart, yEnd int) {
for row := yStart; row < yEnd; row++ {
yOff := row * src.YStride
uvOff := (row >> 1) * src.UStride
uvOffV := (row >> 1) * src.VStride
outIdx := row * w * 4
for col := range w {
y := int(src.Y[yOff+col])
uv := col >> 1
u := int(src.U[uvOff+uv]) - 128
v := int(src.V[uvOffV+uv]) - 128
out[outIdx] = clampByte((256*y + 475*u + 128) >> 8)
out[outIdx+1] = clampByte((256*y - 48*u - 120*v + 128) >> 8)
out[outIdx+2] = clampByte((256*y + 403*v + 128) >> 8)
out[outIdx+3] = 255
outIdx += 4
}
}
}
} else {
rowFn = func(yStart, yEnd int) {
for row := yStart; row < yEnd; row++ {
yOff := row * src.YStride
uvOffU := (row >> 1) * src.UStride
uvOffV := (row >> 1) * src.VStride
outIdx := row * w * 4
for col := range w {
c := int(src.Y[yOff+col]) - 16
uv := col >> 1
u := int(src.U[uvOffU+uv]) - 128
v := int(src.V[uvOffV+uv]) - 128
out[outIdx] = clampByte((298*c + 541*u + 128) >> 8)
out[outIdx+1] = clampByte((298*c - 55*u - 136*v + 128) >> 8)
out[outIdx+2] = clampByte((298*c + 459*v + 128) >> 8)
out[outIdx+3] = 255
outIdx += 4
}
}
}
}
parallelRows(w, h, rowFn)
return out, true
}
// avc444bt709BGRA converts one YCbCr pixel to BGRA using BT.709 coefficients,
// matching FreeRDP's general_YUV444ToBGRX implementation.
// Windows AVC444v2 content is encoded in BT.709; using BT.601 here was the
// cause of red color bleeding on LC=2 chroma-upgrade frames.
// Cb and Cr are raw (0-255); the function subtracts 128 internally.
//
// Full range (Y∈[0,255]): R = Y + 1.5748*(Cr-128) ≈ (256y + 403v) >> 8
// Limited range (Y∈[16,235]): R = 1.164*(Y-16) + 1.793*(Cr-128) ≈ (298c + 459v) >> 8
func avc444bt709BGRA(Y, Cb, Cr byte, fullRange bool, dst []byte) {
u := int(Cb) - 128
v := int(Cr) - 128
var r, g, b int
if fullRange {
y := int(Y)
r = (256*y + 403*v + 128) >> 8
g = (256*y - 48*u - 120*v + 128) >> 8
b = (256*y + 475*u + 128) >> 8
} else {
c := int(Y) - 16
r = (298*c + 459*v + 128) >> 8
g = (298*c - 55*u - 136*v + 128) >> 8
b = (298*c + 541*u + 128) >> 8
}
dst[0] = clampByte(b)
dst[1] = clampByte(g)
dst[2] = clampByte(r)
dst[3] = 255
}
// clampByte clamps an integer to [0, 255] using branchless min/max built-ins.
func clampByte(v int) byte {
return byte(max(0, min(255, v)))
}
// primeAuxDecoder feeds stream2 data from an LC=0 packet to h264dec2 so that
// the decoder's decoded-picture buffer (DPB) stays in sync with the full
// stream2 H.264 sequence. Stream2 frames are always part of one continuous
// H.264 sequence: the IDR is carried in LC=0 (and duplicated in a standalone
// LC=2 packet), and subsequent P-frames arrive via BOTH LC=0 packets and
// standalone LC=2 packets. If primeAuxDecoder only decoded IDRs, h264dec2's
// DPB would be stuck at the IDR while the server advanced the sequence through
// several LC=0 P-frames; the first standalone LC=2 P-frame would then be
// decoded against the wrong reference, producing all-zero chroma (Cb=0,
// Cr=0) and a full-screen green tint. By decoding ALL stream2 frames here
// (output discarded), h264dec2's DPB is always at the correct reference when
// primeH264dec2KeepDPB feeds stream2 data to h264dec2 and discards the
// output. Call this whenever decodeAVC444LC2 must skip the combine step
// (Y cache empty or stale) so that h264dec2's decoded-picture buffer stays
// in sync with the stream2 H.264 sequence. Without this, the next
// standalone LC=2 P-frame would reference a DPB state that is behind
// the expected position, causing FFmpeg to produce all-zero chroma
// (Cb=0, Cr=0) and a full-screen green tint.
func (g *GfxHandler) primeH264dec2KeepDPB(h264Data []byte) {
if g.h264dec2 == nil {
return
}
i420dec, ok := g.h264dec2.(I420Decoder)
if !ok {
return
}
_, _, err := i420dec.DecodeWithI420(h264Data)
if err != nil {
slog.Debug("RDPGFX: LC=2 DPB prime error", "err", err)
}
if g.h264dec2 != nil && g.h264dec2.IsBroken() {
g.h264dec2.Close()
g.h264dec2 = nil
g.startAuxDecoderBrokenTimer()
}
}
// decodeAVC444LC2 decodes a standalone LC=2 P-frame.
func (g *GfxHandler) primeAuxDecoder(h264Data []byte) {
// Mark that stream2 data has appeared in an LC=0 packet. VirtualBox VRDE
// never includes stream2, so this flag distinguishes VirtualBox from Windows.
g.stream2EverSeen = true
isIDR := h264PacketHasIDR(h264Data)
if g.h264dec2 == nil {
if !isIDR {
// No aux decoder yet; wait for the stream2 IDR to create one.
return
}
// A stream2 IDR arrived — clear any permanent-degrade state so LC=2
// can recover (e.g. after a server-side GOP reset much later in the session).
if g.lc2PermanentlyDegraded {
slog.Debug("H.264: stream2 IDR received after LC=2 degrade — recovering aux decoder")
g.lc2PermanentlyDegraded = false
g.auxDecoderNoIDRRetries = 0
}
// Recreate aux decoder on a stream2 IDR so it starts with a clean
// reference frame. This avoids the rapid create/destroy cycle that
// can destabilise the decoder.
slog.Debug("H.264: recreating aux decoder on stream2 IDR")
g.h264dec2 = newH264DecoderSW()
g.stopAuxDecoderBrokenTimer() // LC=0 IDR arrived; cancel recovery timer
// Fall through to prime the freshly-created decoder with this IDR.
}
// If the aux decoder is broken, reset it only on an IDR (P-frames cannot
// start a new decode sequence).
if g.h264dec2.IsBroken() {
if !isIDR {
return
}
// Stream2 IDR received while aux decoder is broken — recreate it now
// and fall through to prime the fresh decoder with this IDR.
// (Previously this closed and waited for a *second* IDR which often
// never arrived, permanently losing LC=2 quality for the session.)
slog.Debug("H.264: recreating broken aux decoder on stream2 IDR")
g.h264dec2.Close()
g.h264dec2 = newH264DecoderSW()
g.stopAuxDecoderBrokenTimer()
}
i420dec, ok := g.h264dec2.(I420Decoder)
if !ok {
return
}
_, i420primed, err := i420dec.DecodeWithI420(h264Data)
if err != nil {
slog.Debug("RDPGFX: AVC444 aux prime error", "err", err)
}
// The pre-flight stall detector inside DecodeWithI420 can set broken=true
// and return nil,nil without an error (broken state invisible to caller).
// Check IsBroken() after the call to catch this case.
if g.h264dec2.IsBroken() {
slog.Debug("H.264: aux decoder broken after prime, waiting for IDR to recreate")
g.h264dec2.Close()
g.h264dec2 = nil
g.startAuxDecoderBrokenTimer()
return
}
// For P-frames, validate the decoded output. If the primed output looks
// blank (near-zero or near-saturated chroma), the DPB is likely corrupted
// (e.g. due to a dropped LC=0 PDU that left h264dec2 out of sync).
// Reset h264dec2 immediately so the DPB corruption does not cascade into
// the subsequent LC=2 standalone decode. The IDR case is excluded because
// near-zero output is expected during codec initialisation.
if !isIDR && i420primed != nil && isAuxChromaBlank(i420primed) {
slog.Debug("H.264: aux decoder DPB desynced during priming (P-frame blank chroma), resetting")
g.h264dec2.Close()
g.h264dec2 = nil
g.startAuxDecoderBrokenTimer()
}
}
// decodeAVC444LC2 decodes an AVC444 LC=2 chroma-upgrade frame.
// It decodes stream2 via the auxiliary decoder, then combines the cached luma
// (Y plane) with the auxiliary chroma (Y2 = U/Cb channel, U2 = V/Cr channel)
// to produce a BGRA frame.
func (g *GfxHandler) decodeAVC444LC2(stream2 *avc420Stream, destW, destH int) (decoded []byte, regions []avcRect, pooled bool) {
// Record LC=2 arrival unconditionally so maybeRenegotiateCapabilities can
// distinguish an active-LC=2-only server from a truly idle server.
g.lastLC2RecvTime.Store(time.Now().UnixNano())
if g.h264dec2 == nil {
if g.lc2PermanentlyDegraded {
// Server has proven it won't deliver stream2 IDRs; skip silently
// without arming the timer to avoid an endless renegotiation loop.
return
}
// If this standalone LC=2 frame carries an IDR, use it to create and
// prime h264dec2 directly. Some servers deliver the ForceRefresh IDR
// response as LC=1 (luma only) rather than LC=0 (both streams), so the
// IDR in the "duplicate" standalone LC=2 packet is the only opportunity
// to initialise the aux decoder without a full reconnect.
if stream2 != nil && len(stream2.h264Data) > 0 && isH264Keyframe(stream2.h264Data) {
slog.Debug("H.264: creating aux decoder from standalone LC=2 IDR")
g.h264dec2 = newH264DecoderSW()
g.stopAuxDecoderBrokenTimer()
g.auxDecoderNoIDRRetries = 0
// Fall through to the decode path below.
} else {
slog.Debug("RDPGFX: AVC444 LC=2 skipped (no aux decoder)")
// Arm the renegotiation timer so maybeRenegotiateCapabilities fires if
// no stream2 IDR arrives to prime h264dec2 within auxDecoderBrokenTimeout.
// This is idempotent — subsequent calls are no-ops while the timer runs.
g.startAuxDecoderBrokenTimer()
return
}
}
if stream2 == nil || len(stream2.h264Data) == 0 {
slog.Debug("RDPGFX: AVC444 LC=2 skipped (empty aux stream)")
return
}
// If the main decoder is broken (e.g. HW stall or no IDR received), trigger
// soft reset so it can recover even when only LC=2 (chroma-only) frames are
// arriving and the LC=0/1 decode path never gets called.
if g.h264dec != nil && g.h264dec.IsBroken() {
g.maybeNotifyDecoderBroken()
return
}
if g.avc444YPlane.w == 0 {
slog.Debug("RDPGFX: AVC444 LC=2 skipped (no cached luma)")
// Still advance h264dec2's DPB so the next standalone LC=2 P-frame
// finds the correct reference. Without this the DPB falls behind and
// FFmpeg outputs all-zero chroma (green tiles) on the next LC=2 decode.
g.primeH264dec2KeepDPB(stream2.h264Data)
g.maybeRequestKeyframe()
return
}
// Skip the combine when the Y cache is stale: the main decoder is likely
// stalling (VideoToolbox null frames). Combining old luma with fresh chroma
// produces visible colour artefacts. We suppress LC=2 output until h264dec
// delivers a fresh frame and refreshes the cache.
if !g.avc444YPlane.updatedAt.IsZero() && time.Since(g.avc444YPlane.updatedAt) > avc444YStaleness {
age := time.Since(g.avc444YPlane.updatedAt).Round(time.Millisecond)
slog.Debug("RDPGFX: AVC444 LC=2 skipped (Y cache stale, main decoder likely stalling)",
"age", age)
// Advance h264dec2's DPB even though we skip the combine, so that it
// stays in sync with the stream2 sequence and recovers cleanly once the
// main decoder exits its stall.
g.primeH264dec2KeepDPB(stream2.h264Data)
// Signal the server to reduce encoding quality/bitrate while the HW
// decoder is stalling. This throttles the stream of LC=2 frames that
// accumulate during VideoToolbox null-frame periods and gives VT more
// headroom to flush its pipeline. The hint is cleared in
// updateAVC444YCache when the HW decoder resumes real-frame output.
g.SetQueueDepthHint(avcHWStallQueueDepthHint)
// During a VideoToolbox stall h264dec.NeedsKeyframe() is false (the
// decoder has not been reset) so maybeRequestKeyframe() returns early.
// Request a keyframe directly here, reusing the shared rate-limiter, so
// the server delivers a fresh IDR that can help break the VT stall.
const keyframeRequestInterval = 2 * time.Second
if g.onKeyframeRequest != nil && time.Since(g.lastKeyframeRequest) >= keyframeRequestInterval {
g.lastKeyframeRequest = time.Now()
go g.onKeyframeRequest()
}
return
}
i420dec, ok := g.h264dec2.(I420Decoder)
if !ok {
slog.Debug("RDPGFX: AVC444 LC=2 skipped (aux decoder lacks I420 support)")
return
}
_, i420aux, err := i420dec.DecodeWithI420(stream2.h264Data)
if err != nil {
slog.Warn("RDPGFX: AVC444 LC=2 aux decode error", "err", err)
if g.h264dec2.IsBroken() {
g.h264dec2.Close()
g.h264dec2 = nil
g.startAuxDecoderBrokenTimer()
}
return
}
if i420aux == nil {
slog.Debug("RDPGFX: AVC444 LC=2 aux decode buffering",
"h264Len", len(stream2.h264Data),
"firstNAL", firstNALType(stream2.h264Data),
"isIDR", isH264Keyframe(stream2.h264Data))
// The pre-flight stall detector inside Decode() may have set broken=true
// and returned nil without an error. Detect and tear down here; the
// decoder will be recreated by primeAuxDecoder when the next stream2
// IDR arrives, avoiding a rapid VT session create/destroy cycle.
if g.h264dec2 != nil && g.h264dec2.IsBroken() {
slog.Debug("H.264: aux decoder broken during LC=2 decode, waiting for IDR to recreate")
g.h264dec2.Close()
g.h264dec2 = nil
// Do NOT call maybeRequestKeyframe() here: ForceRefresh only delivers
// LC=1 luma IDR, not a stream2/chroma IDR. h264dec2 will be re-primed
// naturally when the next LC=0 frame arrives via primeAuxDecoder.
// The aux decoder broken timer will escalate to caps renegotiation if
// no LC=0 IDR arrives within auxDecoderBrokenTimeout.
g.startAuxDecoderBrokenTimer()
}
return
}
// Detect invalid aux chroma: two failure modes trigger this check.
// 1. Near-zero (Cb≈0, Cr≈0): Windows Server initialises stream2 IDR with
// Y≈0 and only refreshes regions that change; combining zero chroma with
// any luma produces BGRA(0,135,0,255) — a bright green screen.
// 2. Near-saturation (Cb≈255 or Cr≈255): DPB mismatch or aux decoder
// corruption that produces near-maximal stream2 Y values; these encode
// as extreme chroma and produce a pink/magenta overlay when combined.
// Determine IDR status before the blank-chroma check so it can drive the
// h264dec2 reset decision below.
stream2IsIDR := isH264Keyframe(stream2.h264Data)
if isAuxChromaBlank(i420aux) {
slog.Debug("RDPGFX: AVC444 LC=2 skipped (stream2 chroma invalid: near-zero or near-saturated)")
// For P-frames, corrupt chroma means h264dec2's DPB has diverged from
// the server's reference (typically from a dropped LC=0 PDU). Decoding
// further P-frames against this wrong DPB would produce equally wrong
// output on every subsequent LC=2, perpetuating the pink/green artefact.
// Reset h264dec2 now so the DPB corruption does not cascade; recovery
// will happen automatically on the next stream2 IDR arriving in an LC=0.
// IDRs are excluded because near-zero chroma is expected at GOP start
// during stream2 codec initialisation and should not trigger a reset.
if !stream2IsIDR && g.h264dec2 != nil {
slog.Debug("H.264: aux decoder reset after P-frame blank chroma (DPB cascade prevention)")
g.h264dec2.Close()
g.h264dec2 = nil
g.startAuxDecoderBrokenTimer()
}
return
}
// Select the luma plane for the combine. When stream2 carries an IDR its
// chroma data corresponds to the GOP-boundary frame, not to the latest
// P-frame. Using avc444IDRYPlane (a snapshot of the luma at the moment
// stream1's IDR was decoded) avoids combining mismatched luma/chroma planes
// and eliminates the transient green tint that appears at GOP boundaries
// when the server delivers the stream2 IDR as a standalone LC=2 packet.
// Fall back to avc444YPlane when no IDR snapshot is available (e.g. the
// VideoToolbox pipeline delayed the IDR output past the P-frame boundary).
yp := &g.avc444YPlane
if stream2IsIDR && g.avc444IDRYPlane.w > 0 {
yp = &g.avc444IDRYPlane
slog.Debug("RDPGFX: AVC444 LC=2 IDR combine using IDR luma snapshot")
}
w, h := yp.w, yp.h
if i420aux.Width < w || i420aux.Height < h {
slog.Debug("RDPGFX: AVC444 LC=2 aux frame too small",
"auxW", i420aux.Width, "auxH", i420aux.Height, "lumaW", w, "lumaH", h)
return
}
// Guard against corrupt cached stream1 chroma. A frame that slipped past
// the decoder's low-chroma guard can poison the U/V cache; combining that
// with any stream2 chroma produces green/pink artefacts on the even-column,
// even-row pixels. Skip the combine and ask for a fresh IDR.
if isAVC444YPlaneChromaBlank(yp) {
slog.Debug("RDPGFX: AVC444 LC=2 skipped (cached stream1 chroma blank/corrupt)")
g.primeH264dec2KeepDPB(stream2.h264Data)
g.maybeRequestKeyframe()
return
}
// Pass dirty regions to combineAVC444v2BGRA so it can skip unchanged rows
// (significant savings for frames where only a small area updates).
// Only do this when shouldUseAVCRegions is true: in that case the caller
// (decodeAVC444 / decodeAVC444WithI420) routes the output through
// blitAndEmitAVCRegions, which reads only within the dirty rectangles, so
// any uninitialized rows in the output buffer are never accessed.
//
// Use destW/destH (surface dimensions) for the shouldUseAVCRegions check,
// not w/h (decoded frame dimensions). The region coordinates are in surface
// space, and the callers (WTS1/WTS2) also call shouldUseAVCRegions with
// surface dimensions. Using different dimensions here could cause
// combineRegions to be set (skipping rows, leaving stale pool garbage) while
// the caller falls through to blitToSurface (reading all rows) — writing
// that garbage to the display. Using destW/destH keeps the two decisions
// in sync and is more correct since the regions are in surface coordinate space.
var combineRegions []avcRect
if len(stream2.regions) > 0 && shouldUseAVCRegions(stream2.regions, destW, destH) {
combineRegions = stream2.regions
}
combined, _ := combineAVC444v2BGRA(
yp.data, yp.stride,
yp.u, yp.v, yp.uvStride,
i420aux,
yp.fullRange,
w, h,
combineRegions,
)
if combined == nil {
return
}
// Mark that LC=2 has produced at least one frame this session.
// maybeRenegotiateCapabilities uses this to distinguish "was working then broke"
// (needs reconnect) from "never worked" (graceful LC=0 degradation).
g.lc2EverDecoded = true
g.auxDecoderNoIDRRetries = 0 // reset so a future break starts retries from scratch
// lc2Sample logs the actual Cb/Cr values used by combineAVC444v2BGRA for
// position (px,py), which depend on the B-area that pixel falls into.
halfW := w / 2
quarterW := w / 4
lc2Sample := func(px, py int) {
if px >= w || py >= h {
return
}
off := (py*w + px) * 4
if off+3 >= len(combined) {
return
}
uvRow := py >> 1
var actualCb, actualCr byte
var barea string
if px&1 == 1 {
// B4/B5: odd column — Cb/Cr packed in stream2 Y plane.
barea = "B4/B5"
k := px >> 1
auxYRow := i420aux.Y[py*i420aux.YStride:]
actualCb = auxYRow[k]
actualCr = auxYRow[halfW+k]
} else if py&1 == 0 {
// B2/B3: even column, even row — from stream1 cached chroma.
barea = "B2/B3"
actualCb = yp.u[uvRow*yp.uvStride+(px>>1)]
actualCr = yp.v[uvRow*yp.uvStride+(px>>1)]
} else {
k2 := px >> 2
if px&2 == 0 {
// B6/B7: even column (col%4==0), odd row.
barea = "B6/B7"
actualCb = i420aux.U[uvRow*i420aux.UStride+k2]
actualCr = i420aux.U[uvRow*i420aux.UStride+quarterW+k2]
} else {
// B8/B9: even column (col%4==2), odd row.
barea = "B8/B9"
actualCb = i420aux.V[uvRow*i420aux.VStride+k2]
actualCr = i420aux.V[uvRow*i420aux.VStride+quarterW+k2]
}
}
slog.Debug("H.264: pixel sample (LC=2 combine)",
"x", px, "y", py,
"area", barea,
"isIDR", stream2IsIDR,
"usedIDRSnapshot", yp == &g.avc444IDRYPlane,
"Y1", yp.data[py*yp.stride+px],
"Cb", actualCb, "Cr", actualCr,
"B", combined[off], "G", combined[off+1], "R", combined[off+2])
}
if !g.lc2SampleLogged {
g.lc2SampleLogged = true
// B2/B3 (even col, even row)
lc2Sample(100, 50)
lc2Sample(500, 50)
// B4/B5 (odd col) — most important for diagnosing tint artifacts
lc2Sample(101, 50)
lc2Sample(501, 50)
lc2Sample(961, 50)
// B6/B7 (col%4==0, odd row)
lc2Sample(100, 51)
lc2Sample(500, 51)
// B8/B9 (col%4==2, odd row)
lc2Sample(102, 51)
lc2Sample(502, 51)
// video area — all four B-areas near the same spot
lc2Sample(960, 600)
lc2Sample(961, 600)
lc2Sample(960, 601)
lc2Sample(962, 601)
} else if !g.lc2PFrameSampleLogged && !stream2IsIDR {
g.lc2PFrameSampleLogged = true
lc2Sample(100, 50)
lc2Sample(101, 50)
lc2Sample(100, 51)
lc2Sample(102, 51)
lc2Sample(500, 50)
lc2Sample(501, 50)
lc2Sample(960, 400)
lc2Sample(961, 400)
lc2Sample(960, 401)
lc2Sample(962, 401)
lc2Sample(960, 600)
lc2Sample(961, 600)
}
decoded, pooled = cropBGRA(combined, w, h, destW, destH)
if w == destW && h == destH {
// cropBGRA returned combined unchanged; mark as pooled so caller releases it.
pooled = true
} else {
// cropBGRA created a new buffer; release the intermediate combined buffer.
releaseBitmapBuf(combined)
}
regions = stream2.regions
slog.Debug("RDPGFX: AVC444 LC=2 decoded", "w", w, "h", h,
"destW", destW, "destH", destH, "h264Len", len(stream2.h264Data))
g.noteSuccessfulDecode()
return
}
// softResetLimit is the number of in-place decoder recreations attempted
// before escalating to a full RDP reconnect.
const softResetLimit = 5
// maybeRequestKeyframe sends a keyframe request to the server when either
// decoder needs a fresh IDR. Requests are rate-limited to once per 2 seconds
// so that repeated nil-frame callbacks (e.g. while waiting for the IDR) don't
// flood the server. This covers both post-flush and post-soft-reset cases,
// including the case where h264dec2 was reset independently of h264dec.
//
// Proactive stall recovery: even when NeedsKeyframe()==false (decoder has not
// yet been reset), we send ForceRefresh early when the HW decoder appears to be
// stalling — packets are arriving but no real frame has been produced for longer
// than avc444YStaleness. This gives the server a ~1 second head-start to
// prepare an IDR before the stall detector fires and triggers SW fallback,
// reducing the visible freeze from ~18 s to a few seconds.
func (g *GfxHandler) maybeRequestKeyframe() {
if g.onKeyframeRequest == nil {
return
}
if g.h264dec == nil || g.h264dec.IsBroken() {
return
}
dec1NeedsKF := g.h264dec.NeedsKeyframe()
// Do NOT include h264dec2 here: ForceRefresh only triggers an LC=1 luma IDR
// from the server. The stream2/chroma IDR is never delivered via
// ForceRefresh — it arrives naturally as an LC=0 frame via primeAuxDecoder.
// Requesting ForceRefresh because h264dec2.NeedsIDR()=true spams the server
// with keyframe requests, causes the server to repeatedly send LC=1 IDRs,
// and can deadlock the main VideoToolbox decoder.
if !dec1NeedsKF {
// Proactive early request: if packets are flowing in but no real frame
// has been produced for avc444YStaleness, the HW decoder is likely
// producing null frames. Request a keyframe now so the server has time
// to respond before the stall detector escalates to SW fallback.
recvTime := g.h264dec.LastReceiveTime()
if recvTime.IsZero() || time.Since(recvTime) >= avc444YStaleness {
// No packets arriving — server is idle, not a HW stall.
return
}
lastNS := g.lastDecodedFrame.Load()
if lastNS == 0 || time.Since(time.Unix(0, lastNS)) < avc444YStaleness {
// Frames are still being produced recently — not stalling.
return
}
}
const keyframeRequestInterval = 2 * time.Second
if time.Since(g.lastKeyframeRequest) < keyframeRequestInterval {
return
}
g.lastKeyframeRequest = time.Now()
go g.onKeyframeRequest()
}
// maybeNotifyDecoderBroken is called whenever the H.264 decoder returns a
// nil frame. It first tries up to softResetLimit in-place decoder resets
// (cheap: just recreate the FFmpeg/VideoToolbox context and ask the server
// for a fresh IDR). Only after all soft resets are exhausted does it call
// onDecoderBroken, which triggers a full RDP reconnect.
func (g *GfxHandler) maybeNotifyDecoderBroken() {
if g.decoderBrokenNotified {
return
}
if g.h264dec == nil || !g.h264dec.IsBroken() {
return
}
reason := g.h264dec.BrokenReason()
if reason == H264BrokenReasonNoIDR && g.h264dec.LastReceiveTime().IsZero() {
// The H.264 decoder's keyframe-wait timer fired, but the decoder has
// never received any data (LastReceiveTime is zero). This means the
// server is using a non-H.264 codec (e.g. CA Progressive / codecId=9)
// for the entire session and will never send H.264 frames.
// Sending ForceRefresh or reconnecting would disrupt the session
// unnecessarily — Ubuntu GNOME Remote Desktop responds to ForceRefresh
// with DEACTIVATEALLPDU followed by a disconnect.
// Disable the H.264 decoder so the watchdog can never fire again.
slog.Debug("H.264: watchdog fired but no H.264 data received — server uses non-H.264 codec, disabling H.264 decoder")
g.h264dec.Close()
g.h264dec = nil
return
}
if reason == H264BrokenReasonNoIDR {
// Allow one no-IDR soft reset before escalating to reconnect, unless
// we are already in SW fallback mode (after a HW stall). In the SW
// fallback case ForceRefresh was already sent multiple times during the
// VT stall and the server has not responded; another retry just prolongs
// the freeze by another keyframeWaitTimeoutSWFallback seconds. Skip
// straight to reconnect so the server can deliver a fresh IDR via the
// normal session-start path, which it reliably does.
//
// For the non-fallback path: ForceRefresh (SuppressOutput toggle) often
// fails to trigger a new AVC444 IDR from Windows servers; repeatedly
// retrying just prolongs the freeze. One attempt gives the server a
// fair chance; after that a full reconnect is faster.
//
// noIDRSoftResetCount is kept separate from softResetCount so that a
// prior HW-stall reset does not consume this budget — after an HW stall
// the SW fallback decoder skips retries (see above); for a pure SW
// session one no-IDR retry is still allowed.
const softResetLimitNoIDR = 1
if !g.usingSWFallback && g.noIDRSoftResetCount < softResetLimitNoIDR {
g.noIDRSoftResetCount++
slog.Debug("H.264: soft decoder reset (no-IDR)",
"attempt", g.noIDRSoftResetCount, "limit", softResetLimitNoIDR,
"reason", reason.String())
g.h264dec.Close()
g.h264dec = newH264DecoderWithWatchdog(g.watchdogCh)
if g.h264dec2 != nil && g.h264dec2.IsBroken() {
slog.Debug("H.264: aux decoder also broken on soft reset, waiting for IDR to recreate")
g.h264dec2.Close()
g.h264dec2 = nil
}
g.lastKeyframeRequest = time.Time{}
g.maybeRequestKeyframe()
return
}
slog.Debug("H.264: escalating to reconnect after no-IDR soft reset exhausted",
"reason", reason.String())
g.decoderBrokenNotified = true
if g.onDecoderBroken != nil {
go g.onDecoderBroken()
}
return
}
if g.softResetCount < softResetLimit {
g.softResetCount++
if reason == H264BrokenReasonHWStall && !g.usingSWFallback {
// Switch to software (FFmpeg) decoding when VideoToolbox stalls.
// Even if a proactive ForceRefresh was already sent, the server
// typically delivers the IDR within ~1-2 s; the SW decoder will
// pick it up and the session continues without a full reconnect.
slog.Debug("H.264: HW stall — falling back to software decoding",
"attempt", g.softResetCount, "limit", softResetLimit)
g.usingSWFallback = true
} else {
slog.Debug("H.264: soft decoder reset",
"attempt", g.softResetCount, "limit", softResetLimit,
"reason", reason.String())
}
g.h264dec.Close()
if g.usingSWFallback {
g.h264dec = newH264DecoderSWWithWatchdog(g.watchdogCh)
} else {
g.h264dec = newH264DecoderWithWatchdog(g.watchdogCh)
}
// Prime the SW fallback decoder with the last cached stream1 IDR so it
// can decode subsequent P-frames immediately, without waiting for the
// server to send a fresh IDR via ForceRefresh.
//
// We always prime when an IDR is cached, regardless of its age. A stale
// IDR is missing the reference frames decoded since then, so moving
// regions may show transient block noise until the next P-frames refresh
// them (or a fresh IDR fully heals the picture) — but for a mostly-static
// desktop the stale IDR is a close approximation and the artifacts are
// minor. Crucially this avoids the alternative: AVC444 servers only send
// an IDR at session start, so the cached IDR is essentially always "stale"
// at stall time; gating priming on freshness meant the SW decoder waited
// for a fresh IDR that never arrives, the watchdog fired, and the whole
// RDP session reconnected. Continuing with a primed SW decoder is far
// less disruptive than a reconnect. maybeRequestKeyframe() below still
// asks the server for a fresh IDR to clean up any residual artifacts.
idrAge := time.Since(g.lastStream1IDRTime)
idrFrameAge := g.framesDecoded.Load() - g.lastStream1IDRFrame
if g.usingSWFallback && len(g.lastStream1IDR) > 0 &&
!g.lastStream1IDRTime.IsZero() {
slog.Debug("H.264: priming SW fallback with cached stream1 IDR to avoid IDR wait",
"idrLen", len(g.lastStream1IDR),
"idrAge", idrAge.Round(time.Millisecond),
"idrFrameAge", idrFrameAge,
)
g.swFallbackPrimed = true
g.swFallbackDroppedCount = 0
g.swFallbackFirstDropTime = time.Time{}
if _, err := g.h264dec.Decode(g.lastStream1IDR); err != nil {
slog.Debug("H.264: cached IDR prime failed, watchdog will wait for natural IDR",
"err", err)
}
} else if g.usingSWFallback {
slog.Debug("H.264: no cached stream1 IDR to prime SW fallback — watchdog will wait for a natural IDR",
"idrAge", idrAge.Round(time.Millisecond),
"idrFrameAge", idrFrameAge,
)
}
// Keep h264dec2 if healthy; tear it down if already broken so
// primeAuxDecoder can recreate it when the next stream2 IDR arrives,
// rather than spinning up a new VT session only to have it break again.
// Always keep avc444YPlane so that LC=2 frames can continue to display
// stale-but-reasonable content during recovery.
if g.h264dec2 != nil && g.h264dec2.IsBroken() {
slog.Debug("H.264: aux decoder also broken on soft reset, waiting for IDR to recreate")
g.h264dec2.Close()
g.h264dec2 = nil
}
// Reset rate-limiter so keyframe request fires immediately after reset.
g.lastKeyframeRequest = time.Time{}
g.maybeRequestKeyframe()
return
}
// All soft resets exhausted — escalate to full reconnect.
g.decoderBrokenNotified = true
if g.onDecoderBroken != nil {
go g.onDecoderBroken()
}
}
// swFallbackDropLimit is the minimum number of consecutive dropped frames,
// and swFallbackResyncTimeout the minimum elapsed time, that must accumulate
// after priming the SW fallback decoder with a stale cached IDR before we give
// up and reconnect. Both conditions must hold: a large frame count alone (a
// fast stall burst) should not reconnect before the ForceRefresh resync IDR has
// had a realistic chance to arrive, and a long idle gap alone should not
// reconnect if only one or two frames were bad. While corruption persists the
// frames are dropped (screen holds the last good frame) rather than shown, and
// maybeRequestKeyframe keeps asking the server for a fresh IDR; only when the
// server fails to heal within the timeout do we fall back to a full reconnect.
const swFallbackDropLimit = 3
const swFallbackResyncTimeout = 2 * time.Second
// trackSWFallbackDroppedFrame counts a dropped frame after a SW fallback IDR
// prime. When corruption from a stale prime persists past swFallbackDropLimit
// consecutive drops AND swFallbackResyncTimeout — i.e. the ForceRefresh resync
// IDR did not arrive in time — it marks the decoder broken so the application
// reconnects instead of showing a frozen/green screen. A genuine fresh IDR
// (maybeCacheStream1IDR) or any clean decode (noteSuccessfulDecode) clears the
// run before it reaches the escalation threshold.
func (g *GfxHandler) trackSWFallbackDroppedFrame() {
if !g.usingSWFallback || !g.swFallbackPrimed {
return
}
if g.swFallbackDroppedCount == 0 {
g.swFallbackFirstDropTime = time.Now()
}
g.swFallbackDroppedCount++
if g.swFallbackDroppedCount >= swFallbackDropLimit &&
!g.swFallbackFirstDropTime.IsZero() &&
time.Since(g.swFallbackFirstDropTime) >= swFallbackResyncTimeout {
slog.Warn("H.264: SW fallback stale IDR prime did not resync in time, escalating to reconnect",
"dropped", g.swFallbackDroppedCount,
"persistedFor", time.Since(g.swFallbackFirstDropTime).Round(time.Millisecond))
g.swFallbackPrimed = false
g.swFallbackDroppedCount = 0
g.swFallbackFirstDropTime = time.Time{}
g.decoderBrokenNotified = true
if g.onDecoderBroken != nil {
go g.onDecoderBroken()
}
}
}
// cropBGRA crops or pads BGRA pixel data to the target dimensions.
// When srcW == dstW and srcH == dstH the input slice is returned unchanged
// and pooled is false. Otherwise a new buffer is acquired from bitmapBufPool
// (pooled == true) and the caller must call releaseBitmapBuf on it.
func cropBGRA(src []byte, srcW, srcH, dstW, dstH int) ([]byte, bool) {
if srcW == dstW && srcH == dstH {
return src, false
}
out := acquireBitmapBuf(dstW * dstH * 4)
copyW := min(dstW, srcW)
copyH := min(dstH, srcH)
srcStride := srcW * 4
dstStride := dstW * 4
rowBytes := copyW * 4
for y := range copyH {
copy(out[y*dstStride:y*dstStride+rowBytes], src[y*srcStride:y*srcStride+rowBytes])
}
return out, true
}
// avcRegionUseThresholdPercent is the upper bound on the *fraction* of the
// decoded frame area that the union of dirty rects can cover before we give
// up and just blit the whole frame. When the dirty area approaches the
// total area, the per-rect bookkeeping (allocation per rect, separate
// BitmapUpdate per rect) costs more than the bytes-copied savings.
const avcRegionUseThresholdPercent = 60
// shouldUseAVCRegions returns true when the per-region partial blit path is
// expected to be cheaper than a single full-frame blit. A single region
// covering everything is treated as "no win"; many tiny regions covering
// most of the frame are similarly bypassed.
func shouldUseAVCRegions(regions []avcRect, frameW, frameH int) bool {
if frameW <= 0 || frameH <= 0 {
return false
}
total := frameW * frameH
if total == 0 {
return false
}
// Sum (with overlap double-counting) — overlap is uncommon in practice
// and the threshold leaves slack for it.
sum := 0
for _, r := range regions {
if r.right <= r.left || r.bottom <= r.top {
continue
}
w := int(r.right - r.left)
h := int(r.bottom - r.top)
sum += w * h
if sum*100 >= total*avcRegionUseThresholdPercent {
return false
}
}
return sum > 0
}
// blitAndEmitAVCRegions copies only the dirty rectangles of a decoded AVC
// frame into the persistent surface and emits a BitmapUpdate per region.
// All region coordinates are in decoded-frame space (i.e. relative to
// (left, top) on the surface).
//
// The emitted Data buffers are borrowed from bitmapBufPool and are returned
// to the pool once the synchronous onBitmap callback completes — see the
// BitmapUpdate lifecycle note.
func (g *GfxHandler) blitAndEmitAVCRegions(s *surface, left, top, frameW, frameH int, decoded []byte, regions []avcRect) {
frameStride := frameW * 4
surfStride := int(s.width) * 4
g.updatesBuf = g.updatesBuf[:0]
for _, rc := range regions {
if rc.right <= rc.left || rc.bottom <= rc.top {
continue
}
rx, ry := int(rc.left), int(rc.top)
rw, rh := int(rc.right-rc.left), int(rc.bottom-rc.top)
if rx+rw > frameW {
rw = frameW - rx
}
if ry+rh > frameH {
rh = frameH - ry
}
if rw <= 0 || rh <= 0 {
continue
}
rowBytes := rw * 4
region := acquireBitmapBuf(rw * rh * 4)
for row := 0; row < rh; row++ {
srcOff := (ry+row)*frameStride + rx*4
if srcOff+rowBytes > len(decoded) {
break
}
copy(region[row*rowBytes:row*rowBytes+rowBytes],
decoded[srcOff:srcOff+rowBytes])
// Mirror the same row into the persistent surface so any
// subsequent codec (RFX progressive etc.) operating on the
// same surface starts from the up-to-date pixels.
dy := top + ry + row
if dy < 0 || dy >= int(s.height) {
continue
}
dstOff := dy*surfStride + (left+rx)*4
if dstOff < 0 || dstOff+rowBytes > len(s.data) {
continue
}
copy(s.data[dstOff:dstOff+rowBytes],
decoded[srcOff:srcOff+rowBytes])
}
if !s.mapped || g.onBitmap == nil {
releaseBitmapBuf(region)
continue
}
destL := int(s.outputX) + left + rx
destT := int(s.outputY) + top + ry
g.updatesBuf = append(g.updatesBuf, BitmapUpdate{
DestLeft: destL, DestTop: destT,
DestRight: destL + rw - 1, DestBottom: destT + rh - 1,
Width: rw, Height: rh, Bpp: 4, Data: region,
})
}
g.emitAndReleaseUpdates(g.updatesBuf)
}
// blitAVCRegionsToSurface copies only the dirty rectangles of a decoded AVC
// frame into the persistent CPU surface shadow (s.data), without allocating
// per-region buffers or emitting BitmapUpdates. It is the shadow-only
// counterpart to blitAndEmitAVCRegions, used on the GPU display path
// (onNV12/onI420) where the display is driven directly from the YUV planes and
// only the CPU shadow needs maintaining for later surface-to-surface / cache /
// mixed-codec operations.
//
// The caller must ensure the shadow is not stale (surface.shadowStale == false)
// before using this partial update; otherwise a full blitToSurface is required
// to repair regions that earlier GPU-only frames advanced without a shadow
// update. Region coordinates are in decoded-frame space (relative to
// (left, top) on the surface); both source and destination are bounds-clamped.
func (g *GfxHandler) blitAVCRegionsToSurface(s *surface, left, top, frameW, frameH int, decoded []byte, regions []avcRect) {
frameStride := frameW * 4
surfStride := int(s.width) * 4
for _, rc := range regions {
if rc.right <= rc.left || rc.bottom <= rc.top {
continue
}
rx, ry := int(rc.left), int(rc.top)
rw, rh := int(rc.right-rc.left), int(rc.bottom-rc.top)
if rx+rw > frameW {
rw = frameW - rx
}
if ry+rh > frameH {
rh = frameH - ry
}
if rw <= 0 || rh <= 0 {
continue
}
rowBytes := rw * 4
for row := 0; row < rh; row++ {
srcOff := (ry+row)*frameStride + rx*4
if srcOff+rowBytes > len(decoded) {
break
}
dy := top + ry + row
if dy < 0 || dy >= int(s.height) {
continue
}
dstOff := dy*surfStride + (left+rx)*4
if dstOff < 0 || dstOff+rowBytes > len(s.data) {
continue
}
copy(s.data[dstOff:dstOff+rowBytes], decoded[srcOff:srcOff+rowBytes])
}
}
}