package rdpgfx // Standard RemoteFX (MS-RDPRFX) codec helpers shared with rfx.go. // Extracted from the progressive decoder rewrite; algorithms unchanged. func rfxGetQuant(quants []rfxQuant, idx int) rfxQuant { if idx < len(quants) { return quants[idx] } return rfxQuant{6, 6, 6, 6, 6, 6, 6, 6, 6, 6} } // rfxDecodeComponent decodes one color component (Y, Cb, or Cr) for a 64×64 tile. // The returned slice is backed by a *coeffArr from coeffPool; the caller must // return it via coeffPool.Put((*coeffArr)(result)) when done. func rfxDecodeComponent(data []byte, quant rfxQuant, rlgrMode int) []int16 { const tilePixels = rfxTileSize * rfxTileSize // 4096 // Get a pooled coefficient buffer. The pool stores *coeffArr (pointer to a // fixed-size array) so the any interface stores a single pointer word with no // heap-boxing allocation. arr := coeffPool.Get().(*coeffArr) coeffs := arr[:] if data == nil { clear(coeffs) return coeffs } // 1. RLGR entropy decode → 4096 coefficients if rlgrMode == 3 { coeffs = rlgr3Decode(data, tilePixels, coeffs) } else { coeffs = rlgr1Decode(data, tilePixels, coeffs) } // 2. Differential decode LL3 and dequantize LL3 in a single pass. // Mathematical identity: cumsum(x) * 2^s == cumsum_of(x * 2^s) // so we can left-shift each element before accumulating. if quant.LL3 > 1 { shift := quant.LL3 - 1 coeffs[4032] <<= shift for i := 4033; i < 4096; i++ { coeffs[i] = coeffs[i-1] + coeffs[i]<>1) prevEvenH := lh[rowOff] - int16((int32(hh[rowOff])*2+1)>>1) tmp[lDstOff] = prevEvenL tmp[hDstOff] = prevEvenH // col=1..n-1: compute even[col], then immediately compute odd[col-1] // using prevEven (=even[col-1], still in register) and the just-computed // even[col] — no re-read of tmp required. for col := 1; col < n; col++ { x := col << 1 evenL := ll[rowOff+col] - int16((int32(hl[rowOff+col-1])+int32(hl[rowOff+col])+1)>>1) evenH := lh[rowOff+col] - int16((int32(hh[rowOff+col-1])+int32(hh[rowOff+col])+1)>>1) tmp[lDstOff+x-1] = int16((int32(hl[rowOff+col-1])<<1) + ((int32(prevEvenL)+int32(evenL))>>1)) tmp[hDstOff+x-1] = int16((int32(hh[rowOff+col-1])<<1) + ((int32(prevEvenH)+int32(evenH))>>1)) tmp[lDstOff+x] = evenL tmp[hDstOff+x] = evenH prevEvenL = evenL prevEvenH = evenH } // last odd[n-1]: right boundary, even[n] = even[n-1]. x := (n - 1) << 1 tmp[lDstOff+x+1] = int16((int32(hl[rowOff+n-1])<<1) + int32(prevEvenL)) tmp[hDstOff+x+1] = int16((int32(hh[rowOff+n-1])<<1) + int32(prevEvenH)) } // Step 2: Vertical IDWT on each column. // Process 8 columns at a time to improve cache utilisation — a cache line // holds 32 int16 values; 8 columns keeps the working set within one or two // lines per row access. All valid sizes (16, 32, 64) divide evenly by 8, // so the scalar tail loop is never reached in practice. const blk = 8 col := 0 for ; col+blk <= size; col += blk { // Row 0: first even output (no previous odd) l0 := tmp[col : col+blk] h0 := tmp[n*size+col : n*size+col+blk] out0 := buf[col : col+blk] for b := range blk { out0[b] = int16(int32(l0[b]) - ((int32(h0[b])*2 + 1) >> 1)) } // Rows 1..n-1: interleaved even/odd outputs for row := 1; row < n; row++ { lBase := row*size + col hBase := (row+n)*size + col hPrevBase := (row-1+n)*size + col evenBase := 2*row*size + col prevEvenBase := (2*row-2)*size + col oddBase := (2*row-1)*size + col l := tmp[lBase : lBase+blk] h := tmp[hBase : hBase+blk] hPrev := tmp[hPrevBase : hPrevBase+blk] evenOut := buf[evenBase : evenBase+blk] prevEvenIn := buf[prevEvenBase : prevEvenBase+blk] oddOut := buf[oddBase : oddBase+blk] for b := range blk { hPrevV := int32(hPrev[b]) even := int32(l[b]) - ((hPrevV + int32(h[b]) + 1) >> 1) evenOut[b] = int16(even) oddOut[b] = int16((hPrevV << 1) + ((int32(prevEvenIn[b]) + even) >> 1)) } } // Last odd row lastEvenBase := (2*n-2)*size + col lastHBase := (2*n-1)*size + col lastEvenSlice := buf[lastEvenBase : lastEvenBase+blk] lastHSlice := tmp[lastHBase : lastHBase+blk] lastOddOut := buf[lastHBase : lastHBase+blk] for b := range blk { lastOddOut[b] = int16((int32(lastHSlice[b]) << 1) + int32(lastEvenSlice[b])) } } for ; col < size; col++ { lVal := int32(tmp[col]) hVal := int32(tmp[n*size+col]) buf[col] = int16(lVal - ((hVal*2 + 1) >> 1)) for row := 1; row < n; row++ { lIdx := row*size + col hIdx := (row+n)*size + col hPrevIdx := (row-1+n)*size + col even := int32(tmp[lIdx]) - ((int32(tmp[hPrevIdx]) + int32(tmp[hIdx]) + 1) >> 1) buf[2*row*size+col] = int16(even) prevEven := int32(buf[(2*row-2)*size+col]) odd := (int32(tmp[hPrevIdx]) << 1) + ((prevEven + even) >> 1) buf[(2*row-1)*size+col] = int16(odd) } lastEven := int32(buf[(2*n-2)*size+col]) lastH := int32(tmp[(2*n-1)*size+col]) buf[(2*n-1)*size+col] = int16((lastH << 1) + lastEven) } } func rfxShiftSubband(data []int16, factor uint8) { if factor <= 1 { return } shift := factor - 1 for i := range data { data[i] <<= shift } }