diff --git a/rules/gr2/cx-save-must-wait-for-hq.md b/rules/gr2/cx-save-must-wait-for-hq.md index 5e41cdaa..0de87ee9 100644 --- a/rules/gr2/cx-save-must-wait-for-hq.md +++ b/rules/gr2/cx-save-must-wait-for-hq.md @@ -9,13 +9,15 @@ the magic check and the incoming context kept the outgoing one's live GL state: another demo (ideas) drew into amesh's window with amesh's window, clip and matrices. Rare: took minutes of two GL demos running together. -Fix (mod.rs `cx_save_pending`, counted like `fin3_pending` on the FIFO -push, decremented after the HQ2 executes the token): an HQ2_GEDMA read -returns bus busy while a save is queued. The VDMA worker and the CPU both -retry busy reads. Also: a rejected restore now resets GlState to defaults -instead of keeping the previous context's state (draws nothing until the -next window / matrix tokens, rather than into someone else's window). +Fix: HQ2_GEDMA reads wait (bus busy) while the port is empty and the HQ2 +still has work; see gedma-read-port.md (the first fix counted queued 0x1E1 +tokens, `cx_save_pending`; replaced by the general read-port rule when pixel +DMA reads needed the same). Also: a rejected restore now resets GlState to +defaults instead of keeping the previous context's state (draws nothing +until the next window / matrix tokens, rather than into someone else's +window). Trace diagnostics: GE_HQMSAV lines show `(live was )` and flag `LEAK?` for state 1 over another context's state; restores report `REJECTED: `. Test: `gl_context_save_read_waits_for_hq`. + diff --git a/src/dev/gr2/gl.rs b/src/dev/gr2/gl.rs index cee41224..a10ecba6 100644 --- a/src/dev/gr2/gl.rs +++ b/src/dev/gr2/gl.rs @@ -186,6 +186,16 @@ pub const T_GET_COLOR: u32 = 0x0e9; pub const T_GET_NORMAL: u32 = 0x0ea; pub const T_GET_RASTERPOS: u32 = 0x107; pub const T_READ_DONE: u32 = 0x0bd; +/// Pixel read setup (__glExpReadPixelsKDMA, Fetch, ReadColor): read setup +/// 0x0A7 / 0x0A8 = 0 and read mode 0x0BC (1 before a pixel DMA read, 0 +/// before READ_RECT). Meaning unknown; the HLE ignores them. +pub const T_READ_SETUP_A: u32 = 0x0a7; +pub const T_READ_SETUP_B: u32 = 0x0a8; +pub const T_READ_MODE: u32 = 0x0bc; +/// Read source (__glExpSetReadBuffer): two words to the token, kind and +/// buffer: 0, 0 = front; 0, 1 = back; 1, n = with aux buffers; 2, 0 = the +/// Z buffer (depth and stencil reads). +pub const T_READ_BUFFER: u32 = 0x10a; const READBACK_SHRAM: usize = 0x4022; // Vertex routines (LOADV token) and Begin/End tokens (gr2_prim.c). @@ -367,6 +377,8 @@ pub struct GlState { /// User clip planes (eye space) and their enable bits. uclip: [[f32; 4]; 6], uclip_on: u32, + /// Read source from 0x10A: kind, buffer (0, 0 = front). + read_src: [u32; 2], /// Vertices drawn / primitives emitted since the last trace note. pub stats_vertices: u32, } @@ -659,7 +671,7 @@ fn port_words(index: u32) -> Option { T_IRIS_CHAR16_TALL | T_IRIS_CHAR32 => Some(21), T_NORMAL_MATRIX => Some(9), T_NORMAL | T_IRIS_NORMAL => Some(3), - T_COLOR_WRITEMASK => Some(2), + T_COLOR_WRITEMASK | T_READ_BUFFER => Some(2), T_BLEND_FACTOR => Some(3), t => light::Lighting::port_words(t), } @@ -720,7 +732,53 @@ pub struct GlCx { impl Hq2Engine { /// 0x1E1 save main: the image the kernel now reads from HQ2_GEDMA. pub(super) fn gl_cx_save(&mut self, out: &mut dyn Re3Sink) { - out.cx_publish(&self.gl_ctx.save); + out.gedma_out(&self.gl_ctx.save); + } + + /// How 0x0AC reads the current read source (0x10A) in this window's + /// pixel format. Double-buffered 12-bit: after the swap to state s the + /// front is bank s and the back bank s ^ 1 (the inverse of GL_BACK's + /// write masks 0xFFF000 / 0x000FFF for states 0 / 1). + fn gl_read_decode(&self) -> super::ReadDecode { + use super::ReadDecode; + let g = &self.gl; + if g.read_src[0] == 2 { + return ReadDecode::Depth; + } + let bank = (g.swap ^ (g.read_src[1] & 1)) & 1; + match g.pixfmt { + re3::PIXFMT_RGB12 => ReadDecode::Rgb12 { bank }, + re3::PIXFMT_CI12 => ReadDecode::Ci12 { bank }, + _ => ReadDecode::Rgb24, + } + } + + pub(super) fn gl_read_desc(&self) -> String { + format!("src {} {} {:?} window ({}, {})", self.gl.read_src[0], self.gl.read_src[1], + self.gl_read_decode(), self.gl.win[0], self.gl.win[1]) + } + + /// 0x0AC pixel DMA read (lrectread, glReadPixels KDMA): x on the token; + /// y, width, height, words/row, flag, 0. Window-relative, y = the + /// bottom row; rows go out top first, as 0x0B5 takes them in. Pixels + /// per word from width / words per row, MSB first. + pub(super) fn gl_dma_read(&mut self, a: &[u32], out: &mut dyn Re3Sink) { + use super::{ReadDest, ReadImage}; + self.gl.ensure_init(); + let (w, rows, wpr) = (a[2], a[3], a[4].max(1)); + let mut s2d = self.s2d; + s2d.buf_select = match (w + wpr - 1) / wpr { 4 => 2, 2 => 1, _ => 0 }; + s2d.buf_offset = 0; + let req = ReadImage { + x: self.gl.win[0] + a[0] as i32, + top: self.gl.win[1] + a[1] as i32 + rows as i32 - 1, + w, + rows, + words_per_row: wpr, + s2d, + decode: self.gl_read_decode(), + }; + out.read_image(&req, ReadDest::Gedma); } /// The live GL state as a context image. @@ -841,7 +899,8 @@ impl Hq2Engine { T_SHADE_MODEL | T_FRONT_FACE | T_MAKECURRENT | T_CULL_FRONT | T_CULL_BACK | T_POLYGON_MODE | T_SWAP_BUFFERS | T_DITHER | T_STIPPLE_OFF | T_DEPTH_TEST | T_DEPTH_FUNC | T_DEPTH_MASK | T_STENCIL_WMASK | T_STENCIL_CONFIG - | T_BLEND_MODE | T_LOGIC_OP | T_GET_COLOR | T_GET_NORMAL | T_GET_RASTERPOS | T_READ_DONE => 1, + | T_BLEND_MODE | T_LOGIC_OP | T_GET_COLOR | T_GET_NORMAL | T_GET_RASTERPOS | T_READ_DONE + | T_READ_SETUP_A | T_READ_SETUP_B | T_READ_MODE => 1, T_FRAGMENT => 7, T_STENCIL_CLEAR | T_IRIS_BUFFER | T_IRIS_DB => 2, T_IRIS_WRITEMASK | T_IRIS_CLEAR | T_IRIS_ZCLEAR | T_IRIS_BGNPOLYGON | T_IRIS_ENDPOLYGON @@ -1040,6 +1099,12 @@ impl Hq2Engine { } }; } + T_READ_BUFFER => { + g.read_src = [b[0], b[1]]; + if let Some(d) = done.as_mut() { + d(format!("GL_READ_BUFFER {} {}", b[0], b[1])); + } + } T_COLOR_WRITEMASK => { // Word 0 = plane mask for swap state 0, word 1 = state 1. // (Swapping them makes ideas flicker from the start: tested.) @@ -1126,7 +1191,7 @@ impl Hq2Engine { g.lt.command(cmd, &a[..]); } T_LOGIC_OP => g.logic_op = a[0] & 0xf, - T_READ_DONE => {} + T_READ_DONE | T_READ_SETUP_A | T_READ_SETUP_B | T_READ_MODE => {} T_GET_COLOR => { let c = g.color; for (k, v) in c.iter().enumerate() { @@ -1280,6 +1345,8 @@ impl Hq2Engine { T_IRIS_GETCPOS => format!("IRISGL_GETCPOS -> ({}, {}){}", g.cpos[0], g.cpos[1], if g.cpos[2] != 0 { " invalid" } else { "" }), T_GET_COLOR | T_GET_NORMAL | T_GET_RASTERPOS => format!("{} -> shram[{:#x}]", super::index_label(cmd), READBACK_SHRAM), T_READ_DONE => "GL_READ_DONE".to_string(), + T_READ_SETUP_A | T_READ_SETUP_B => format!("GL_READ_SETUP {:#x} {:#x}", cmd, a[0]), + T_READ_MODE => format!("GL_READ_MODE {}", a[0]), T_LOGIC_OP => format!("GL_LOGIC_OP {}", g.logic_op), light::T_LIGHTING => format!("GL_LIGHTING {}", if g.lt.on != 0 { "on" } else { "off" }), light::T_TWO_SIDED => format!("GL_LIGHT_MODEL_TWO_SIDE {}", g.lt.two_sided), @@ -1412,8 +1479,9 @@ impl Hq2Engine { out.reg(re3::REG_ALIGNPAT, 0); let mask = self.gl.masks[self.gl.swap as usize & 1] & self.gl.colormask; out.reg(re3::REG_PIXMASK, mask & 0x00ff_ffff); + // Always: a 2D 12-bit draw (MODE 2) may have left RGB12 behind. + out.op(re3::RE3_OP_PIXFMT, self.gl.pixfmt as u64); if self.gl.pixfmt != 0 { - out.op(re3::RE3_OP_PIXFMT, self.gl.pixfmt as u64); out.reg(re3::REG_ENABDITH, self.gl.dither); } if self.gl_zs_active() { diff --git a/src/dev/gr2/hq2.rs b/src/dev/gr2/hq2.rs index 4031bda7..c1a923d1 100644 --- a/src/dev/gr2/hq2.rs +++ b/src/dev/gr2/hq2.rs @@ -151,6 +151,13 @@ pub const HQ2_DMA_WRITE_PIXELS: u32 = 327; /// 0x147, drawn in the current GL context (window-relative, GL y up). pub const HQ2_GL_DMA_WRITE: u32 = 0x0b5; pub const HQ2_GL_DMA_WRITE_ZOOM: u32 = 0x0b8; +/// Pixel DMA reads (HQ2.h "PIXEL DMA DIRECTION"): the same kernel ioctl and +/// header as the writes, but the VDMA reads height * words/row words from +/// HQ2_GEDMA; FIN2 when done. 0x152 = Xsgi expReadImage* (screen, X-style +/// y = top row, format from BUF_SELECT); 0x0AC = GL lrectread / +/// glReadPixels (window-relative, y = bottom row, source from 0x10A). +pub const HQ2_2D_DMA_READ_PIXELS: u32 = 0x152; +pub const HQ2_GL_DMA_READ: u32 = 0x0ac; /// Pixel rectangles streamed by the kernel's pixel DMA (_Gr2DMAtrigger): /// x on the token; y, width, height, words/row, flag, 0 on GE_DATA; then @@ -226,16 +233,45 @@ pub struct State2d { pub buf_offset: u32, } -/// A READ_IMAGE request, as the sink needs it. +/// How a screen-to-host read turns a VRAM word into a pixel value. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum ReadDecode { + /// Xsgi: the plane group selected by the 2D MODE (`read_value`). + Planes2d, + /// GL 24-bit RGB: 0xAABBGGRR, alpha 0xFF (no alpha planes). + Rgb24, + /// GL 12-bit RGB in `bank` (0 = bits 11:0, 1 = bits 23:12), nibbles + /// widened to 8 bits (x 17), as Rgb24. + Rgb12 { bank: u32 }, + /// GL colour index in 12-bit `bank`. + Ci12 { bank: u32 }, + /// GL depth: Z right-justified (libglcore shifts it up by 32 - bits). + Depth, +} + +/// Where a screen-to-host read goes. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum ReadDest { + /// READ_IMAGE: shram from READ_IMAGE_SHRAM, then FIN3. + Shram, + /// Pixel DMA read: the HQ2_GEDMA read port, then FIN2. + Gedma, +} + +/// A screen-to-host read (READ_IMAGE, 0x152, 0x0AC), as the sink needs it. #[derive(Clone, Copy)] pub struct ReadImage { - /// Left x and GL row of the top line (y = 0 at the bottom). + /// Left x and GL row of the top line (y = 0 at the bottom); rows go + /// down from there. pub x: i32, pub top: i32, pub w: u32, pub rows: u32, pub words_per_row: u32, + /// Pixel format (buf_select: 2 = 8-bit, 1 = 16-bit, 0 = 32-bit) and + /// first-pixel slot (buf_offset); the planes for Planes2d. pub s2d: State2d, + pub decode: ReadDecode, } impl ReadImage { @@ -260,7 +296,7 @@ impl ReadImage { for p in 0..per_word { let i = (k as u32 * per_word + p) as i32 - off; if i >= 0 && (i as u32) < self.w { - let v = read_value(&self.s2d, read(self.x + i, gy)) & mask; + let v = self.value(read(self.x + i, gy)) & mask; word |= v << (32 - bits * (p + 1)); } } @@ -270,6 +306,23 @@ impl ReadImage { } } +impl ReadImage { + /// The pixel value of VRAM word `vram` (for Depth: the Z buffer word). + pub fn value(&self, vram: u32) -> u32 { + let wide = |n: u32| (n & 0xf) * 17; + match self.decode { + ReadDecode::Planes2d => read_value(&self.s2d, vram), + ReadDecode::Rgb24 => 0xff00_0000 | (vram & 0x00ff_ffff), + ReadDecode::Rgb12 { bank } => { + let v = (vram >> (12 * (bank & 1))) & 0xfff; + 0xff00_0000 | wide(v) | (wide(v >> 4) << 8) | (wide(v >> 8) << 16) + } + ReadDecode::Ci12 { bank } => (vram >> (12 * (bank & 1))) & 0xfff, + ReadDecode::Depth => vram & 0x00ff_ffff, + } + } +} + /// The value a VRAM word holds for the active plane group (the inverse of /// `value2d`): 4-bit overlay (MODE 8), 2-bit overlay or popup (MODE 0xB), /// or the colour planes (R in bits 7:0). @@ -278,6 +331,16 @@ pub fn read_value(s: &State2d, vram: u32) -> u32 { match s.mode & 0xff { 8 => aux, 0xb => if s.aux_mask & 0xc != 0 { (aux >> 2) & 3 } else { aux & 3 }, + // 12-bit RGB (expReadImage12TC: MODE 0x1002, ROP flag = the + // window's buffer * 8): the bank the flag selects, R 3:0, G 7:4, + // B 11:8, as 8:8:8 with each nibble replicated. The DDX takes the + // low nibble of each byte back ((w & 0xF) | (w & 0xF00) >> 4 | + // (w & 0xF0000) >> 8); expDrawImage12TC sends the high nibbles. + 2 => { + let v = (vram >> (12 * ((s.flag >> 3) & 1))) & 0xfff; + let wide = |n: u32| (n & 0xf) * 17; + wide(v) | (wide(v >> 4) << 8) | (wide(v >> 8) << 16) + } _ => vram & 0x00ff_ffff, } } @@ -303,6 +366,7 @@ pub fn token_name(index: u32) -> Option<&'static str> { 323 => "2D_GLYPH_8", 324 => "2D_GLYPH_16", 325 => "2D_GLYPH_32", 326 => "2D_GLYPH_64", HQ2_2D_LINE_MODE => "2D_LINE_MODE", HQ2_2D_LINE_CLIP => "2D_LINE_CLIP", HQ2_2D_MODE => "2D_MODE", HQ2_2D_ROP => "2D_ROP", HQ2_2D_CID_WRITE => "2D_CID_WRITE", HQ2_2D_BUF_SELECT => "2D_BUF_SELECT", + HQ2_2D_DMA_READ_PIXELS => "2D_DMA_READ_PIXELS", HQ2_GL_DMA_READ => "GL_DMA_READ", 340 => "2D_COPY_RECT", 342 => "2D_DRAW_IMAGE", 343 => "2D_DRAW_IMAGE_SMALL", HQ2_2D_READ_IMAGE => "2D_READ_IMAGE", 346 | 347 => "2D_STIPPLED_SPAN", HQ2_2D_STIPPLE_RECT => "2D_STIPPLE_RECT", 349 => "2D_MONO_IMAGE_8", 350 => "2D_MONO_IMAGE_16", 351 => "2D_MONO_IMAGE_32", @@ -351,7 +415,7 @@ pub fn gl_token_name(tok: u32) -> Option<&'static str> { 0x0f3 => "GL_BEGIN_QUADS", 0x0f4 => "GL_END_QUADS", 0x0f5 => "GL_VTX_QUADS", 0x0f6 => "GL_EDGE_FLAG", 0x0f7 => "GL_BEGIN_POLYGON", 0x0f8 => "GL_VTX_POLYGON", 0x0fc | 0x0fd => "GL_END_POLYGON", 0x104 => "GL_CLEAR_COLOR", 0x108 => "GL_FRONT_FACE", - 0x10a => "GL_BUFFER_MODE", 0x10b => "GL_COLOR_WRITEMASK", 0x10d => "GL_NORMAL", + 0x10a => "GL_READ_BUFFER", 0x10b => "GL_COLOR_WRITEMASK", 0x10d => "GL_NORMAL", 0x10e => "GL_LIGHTPATH", 0x11d => "GL_BEGIN_POINTS", 0x123 => "GL_VTX_POINTS", 0x168 => "GL_END_POINTS", 0x17c => "GL_BEGIN_LINES", 0x182 => "COLOR", 0x1e5 => "CX_RESTORE_1E5", 0x1e7 => "GL_SWAP_BUFFERS", @@ -481,15 +545,18 @@ pub trait Re3Sink { /// Raise a finish flag (FIN1/FIN2/FIN3), as the microcode does when it /// completes a request the host is waiting on. fn finish(&mut self, flag: usize); - /// READ_IMAGE: once all drawing is done, pack the rectangle into shram - /// and raise FIN3. - fn read_image(&mut self, req: &ReadImage); + /// Screen-to-host read: once all drawing is done, pack the rectangle + /// into shram and raise FIN3 (READ_IMAGE), or feed it to the HQ2_GEDMA + /// read port and raise FIN2 (pixel DMA reads). + fn read_image(&mut self, req: &ReadImage, dest: ReadDest); /// Emulator-private RE3 operation (re3::RE3_OP_*). fn op(&mut self, op: u32, val: u64); /// Store a word into shared RAM, as the microcode does for readbacks. fn shram(&mut self, word: usize, val: u32); - /// Make a saved context image available on HQ2_GEDMA reads (0x1E1). - fn cx_publish(&mut self, words: &[u32]); + /// Start a new transfer on the HQ2_GEDMA read port and queue `words` + /// (a context image, 0x1E1). Words a previous transfer left unread are + /// dropped. Blocks while the port is full. + fn gedma_out(&mut self, words: &[u32]); } /// HLE command interpreter state. Owned by the HQ2 thread. Plain data. @@ -559,6 +626,8 @@ impl Hq2Engine { | HQ2_2D_CID_WRITE | HQ2_2D_END_PRIMITIVE => 1, HQ2_2D_BUF_SELECT => 2, HQ2_2D_READ_IMAGE => 5, + // Token x, then y, width, height, words/row, flag, 0 on GE_DATA. + HQ2_2D_DMA_READ_PIXELS | HQ2_GL_DMA_READ => 7, HQ2_2D_ROP => 4, HQ2_2D_TILE_SETUP => 5, HQ2_2D_MONO_IMAGE_8 => 8, @@ -897,8 +966,30 @@ impl Hq2Engine { } } - /// Set the RE3 colour for `pixel` (colour planes: R in bits 7:0). + /// Set the RE3 colour for a GC pixel (fg, COLOR_ON / OFF; colour planes: + /// R in bits 7:0). In 12-bit mode (MODE 2) the pixel is the X visual's + /// 12-bit value (R 3:0, G 7:4, B 11:8): the DDX passes GC pixels + /// unconverted (expDrawPoints), so widen it here (inferred). fn color2d(&self, pixel: u32, out: &mut dyn Re3Sink) { + if self.s2d.mode & 0xff == 2 { + let wide = |n: u32| (n & 0xf) * 17; + return self.rgb2d(wide(pixel) | (wide(pixel >> 4) << 8) | (wide(pixel >> 8) << 16), out); + } + self.rgb2d(pixel, out); + } + + /// Colour of an image / tile pixel word: in 12-bit mode the DDX has + /// already widened it to 8:8:8 (expDrawImage12TC, expTileRects: the + /// nibbles in the high halves), which the RE3 12-bit packer takes as is. + fn image2d(&self, pixel: u32, out: &mut dyn Re3Sink) { + if self.s2d.mode & 0xff == 2 && self.s2d.cid_write & 0xf000 == 0 { + return self.rgb2d(pixel, out); + } + self.value2d(pixel, out); + } + + /// RE3 colour registers from an 8:8:8 value (R in bits 7:0). + fn rgb2d(&self, pixel: u32, out: &mut dyn Re3Sink) { out.reg(re3::REG_R, (pixel & 0xff) << 11); out.reg(re3::REG_G, ((pixel >> 8) & 0xff) << 11); out.reg(re3::REG_B, ((pixel >> 16) & 0xff) << 11); @@ -1022,7 +1113,7 @@ impl Hq2Engine { } let sx = x + (n - start) as i32; if sx < re3::FB_W as i32 && sx + (e - n) as i32 > 0 { - self.value2d(c, out); + self.image2d(c, out); self.span(sx, gy, (e - n) as u32, None, out); } n = e; @@ -1177,7 +1268,7 @@ impl Hq2Engine { while e < x2 && tile_px((e - ox).rem_euclid(tw), ty) == c { e += 1; } - self.value2d(c, out); + self.image2d(c, out); self.span(x, gy, (e - x) as u32, None, out); x = e; } @@ -1207,11 +1298,33 @@ impl Hq2Engine { out.reg(re3::REG_PIXMASK, 0); self.value2d(fg, out); } + // 12-bit RGB (MODE 2, 12-bit TrueColor windows): the RE3 + // writes the value to both 12-bit buffers and the plane mask + // picks one (RE3.h). The buffers never move; ROP flag bit 3 + // (the window's dbc buffer * 8) names buffer 1. A mask the + // DDX already moved or doubled (dbc -3 / -1, expTileRects) + // is used as it is. + 2 => { + let pm = s.planemask & 0x00ff_ffff; + let mask = if s.flag & 8 == 0 { + pm & 0xfff + } else if pm & 0xfff000 != 0 { + pm + } else { + pm << 12 + }; + out.reg(re3::REG_UAUXDATA, 0); + out.reg(re3::REG_AUXMASK, 0); + out.reg(re3::REG_PIXMASK, mask); + out.op(re3::RE3_OP_PIXFMT, re3::PIXFMT_RGB12 as u64); + out.reg(re3::REG_ENABDITH, 0); + } // Colour planes. _ => { out.reg(re3::REG_UAUXDATA, 0); out.reg(re3::REG_AUXMASK, 0); out.reg(re3::REG_PIXMASK, s.planemask & 0x00ff_ffff); + out.op(re3::RE3_OP_PIXFMT, 0); } } } @@ -1263,6 +1376,10 @@ impl Hq2Engine { HQ2_2D_MODE | HQ2_2D_COLOR_AUX | HQ2_2D_COLOR_OFF | HQ2_2D_COLOR_ON | HQ2_2D_CID_WRITE | HQ2_2D_BEGIN | HQ2_2D_END_PRIMITIVE => format!("{name} {:#x}", a[0]), HQ2_2D_BUF_SELECT => format!("{name} format={} offset={}", i(0), i(1)), + HQ2_2D_DMA_READ_PIXELS => format!("{name} ({}, {}) {}x{} words/row={} flag={} format={} offset={} mode={:#x} -> GEDMA, FIN2", + i(0), i(1), i(2), i(3), i(4), i(5), self.s2d.buf_select, self.s2d.buf_offset, self.s2d.mode), + HQ2_GL_DMA_READ => format!("{name} ({}, {}) {}x{} words/row={} flag={} {} -> GEDMA, FIN2", + i(0), i(1), i(2), i(3), i(4), i(5), self.gl_read_desc()), HQ2_2D_READ_IMAGE => format!("{name} ({}, {}) {}x{} words/row={} format={} offset={} mode={:#x} -> shram[{:#x}], FIN3", i(1), i(2), i(3), i(4), i(0), self.s2d.buf_select, self.s2d.buf_offset, self.s2d.mode, READ_IMAGE_SHRAM), PUC_COLOR => format!("{name} ci={}", i(0)), @@ -1316,7 +1433,7 @@ impl Hq2Engine { out.finish(FIN2); } GE_CX_SAVE_EXT => { - out.cx_publish(&[]); + out.gedma_out(&[]); out.finish(FIN2); } HQ_PCX_1E0 | HQ_PCX_1E6 => out.finish(FIN2), @@ -1358,9 +1475,23 @@ impl Hq2Engine { rows: a[4], words_per_row: a[0], s2d: self.s2d, + decode: ReadDecode::Planes2d, + }; + out.read_image(&req, ReadDest::Shram); + } + HQ2_2D_DMA_READ_PIXELS => { + let req = ReadImage { + x: a[0] as i32, + top: SCREEN_H - 1 - a[1] as i32, + w: a[2], + rows: a[3], + words_per_row: a[4], + s2d: self.s2d, + decode: ReadDecode::Planes2d, }; - out.read_image(&req); + out.read_image(&req, ReadDest::Gedma); } + HQ2_GL_DMA_READ => self.gl_dma_read(&a, out), HQ2_2D_TILE_SETUP => { self.tile_w = a[1]; self.tile_h = a[2]; diff --git a/src/dev/gr2/hq2_tests.rs b/src/dev/gr2/hq2_tests.rs index 0c28f64c..d1fad0f3 100644 --- a/src/dev/gr2/hq2_tests.rs +++ b/src/dev/gr2/hq2_tests.rs @@ -2304,3 +2304,192 @@ fn gl_line_steep_downwards() { } assert_eq!(gl_px(g, 20, 10), 0, "last point not drawn"); } + +/// Kernel pixel-DMA read as Xsgi's expReadImage drives it for large +/// rectangles (XGetImage, readximage.log): BUF_SELECT; fin2 = 0; 0x152 x; +/// GE_DATA y, width, rows, words/row, flag, 0; the VDMA then reads the +/// packed rows (top first, X-style y) from HQ2_GEDMA and waits for FIN2. +/// Before IRIS knew 0x152 the reads got zeros, FIN2 never came and the +/// kernel reset the board. +#[test] +fn ddx_dma_read_pixels_streams_gedma_and_sets_fin2() { + use super::hq2::*; + let g = live_gr2(Gr2Variant::Xz); + cmd(g, HQ2_2D_BEGIN, 0); + cmd(g, HQ2_2D_MODE, 0x1009); + cmd(g, HQ2_2D_ROP, 0); + for v in [0xff, 3, 0] { data(g, v); } + cmd(g, HQ2_2D_DRAW_IMAGE, 100); + for v in [50, 6, 2, 2, 2, 0] { data(g, v); } + for v in [0x01020304u32, 0x05060000, 0x11121314, 0x15160000, 0xdeadbeef, 0xdeadbeef] { + data(g, v); + } + cmd(g, HQ2_2D_END_PRIMITIVE, 0); + cmd(g, HQ2_2D_BUF_SELECT, 2); + data(g, 0); + w32(g, 0x6a04c, 0); + cmd(g, HQ2_2D_DMA_READ_PIXELS, 100); + for v in [50, 6, 2, 2, 0, 0] { cmd(g, 0, v); } + // No wait: read the way the VDMA does, right after the header. + let words: Vec = (0..4).map(|_| r32(g, 0x6a068)).collect(); + assert_eq!(words, vec![0x0102_0304, 0x0506_0000, 0x1112_1314, 0x1516_0000]); + assert_eq!(r32(g, 0x6a040) & 2, 2, "FIN2 after the transfer"); +} + +/// A GEDMA read with nothing produced and the HQ2 idle is an overrun: 0, +/// not a hang. (With the HQ2 busy it waits instead; see the save test.) +#[test] +fn gedma_read_with_hq_idle_is_an_overrun() { + let g = live_gr2(Gr2Variant::Xz); + g.wait_idle(); + assert_eq!(r32(g, 0x6a068), 0); +} + +fn gl_dma_read(g: &Gr2, x: u32, y: u32, w: u32, rows: u32) -> Vec { + w32(g, 0x6a04c, 0); + cmd(g, super::hq2::HQ2_GL_DMA_READ, x); + for v in [y, w, rows, w, 0, 0] { cmd(g, 0, v); } + let words = (0..w * rows).map(|_| r32(g, 0x6a068)).collect(); + assert_eq!(r32(g, 0x6a040) & 2, 2, "FIN2 after the transfer"); + words +} + +/// glReadPixels RGBA through the kernel pixel DMA (libglcore +/// __glExpReadPixelsKDMARGBA, readback.log): 0x10A read source; 0x0AC x; +/// GE_DATA y (window-relative, bottom row), width, rows, words/row, 0, 0. +/// 24-bit RGB: one pixel per word as 0xAABBGGRR (the order 0x0B5 takes), +/// alpha 0xFF; rows go out top first. +#[test] +fn gl_dma_read_rgb24_top_row_first() { + let g = live_gr2(Gr2Variant::Xz); + gl_setup_window(g); + gl_quad3(g, [1., 0., 0.], [[0., 0., 0.], [400., 0., 0.], [400., 150., 0.], [0., 150., 0.]]); + gl_quad3(g, [0., 0., 1.], [[0., 150., 0.], [400., 150., 0.], [400., 300., 0.], [0., 300., 0.]]); + for v in [0, 0] { cmd(g, 0x10a, v); } + let (red, blue) = (0xff00_00ff, 0xffff_0000); + assert_eq!(gl_dma_read(g, 10, 148, 2, 4), vec![blue, blue, blue, blue, red, red, red, red]); +} + +/// Double-buffered 12-bit RGB (MAKECURRENT 2): GL_BACK draws with the write +/// masks 0xFFF000 / 0x000FFF, so in swap state 0 the back buffer is bank 1. +/// A back read (0x10A = 0, 1) returns it with the nibbles widened; a front +/// read returns bank 0. +#[test] +fn gl_dma_read_rgb12_back_and_front_banks() { + let g = live_gr2(Gr2Variant::Xz); + gl_setup_window(g); + cmd(g, 0x004, 2); + for v in [0x00ff_f000, 0x0000_0fff] { cmd(g, 0x10b, v); } + cmd(g, 0x011, 0); + gl_quad3(g, [1., 0.5, 0.], [[0., 0., 0.], [400., 0., 0.], [400., 300., 0.], [0., 300., 0.]]); + for v in [0, 1] { cmd(g, 0x10a, v); } + let back = gl_dma_read(g, 20, 20, 1, 1); + assert_eq!(back[0] & 0xff00_00ff, 0xff00_00ff, "back: red nibble 0xF widened to 0xFF, alpha 0xFF"); + assert_eq!(back[0] & 0x00ff_0000, 0, "back: no blue"); + for v in [0, 0] { cmd(g, 0x10a, v); } + assert_eq!(gl_dma_read(g, 20, 20, 1, 1), vec![0xff00_0000], "front bank untouched"); +} + +/// XGetImage of a 12-bit double-buffered window (cap.log, glprim --scene +/// quadrants --db): expReadImage12TC sets MODE 0x1002 and ROP flag = the +/// window's buffer * 8, reads 32-bit words by 0x152 and keeps the low +/// nibble of each byte. So the read returns the flagged bank as 8:8:8 with +/// the nibbles replicated, not the raw 24 bits of both banks. +#[test] +fn ddx_dma_read_rgb12_takes_the_flagged_bank() { + use super::hq2::*; + let g = live_gr2(Gr2Variant::Xz); + gl_setup_window(g); + cmd(g, 0x004, 2); + cmd(g, 0x011, 0); + // Bank 0 red, bank 1 blue (12-bit: R 3:0, G 7:4, B 11:8). + for v in [0x0000_0fff, 0x0000_0fff] { cmd(g, 0x10b, v); } + gl_quad3(g, [1., 0., 0.], [[0., 0., 0.], [400., 0., 0.], [400., 300., 0.], [0., 300., 0.]]); + for v in [0x00ff_f000, 0x00ff_f000] { cmd(g, 0x10b, v); } + gl_quad3(g, [0., 0., 1.], [[0., 0., 0.], [400., 0., 0.], [400., 300., 0.], [0., 300., 0.]]); + let read = |buffer: u32| { + cmd(g, HQ2_2D_BEGIN, 0); + cmd(g, HQ2_2D_MODE, 0x1002); + cmd(g, HQ2_2D_ROP, 0); + for v in [0xff_ffff, 3, buffer * 8] { data(g, v); } + cmd(g, HQ2_2D_BUF_SELECT, 0); + data(g, 0); + w32(g, 0x6a04c, 0); + // Screen x 100, X-style y 1023 - 700 (inside the window). + cmd(g, HQ2_2D_DMA_READ_PIXELS, 100); + for v in [1023 - 700, 1, 1, 1, 0, 0] { cmd(g, 0, v); } + let w = r32(g, 0x6a068); + (w & 0xf) | ((w & 0xf00) >> 4) | ((w & 0xf0000) >> 8) + }; + assert_eq!(read(0), 0x00f, "buffer 0: red as an X 12-bit pixel"); + assert_eq!(read(1), 0xf00, "buffer 1: blue"); +} + +fn ddx_12bit_setup(g: &Gr2, planemask: u32, buffer: u32) { + use super::hq2::*; + cmd(g, HQ2_2D_BEGIN, 0); + cmd(g, HQ2_2D_MODE, 0x1002); + cmd(g, HQ2_2D_ROP, 0); + for v in [planemask, 3, buffer.wrapping_mul(8)] { data(g, v); } +} + +/// 12-bit TrueColor windows (MODE 0x1002): the buffers stay put in VRAM +/// (bank 0 = bits 11:0, bank 1 = bits 23:12) and the ROP flag (dbc buffer +/// * 8) says which one X draws into. A GC pixel is the visual's 12-bit +/// value (R 3:0, G 7:4, B 11:8). +#[test] +fn ddx_12bit_fill_goes_to_the_flagged_bank() { + use super::hq2::*; + let g = live_gr2(Gr2Variant::Xz); + // Solid rect, X-style y: (10, 1023 - 20) .. (13, 1023 - 20). + let rect = |g: &Gr2, fg: u32, pm: u32, buffer: u32| { + ddx_12bit_setup(g, pm, buffer); + cmd(g, HQ2_2D_ROP, fg); + for v in [pm, 3, buffer.wrapping_mul(8)] { data(g, v); } + cmd(g, HQ2_2D_SOLID_RECT, 0); + for v in [10, 1023 - 20, 14, 1023 - 19] { data(g, v); } + cmd(g, HQ2_2D_END_PRIMITIVE, 0); + g.wait_idle(); + }; + rect(g, 0x00f, 0xff_ffff, 0); + assert_eq!(pixel(g, 11, 20), 0x00_000f, "buffer 0 only, even with an all-ones plane mask"); + rect(g, 0xf00, 0xfff, 1); + assert_eq!(pixel(g, 11, 20), 0xf0_000f, "buffer 1: the low mask moved up"); + // dbc -1: the DDX doubles the mask itself. + rect(g, 0x0f0, 0xff_ffff, u32::MAX); + assert_eq!(pixel(g, 11, 20), 0x0f_00f0, "both buffers"); +} + +/// Image words in 12-bit mode come widened by the DDX (expDrawImage12TC: +/// (x & 0xF) << 4 | (x & 0xF0) << 8 | (x & 0xF00) << 12) and land as the +/// 12-bit pixel. +#[test] +fn ddx_12bit_image_word_is_widened_pixel() { + use super::hq2::*; + let g = live_gr2(Gr2Variant::Xz); + ddx_12bit_setup(g, 0xfff, 0); + let widen = |x: u32| ((x & 0xf) << 4) | ((x & 0xf0) << 8) | ((x & 0xf00) << 12); + cmd(g, HQ2_2D_DRAW_IMAGE, 100); + for v in [50, 2, 1, 2, 0, 0] { data(g, v); } + for v in [widen(0x123), widen(0xabc)] { data(g, v); } + cmd(g, HQ2_2D_END_PRIMITIVE, 0); + g.wait_idle(); + assert_eq!(pixel(g, 100, 1023 - 50), 0x123); + assert_eq!(pixel(g, 101, 1023 - 50), 0xabc); +} + +/// A 2D 12-bit draw must not leave the RE3 in 12-bit mode for a 24-bit GL +/// window drawn next. +#[test] +fn gl_24bit_after_ddx_12bit_draw() { + use super::hq2::*; + let g = live_gr2(Gr2Variant::Xz); + gl_setup_window(g); + ddx_12bit_setup(g, 0xfff, 0); + cmd(g, HQ2_2D_SOLID_RECT, 0); + for v in [0, 0, 2, 2] { data(g, v); } + cmd(g, HQ2_2D_END_PRIMITIVE, 0); + gl_quad3(g, [1., 0.5, 0.25], [[0., 0., 0.], [400., 0., 0.], [400., 300., 0.], [0., 300., 0.]]); + g.wait_idle(); + assert_eq!(gl_px(g, 20, 20), 0x40_80ff); +} diff --git a/src/dev/gr2/mod.rs b/src/dev/gr2/mod.rs index 6b17e345..77132ed8 100644 --- a/src/dev/gr2/mod.rs +++ b/src/dev/gr2/mod.rs @@ -92,6 +92,10 @@ pub const TP_PROBE_ID: usize = 0x7fff; pub const HQ_FIFO_DEPTH: usize = 65536; pub const RE3_FIFO_DEPTH: usize = 65536; +/// Words the HQ2_GEDMA read port holds ahead of its reader (power of two). +pub const GEDMA_OUT_WORDS: usize = 16384; +/// How long the HQ2 waits for a reader when the read port is full. +const GEDMA_OUT_STALL_LIMIT_NS: u64 = 2_000_000_000; /// Which board is installed. Same hardware family; only the probe answers differ. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -167,14 +171,6 @@ pub struct Gr2 { /// kernel sees FIN2 as soon as the HQ2 gets there. See /// rules/gr2/fin2-wait-must-stall.md. fin2_wait: AtomicU32, - /// Context saves (0x1E1) the CPU has queued that the HQ2 has not executed - /// yet. HQ2_GEDMA reads stall while any is pending: the kernel starts its - /// VDMA read right after queuing 0x1E1, and a read that beat the HQ2 - /// returned the end of the previous image, shifting the saved image one - /// word; its restore was then rejected and the previous context's state - /// stayed live (another demo drawn in amesh's window). See - /// rules/gr2/cx-save-must-wait-for-hq.md. - cx_save_pending: AtomicU32, /// Host time (fin2_clock ns) of the first stalled FIN2 poll since the /// wait was armed (0 = none yet). The stall gives up FIN2_STALL_LIMIT /// after that, so a pipeline that can never finish (RE3 held by a @@ -182,10 +178,15 @@ pub struct Gr2 { /// poll, not from the ack: the DMA between ack and poll may take long on /// a busy host (a 400-row pixel DMA on a GitHub CI runner did). fin2_armed_ns: AtomicU64, - /// Context image handed to the kernel through HQ2_GEDMA reads (0x1E1). - cx_out: [AtomicU32; hq2::CX_WORDS], - cx_out_len: AtomicU32, - cx_out_pos: AtomicU32, + /// HQ2_GEDMA read port: HQ2 -> host words (context saves 0x1E1, pixel + /// DMA reads 0x152 / 0x0AC). A ring the HQ2 thread fills and the reader + /// (the kernel's VDMA) drains; head / tail count words since reset. + /// A read with the ring empty waits (bus busy) while the HQ2 still has + /// work, since the kernel starts the VDMA right after queuing the + /// request. See rules/gr2/gedma-read-port.md. + gedma_out: [AtomicU32; GEDMA_OUT_WORDS], + gedma_out_head: AtomicU32, + gedma_out_tail: AtomicU32, hq_fifo: GFifo, re3_fifo: GFifo, /// Pending half of a two-entry COPY op (RE3 thread only). @@ -324,6 +325,58 @@ impl Gr2 { } } + /// HQ2 thread: a new HQ2_GEDMA read transfer starts. Words an earlier + /// one left unread would shift this one, so drop them. + fn gedma_begin(&self) { + let head = self.gedma_out_head.load(Ordering::Relaxed); + let tail = self.gedma_out_tail.load(Ordering::Acquire); + if head != tail { + crate::dlog_dev!(crate::devlog::LogModule::Gr2, + "GR2: GEDMA: {} words of the previous transfer never read, dropped", head.wrapping_sub(tail)); + let _ = self.gedma_out_tail.compare_exchange(tail, head, Ordering::AcqRel, Ordering::Relaxed); + } + } + + /// HQ2 thread: queue one word on the HQ2_GEDMA read port, waiting while + /// it is full. False if no reader drained it within the limit (or the + /// engine stops): the rest of the transfer is dropped. + fn gedma_push(&self, w: u32) -> bool { + let head = self.gedma_out_head.load(Ordering::Relaxed); + let mut since = 0u64; + while head.wrapping_sub(self.gedma_out_tail.load(Ordering::Acquire)) as usize >= GEDMA_OUT_WORDS { + if !self.running.load(Ordering::Relaxed) { + return false; + } + let now = fin2_clock(); + if since == 0 { + since = now; + } else if now - since > GEDMA_OUT_STALL_LIMIT_NS { + crate::dlog_dev!(crate::devlog::LogModule::Gr2, "GR2: GEDMA read port full, no reader: transfer dropped"); + return false; + } + thread::yield_now(); + } + self.gedma_out[head as usize & (GEDMA_OUT_WORDS - 1)].store(w, Ordering::Relaxed); + self.gedma_out_head.store(head.wrapping_add(1), Ordering::Release); + true + } + + /// Reader: the next HQ2_GEDMA word, if the HQ2 has produced one. + fn gedma_pop(&self) -> Option { + let mut tail = self.gedma_out_tail.load(Ordering::Acquire); + loop { + if tail == self.gedma_out_head.load(Ordering::Acquire) { + return None; + } + let v = self.gedma_out[tail as usize & (GEDMA_OUT_WORDS - 1)].load(Ordering::Relaxed); + // CAS: the CPU and the VDMA worker may both read the port. + match self.gedma_out_tail.compare_exchange(tail, tail.wrapping_add(1), Ordering::AcqRel, Ordering::Acquire) { + Ok(_) => return Some(v), + Err(t) => tail = t, + } + } + } + fn wake(slot: &Mutex>) { if let Some(t) = slot.lock().as_ref() { t.unpark(); @@ -360,20 +413,21 @@ impl Gr2 { FIFO..FIFO_END => 0, HQUCODE..HQUCODE_END => r.hq.ucode[((off - HQUCODE) >> 2) as usize], GEWIN..GEWIN_END => r.ge.read(((off - GEWIN) >> 10) as usize, ((off >> 2) & 0xff) as usize), - // HQ2_GEDMA read: the next word of a context image (0x1E1 save, - // read by the kernel's VDMA). Past the end: 0. + // HQ2_GEDMA read: the next word the HQ2 produced (context image, + // pixel DMA read), read by the kernel's VDMA. 0x6a068 => { - if self.cx_save_pending.load(Ordering::Acquire) != 0 { - // The save the kernel is reading has not run yet. - return BusRead32::busy(); - } - let len = self.cx_out_len.load(Ordering::Acquire); - let pos = self.cx_out_pos.fetch_add(1, Ordering::Relaxed); - if pos < len { - self.cx_out[pos as usize].load(Ordering::Relaxed) - } else { - crate::dlog_dev!(crate::devlog::LogModule::Gr2, "GR2: GEDMA read past the context image ({pos}/{len})"); - 0 + // Sample "HQ2 has work" before looking at the port: a word + // is pushed before the HQ2 goes idle, so an idle HQ2 seen + // here means every word it produced is visible below. + let working = !self.hq_fifo.is_empty() || self.hq_busy.load(Ordering::Acquire); + match self.gedma_pop() { + Some(v) => v, + // Not produced yet: the request is queued or running. + None if working => return BusRead32::busy(), + None => { + crate::dlog_dev!(crate::devlog::LogModule::Gr2, "GR2: GEDMA read overrun (HQ2 idle, nothing to read)"); + 0 + } } } 0x6b000 => if self.fin3_pending.load(Ordering::Acquire) != 0 { 0 } else { r.hq.fin[hq2::FIN3].load(Ordering::Acquire) }, @@ -452,22 +506,15 @@ impl Gr2 { FIFO..FIFO_END => { let idx = (off - FIFO) >> 2; let fin = is_fin3_token(idx); - let save = is_cx_save_token(idx); if fin { // Counted before the push so the HQ can never see the // token before the counter includes it. self.fin3_pending.fetch_add(1, Ordering::AcqRel); } - if save { - self.cx_save_pending.fetch_add(1, Ordering::AcqRel); - } if !self.hq_fifo.try_push(idx, val as u64) { if fin { self.fin3_pending.fetch_sub(1, Ordering::AcqRel); } - if save { - self.cx_save_pending.fetch_sub(1, Ordering::AcqRel); - } return BUS_BUSY; } Self::wake(&self.hq_thread); @@ -611,34 +658,46 @@ impl Gr2 { *w = val; } } - fn cx_publish(&mut self, words: &[u32]) { - let g = self.0; - for (slot, &w) in g.cx_out.iter().zip(words) { - slot.store(w, Ordering::Relaxed); + fn gedma_out(&mut self, words: &[u32]) { + self.0.gedma_begin(); + for &w in words { + if !self.0.gedma_push(w) { + break; + } } - g.cx_out_pos.store(0, Ordering::Relaxed); - g.cx_out_len.store(words.len().min(g.cx_out.len()) as u32, Ordering::Release); } fn op(&mut self, op: u32, val: u64) { self.0.re3_fifo.push(op, val); Gr2::wake(&self.0.re3_thread); } - fn read_image(&mut self, req: &hq2::ReadImage) { + fn read_image(&mut self, req: &hq2::ReadImage, dest: hq2::ReadDest) { while !self.0.re3_fifo.is_empty() || self.0.re3_busy.load(Ordering::Acquire) { std::hint::spin_loop(); } - // SAFETY: RE3 is drained and idle; VRAM is only read here, as - // the display thread does. - let vram = unsafe { &(*self.0.re3.get()).vram }; + // SAFETY: RE3 is drained and idle; VRAM and Z are only read + // here, as the display thread does. + let re3 = unsafe { &*self.0.re3.get() }; + let plane: &[u32] = if req.decode == hq2::ReadDecode::Depth { &re3.zbuf } else { &re3.vram }; let read = |x: i32, y: i32| { if x < 0 || y < 0 || x as usize >= re3::FB_W || y as usize >= re3::FB_H { 0 } else { - vram[y as usize * re3::FB_W + x as usize] + plane[y as usize * re3::FB_W + x as usize] } }; - req.pack(read, &mut self.0.regs().shram[hq2::READ_IMAGE_SHRAM..]); - self.0.regs().hq.fin[hq2::FIN3].store(1, Ordering::Release); + match dest { + hq2::ReadDest::Shram => { + req.pack(read, &mut self.0.regs().shram[hq2::READ_IMAGE_SHRAM..]); + self.0.regs().hq.fin[hq2::FIN3].store(1, Ordering::Release); + } + hq2::ReadDest::Gedma => { + let mut words = vec![0u32; req.rows as usize * req.words_per_row as usize]; + req.pack(read, &mut words); + self.gedma_out(&words); + // The kernel polls FIN2 once its VDMA has read them. + self.0.regs().hq.fin[hq2::FIN2].store(1, Ordering::Release); + } + } } } *self.hq_thread.lock() = Some(thread::current()); @@ -665,11 +724,6 @@ impl Gr2 { let _ = self.fin3_pending.fetch_update(Ordering::AcqRel, Ordering::Acquire, |n| Some(n.saturating_sub(1))); } - if is_cx_save_token(index) { - // Executed: the image is published (cx_publish). - let _ = self.cx_save_pending.fetch_update(Ordering::AcqRel, Ordering::Acquire, - |n| Some(n.saturating_sub(1))); - } self.hq_fifo.consume(); backoff.reset(); } else { @@ -824,20 +878,13 @@ impl BusDevice for Gr2 { // Both words or neither: a retried store must not push the first twice. let idx = (off - FIFO) >> 2; let fins = is_fin3_token(idx) as u32 + is_fin3_token(idx + 1) as u32; - let saves = is_cx_save_token(idx) as u32 + is_cx_save_token(idx + 1) as u32; if fins != 0 { self.fin3_pending.fetch_add(fins, Ordering::AcqRel); } - if saves != 0 { - self.cx_save_pending.fetch_add(saves, Ordering::AcqRel); - } if !self.hq_fifo.try_push2(idx, hi as u64, idx + 1, lo as u64) { if fins != 0 { self.fin3_pending.fetch_sub(fins, Ordering::AcqRel); } - if saves != 0 { - self.cx_save_pending.fetch_sub(saves, Ordering::AcqRel); - } return BUS_BUSY; } Self::wake(&self.hq_thread); @@ -890,7 +937,8 @@ impl Device for Gr2 { self.hq_fifo.reset(); self.fin3_pending.store(0, Ordering::Release); self.fin2_wait.store(0, Ordering::Release); - self.cx_save_pending.store(0, Ordering::Release); + let head = self.gedma_out_head.load(Ordering::Acquire); + self.gedma_out_tail.store(head, Ordering::Release); self.re3_fifo.reset(); } @@ -984,12 +1032,6 @@ fn fin2_clock() -> u64 { START.get_or_init(std::time::Instant::now).elapsed().as_nanos() as u64 } -/// Context save main (0x1E1): publishes the image the kernel then reads -/// from HQ2_GEDMA. -fn is_cx_save_token(idx: u32) -> bool { - idx == hq2::GE_CX_SAVE_MAIN -} - fn is_fin3_token(idx: u32) -> bool { idx == hq2::GL_FINISH || idx == hq2::HQ_GL_FIN3 } diff --git a/test/gltest/glprim b/test/gltest/glprim index 50b85f2f..cc5adc53 100755 Binary files a/test/gltest/glprim and b/test/gltest/glprim differ diff --git a/test/gltest/glprim.c b/test/gltest/glprim.c index 18fd6b53..7fde0774 100644 --- a/test/gltest/glprim.c +++ b/test/gltest/glprim.c @@ -16,8 +16,10 @@ #include #include #include +#include #include #include +#include #include #include @@ -59,6 +61,33 @@ static char scene[16] = ""; /* --scene depth|stencil|alphatest|blend * static GLenum depth_func = GL_LESS; /* --depthfunc */ static int want_stencil = 0; +/* ---- readback (--read) --------------------------------------------------- + After the draw (and glFinish), read pixels back and print what came back: + ximage XGetImage of the window (Xsgi, front buffer) + front glReadBuffer(GL_FRONT) + glReadPixels colour + back glReadBuffer(GL_BACK) + glReadPixels colour (needs --db) + depth glReadPixels GL_DEPTH_COMPONENT (needs a depth buffer) + stencil glReadPixels GL_STENCIL_INDEX (needs a stencil buffer) + With --db the scene is swapped to the front, then the back buffer is + cleared to --backclear with a white marker (bottom left), so front and + back reads differ. libglcore (EXPRESS gr2_pixel.c) picks the path: + colour GL_RGBA/UNSIGNED_BYTE kernel pixel DMA (0x0AC), rows swapped + colour GL_ABGR_EXT/UNSIGNED_BYTE kernel pixel DMA (0x0AC), as is + depth UNSIGNED_INT kernel pixel DMA (0x0AC), buffer_mode 2 + unaligned buffer / odd stride PIO READ_RECT (0x0AD) via the mailbox + GL_FLOAT, stencil slow path, one READ_RECT per pixel + --readfmt/--readtype/--readunaligned choose among them. */ +#define MAX_READS 8 +static char reads[MAX_READS][8]; +static int nreads = 0; +static int rrect[4] = { -1, 0, 0, 0 }; /* --readrect x,y,w,h (GL window coords) */ +static char readfmt[8] = "rgba"; /* --readfmt rgba|abgr */ +static char readtype[8] = "ub"; /* --readtype ub|float (colour), uint|float (depth) */ +static int read_unaligned = 0; /* --readunaligned: force the PIO path */ +static int pack_row = 0; /* --packrow N: GL_PACK_ROW_LENGTH */ +static const char *read_out = 0; /* --readout PREFIX: write PREFIX_.ppm/.pgm */ +static float back_rgb[3] = { 0.0f, 0.75f, 0.75f }; /* --backclear */ + /* ---- multi-primitive scenes (window pixel coordinates, glOrtho) --------- Each is small and has a known expected image: depth: red triangle at z = 0, then a blue band whose z runs from -0.5 @@ -254,6 +283,47 @@ static void draw_scene(float w, float h) { glColor4f(0, 0, 1, 1.0f); glVertex2f(0.5f * w, 0.9f * h); glEnd(); glDisable(GL_BLEND); + } else if (!strcmp(scene, "quadrants")) { + /* Readback pattern: every orientation error shows. Border and gaps + keep the clear colour; depth and stencil differ per quadrant. + bottom left red, z 0.5 (depth 0.25), stencil 1 + bottom right green, z 0.0 (depth 0.50), stencil 2 + top left blue, z -0.5 (depth 0.75), stencil 3 + top right black -> yellow ramp left to right, z 0.9 -> -0.9 + (depth 0.05 -> 0.95), stencil 4 */ + float x0 = 8, x1 = w * 0.5f - 4, x2 = w * 0.5f + 4, x3 = w - 8; + float y0 = 8, y1 = h * 0.5f - 4, y2 = h * 0.5f + 4, y3 = h - 8; + if (want_stencil) { + glEnable(GL_STENCIL_TEST); + glStencilFunc(GL_ALWAYS, 0, 0xff); + glStencilOp(GL_KEEP, GL_KEEP, GL_REPLACE); + } + glShadeModel(GL_FLAT); + if (want_stencil) glStencilFunc(GL_ALWAYS, 1, 0xff); + glBegin(GL_QUADS); + glColor3f(1, 0, 0); + glVertex3f(x0, y0, 0.5f); glVertex3f(x1, y0, 0.5f); glVertex3f(x1, y1, 0.5f); glVertex3f(x0, y1, 0.5f); + glEnd(); + if (want_stencil) glStencilFunc(GL_ALWAYS, 2, 0xff); + glBegin(GL_QUADS); + glColor3f(0, 1, 0); + glVertex3f(x2, y0, 0.0f); glVertex3f(x3, y0, 0.0f); glVertex3f(x3, y1, 0.0f); glVertex3f(x2, y1, 0.0f); + glEnd(); + if (want_stencil) glStencilFunc(GL_ALWAYS, 3, 0xff); + glBegin(GL_QUADS); + glColor3f(0, 0, 1); + glVertex3f(x0, y2, -0.5f); glVertex3f(x1, y2, -0.5f); glVertex3f(x1, y3, -0.5f); glVertex3f(x0, y3, -0.5f); + glEnd(); + if (want_stencil) glStencilFunc(GL_ALWAYS, 4, 0xff); + glShadeModel(GL_SMOOTH); + glBegin(GL_QUADS); + glColor3f(0, 0, 0); glVertex3f(x2, y2, 0.9f); + glColor3f(1, 1, 0); glVertex3f(x3, y2, -0.9f); + glColor3f(1, 1, 0); glVertex3f(x3, y3, -0.9f); + glColor3f(0, 0, 0); glVertex3f(x2, y3, 0.9f); + glEnd(); + glShadeModel(smooth ? GL_SMOOTH : GL_FLAT); + glDisable(GL_STENCIL_TEST); } else { printf("unknown scene %s\n", scene); } @@ -365,6 +435,15 @@ static void usage(void) { printf(" --scene NAME depth | stencil | alphatest | blend | blendsmooth |\n" " lit | litlocal | twoside | fog (see source)\n" " --depthfunc F never less equal lequal greater notequal gequal always\n"); + printf(" --read M[,M...] after drawing, read back: ximage front back depth stencil\n" + " --readrect X,Y,W,H area to read (GL window coords; default whole window)\n" + " --readfmt rgba|abgr colour format for glReadPixels (default rgba)\n" + " --readtype T ub|float for colour, uint|float for depth (default ub/uint)\n" + " --readunaligned read into a misaligned buffer (forces the PIO path)\n" + " --packrow N GL_PACK_ROW_LENGTH N (strided pixel DMA)\n" + " --readout PREFIX also write PREFIX_.ppm (colour) / .pgm (depth, stencil)\n" + " --backclear R,G,B with --db: back buffer colour for the back read (default 0,.75,.75)\n" + " (--scene quadrants is the readback pattern: depth + stencil, see source)\n"); } static int parse_ints(const char *s, int *out, int n) { @@ -449,7 +528,27 @@ static void parse_args(int argc, char **argv) { strncpy(scene, next, sizeof(scene) - 1); if (!strcmp(scene, "depth")) depth = 1; if (!strcmp(scene, "stencil")) want_stencil = 1; - } else if (!strcmp(a, "--depthfunc")) { + if (!strcmp(scene, "quadrants")) { depth = 1; want_stencil = 1; } + } else if (!strcmp(a, "--read")) { + char buf[64], *t; + NEED(); + strncpy(buf, next, sizeof(buf) - 1); + buf[sizeof(buf) - 1] = 0; + for (t = strtok(buf, ","); t; t = strtok(0, ",")) { + if (strcmp(t, "ximage") && strcmp(t, "front") && strcmp(t, "back") + && strcmp(t, "depth") && strcmp(t, "stencil")) { + fprintf(stderr, "bad --read mode %s\n", t); exit(2); + } + if (nreads < MAX_READS) strncpy(reads[nreads++], t, 7); + } + } else if (!strcmp(a, "--readrect")) { NEED(); if (!parse_ints(next, rrect, 4)) { fprintf(stderr, "bad --readrect\n"); exit(2); } } + else if (!strcmp(a, "--readfmt")) { NEED(); strncpy(readfmt, next, sizeof(readfmt) - 1); } + else if (!strcmp(a, "--readtype")) { NEED(); strncpy(readtype, next, sizeof(readtype) - 1); } + else if (!strcmp(a, "--readunaligned")) read_unaligned = 1; + else if (!strcmp(a, "--packrow")) { NEED(); pack_row = atoi(next); } + else if (!strcmp(a, "--readout")) { NEED(); read_out = next; } + else if (!strcmp(a, "--backclear")) { NEED(); if (!parse_floats(next, back_rgb, 3)) { fprintf(stderr, "bad --backclear\n"); exit(2); } } + else if (!strcmp(a, "--depthfunc")) { static const char *names[] = { "never", "less", "equal", "lequal", "greater", "notequal", "gequal", "always" }; NEED(); for (k = 0; k < 8; k++) if (!strcmp(next, names[k])) break; @@ -462,6 +561,207 @@ static void parse_args(int argc, char **argv) { } } +/* ---- readback ------------------------------------------------------------ */ + +enum { RB_COLOR, RB_DEPTH, RB_STENCIL }; + +static double now_ms(void) { + struct timeval tv; + gettimeofday(&tv, 0); + return tv.tv_sec * 1000.0 + tv.tv_usec / 1000.0; +} + +/* Width in bits and shift of a visual channel mask. */ +static void mask_bits(unsigned long m, int *shift, int *bits) { + *shift = 0; *bits = 0; + if (!m) return; + while (!(m & 1)) { m >>= 1; (*shift)++; } + while (m & 1) { m >>= 1; (*bits)++; } +} + +/* Scale an n-bit channel value to 8 bits by bit replication. */ +static unsigned long to8(unsigned long v, int bits) { + unsigned long r = 0; + int have = 0; + if (bits <= 0) return 0; + if (bits >= 8) return v >> (bits - 8); + while (have < 8) { r = (r << bits) | v; have += bits; } + return r >> (have - 8); +} + +/* Read one buffer into vals[] (GL order: row 0 = bottom), as 0xRRGGBBAA + for colour, the raw 32-bit value for depth, the index for stencil. + Returns the kind, or -1 when the read is not possible. */ +static int read_pixels(Display *dpy, Window win, XVisualInfo *vi, const char *m, + int x, int y, int w, int h, unsigned long *vals) { + int row = pack_row > w ? pack_row : w; + int r, c; + if (!strcmp(m, "ximage")) { + XImage *im; + int rs, rb, gs, gb, bs, bb; + mask_bits(vi->red_mask, &rs, &rb); + mask_bits(vi->green_mask, &gs, &gb); + mask_bits(vi->blue_mask, &bs, &bb); + im = XGetImage(dpy, win, x, win_h - y - h, (unsigned)w, (unsigned)h, AllPlanes, ZPixmap); + if (!im) { printf(" XGetImage failed\n"); return -1; } + for (r = 0; r < h; r++) + for (c = 0; c < w; c++) { + unsigned long p = XGetPixel(im, c, h - 1 - r); + vals[r * w + c] = (to8((p & vi->red_mask) >> rs, rb) << 24) + | (to8((p & vi->green_mask) >> gs, gb) << 16) + | (to8((p & vi->blue_mask) >> bs, bb) << 8) | 0xff; + } + XDestroyImage(im); + return RB_COLOR; + } + glPixelStorei(GL_PACK_ALIGNMENT, read_unaligned ? 1 : 4); + glPixelStorei(GL_PACK_ROW_LENGTH, pack_row); + if (!strcmp(m, "front") || !strcmp(m, "back")) { + int isf = !strcmp(readtype, "float"); + int abgr = !strcmp(readfmt, "abgr"); + size_t bpp = isf ? 16 : 4; + char *mem = malloc((size_t)row * h * bpp + 8); + unsigned char *b = (unsigned char *)mem + (read_unaligned ? 1 : 0); + if (!mem) return -1; + glReadBuffer(!strcmp(m, "front") ? GL_FRONT : GL_BACK); +#ifdef GL_ABGR_EXT + glReadPixels(x, y, w, h, abgr ? GL_ABGR_EXT : GL_RGBA, isf ? GL_FLOAT : GL_UNSIGNED_BYTE, b); +#else + if (abgr) printf(" (no GL_ABGR_EXT, reading GL_RGBA)\n"); + abgr = 0; + glReadPixels(x, y, w, h, GL_RGBA, isf ? GL_FLOAT : GL_UNSIGNED_BYTE, b); +#endif + for (r = 0; r < h; r++) + for (c = 0; c < w; c++) { + unsigned long ch[4]; + int k; + for (k = 0; k < 4; k++) { + if (isf) { + float f; + memcpy(&f, b + ((size_t)r * row + c) * 16 + k * 4, 4); + ch[k] = (unsigned long)(f * 255.0f + 0.5f) & 0xff; + } else ch[k] = b[((size_t)r * row + c) * 4 + k]; + } + /* ABGR_EXT is A, B, G, R in memory. */ + vals[r * w + c] = abgr + ? (ch[3] << 24) | (ch[2] << 16) | (ch[1] << 8) | ch[0] + : (ch[0] << 24) | (ch[1] << 16) | (ch[2] << 8) | ch[3]; + } + free(mem); + return RB_COLOR; + } + if (!strcmp(m, "depth")) { + int isf = !strcmp(readtype, "float"); + char *mem = malloc((size_t)row * h * 4 + 8); + unsigned char *b = (unsigned char *)mem + (read_unaligned ? 1 : 0); + if (!mem) return -1; + glReadPixels(x, y, w, h, GL_DEPTH_COMPONENT, isf ? GL_FLOAT : GL_UNSIGNED_INT, b); + for (r = 0; r < h; r++) + for (c = 0; c < w; c++) { + if (isf) { + float f; + memcpy(&f, b + ((size_t)r * row + c) * 4, 4); + vals[r * w + c] = (unsigned long)((double)f * 4294967295.0); + } else { + unsigned int u; + memcpy(&u, b + ((size_t)r * row + c) * 4, 4); + vals[r * w + c] = u; + } + } + free(mem); + return RB_DEPTH; + } + if (!strcmp(m, "stencil")) { + unsigned char *b = malloc((size_t)row * h + 8); + if (!b) return -1; + glPixelStorei(GL_PACK_ALIGNMENT, 1); + glReadPixels(x, y, w, h, GL_STENCIL_INDEX, GL_UNSIGNED_BYTE, b); + for (r = 0; r < h; r++) + for (c = 0; c < w; c++) vals[r * w + c] = b[(size_t)r * row + c]; + free(b); + return RB_STENCIL; + } + return -1; +} + +static void print_val(int kind, unsigned long v) { + if (kind == RB_COLOR) printf("%08lx", v); + else if (kind == RB_DEPTH) printf("%08lx (%.3f)", v, (double)v / 4294967295.0); + else printf("%lu", v); +} + +/* PPM/PGM, top row first (image file convention). */ +static void write_image(const char *m, int kind, int w, int h, const unsigned long *vals) { + char name[256]; + FILE *f; + int r, c; + sprintf(name, "%.200s_%s.%s", read_out, m, kind == RB_COLOR ? "ppm" : "pgm"); + f = fopen(name, "wb"); + if (!f) { printf(" cannot write %s\n", name); return; } + fprintf(f, "%s\n%d %d\n255\n", kind == RB_COLOR ? "P6" : "P5", w, h); + for (r = h - 1; r >= 0; r--) + for (c = 0; c < w; c++) { + unsigned long v = vals[r * w + c]; + if (kind == RB_COLOR) { + fputc((int)(v >> 24) & 0xff, f); + fputc((int)(v >> 16) & 0xff, f); + fputc((int)(v >> 8) & 0xff, f); + } else if (kind == RB_DEPTH) fputc((int)(v >> 24) & 0xff, f); + else fputc(v > 4 ? 255 : (int)v * 60, f); + } + fclose(f); + printf(" wrote %s\n", name); +} + +static void do_reads(Display *dpy, Window win, XVisualInfo *vi) { + int x = 0, y = 0, w = win_w, h = win_h, i; + unsigned long *vals; + if (rrect[0] >= 0) { x = rrect[0]; y = rrect[1]; w = rrect[2]; h = rrect[3]; } + vals = malloc(sizeof(unsigned long) * (size_t)w * h); + if (!vals) return; + for (i = 0; i < nreads; i++) { + /* Sample points, relative to the read rectangle: corners 2 px in, + the centre, the quadrant centres (the --scene quadrants colours). */ + static const char *lbl[9] = { "bl", "br", "tl", "tr", "centre", "q.bl", "q.br", "q.tl", "q.tr" }; + int sx[9], sy[9], kind, k; + unsigned long sum = 2166136261UL; + GLenum err; + double t0, t1; + sx[0] = 2; sy[0] = 2; sx[1] = w - 3; sy[1] = 2; sx[2] = 2; sy[2] = h - 3; + sx[3] = w - 3; sy[3] = h - 3; sx[4] = w / 2; sy[4] = h / 2; + sx[5] = w / 4; sy[5] = h / 4; sx[6] = 3 * w / 4; sy[6] = h / 4; + sx[7] = w / 4; sy[7] = 3 * h / 4; sx[8] = 3 * w / 4; sy[8] = 3 * h / 4; + memset(vals, 0, sizeof(unsigned long) * (size_t)w * h); + while (glGetError() != GL_NO_ERROR) ; + t0 = now_ms(); + kind = read_pixels(dpy, win, vi, reads[i], x, y, w, h, vals); + t1 = now_ms(); + err = glGetError(); + if (kind < 0) continue; + for (k = 0; k < w * h; k++) { + /* Colour: RGB only, so XGetImage and glReadPixels compare. */ + unsigned long v = kind == RB_COLOR ? vals[k] >> 8 : vals[k]; + sum = ((sum ^ (v & 0xff)) * 16777619UL) & 0xffffffffUL; + sum = ((sum ^ ((v >> 8) & 0xff)) * 16777619UL) & 0xffffffffUL; + sum = ((sum ^ ((v >> 16) & 0xff)) * 16777619UL) & 0xffffffffUL; + sum = ((sum ^ ((v >> 24) & 0xff)) * 16777619UL) & 0xffffffffUL; + } + printf("glprim: read %s %dx%d at %d,%d%s%s%s: %.1f ms, gl error 0x%x, sum %08lx\n", + reads[i], w, h, x, y, + kind == RB_COLOR && strcmp(reads[i], "ximage") ? (strcmp(readfmt, "abgr") ? " rgba" : " abgr") : "", + strcmp(reads[i], "ximage") && strcmp(reads[i], "stencil") ? (!strcmp(readtype, "float") ? " float" : "") : "", + read_unaligned ? " unaligned" : "", t1 - t0, (unsigned)err, sum); + for (k = 0; k < 9; k++) { + printf(" %-6s (%3d,%3d) ", lbl[k], sx[k], sy[k]); + print_val(kind, vals[sy[k] * w + sx[k]]); + printf("\n"); + } + if (read_out) write_image(reads[i], kind, w, h, vals); + fflush(stdout); + } + free(vals); +} + /* ---- main ---------------------------------------------------------------- */ int main(int argc, char **argv) { @@ -611,6 +911,24 @@ int main(int argc, char **argv) { else glFlush(); glFinish(); + if (nreads > 0) { + if (dbl) { + /* The scene is now in front; give the back buffer its own + content: --backclear with a white 40x40 marker at the bottom left + (the "bl" sample). + Depth and stencil keep the scene's values. */ + glDrawBuffer(GL_BACK); + glDisable(GL_DEPTH_TEST); + glDisable(GL_SCISSOR_TEST); + glClearColor(back_rgb[0], back_rgb[1], back_rgb[2], 1.0f); + glClear(GL_COLOR_BUFFER_BIT); + glColor3f(1, 1, 1); + glRecti(0, 0, 40, 40); + glFinish(); + } + do_reads(dpy, win, vi); + } + if (hold_ms > 0) { int left = hold_ms; XSync(dpy, False); diff --git a/test/gltest/gltest b/test/gltest/gltest index 7a8f6a52..97cbc8d2 100755 Binary files a/test/gltest/gltest and b/test/gltest/gltest differ