From 12f1bb05ef5c2d7aa314a184dbff51ee34a7024d Mon Sep 17 00:00:00 2001 From: atomchild411 <143453386+atomchild411@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:38:34 -0800 Subject: [PATCH 1/5] mgras: an IMPACT board model in place of the ID stub src/mgras.rs answered the board-ID probe and nothing else. This is a model of the Indigo2 IMPACT (MGRAS) graphics board that the PROM and the IRIX X server draw on, as a module in src/mgras/: - dcb.rs: the display control bus devices -- the VC3 timing and cursor, the XMAP, the colour maps and the gamma DAC. - raster.rs: the raster engine -- lines (stippled too), rectangles and block fills, pixel logic ops, RGB and colour-index pixels, the overlay planes, window IDs -- plus its tests. - mod.rs: the GIO interface, the command FIFO and geometry engine port, host DMA for pixel transfers, the three interrupt lines (FIFO, general, vertical retrace), and scanout into a finished frame. The board presents like GR2 does: it implements `GfxDisplay` and hands the renderer a prebuilt frame (`Rex3Screen::prebuilt`), keeping `rgba` current so CI screenshots work headless. Register names follow OpenBSD's impact(4) driver, or say what the register does. `[impact]` accepts a board in the graphics slot only; a second head is not modelled, and exp0/exp1 are refused rather than silently ignored. Tested by the model's own unit tests here; it boots to the IRIX desktop on the IP28 that the next commit adds. Co-Authored-By: Claude Opus 5.5 --- src/config.rs | 19 +- src/machine.rs | 14 +- src/mgras.rs | 347 ---------- src/mgras/dcb.rs | 422 ++++++++++++ src/mgras/mod.rs | 1222 +++++++++++++++++++++++++++++++++ src/mgras/raster.rs | 823 ++++++++++++++++++++++ src/platform_profile_tests.rs | 37 +- 7 files changed, 2488 insertions(+), 396 deletions(-) delete mode 100644 src/mgras.rs create mode 100644 src/mgras/dcb.rs create mode 100644 src/mgras/mod.rs create mode 100644 src/mgras/raster.rs diff --git a/src/config.rs b/src/config.rs index f1c7fa78..b00fcbaa 100644 --- a/src/config.rs +++ b/src/config.rs @@ -469,23 +469,10 @@ impl ImpactSection { || self.exp1 != ImpactSlot::None } - /// Hardware-valid slot population (rejects High+High and orphan expansion boards). + /// One IMPACT board, in the graphics slot; a second head is not modelled yet. pub fn validate(&self) -> Result<(), String> { - let slots = [self.gfx, self.exp0, self.exp1]; - let high_count = slots.iter().filter(|&&s| s == ImpactSlot::High).count(); - if high_count >= 2 { - return Err( - "[impact] High+High is invalid — at most one High IMPACT board per system".into(), - ); - } - if self.exp0 != ImpactSlot::None && self.gfx == ImpactSlot::None { - return Err("[impact] exp0 requires gfx slot populated".into()); - } - if self.exp1 != ImpactSlot::None && self.exp0 == ImpactSlot::None { - return Err("[impact] exp1 requires exp0 populated (Maximum IMPACT uses all three slots)".into()); - } - if self.exp1 == ImpactSlot::Max && self.exp0 != ImpactSlot::High { - return Err("[impact] Maximum IMPACT expects exp0=high when exp1=max".into()); + if self.exp0 != ImpactSlot::None || self.exp1 != ImpactSlot::None { + return Err("[impact] only the graphics slot (gfx) is supported so far".into()); } Ok(()) } diff --git a/src/machine.rs b/src/machine.rs index dfde50b7..1ba79e46 100644 --- a/src/machine.rs +++ b/src/machine.rs @@ -520,7 +520,8 @@ impl Machine { let nvram_provenance = cfg.nvram.clone(); // REX3 Graphics — Newport only; skipped in headless mode or when XZ board selected - let rex3: Option> = if cfg.headless || cfg.graphics.board != crate::config::GraphicsBoard::Newport { + // An IMPACT board takes the graphics slot and the window; no Newport then. + let rex3: Option> = if cfg.headless || cfg.graphics.board != crate::config::GraphicsBoard::Newport || cfg.impact.any_enabled() { None } else { let r = Arc::new(Rex3::new(heartbeat.clone(), fasttick_count.clone(), decoded_count.clone(), Arc::clone(&l1i_hit_count), Arc::clone(&l1i_fetch_count), Arc::clone(&uncached_fetch_count))); @@ -578,9 +579,9 @@ impl Machine { } }; - // Indigo2 IMPACT/MGRAS preview — multi-slot GIO stub. + // Indigo2 IMPACT graphics in the GIO graphics slot. let mgras: Option> = if !guinness && cfg.impact.any_enabled() { - Some(Arc::new(crate::mgras::Mgras::new(&cfg.impact))) + Some(Arc::new(crate::mgras::Mgras::new(&cfg.impact, ioc.clone(), heartbeat.clone(), fasttick_count.clone()))) } else { None }; @@ -690,6 +691,7 @@ impl Machine { // Connect VINO to System Memory, install a video source, start DMA. // Source kind + broadcast standard come from `[vino]` in iris.toml. phys.vino.set_phys(phys.clone()); + if let Some(mgras) = &phys.mgras { mgras.set_phys(phys.clone()); } let standard = match cfg.vino.standard { crate::config::VinoStandard::Ntsc => crate::video_source::VideoStandard::Ntsc, crate::config::VinoStandard::Pal => crate::video_source::VideoStandard::Pal, @@ -827,6 +829,7 @@ impl Machine { // happen before hpc3.scsi().start() (called later, from // Machine::start) actually spawns the worker thread that reads it. if let Some(rex3) = &phys.rex3 { rex3.set_cpu_cycles(cpu.cycles_ptr()); } + if let Some(mgras) = &phys.mgras { mgras.set_cpu_cycles(cpu.cycles_ptr()); } if let Some(rex3) = &phys.rex3_head1 { rex3.set_cpu_cycles(cpu.cycles_ptr()); } if let Some(gr2) = &phys.gr2 { gr2.set_cpu_cycles(cpu.cycles_ptr()); } if let Some(td) = &phys.testdev { td.attach_core(cpu.core_ptr()); } @@ -1014,6 +1017,7 @@ impl Machine { // Program VC2 before the refresh thread runs so the first frame has size. self.apply_host_display_resolution(); if let Some(rex3) = &self._phys.rex3 { rex3.start(); } + if let Some(mgras) = &self._phys.mgras { mgras.start_display(); } if let Some(rex3) = &self._phys.rex3_head1 { rex3.start(); } if let Some(gr2) = &self._phys.gr2 { gr2.start(); } #[cfg(feature = "ultra64")] @@ -1102,6 +1106,7 @@ impl Machine { pub fn stop(&mut self) { self.cpu.stop(); if let Some(rex3) = &self._phys.rex3 { rex3.stop(); } + if let Some(mgras) = &self._phys.mgras { mgras.stop_display(); } if let Some(rex3) = &self._phys.rex3_head1 { rex3.stop(); } if let Some(gr2) = &self._phys.gr2 { gr2.stop(); } self.hpc3.stop(); @@ -1176,6 +1181,9 @@ impl Machine { if let Some(g) = &self._phys.gr2 { return Some(g.clone() as Arc); } + if let Some(m) = &self._phys.mgras { + return Some(m.clone() as Arc); + } self._phys.rex3.clone().map(|r| r as Arc) } diff --git a/src/mgras.rs b/src/mgras.rs deleted file mode 100644 index feb5e2b7..00000000 --- a/src/mgras.rs +++ /dev/null @@ -1,347 +0,0 @@ -/// IMPACT / MGRAS graphics — preview stub (Indigo2 IP22) -/// -/// Post-1995 IMPACT boards use the MGRAS ASIC set (geometry engine, raster -/// engine, TRAM controllers) spread across one to three GIO64 slots depending on -/// the option (Solid / High / Maximum). -/// -/// **Preview only:** per-slot register files with probe-friendly board IDs and -/// idle status. No TRAM, no DMA, no GL/command processing. -/// -/// GIO slot bases (physical, IP22): -/// gfx — 0x1F000000 (4 MB) -/// exp0 — 0x1F400000 (2 MB) -/// exp1 — 0x1F600000 (4 MB) -/// -/// See `docs/impact-mgras-research.md` for hardware notes and implementation status. - -use parking_lot::Mutex; -use std::io::Write as IoWrite; - -use crate::config::{ImpactSection, ImpactSlot}; -use crate::devlog::LogModule; -use crate::snapshot::{get_field, hex_u32, toml_u32}; -use crate::traits::{BusDevice, BusRead8, BusRead16, BusRead32, BusRead64, BUS_OK, Device, Saveable}; - -// ─── GIO slot geometry ─────────────────────────────────────────────────────── - -pub const MGRAS_SLOT_GFX_BASE: u32 = 0x1F00_0000; -pub const MGRAS_SLOT_GFX_SIZE: u32 = 0x0040_0000; -pub const MGRAS_SLOT_EXP0_BASE: u32 = 0x1F40_0000; -pub const MGRAS_SLOT_EXP0_SIZE: u32 = 0x0020_0000; -pub const MGRAS_SLOT_EXP1_BASE: u32 = 0x1F60_0000; -pub const MGRAS_SLOT_EXP1_SIZE: u32 = 0x0040_0000; - -/// MGRAS CPU register window within each populated slot (research placeholder; -/// mirrors the Newport/XZ `+0x0F0000` pattern until a verified map lands). -pub const MGRAS_REG_OFF: u32 = 0x000F_0000; -pub const MGRAS_REG_SIZE: u32 = 0x2000; - -// ─── Register offsets (relative to slot base + MGRAS_REG_OFF) ─────────────── - -pub mod reg { - use crate::config::ImpactSlot; - - pub const BOARD_ID: u32 = 0x0000; - pub const REVISION: u32 = 0x0004; - pub const STATUS: u32 = 0x0008; - pub const INTR_STATUS: u32 = 0x000C; - pub const INTR_ENABLE: u32 = 0x0010; - pub const FIFO_WRITE: u32 = 0x0018; - pub const SLOT_ROLE: u32 = 0x0024; // which GE/TRAM slice (multi-slot boards) - - pub const STATUS_IDLE: u32 = 0x0000_0007; // FIFO empty + engines idle (preview) - - pub fn board_id_for(slot: ImpactSlot) -> u32 { - match slot { - ImpactSlot::None => 0, - ImpactSlot::Solid => 0x004D_4752, // "MGR" + Solid class tag (ASCII-ish) - ImpactSlot::High => 0x004D_4748, // High IMPACT - ImpactSlot::Max => 0x004D_474D, // Maximum IMPACT - } - } - - pub fn revision_for(slot: ImpactSlot) -> u32 { - match slot { - ImpactSlot::None => 0, - ImpactSlot::Solid => 0x0000_0100, - ImpactSlot::High => 0x0000_0200, - ImpactSlot::Max => 0x0000_0300, - } - } -} - -#[derive(Clone, Copy)] -struct SlotMap { - kind: ImpactSlot, - base: u32, - size: u32, -} - -struct SlotState { - intr_status: u32, - intr_enable: u32, - fifo_depth: u32, -} - -struct MgrasState { - slots: [Option; 3], -} - -#[derive(Clone)] -pub struct Mgras { - map: [SlotMap; 3], - state: std::sync::Arc>, -} - -impl Mgras { - pub fn new(cfg: &ImpactSection) -> Self { - let kinds = [cfg.gfx, cfg.exp0, cfg.exp1]; - let bases = [MGRAS_SLOT_GFX_BASE, MGRAS_SLOT_EXP0_BASE, MGRAS_SLOT_EXP1_BASE]; - let sizes = [MGRAS_SLOT_GFX_SIZE, MGRAS_SLOT_EXP0_SIZE, MGRAS_SLOT_EXP1_SIZE]; - let mut map = [SlotMap { kind: ImpactSlot::None, base: 0, size: 0 }; 3]; - for i in 0..3 { - map[i] = SlotMap { kind: kinds[i], base: bases[i], size: sizes[i] }; - } - let mut slots = [None, None, None]; - for (i, m) in map.iter().enumerate() { - if m.kind != ImpactSlot::None { - slots[i] = Some(SlotState { intr_status: 0, intr_enable: 0, fifo_depth: 0 }); - } - } - Self { - map, - state: std::sync::Arc::new(Mutex::new(MgrasState { slots })), - } - } - - pub fn power_on(&self) { - let mut st = self.state.lock(); - for (i, m) in self.map.iter().enumerate() { - st.slots[i] = if m.kind != ImpactSlot::None { - Some(SlotState { intr_status: 0, intr_enable: 0, fifo_depth: 0 }) - } else { - None - }; - } - } - - pub fn any_slot(&self) -> bool { - self.map.iter().any(|m| m.kind != ImpactSlot::None) - } - - fn locate(&self, addr: u32) -> Option<(usize, u32)> { - for (i, m) in self.map.iter().enumerate() { - if m.kind == ImpactSlot::None { - continue; - } - let reg_base = m.base.wrapping_add(MGRAS_REG_OFF); - if (addr & 0xFFFF_E000) == reg_base { - return Some((i, addr & (MGRAS_REG_SIZE - 1))); - } - // Rest of the slot aperture: reads as 0, writes ignored. - if addr >= m.base && addr < m.base.wrapping_add(m.size) { - return None; - } - } - None - } - - fn read_reg(&self, slot_idx: usize, off: u32) -> u32 { - let kind = self.map[slot_idx].kind; - let st = self.state.lock(); - let Some(s) = &st.slots[slot_idx] else { return 0 }; - match off { - reg::BOARD_ID => reg::board_id_for(kind), - reg::REVISION => reg::revision_for(kind), - reg::STATUS => reg::STATUS_IDLE, - reg::INTR_STATUS => s.intr_status, - reg::INTR_ENABLE => s.intr_enable, - reg::SLOT_ROLE => slot_idx as u32, - _ => 0, - } - } - - fn write_reg(&self, slot_idx: usize, off: u32, val: u32) { - let mut st = self.state.lock(); - let Some(s) = &mut st.slots[slot_idx] else { return }; - match off { - reg::INTR_STATUS => s.intr_status &= !val, - reg::INTR_ENABLE => s.intr_enable = val, - reg::FIFO_WRITE => { - s.fifo_depth = s.fifo_depth.saturating_add(4); - dlog_dev!( - LogModule::Rex3, - "MGRAS slot{}: FIFO write {:08x} (stub, depth={})", - slot_idx, - val, - s.fifo_depth - ); - } - _ => { - dlog_dev!( - LogModule::Rex3, - "MGRAS slot{}: write reg {:04x} = {:08x} (ignored)", - slot_idx, - off, - val - ); - } - } - } -} - -impl Device for Mgras { - fn step(&self, _cycles: u64) {} - fn stop(&self) {} - fn start(&self) {} - fn is_running(&self) -> bool { false } - fn get_clock(&self) -> u64 { 0 } - - fn register_commands(&self) -> Vec<(String, String)> { - vec![ - ("mgras".into(), "IMPACT/MGRAS preview stub (status)".into()), - ("impact".into(), "IMPACT inventory summary (hinv-style preview)".into()), - ] - } - - fn execute_command(&self, cmd: &str, args: &[&str], mut writer: Box) -> Result<(), String> { - if cmd != "mgras" && cmd != "impact" { - return Err(format!("unknown mgras command: {cmd}")); - } - if !args.is_empty() { - return Err(format!("usage: {cmd}")); - } - let st = self.state.lock(); - let names = ["gfx", "exp0", "exp1"]; - if cmd == "impact" { - writeln!(writer, "Graphics inventory (IMPACT preview — driver attach not implemented):").map_err(|e| e.to_string())?; - } - for (i, m) in self.map.iter().enumerate() { - if m.kind == ImpactSlot::None { - if cmd == "impact" { - continue; - } - writeln!(writer, " {}: (empty)", names[i]).map_err(|e| e.to_string())?; - continue; - } - let depth = st.slots[i].as_ref().map(|s| s.fifo_depth).unwrap_or(0); - let label = match m.kind { - ImpactSlot::Solid => "IMPACT Solid", - ImpactSlot::High => "IMPACT High", - ImpactSlot::Max => "IMPACT Maximum", - ImpactSlot::None => unreachable!(), - }; - if cmd == "impact" { - writeln!( - writer, - " Graphics board {}: {} (GIO @ {:#010x})", - i, - label, - m.base.wrapping_add(MGRAS_REG_OFF), - ) - .map_err(|e| e.to_string())?; - } else { - writeln!( - writer, - " {}: {:?} @ {:#010x} fifo_bytes={}", - names[i], - m.kind, - m.base.wrapping_add(MGRAS_REG_OFF), - depth, - ) - .map_err(|e| e.to_string())?; - } - } - if cmd == "impact" && !self.map.iter().any(|m| m.kind != ImpactSlot::None) { - writeln!(writer, " (no IMPACT boards configured in [impact])").map_err(|e| e.to_string())?; - } - Ok(()) - } -} - -impl BusDevice for Mgras { - fn read32(&self, addr: u32) -> BusRead32 { - match self.locate(addr) { - Some((idx, off)) => BusRead32::ok(self.read_reg(idx, off)), - None => BusRead32::ok(0), - } - } - - fn write32(&self, addr: u32, val: u32) -> u32 { - if let Some((idx, off)) = self.locate(addr) { - self.write_reg(idx, off, val); - } - BUS_OK - } - - fn read8(&self, addr: u32) -> BusRead8 { - BusRead8::ok(self.read32(addr & !3).data as u8) - } - fn write8(&self, addr: u32, val: u8) -> u32 { - let shift = (addr & 3) * 8; - let cur = self.read32(addr & !3).data; - self.write32(addr & !3, (cur & !(0xFF << shift)) | ((val as u32) << shift)); - BUS_OK - } - fn read16(&self, addr: u32) -> BusRead16 { - BusRead16::ok(self.read32(addr & !3).data as u16) - } - fn write16(&self, addr: u32, val: u16) -> u32 { - let shift = if (addr & 2) != 0 { 16 } else { 0 }; - let cur = self.read32(addr & !3).data; - self.write32(addr & !3, (cur & !(0xFFFF << shift)) | ((val as u32) << shift)); - BUS_OK - } - fn read64(&self, addr: u32) -> BusRead64 { - let lo = self.read32(addr).data as u64; - let hi = self.read32(addr.wrapping_add(4)).data as u64; - BusRead64::ok((hi << 32) | lo) - } - fn write64(&self, addr: u32, val: u64) -> u32 { - self.write32(addr, val as u32); - self.write32(addr.wrapping_add(4), (val >> 32) as u32); - BUS_OK - } -} - -impl Saveable for Mgras { - fn save_state(&self) -> toml::Value { - let st = self.state.lock(); - let mut slots = toml::map::Map::new(); - let names = ["gfx", "exp0", "exp1"]; - for (i, name) in names.iter().enumerate() { - if self.map[i].kind == ImpactSlot::None { - continue; - } - let Some(s) = st.slots[i].as_ref() else { continue }; - let mut slot_tbl = toml::map::Map::new(); - slot_tbl.insert("intr_status".into(), hex_u32(s.intr_status)); - slot_tbl.insert("intr_enable".into(), hex_u32(s.intr_enable)); - slot_tbl.insert("fifo_depth".into(), toml::Value::Integer(s.fifo_depth as i64)); - slots.insert((*name).into(), toml::Value::Table(slot_tbl)); - } - toml::Value::Table(slots) - } - - fn load_state(&self, v: &toml::Value) -> Result<(), String> { - let Some(tbl) = v.as_table() else { return Ok(()) }; - let names = ["gfx", "exp0", "exp1"]; - let mut st = self.state.lock(); - for (i, name) in names.iter().enumerate() { - let Some(slot_v) = tbl.get(*name) else { continue }; - let Some(s) = st.slots[i].as_mut() else { continue }; - if let Some(x) = get_field(slot_v, "intr_status") { - s.intr_status = toml_u32(x).unwrap_or(s.intr_status); - } - if let Some(x) = get_field(slot_v, "intr_enable") { - s.intr_enable = toml_u32(x).unwrap_or(s.intr_enable); - } - if let Some(x) = get_field(slot_v, "fifo_depth") { - if let Some(n) = x.as_integer() { - s.fifo_depth = n as u32; - } - } - } - Ok(()) - } -} diff --git a/src/mgras/dcb.rs b/src/mgras/dcb.rs new file mode 100644 index 00000000..42b88470 --- /dev/null +++ b/src/mgras/dcb.rs @@ -0,0 +1,422 @@ +//! The display control bus (DCB): the board's slow side bus to its video +//! chips, reached through a 32 KB window at slot offset `0x60000`. +//! +//! Each device owns a 1 KB window (`0x60000 + dev * 0x400`), and the address +//! within it encodes the transaction: bits 9:7 select the chip's register +//! (its "CRS" line), bits 4:3 the transfer width in bytes (0 = 4), bit 5 asks +//! the chip to increment its register select, bit 6 packs data. Data rides in +//! the most significant bytes of a 32-bit bus access, so a one-byte register +//! read with a word load comes back in bits 31:24. + +use std::collections::HashMap; + +/// Device numbers on the bus. +pub const DEV_CMAP_ALL: u32 = 3; +pub const DEV_CMAP0: u32 = 4; +pub const DEV_CMAP1: u32 = 5; +pub const DEV_DAC: u32 = 6; +pub const DEV_XMAP: u32 = 7; +pub const DEV_VC3: u32 = 8; +pub const DEV_BDVERS: u32 = 9; +pub const DEV_I2C: u32 = 11; + +/// One decoded bus transaction. +#[derive(Clone, Copy, Debug)] +pub struct Txn { + pub dev: u32, + pub crs: u32, + /// Transfer width in bytes, 1-4. + pub width: u32, +} + +impl Txn { + /// `off` is the offset within the slot, in `0x60000..0x68000`. + pub fn decode(off: u32) -> Self { + let w = off & 0x3FF; + let width = match (w >> 3) & 3 { 0 => 4, n => n }; + Txn { dev: (off - 0x60000) >> 10, crs: (w >> 7) & 7, width } + } + + /// The data a CPU store of `bits` carries for this transaction. + pub fn store_data(&self, bits: u32, val: u64) -> u32 { + match bits { + 8 => val as u32 & 0xFF, + 16 => val as u32 & 0xFFFF, + 32 => (val as u32) >> (8 * (4 - self.width)), + _ => ((val >> 32) as u32) >> (8 * (4 - self.width)), + } + } + + /// Place `data` where a CPU load of `bits` expects it. + pub fn load_value(&self, bits: u32, data: u32) -> u64 { + match bits { + 8 => (data & 0xFF) as u64, + 16 => (data & 0xFFFF) as u64, + 32 => (data << (8 * (4 - self.width))) as u64 & 0xFFFF_FFFF, + _ => ((data << (8 * (4 - self.width))) as u64) << 32, + } + } +} + +/// A colormap chip: 8192 entries of 24-bit RGB. +pub struct Cmap { + pub pal: Vec, + addr: u32, + rev: u32, + cmd: u32, +} + +impl Cmap { + fn new(rev: u32) -> Self { + Cmap { pal: vec![0; 8192], addr: 0, rev, cmd: 0 } + } + + fn write(&mut self, t: Txn, d: u32) { + match (t.crs, t.width) { + // Address: one byte at a time (low, then high), or 16 bits sent low + // byte first. + (0, 1) => self.addr = (self.addr & 0x1F00) | d, + (0, _) => self.addr = (((d & 0xFF) << 8) | (d >> 8 & 0xFF)) & 0x1FFF, + (1, _) => self.addr = ((d & 0x1F) << 8) | (self.addr & 0xFF), + // Palette entry: red, green, blue; the address advances. + (2, _) => { + self.pal[self.addr as usize] = d & 0xFF_FFFF; + self.addr = (self.addr + 1) & 0x1FFF; + } + (3, _) => self.cmd = d, + _ => {} + } + } + + fn read(&self, t: Txn) -> u32 { + match t.crs { + 0 => self.addr & 0xFF, + 1 => self.addr >> 8, + 2 => self.pal[self.addr as usize], + 3 => self.cmd, + 4 => 0x08, // status: ready for writes + 6 => self.rev, + _ => 0, + } + } +} + +/// The RAMDAC: an address register selecting internal registers, and a +/// 256-entry, 10-bit gamma table written as red, green, blue in turn. +pub struct Dac { + addr: u32, + regs: HashMap, + pub gamma: Vec<[u16; 3]>, + gamma_comp: usize, + mode: u32, +} + +/// DAC register: the pixel read mask; zero blanks the screen. +pub const DAC_PIXMASK: u32 = 4; + +impl Dac { + fn new() -> Self { + let gamma = (0..256).map(|i| { let v = (i << 2) as u16; [v, v, v] }).collect(); + Dac { addr: 0, regs: HashMap::new(), gamma, gamma_comp: 0, mode: 0 } + } + + fn write(&mut self, t: Txn, d: u32) { + match t.crs { + // 16 bits, low byte first; a single byte sets the low half. + 0 if t.width >= 2 => { self.addr = ((d & 0xFF) << 8) | (d >> 8 & 0xFF); self.gamma_comp = 0; } + 0 => { self.addr = d & 0xFF; self.gamma_comp = 0; } + 1 => { + let i = (self.addr & 0xFF) as usize; + // Ten-bit entries: a byte write gives the top eight bits; a + // 16-bit write gives bits 9:2 in its first byte and 1:0 in + // the second. + let v = if t.width >= 2 { (((d >> 8 & 0xFF) << 2) | (d & 3)) as u16 } else { (d as u16) << 2 }; + self.gamma[i][self.gamma_comp] = v & 0x3FF; + self.gamma_comp += 1; + if self.gamma_comp == 3 { + self.gamma_comp = 0; + self.addr = self.addr.wrapping_add(1); + } + } + 2 => { self.regs.insert(self.addr, d & 0xFF); } + 3 => self.mode = d, + _ => {} + } + } + + fn read(&self, t: Txn) -> u32 { + match t.crs { + 0 => self.addr, + 2 => self.regs.get(&self.addr).copied().unwrap_or(0), + 3 => self.mode, + _ => 0, + } + } + + pub fn pixmask(&self) -> u32 { + self.regs.get(&DAC_PIXMASK).copied().unwrap_or(0xFF) + } +} + +/// The XMAP: the pixel processors' display side (display modes per window ID, +/// buffer selects, scanout pointers), reached as an index register (`INDEX`) +/// plus register files on the selects after it. +pub struct Xmap { + index: u32, + regs: HashMap<(u32, u32), u32>, +} + +/// XMAP registers, by select (named as in OpenBSD's impact(4)). +mod xmap { + /// Which pixel processor hears writes. + pub const PP1SELECT: u32 = 0; + pub const INDEX: u32 = 1; + pub const CONFIG: u32 = 2; + pub const BUF_SELECT: u32 = 3; + /// Display mode per window ID, at index `did * 4`. + pub const MAIN_MODE: u32 = 4; + pub const OVERLAY_MODE: u32 = 5; + /// The raster engine to pixel processor link. + pub const RE_RAC: u32 = 7; +} + +/// `CONFIG` index of the byte whose bit 3 makes the index auto-increment. +const XMAP_CONFIG_BYTE: u32 = 1; +const XMAP_AUTOINC: u32 = 0x08; +/// `CONFIG` index 4: the pixel processor revision. +const XMAP_REV_INDEX: u32 = 4; +/// `CONFIG` index read before each display-mode write. +const XMAP_MODE_ROOM_INDEX: u32 = 8; + +impl Xmap { + fn new() -> Self { + Xmap { index: 0, regs: HashMap::new() } + } + + fn autoinc(&mut self, t: Txn) { + let cfg = self.regs.get(&(xmap::CONFIG, XMAP_CONFIG_BYTE)).copied().unwrap_or(0); + if t.crs >= xmap::BUF_SELECT && cfg & XMAP_AUTOINC != 0 { + self.index = self.index.wrapping_add(t.width); + } + } + + fn write(&mut self, t: Txn, d: u32) { + match t.crs { + xmap::PP1SELECT => { self.regs.insert((xmap::PP1SELECT, 0), d); } + xmap::INDEX => self.index = d, + crs => { + self.regs.insert((crs, self.index), d); + self.autoinc(t); + } + } + } + + fn read(&mut self, t: Txn) -> u32 { + let v = match t.crs { + xmap::INDEX => self.index, + // The raster engine to pixel processor link reports its sync + // signature here; 1 in each nibble lane means "in sync". + xmap::RE_RAC => 0x0001_0101, + xmap::CONFIG if self.index == XMAP_REV_INDEX => 1, + // Polled nonzero before every display-mode write: room for a + // mode update. + xmap::CONFIG if self.index == XMAP_MODE_ROOM_INDEX => 0x10, + crs => self.regs.get(&(crs, self.index)).copied().unwrap_or(0), + }; + if t.crs >= xmap::CONFIG { self.autoinc(t); } + v + } + + /// Display mode of overlay window ID `did` (`OVERLAY_MODE`, index `did * 4`); + /// 0 is "overlay off". + pub fn overlay_mode(&self, did: u32) -> u32 { + self.regs.get(&(xmap::OVERLAY_MODE, (did & 0x1F) << 2)).copied().unwrap_or(0) + } + + /// Colormap address of cursor colour 0: the config register (`CONFIG`, + /// index 0) holds it divided by four. + pub fn cursor_cmap_base(&self) -> usize { + (self.regs.get(&(xmap::CONFIG, 0)).copied().unwrap_or(0) as usize & 0x7FF) << 2 + } + + /// Display mode of window ID `did` (`MAIN_MODE`, index `did * 4`). + pub fn main_mode(&self, did: u32) -> u32 { + self.regs.get(&(xmap::MAIN_MODE, did << 2)).copied().unwrap_or(0) + } +} + +/// The video timing chip: indexed 16-bit registers and a 32K x 16 SRAM holding +/// line and frame tables, the cursor glyph and window-ID tables. +/// VC3 cursor registers: glyph address, position, and control (bit 0 +/// enable, bit 1 display, bit 3 64x64 glyph). +const CURSOR_GLYPH_REG: usize = 1; +const CURSOR_X_REG: usize = 2; +const CURSOR_Y_REG: usize = 3; +const CURSOR_CONTROL_REG: usize = 0x1D; +/// VC3 register: SRAM address of the main planes' window-ID frame table. +const MAIN_WID_FRAME_REG: usize = 4; +const OVERLAY_WID_FRAME_REG: usize = 5; + +pub struct Vc3 { + index: u32, + pub regs: [u16; 32], + pub sram: Vec, + line_counter: u16, +} + +pub const VC3_SRAM_POINTER: usize = 7; +const VC3_CURRENT_LINE: usize = 0xB; + +impl Vc3 { + fn new() -> Self { + Vc3 { index: 0, regs: [0; 32], sram: vec![0; 0x8000], line_counter: 0 } + } + + /// The hardware cursor, when shown: the screen position of its top-left + /// corner, its size (32 or 64), and the SRAM address of its glyph. The + /// glyph is two bit planes, one after the other, a row at a time, most + /// significant bit leftmost; the planes make a two-bit colour, 0 being + /// transparent. The position registers are offset by 31. + pub fn cursor(&self) -> Option<(i32, i32, usize, usize)> { + let ctl = self.regs[CURSOR_CONTROL_REG]; + if ctl & 3 != 3 { + return None; + } + let size = if ctl & 8 != 0 { 64 } else { 32 }; + Some(( + self.regs[CURSOR_X_REG] as i32 - 31, + self.regs[CURSOR_Y_REG] as i32 - 31, + size, + self.regs[CURSOR_GLYPH_REG] as usize, + )) + } + + /// Window-ID runs of scanline `y` (0 is the top) in the main planes, as + /// `(first x, did)`. Register 4 points at the frame table, one line-table + /// address per scanline; a line table lists `(x << 5) | did` entries, + /// each starting a run, and ends at x = 0x7FF. + pub fn main_did_runs(&self, y: usize, out: &mut Vec<(u16, u8)>) { + self.did_runs(MAIN_WID_FRAME_REG, y, out) + } + + /// The same for the overlay planes, from register 5's frame table. + pub fn overlay_did_runs(&self, y: usize, out: &mut Vec<(u16, u8)>) { + self.did_runs(OVERLAY_WID_FRAME_REG, y, out) + } + + fn did_runs(&self, reg: usize, y: usize, out: &mut Vec<(u16, u8)>) { + out.clear(); + let frame = self.regs[reg] as usize; + let line = self.sram[(frame + y) & 0x7FFF]; + if line == 0xFFFF { + return; + } + for k in 0..64 { + let e = self.sram[(line as usize + k) & 0x7FFF]; + let x = e >> 5; + if x == 0x7FF { + break; + } + out.push((x, (e & 0x1F) as u8)); + } + } + + fn sram_write(&mut self, v: u16) { + let a = self.regs[VC3_SRAM_POINTER] as usize & 0x7FFF; + self.sram[a] = v; + self.regs[VC3_SRAM_POINTER] = self.regs[VC3_SRAM_POINTER].wrapping_add(1); + } + + fn sram_read(&mut self) -> u16 { + let a = self.regs[VC3_SRAM_POINTER] as usize & 0x7FFF; + self.regs[VC3_SRAM_POINTER] = self.regs[VC3_SRAM_POINTER].wrapping_add(1); + self.sram[a] + } + + fn write(&mut self, t: Txn, d: u32) { + match (t.crs, t.width) { + (0, 1) => self.index = d & 0x1F, + // Index and 16-bit value in one three-byte transfer. + (0, _) => { + self.index = (d >> 16) & 0x1F; + self.regs[self.index as usize] = d as u16; + } + (1, _) => self.regs[self.index as usize] = d as u16, + (3, 4) => { self.sram_write((d >> 16) as u16); self.sram_write(d as u16); } + (3, _) => self.sram_write(d as u16), + _ => {} + } + } + + fn read(&mut self, t: Txn) -> u32 { + match (t.crs, t.width) { + (0, _) => self.index, + (1, _) if self.index as usize == VC3_CURRENT_LINE => { + // Keep moving so vertical-blank polls terminate. + self.line_counter = (self.line_counter + 1) % 1066; + self.line_counter as u32 + } + (1, _) => self.regs[self.index as usize] as u32, + (3, 4) => { let hi = self.sram_read() as u32; (hi << 16) | self.sram_read() as u32 } + (3, _) => self.sram_read() as u32, + _ => 0, + } + } +} + +/// The whole bus: all devices of one board. +pub struct Dcb { + pub cmap: [Cmap; 2], + pub dac: Dac, + pub xmap: Xmap, + pub vc3: Vc3, + bdvers: [u32; 2], + bc1: u32, + other: HashMap<(u32, u32), u32>, +} + +impl Dcb { + /// `bdvers` are the two board-version bytes; `cmap_rev` the colormaps' + /// revision registers. + pub fn new(bdvers: [u32; 2], cmap_rev: [u32; 2]) -> Self { + Dcb { + cmap: [Cmap::new(cmap_rev[0]), Cmap::new(cmap_rev[1])], + dac: Dac::new(), + xmap: Xmap::new(), + vc3: Vc3::new(), + bdvers, + bc1: 0, + other: HashMap::new(), + } + } + + pub fn write(&mut self, t: Txn, d: u32) { + match t.dev { + DEV_CMAP_ALL => { self.cmap[0].write(t, d); self.cmap[1].write(t, d); } + DEV_CMAP0 => self.cmap[0].write(t, d), + DEV_CMAP1 => self.cmap[1].write(t, d), + DEV_DAC => self.dac.write(t, d), + DEV_XMAP => self.xmap.write(t, d), + DEV_VC3 => self.vc3.write(t, d), + DEV_BDVERS => { + if t.crs == 1 { self.bc1 = d; } + } + DEV_I2C => {} + dev => { self.other.insert((dev, t.crs), d); } + } + } + + pub fn read(&mut self, t: Txn) -> u32 { + match t.dev { + DEV_CMAP_ALL | DEV_CMAP0 => self.cmap[0].read(t), + DEV_CMAP1 => self.cmap[1].read(t), + DEV_DAC => self.dac.read(t), + DEV_XMAP => self.xmap.read(t), + DEV_VC3 => self.vc3.read(t), + DEV_BDVERS => self.bdvers.get(t.crs as usize).copied().unwrap_or(0), + // No flat panel: the I2C controller reads back nothing. + DEV_I2C => 0, + dev => self.other.get(&(dev, t.crs)).copied().unwrap_or(0), + } + } +} diff --git a/src/mgras/mod.rs b/src/mgras/mod.rs new file mode 100644 index 00000000..b062389a --- /dev/null +++ b/src/mgras/mod.rs @@ -0,0 +1,1222 @@ +//! IMPACT (MGRAS) graphics for the Indigo2. +//! +//! One board occupies the GIO graphics slot (`0x1F000000`) and decodes a 1 MB +//! window there. The host talks to one chip on it, the host interface, +//! which offers: +//! +//! | offset | what | +//! |-------------------|---------------------------------------------------| +//! | `0x00000` | GIO ID | +//! | `0x40000-0x45FFF` | command-processor microcode RAM (24-bit words) | +//! | `0x50000-0x5FFFF` | privileged registers, privileged command FIFO | +//! | `0x60000-0x67FFF` | display control bus devices (see `dcb`) | +//! | `0x68000-0x6FFFF` | per-device bus protocol registers | +//! | `0x70000-0x7BFFF` | user registers: status, flags, command FIFO | +//! | `0x7C000-0x7FFFF` | raster registers, direct access (see `raster`); | +//! | | the alias at `+0x1000` also executes the IR | +//! +//! Drawing arrives through the command FIFO as (command, data) pairs. Commands +//! at `0x1000` and up write raster registers directly, with bit `0x400` +//! meaning "execute"; lower numbers go to the command processor's microcode, +//! which this model does not run (the PROM, the kernel's console and the X +//! server's 2D paths never need it). The FIFO is drained as it is written, so +//! it always reads as empty. +//! +//! The model scans its framebuffer out through the colormap and DAC gamma into +//! the same window and status bar Newport uses (`GfxDisplay`). +//! +//! `IRIS_MGRAS_TRACE=` logs every access to the board. + +mod dcb; +mod raster; + +use parking_lot::Mutex; +use std::collections::{HashMap, HashSet}; +use std::io::Write as IoWrite; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::Arc; + +use crate::config::{ImpactSection, ImpactSlot}; +use crate::rex3::Renderer; +use crate::traits::{BusDevice, BusRead8, BusRead16, BusRead32, BusRead64, BUS_OK, Device, Saveable}; + +/// The graphics slot the board decodes. +pub const MGRAS_SLOT_GFX_BASE: u32 = 0x1F00_0000; +pub const MGRAS_SLOT_GFX_SIZE: u32 = 0x0040_0000; +/// The board's register window within the slot. +const MAP_SIZE: u32 = 0x10_0000; + +/// GIO ID: product 0x10, 32-bit ID, revision 1, GIO64, no ROM, manufacturer 1. +pub const GIO_ID: u32 = 0x0005_0190; + +/// Host interface register offsets. +mod host { + pub const UCODE: u32 = 0x40000; + pub const UCODE_END: u32 = 0x46000; + pub const SET_FLAGS_PRIVILEGED: u32 = 0x50008; + pub const CLEAR_FLAGS_PRIVILEGED: u32 = 0x5000C; + pub const CFIFO_PRIVILEGED: u32 = 0x50080; + pub const STATUS: u32 = 0x70000; + pub const FIFOSTATUS: u32 = 0x70004; + pub const SET_FLAGS: u32 = 0x70008; + pub const CLEAR_FLAGS: u32 = 0x7000C; + pub const GE_READBACK_HI: u32 = 0x70010; + pub const GE_READBACK_LO: u32 = 0x70014; + pub const CFIFO: u32 = 0x70080; + pub const GIOSTATUS: u32 = 0x70100; + pub const DMABUSY: u32 = 0x70104; + pub const RASTER: u32 = 0x7C000; + pub const RASTER_END: u32 = 0x80000; + + /// Status: raster idle and host idle, command and data FIFOs at or below + /// their low-water marks. + pub const STATUS_IDLE: u32 = 0x01 | 0x02 | 0x10 | 0x40; + pub const STATUS_VBLANK: u32 = 0x04; + + /// Flag set by the "set flag" command (0xE04): DMA/sync completion. + pub const FLAG_DONE: u32 = 1 << 16; + /// Flag set when the geometry engine has data waiting to be read back. + pub const FLAG_GE_DATA: u32 = 1 << 17; + /// Flag set while a geometry engine diagnostic readback has data waiting. + pub const FLAG_GE_DIAG: u32 = 1 << 18; + /// Context switch: outgoing context saved (phase 1) and incoming context + /// loaded (phase 2). + pub const FLAG_CONTEXT_SAVED: u32 = 1 << 19; + pub const FLAG_CONTEXT_LOADED: u32 = 1 << 6; + /// Command-processor flag 0: a scheduled buffer swap has happened. + pub const FLAG_CP0: u32 = 1 << 10; + /// Flags that can raise the general interrupt. + pub const INTR_CAUSES: u32 = 0x7F_FFFF; + + /// Flag and interrupt enables: read at the first address, write-1-to-set + /// there, write-1-to-clear at the second. + pub const FLAG_ENABLE_SET: u32 = 0x50010; + pub const FLAG_ENABLE_CLEAR: u32 = 0x50014; + pub const INTERRUPT_ENABLE_SET: u32 = 0x50018; + pub const INTERRUPT_ENABLE_CLEAR: u32 = 0x5001C; + /// Context switch request: starts the switch routine at the written + /// microcode address. + pub const CONTEXT_SWITCH: u32 = 0x50050; + /// Words of incoming context the host pushes after the save phase. + pub const CONTEXT_SWITCH_WORDS: u32 = 63; + + /// Geometry engine diagnostic ports, per engine: data, then address. + pub const GE_DIAG: [(u32, u32); 2] = [(0x50040, 0x50044), (0x50048, 0x5004C)]; + /// Diagnostic readback words: a discarded word, then the data word. + pub const GE_DIAG_READ_PAD: u32 = 0x50230; + pub const GE_DIAG_READ: u32 = 0x5022C; + + /// Host DMA engine and raster-interface context, read back per register. + pub const DMA_CONTEXT: u32 = 0x50300; + pub const RASTER_IF_CONTEXT: u32 = 0x50200; + /// PIO pixel reads: the raster char registers, high word at the execute + /// alias (which takes the next doubleword), low word at the plain one. + pub const PIO_READ_HI: u32 = 0x7D1C0; + pub const PIO_READ_LO: u32 = 0x7C1C4; +} + +/// A geometry engine as its diagnostic port sees it. The engine itself does +/// not run; it only has to hold downloaded microcode, read it back, and +/// answer "started". +#[derive(Default)] +struct GeDiag { + /// Current diagnostic address: a microcode line (from `UCODE_BASE`) or an + /// internal register. + addr: u32, + /// Which 32-bit word of the current 72-bit microcode line comes next. + word: usize, + ucode: HashMap, +} + +impl GeDiag { + const UCODE_BASE: u32 = 0x20_0000; + /// Internal register: execution control; bit 0 starts the engine. + const EXEC_CONTROL: u32 = 0x4_0000; + + /// Words a readback starting at `addr` delivers, in order: for each pair + /// of lines, word 0 and the top byte of the first, then word 1 of the + /// second. A readback starting on an odd line is preceded by two words + /// that carry nothing. (The driver's verifier reads exactly this; the + /// order is inferred from it.) + fn readback(&self, lines: u32) -> Vec { + let w = |l: u32, i: usize| self.ucode.get(&l).map(|x| x[i]).unwrap_or(0); + let mut out = Vec::new(); + let mut l = self.addr; + if l.wrapping_sub(Self::UCODE_BASE) & 1 == 1 { + out.extend([0, 0]); + } + for _ in 0..lines / 2 { + out.extend([w(l, 0), w(l, 2) & 0xFF, w(l + 1, 1)]); + l += 2; + } + out + } +} + +/// Command FIFO command numbers. +mod cmd { + /// Command-processor token: schedule a buffer swap for the next retrace. + pub const CP_SCHEDULE_SWAP: u32 = 0x37; + pub const SET_DONE_FLAG: u32 = 0xE04; + pub const RASTER_BASE: u32 = 0x1000; + pub const RASTER_EXECUTE: u32 = 0x400; + pub const DMA_BASE: u32 = 0x800; + pub const RASTER_IF_BASE: u32 = 0xA00; + pub const FORMATTER: u32 = 0xC00; + /// Below this, commands are command-processor microcode tokens. + pub const CP_LIMIT: u32 = 0x200; +} + +/// The command FIFO's word-stream parser: a command word, then the data words +/// its byte count announces. +#[derive(Default)] +struct Cfifo { + cmd: u32, + pixel: bool, + need: u32, + data: Vec, +} + +/// Host DMA engine registers. +mod dma { + pub const PAGE_LIST: usize = 0x00; + pub const STRIDE: usize = 0x04; + pub const ROW_OFFSET: usize = 0x05; + pub const ROW_START: usize = 0x06; + pub const LINES: usize = 0x07; + pub const LINE_BYTES: usize = 0x08; + /// The start word: bit 0 run, bits 2:1 pool, bit 3 board to host. + pub const START: usize = 0x0B; + /// Page-table base of pool `p`: eight bytes at `TABLE_BASE + 2p`, the + /// address in the low word. + pub const TABLE_BASE: usize = 0x20; +} + +/// Board state behind one lock. +struct Board { + regs: HashMap, + ucode: Vec, + flags: u32, + flag_enable: u32, + interrupt_enable: u32, + /// Context-switch packet words still to swallow from the FIFO. + context_words: u32, + ge_readback: [u32; 2], + ge: [GeDiag; 2], + /// Pending diagnostic readback words, oldest first. + ge_out: std::collections::VecDeque, + /// One parser per FIFO port (user, privileged): the two may interleave. + cfifo: [Cfifo; 2], + /// Host DMA engine registers as 32-bit words; an eight-byte register + /// takes two, high word first. + dma_regs: [u32; 0x80], + raster_if_regs: [u32; 0x10], + formatter: u32, + /// A board-to-host DMA started on the host side, waiting for the raster + /// engine to be started (xfrcontrol = 9). + dma_read_pending: Option, + /// System memory, for DMA. + mem: Option>, + /// DMA transfers logged so far (bring-up). + dma_logged: u32, + dcb: dcb::Dcb, + raster: raster::Raster, + /// Commands and registers not modelled yet, reported once each. + unhandled: HashSet, +} + +impl Board { + fn new(kind: ImpactSlot) -> Self { + // Board version bytes: [RA/RB boards + TRAMs, product + GE count]. + let bdvers = match kind { + ImpactSlot::Solid => [0x70, 0x21], + ImpactSlot::High => [0x70, 0x01], + ImpactSlot::Max => [0x00, 0x02], + ImpactSlot::None => [0, 0], + }; + Board { + regs: HashMap::new(), + ucode: vec![0; ((host::UCODE_END - host::UCODE) / 4) as usize], + flags: 0, + flag_enable: 0, + interrupt_enable: 0, + context_words: 0, + ge_readback: [0; 2], + ge: [GeDiag::default(), GeDiag::default()], + ge_out: std::collections::VecDeque::new(), + cfifo: [Cfifo::default(), Cfifo::default()], + dma_regs: [0; 0x80], + raster_if_regs: [0; 0x10], + formatter: 0, + dma_read_pending: None, + mem: None, + dma_logged: 0, + dcb: dcb::Dcb::new(bdvers, [0xFB, 0xFB]), + raster: raster::Raster::default(), + unhandled: HashSet::new(), + } + } + + fn note(&mut self, what: String) { + if self.unhandled.len() < 256 && self.unhandled.insert(what.clone()) { + eprintln!("mgras: not modelled yet: {what}"); + } + } + + /// Push one 32-bit word into the command FIFO. Returns true when the + /// framebuffer changed. + fn cfifo_word(&mut self, port: usize, w: u32) -> bool { + // A context switch's incoming state follows the save phase as raw + // words; the command processor would load it. Consume it here, and + // report the load done after the last word. + if self.context_words > 0 { + self.context_words -= 1; + if self.context_words == 0 { + self.flags |= host::FLAG_CONTEXT_LOADED; + } + return false; + } + let f = &mut self.cfifo[port]; + if f.need == 0 { + if w & 0x8000_0000 != 0 { + // Pixel data: byte count in bits 19:0, sent as doublewords. + f.pixel = true; + f.cmd = w; + f.need = (((w & 0xF_FFFF) + 7) / 8) * 2; + } else { + f.pixel = false; + f.cmd = (w >> 8) & 0x1FFF; + f.need = ((w & 0xFF) + 3) / 4; + } + f.data.clear(); + if f.need == 0 { + return self.dispatch(port); + } + return false; + } + f.need -= 1; + if f.data.len() < 64 { + f.data.push(w); + } + if f.need == 0 { self.dispatch(port) } else { false } + } + + fn dispatch(&mut self, port: usize) -> bool { + let cmd = self.cfifo[port].cmd; + let data = std::mem::take(&mut self.cfifo[port].data); + if self.cfifo[port].pixel { + self.note(format!("pixel data command {cmd:#010x}")); + return false; + } + let changed = if cmd >= cmd::RASTER_BASE { + let r = cmd & 0x3FF; + let exec = cmd & cmd::RASTER_EXECUTE != 0; + let changed = match data.as_slice() { + [] => self.raster.write(r, 0, exec), + [d] => self.raster.write(r, *d, exec), + [hi, lo, ..] => { + self.raster.write(r, *hi, false); + self.raster.write(r + 1, *lo, exec) + } + }; + if r == raster::reg::XFRCONTROL && data.first() == Some(&9) { + changed | self.start_read_dma() + } else { + changed + } + } else if cmd == cmd::SET_DONE_FLAG { + self.flags |= host::FLAG_DONE; + false + } else if cmd >= cmd::DMA_BASE { + let n = (cmd & 0x1FF) as usize; + let v = data.first().copied().unwrap_or(0); + if cmd >= cmd::FORMATTER { + self.formatter = v; + false + } else if cmd >= cmd::RASTER_IF_BASE { + self.raster_if_regs[n & 0xF] = v; + false + } else { + // Eight-byte registers arrive as a high word, then a low word, + // and fill two register slots. + let n = n & 0x7F; + for (i, d) in data.iter().take(2).enumerate() { + self.dma_regs[(n + i) & 0x7F] = *d; + } + if data.is_empty() { + self.dma_regs[n] = 0; + } + if n == dma::START { self.dma_start(v) } else { false } + } + } else if cmd == cmd::CP_SCHEDULE_SWAP { + // No retrace wait: the swap is reported done at once. + self.flags |= host::FLAG_CP0; + false + } else if cmd < cmd::CP_LIMIT { + self.note(format!("command-processor token {cmd:#x} ({} data words)", data.len())); + false + } else { + self.note(format!("command {cmd:#x}")); + false + }; + self.cfifo[port].data = data; + changed + } + + /// The host DMA engine's start word. Host to board runs now: the raster + /// engine was armed first. Board to host waits for the raster engine. + fn dma_start(&mut self, word: u32) -> bool { + if word & 1 == 0 { + return false; + } + if word & 8 != 0 { + self.dma_read_pending = Some(word); + return false; + } + match self.raster.transfer_armed() { + Some(false) => self.dma(word, false), + _ => { + self.note(format!("host DMA start {word:#x} with no write transfer armed")); + false + } + } + } + + fn start_read_dma(&mut self) -> bool { + let Some(word) = self.dma_read_pending.take() else { + self.note("raster DMA read started with no host DMA pending".into()); + return false; + }; + if self.raster.transfer_armed() != Some(true) { + self.note(format!("host DMA read {word:#x} with no read transfer armed")); + return false; + } + self.dma(word, true) + } + + /// Run a DMA between host memory and the armed raster transfer, a line at + /// a time. Host addresses are logical within the pool and translate + /// through its page table: one 32-bit frame number per 4 KB page. + fn dma(&mut self, word: u32, read: bool) -> bool { + let Some(mem) = self.mem.clone() else { return false }; + let pool = ((word >> 1) & 3) as usize; + let table = self.dma_regs[dma::TABLE_BASE + 2 * pool + 1] & !3; + let base = self.dma_regs[dma::ROW_START].wrapping_add(self.dma_regs[dma::ROW_OFFSET]); + let stride = self.dma_regs[dma::STRIDE]; + let lines = self.dma_regs[dma::LINES]; + let len = self.dma_regs[dma::LINE_BYTES]; + if self.dma_logged < 16 { + self.dma_logged += 1; + eprintln!( + "mgras: DMA {} pool {pool} table {table:#x} base {base:#x} stride {stride} lines {lines} bytes {len} pglist {:#x} shape {:?}", + if read { "read" } else { "write" }, + self.dma_regs[dma::PAGE_LIST], + self.raster.transfer_shape() + ); + } + let frame_of = |page: u32| -> Option { + let r = mem.read32(table.wrapping_add(4 * page)); + r.is_ok().then_some(r.data << 12) + }; + let mut cached: Option<(u32, u32)> = None; + let mut phys = |l: u32| -> Option { + let page = l >> 12; + let f = match cached { + Some((p, f)) if p == page => f, + _ => { + let f = frame_of(page)?; + cached = Some((page, f)); + f + } + }; + Some(f | (l & 0xFFF)) + }; + let mut changed = false; + for i in 0..lines { + let a = base.wrapping_add(i.wrapping_mul(stride)); + if read { + let bytes = self.raster.dma_read_line(i); + for (k, b) in bytes.iter().take(len as usize).enumerate() { + let Some(pa) = phys(a + k as u32) else { return changed }; + mem.write8(pa, *b); + } + } else { + let mut bytes = Vec::with_capacity(len as usize); + for k in 0..len { + let Some(pa) = phys(a + k) else { return changed }; + let r = mem.read8(pa); + bytes.push(if r.is_ok() { r.data } else { 0 }); + } + self.raster.dma_write_line(i, &bytes); + changed = true; + } + } + changed + } + + /// Flags as the host reads them, including those derived from state. + fn all_flags(&self) -> u32 { + self.flags | if self.ge_out.is_empty() { 0 } else { host::FLAG_GE_DIAG } + } + + /// Whether the general interrupt (GIO line 1) is asserted: some enabled + /// flag is set. It is a level, held until the handler clears the flag or + /// its enable. + fn general_irq(&self) -> bool { + self.all_flags() & self.interrupt_enable & host::INTR_CAUSES != 0 + } + + /// Video timing chip display control bit 0: vertical retrace interrupts + /// enabled. + fn retrace_enabled(&self) -> bool { + self.dcb.vc3.regs[0x1E] & 1 != 0 + } + + /// A write to a geometry engine's diagnostic data or address port. + fn ge_diag_write(&mut self, off: u32, val: u32) { + let n = host::GE_DIAG.iter().position(|&(d, a)| off == d || off == a).unwrap(); + let (data_port, _) = host::GE_DIAG[n]; + let ge = &mut self.ge[n]; + if off != data_port { + if val & 0x8000_0000 != 0 { + // Read request for `val & 0x7FFF_FFFF` lines from `addr`; it + // replaces anything a previous request left unread. + let words = ge.readback(val & 0x7FFF_FFFF); + self.ge_out.clear(); + self.ge_out.extend(words); + } else { + ge.addr = val; + ge.word = 0; + } + return; + } + if ge.addr >= GeDiag::UCODE_BASE { + let line = ge.ucode.entry(ge.addr).or_insert([0; 3]); + line[ge.word] = val; + ge.word += 1; + if ge.word == 3 { + ge.word = 0; + ge.addr += 1; + } + } else if ge.addr == GeDiag::EXEC_CONTROL && val & 1 != 0 { + // Started: the version program's answer is waiting (revision 1). + self.ge_readback = [0, 1]; + self.flags |= host::FLAG_GE_DATA; + } + } + + fn status(&self) -> u32 { + // A vertical blank of about 1 ms in every 60 Hz frame. + let us = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_micros() as u64 % 16_667) + .unwrap_or(0); + host::STATUS_IDLE | if us < 1000 { host::STATUS_VBLANK } else { 0 } + } + + fn read(&mut self, off: u32, bits: u32) -> u64 { + match off { + 0 => GIO_ID as u64, + host::UCODE..=0x45FFF => self.ucode[((off - host::UCODE) / 4) as usize] as u64, + 0x60000..=0x67FFF => { + let t = dcb::Txn::decode(off); + let d = self.dcb.read(t); + t.load_value(bits, d) + } + host::STATUS => self.status() as u64, + host::FIFOSTATUS | host::GIOSTATUS | host::DMABUSY => 0, + host::SET_FLAGS | host::CLEAR_FLAGS | host::SET_FLAGS_PRIVILEGED | host::CLEAR_FLAGS_PRIVILEGED => self.all_flags() as u64, + host::FLAG_ENABLE_SET | host::FLAG_ENABLE_CLEAR => self.flag_enable as u64, + host::INTERRUPT_ENABLE_SET | host::INTERRUPT_ENABLE_CLEAR => self.interrupt_enable as u64, + host::GE_DIAG_READ => self.ge_out.pop_front().unwrap_or(0) as u64, + host::RASTER_IF_CONTEXT..=0x50228 => self.raster_if_regs[((off - host::RASTER_IF_CONTEXT) / 4) as usize] as u64, + host::DMA_CONTEXT..=0x504FF => self.dma_regs[((off - host::DMA_CONTEXT) / 4) as usize] as u64, + host::PIO_READ_HI => self.raster.pio_read_hi() as u64, + host::PIO_READ_LO => self.raster.pio_read_lo() as u64, + host::GE_READBACK_HI => self.ge_readback[0] as u64, + host::GE_READBACK_LO => self.ge_readback[1] as u64, + host::RASTER..=0x7FFFF => { + let r = (off & 0xFFC) >> 2; + if bits == 64 { + ((self.raster.read(r) as u64) << 32) | self.raster.read(r + 1) as u64 + } else { + self.raster.read(r) as u64 + } + } + _ => self.regs.get(&off).copied().unwrap_or(0) as u64, + } + } + + /// Returns true when the framebuffer changed. + fn write(&mut self, off: u32, bits: u32, val: u64) -> bool { + match off { + host::UCODE..=0x45FFF => { + self.ucode[((off - host::UCODE) / 4) as usize] = val as u32 & 0xFF_FFFF; + false + } + 0x60000..=0x67FFF => { + let t = dcb::Txn::decode(off); + self.dcb.write(t, t.store_data(bits, val)); + // Display-side state (colormaps, modes, the cursor) changes + // what is shown without touching the framebuffer. Index + // writes (select 1 and below) change nothing by themselves. + t.crs >= 2 || t.dev == dcb::DEV_VC3 + } + off if host::GE_DIAG.iter().any(|&(d, a)| off == d || off == a) => { + self.ge_diag_write(off, val as u32); + false + } + host::FLAG_ENABLE_SET => { self.flag_enable |= val as u32; false } + host::FLAG_ENABLE_CLEAR => { self.flag_enable &= !(val as u32); false } + host::INTERRUPT_ENABLE_SET => { self.interrupt_enable |= val as u32; false } + host::INTERRUPT_ENABLE_CLEAR => { self.interrupt_enable &= !(val as u32); false } + host::CONTEXT_SWITCH => { + // The save phase completes at once; the incoming context + // follows through the FIFO (see `cfifo_word`). + self.flags |= host::FLAG_CONTEXT_SAVED; + self.context_words = host::CONTEXT_SWITCH_WORDS; + false + } + host::SET_FLAGS | host::SET_FLAGS_PRIVILEGED => { self.flags |= val as u32; false } + host::CLEAR_FLAGS | host::CLEAR_FLAGS_PRIVILEGED => { self.flags &= !(val as u32); false } + host::CFIFO | 0x70084 | host::CFIFO_PRIVILEGED | 0x50084 => { + let port = if off >= host::STATUS { 0 } else { 1 }; + if bits == 64 { + let a = self.cfifo_word(port, (val >> 32) as u32); + let b = self.cfifo_word(port, val as u32); + a | b + } else { + self.cfifo_word(port, val as u32) + } + } + host::RASTER..=0x7FFFF => { + let r = (off & 0xFFC) >> 2; + // Registers at 0x7C000 + 4r; the same register at +0x1000 + // also executes the primitive in the IR after the write. + let exec = off & 0x1000 != 0; + if bits == 64 { + self.raster.write(r, (val >> 32) as u32, false); + self.raster.write(r + 1, val as u32, exec) + } else { + self.raster.write(r, val as u32, exec) + } + } + 0x80000..=0xFFFFF => { + self.note(format!("fast-path command window write at {off:#x}")); + false + } + _ => { + self.regs.insert(off, val as u32); + false + } + } + } + + /// The window ID a frame for the window at (`x`, `y`), `w` x `h` (screen + /// coordinates, top-down) should be painted through: the commonest ID + /// over a grid of samples inside it that has an RGB display mode. None + /// if no sampled pixel is in an RGB window. + fn window_did(&self, x: i32, y: i32, w: usize, h: usize) -> Option { + let mut counts = [0u32; 32]; + let mut runs = Vec::new(); + for sy in 0..16 { + let py = y + (h as i32 * (2 * sy + 1)) / 32; + if !(0..raster::HEIGHT as i32).contains(&py) { + continue; + } + self.dcb.vc3.main_did_runs(py as usize, &mut runs); + for sx in 0..16 { + let px = x + (w as i32 * (2 * sx + 1)) / 32; + if !(0..raster::WIDTH as i32).contains(&px) { + continue; + } + let did = runs.iter().rev().find(|r| r.0 as i32 <= px).map_or(0, |r| r.1); + if self.dcb.xmap.main_mode(did as u32) & 0x1F >= 4 { + counts[did as usize & 31] += 1; + } + } + } + let (did, n) = counts.iter().enumerate().max_by_key(|&(_, n)| *n)?; + (*n > 0).then_some(did as u8) + } + + /// Paint a host GL frame (`bgra`: `h` rows, top first, `stride` bytes + /// each) into the framebuffer at screen position (`x`, `y`), wherever the + /// pixel belongs to the frame's window ID, so windows over it stay over it. + /// The pixels become part of the framebuffer, as GL's would on the board, + /// for anything that reads them back. False when no window ID fits. + fn composite(&mut self, x: i32, y: i32, bgra: &[u8], stride: usize, w: usize, h: usize) -> bool { + let Some(target) = self.window_did(x, y, w, h) else { return false }; + let mut runs = Vec::new(); + for row in 0..h { + let sy = y + row as i32; + if !(0..raster::HEIGHT as i32).contains(&sy) { + continue; + } + self.dcb.vc3.main_did_runs(sy as usize, &mut runs); + if runs.is_empty() || runs[0].0 != 0 { + runs.insert(0, (0, 0)); + } + let fb_row = (raster::HEIGHT - 1 - sy as usize) * raster::WIDTH; + for (k, &(x0, did)) in runs.iter().enumerate() { + if did != target { + continue; + } + let x1 = runs.get(k + 1).map_or(raster::WIDTH as i32, |r| r.0 as i32); + let lo = (x0 as i32).max(x).max(0); + let hi = x1.min(x + w as i32).min(raster::WIDTH as i32); + for sx in lo..hi { + let i = row * stride + (sx - x) as usize * 4; + let Some(p) = bgra.get(i..i + 4) else { break }; + self.raster.fb[fb_row + sx as usize] = p[2] as u32 | (p[1] as u32) << 8 | (p[0] as u32) << 16; + } + } + } + true + } + + /// Scan the framebuffer out to `0xFF_BB_GG_RR` (the compositor's order, + /// red in the low byte), stride 2048, top row first. Each pixel's window + /// ID (from the video timing chip's tables) picks its display mode. Modes + /// with a pixel format (bits 4:0) of 4 and up are RGB; the others are + /// colour index, into the colormap block that bits 9:5 choose. Both go + /// through the DAC gamma. + fn scanout(&self, out: &mut [u32]) { + if self.dcb.dac.pixmask() == 0 { + out.iter_mut().for_each(|p| *p = 0xFF00_0000); + return; + } + let pal = &self.dcb.cmap[0].pal; + let gamma = &self.dcb.dac.gamma; + let gamma_rgb = |r: u32, g: u32, b: u32| -> u32 { + let g1 = |v: u32, comp: usize| (gamma[(v & 0xFF) as usize][comp] >> 2) as u32; + 0xFF00_0000 | (g1(b, 2) << 16) | (g1(g, 1) << 8) | g1(r, 0) + }; + // Per window ID: None for an RGB mode, else the colormap block's + // entries through the gamma tables. + let luts: Vec>> = (0..32u32) + .map(|did| { + let mode = self.dcb.xmap.main_mode(did); + if mode & 0x1F >= 4 { + return None; + } + let base = ((mode >> 5) & 0x1F) as usize * 256; + Some( + (0..4096usize) + .map(|i| { + let c = pal.get((base + i) % pal.len().max(1)).copied().unwrap_or(0); + gamma_rgb(c >> 16, c >> 8, c) + }) + .collect(), + ) + }) + .collect(); + let mut runs = Vec::new(); + for row in 0..raster::HEIGHT { + let y = raster::HEIGHT - 1 - row; + let src = &self.raster.fb[y * raster::WIDTH..(y + 1) * raster::WIDTH]; + let dst = &mut out[row * 2048..row * 2048 + raster::WIDTH]; + self.dcb.vc3.main_did_runs(row, &mut runs); + if runs.is_empty() || runs[0].0 != 0 { + runs.insert(0, (0, 0)); + } + for (k, &(x0, did)) in runs.iter().enumerate() { + let x0 = (x0 as usize).min(raster::WIDTH); + let x1 = runs.get(k + 1).map(|r| (r.0 as usize).min(raster::WIDTH)).unwrap_or(raster::WIDTH); + if x1 <= x0 { + continue; + } + match &luts[did as usize & 31] { + Some(lut) => { + for x in x0..x1 { + dst[x] = lut[(src[x] & 0xFFF) as usize]; + } + } + None => { + for x in x0..x1 { + let v = src[x]; + dst[x] = gamma_rgb(v, v >> 8, v >> 16); + } + } + } + } + } + self.draw_overlay(out, &gamma_rgb); + self.draw_cursor(out, &gamma_rgb); + } + + /// Overlay planes over the main scanout: a nonzero pixel is a colour + /// index into the block its overlay mode names (bits 7:3); zero, or a + /// window ID whose overlay is off, shows the main planes. + fn draw_overlay(&self, out: &mut [u32], gamma_rgb: &dyn Fn(u32, u32, u32) -> u32) { + let pal = &self.dcb.cmap[0].pal; + let bases: Vec> = (0..32u32) + .map(|did| { + let mode = self.dcb.xmap.overlay_mode(did); + (mode != 0).then(|| ((mode >> 3) & 0x1F) as usize * 256) + }) + .collect(); + if bases.iter().all(Option::is_none) { + return; + } + let mut runs = Vec::new(); + for row in 0..raster::HEIGHT { + let y = raster::HEIGHT - 1 - row; + let src = &self.raster.overlay[y * raster::WIDTH..(y + 1) * raster::WIDTH]; + let dst = &mut out[row * 2048..row * 2048 + raster::WIDTH]; + self.dcb.vc3.overlay_did_runs(row, &mut runs); + if runs.is_empty() || runs[0].0 != 0 { + runs.insert(0, (0, 0)); + } + for (k, &(x0, did)) in runs.iter().enumerate() { + let Some(base) = bases[did as usize & 31] else { continue }; + let x0 = (x0 as usize).min(raster::WIDTH); + let x1 = runs.get(k + 1).map(|r| (r.0 as usize).min(raster::WIDTH)).unwrap_or(raster::WIDTH); + for x in x0..x1 { + let v = src[x] & 0xFF; + if v != 0 { + let c = pal.get((base + v as usize) % pal.len().max(1)).copied().unwrap_or(0); + dst[x] = gamma_rgb(c >> 16, c >> 8, c); + } + } + } + } + } + + /// Overlay the hardware cursor, its colours from the cursor colormap. + fn draw_cursor(&self, out: &mut [u32], gamma_rgb: &dyn Fn(u32, u32, u32) -> u32) { + let Some((cx, cy, size, glyph)) = self.dcb.vc3.cursor() else { return }; + let sram = &self.dcb.vc3.sram; + let pal = &self.dcb.cmap[0].pal; + let base = self.dcb.xmap.cursor_cmap_base(); + let words_per_row = size / 16; + let plane_words = size * words_per_row; + let bit = |plane: usize, row: usize, col: usize| -> u32 { + let w = sram[(glyph + plane * plane_words + row * words_per_row + col / 16) & 0x7FFF]; + (w >> (15 - col % 16)) as u32 & 1 + }; + for row in 0..size { + let y = cy + row as i32; + if !(0..raster::HEIGHT as i32).contains(&y) { + continue; + } + for col in 0..size { + let x = cx + col as i32; + if !(0..raster::WIDTH as i32).contains(&x) { + continue; + } + let c = bit(0, row, col) | bit(1, row, col) << 1; + if c != 0 { + let rgb = pal.get(base + c as usize).copied().unwrap_or(0xFF_FFFF); + out[y as usize * 2048 + x as usize] = gamma_rgb(rgb >> 16, rgb >> 8, rgb); + } + } + } + } +} + +/// Access trace for bring-up: every access to the board, one line each +/// (`R`/`W`, width in bits, physical address, value). Started from +/// `IRIS_MGRAS_TRACE=` or the monitor (`mgras trace `, `mgras +/// trace off`). Off, it costs one relaxed load per access. +mod trace { + use std::io::Write; + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::OnceLock; + use parking_lot::Mutex; + + static ON: AtomicBool = AtomicBool::new(false); + + fn sink() -> &'static Mutex>> { + static SINK: OnceLock>>> = OnceLock::new(); + SINK.get_or_init(|| { + let w = std::env::var_os("IRIS_MGRAS_TRACE").and_then(|p| open(&p).ok()); + ON.store(w.is_some(), Ordering::Relaxed); + Mutex::new(w) + }) + } + + fn open(path: &std::ffi::OsStr) -> std::io::Result> { + let f = std::fs::OpenOptions::new().create(true).append(true).open(path)?; + Ok(std::io::BufWriter::new(f)) + } + + /// Start tracing to `path`, or stop with `None`. + pub fn set(path: Option<&str>) -> std::io::Result<()> { + let mut s = sink().lock(); + if let Some(mut w) = s.take() { + let _ = w.flush(); + } + if let Some(p) = path { + *s = Some(open(std::ffi::OsStr::new(p))?); + } + ON.store(s.is_some(), Ordering::Relaxed); + Ok(()) + } + + pub fn init() { + let _ = sink(); + } + + pub fn note(dir: char, bits: u32, addr: u32, val: u64) { + if !ON.load(Ordering::Relaxed) { + return; + } + if let Some(w) = sink().lock().as_mut() { + let _ = writeln!(w, "{dir}{bits} {addr:08x} {val:x}"); + } + } +} + +/// The board's three interrupt lines to the GIO slot. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Line { + /// GIO interrupt 0: command FIFO high/low water. + Fifo, + /// GIO interrupt 1: the general graphics interrupt. + General, + /// GIO interrupt 2: vertical retrace. + Retrace, +} + +pub struct Mgras { + kind: ImpactSlot, + ioc: crate::ioc::Ioc, + /// System memory, for the board's DMA (it is a GIO bus master). + sys_mem: Mutex>>, + /// Current level of the general interrupt line. + general: AtomicBool, + board: Mutex, + dirty: AtomicBool, + running: AtomicBool, + refresh: Mutex>>, + renderer: Mutex>>, + /// The finished frame, handed to the renderer as a prebuilt picture + /// (`Rex3Screen::prebuilt`), as GR2 does. + screen: Mutex, + screenshot_pending: AtomicBool, + heartbeat: Arc, + fasttick: Arc, + cycles: Mutex, +} + +impl Mgras { + pub fn new(cfg: &ImpactSection, ioc: crate::ioc::Ioc, heartbeat: Arc, fasttick: Arc) -> Self { + trace::init(); + Mgras { + kind: cfg.gfx, + ioc, + sys_mem: Mutex::new(None), + general: AtomicBool::new(false), + board: Mutex::new(Board::new(cfg.gfx)), + dirty: AtomicBool::new(true), + running: AtomicBool::new(false), + refresh: Mutex::new(None), + renderer: Mutex::new(None), + screen: Mutex::new(crate::disp::Rex3Screen::new()), + screenshot_pending: AtomicBool::new(false), + heartbeat, + fasttick, + cycles: Mutex::new(crate::mips_core::CyclesPtr::dangling()), + } + } + + /// Give the board its path to system memory, for DMA. + pub fn set_phys(&self, mem: Arc) { + self.board.lock().mem = Some(mem.clone()); + *self.sys_mem.lock() = Some(mem); + } + + /// Drive one of the board's interrupt lines (graphics slot wiring). + fn set_line(&self, line: Line, active: bool) { + use crate::ioc::IocInterrupt; + let src = match line { + Line::Fifo => IocInterrupt::GioSgFifo, + Line::General => IocInterrupt::GioSgGraphics, + Line::Retrace => IocInterrupt::GioSgRetrace, + }; + self.ioc.set_interrupt(src, active); + } + + /// Wire up the CPU cycle counter for the status bar's MIPS figure. + pub fn set_cpu_cycles(&self, ptr: crate::mips_core::CyclesPtr) { + *self.cycles.lock() = ptr; + } + + fn offset(&self, addr: u32) -> Option { + let off = addr.wrapping_sub(MGRAS_SLOT_GFX_BASE); + (off < MAP_SIZE).then_some(off) + } + + /// Bring the general interrupt line in line with the board's flags. + fn update_general(&self, level: bool) { + if self.general.swap(level, Ordering::AcqRel) != level { + self.set_line(Line::General, level); + } + } + + fn do_read(&self, addr: u32, bits: u32) -> u64 { + let v = match self.offset(addr) { + Some(off) => { + let mut b = self.board.lock(); + let v = b.read(off, bits); + let irq = b.general_irq(); + drop(b); + self.update_general(irq); + v + } + None => 0, + }; + trace::note('R', bits, addr, v); + v + } + + fn do_write(&self, addr: u32, bits: u32, val: u64) { + trace::note('W', bits, addr, val); + if let Some(off) = self.offset(addr) { + let mut b = self.board.lock(); + let changed = b.write(off, bits, val); + let irq = b.general_irq(); + drop(b); + if changed { + self.dirty.store(true, Ordering::Release); + } + self.update_general(irq); + } + } + + fn refresh_loop(self: &Arc) { + let frame = std::time::Duration::from_micros(16_667); + { + let mut screen = self.screen.lock(); + screen.width = raster::WIDTH; + screen.height = raster::HEIGHT; + screen.fb_rgb.fill(0xFF00_0000); + screen.prebuilt = true; + } + let mut overlay = crate::debug_overlay::DebugOverlay::new(); + let mut status_bar = crate::disp::StatusBar::new(); + let mut sbtex = crate::disp::StatusBarTexture::new(); + let mut sized = false; + let mut last_pending = 0u64; + let mut idle_frames = 0u32; + const PERSISTENT: u64 = crate::rex3::Rex3::HB_LED_RED | crate::rex3::Rex3::HB_LED_GREEN; + + while self.running.load(Ordering::Relaxed) { + let start = std::time::Instant::now(); + let stats = crate::disp::BarStats { + now: start, + hb: self.heartbeat.fetch_and(PERSISTENT, Ordering::Relaxed), + cycles: self.cycles.lock().get(), + fasttick: self.fasttick.load(Ordering::Relaxed), + decoded_delta: 0, + l1i_hits: 0, + l1i_fetches: 0, + uncached: 0, + count_hz: 0, + gfifo_pending: 0, + }; + let retrace = { + let mut b = self.board.lock(); + if b.raster.flush_if_stale(&mut last_pending) { + self.dirty.store(true, Ordering::Release); + } + b.retrace_enabled() + }; + // Vertical retrace: a pulse per frame. The handler acknowledges + // nothing on the board, so the line must drop again by itself. + if retrace { + self.set_line(Line::Retrace, true); + std::thread::sleep(std::time::Duration::from_micros(500)); + self.set_line(Line::Retrace, false); + } + let dirty = self.dirty.swap(false, Ordering::AcqRel); + let shot = self.screenshot_pending.swap(false, Ordering::Relaxed); + idle_frames += 1; + if dirty || shot || idle_frames >= 6 { + idle_frames = 0; + let mut screen = self.screen.lock(); + if dirty || shot { + let screen = &mut *screen; + self.board.lock().scanout(&mut screen.fb_rgb); + // The frame is already final RGB, so keep `rgba` (what CI + // screenshots read) current without a renderer readback, + // as GR2 does. This also makes screenshots work headless. + for y in 0..raster::HEIGHT { + let row = y * 2048; + screen.rgba[row..row + raster::WIDTH].copy_from_slice(&screen.fb_rgb[row..row + raster::WIDTH]); + } + screen.status_bar_only = false; + } else { + screen.status_bar_only = true; + } + if let Some(r) = self.renderer.lock().as_mut() { + if !sized { + r.resize(raster::WIDTH, raster::HEIGHT); + sized = true; + } + r.present(&mut screen, &mut overlay, &mut status_bar, &mut sbtex, &stats, shot, None, None); + } + } + if let Some(rest) = frame.checked_sub(start.elapsed()) { + std::thread::sleep(rest); + } + } + // The renderer's GL state belongs to this thread (its context is + // current here), so it is torn down here and nowhere else. + if let Some(r) = self.renderer.lock().as_mut() { + r.stop(); + } + } + + /// Start the display refresh thread. + pub fn start_display(self: &Arc) { + if self.running.swap(true, Ordering::AcqRel) { + return; + } + let me = Arc::clone(self); + *self.refresh.lock() = Some( + std::thread::Builder::new() + .name("MGRAS-Refresh".into()) + .spawn(move || me.refresh_loop()) + .expect("spawn MGRAS refresh thread"), + ); + } + + pub fn stop_display(&self) { + self.running.store(false, Ordering::Release); + if let Some(h) = self.refresh.lock().take() { + let _ = h.join(); + } + } +} + +impl crate::gfx_display::GfxDisplay for Mgras { + fn renderer_slot(&self) -> &Mutex>> { &self.renderer } + fn screen(&self) -> &Mutex { &self.screen } + fn request_screenshot(&self) { self.screenshot_pending.store(true, Ordering::Relaxed); } + fn cycles(&self) -> crate::mips_core::CyclesPtr { *self.cycles.lock() } +} + +impl Device for Mgras { + fn step(&self, _cycles: u64) {} + fn stop(&self) {} + fn start(&self) {} + fn is_running(&self) -> bool { self.running.load(Ordering::Relaxed) } + fn get_clock(&self) -> u64 { 0 } + + fn register_commands(&self) -> Vec<(String, String)> { + vec![("mgras".into(), "IMPACT graphics: mgras (board state) | mgras shot (save the displayed frame) | mgras dump (raw display state) | mgras trace |off".into())] + } + + fn execute_command(&self, cmd: &str, args: &[&str], mut w: Box) -> Result<(), String> { + if cmd != "mgras" { + return Err(format!("unknown command: {cmd}")); + } + if let ["trace", what] = args { + let r = if *what == "off" { trace::set(None) } else { trace::set(Some(what)) }; + r.map_err(|e| format!("mgras trace: {e}"))?; + return writeln!(w, "trace {what}").map_err(|e| e.to_string()); + } + if let ["dump", path] = args { + let b = self.board.lock(); + dump_state(path, &b).map_err(|e| format!("mgras dump: {e}"))?; + return writeln!(w, "dumped {path}").map_err(|e| e.to_string()); + } + if let ["shot", path] = args { + let mut frame = vec![0u32; 2048 * raster::HEIGHT]; + self.board.lock().scanout(&mut frame); + save_png(path, &frame).map_err(|e| format!("mgras shot: {e}"))?; + return writeln!(w, "saved {path}").map_err(|e| e.to_string()); + } + let b = self.board.lock(); + let e = |r: std::io::Result<()>| r.map_err(|e| e.to_string()); + e(writeln!(w, "IMPACT {:?} at {:#010x}", self.kind, MGRAS_SLOT_GFX_BASE))?; + e(writeln!(w, " flags {:#010x} DAC pixmask {:#04x} XMAP DID0 mode {:#x}", + b.flags, b.dcb.dac.pixmask(), b.dcb.xmap.main_mode(0)))?; + e(writeln!(w, " fill modes seen: {:x?}", b.raster.fillmodes_seen))?; + let mut hist: HashMap = HashMap::new(); + for v in &b.raster.fb { + *hist.entry(*v).or_default() += 1; + } + let mut top: Vec<_> = hist.into_iter().collect(); + top.sort_by(|a, b| b.1.cmp(&a.1)); + e(writeln!(w, " framebuffer values (value, pixels): {:x?}", &top[..top.len().min(12)]))?; + let pal = &b.dcb.cmap[0].pal; + let blocks: Vec = (0..pal.len() / 256) + .filter(|k| pal[k * 256..(k + 1) * 256].iter().any(|c| *c != 0)) + .collect(); + e(writeln!(w, " colormap blocks in use (of {}): {:?}", pal.len() / 256, blocks))?; + for k in blocks.iter().take(4) { + e(writeln!(w, " block {k}: {:06x?}", &pal[k * 256..k * 256 + 8]))?; + } + let modes: Vec<(u32, u32)> = (0..32).map(|d| (d, b.dcb.xmap.main_mode(d))).filter(|m| m.1 != 0).collect(); + e(writeln!(w, " XMAP main modes (did, mode): {:x?}", modes))?; + let mut runs = Vec::new(); + for y in [0usize, 100, 400, 700, 1023] { + b.dcb.vc3.main_did_runs(y, &mut runs); + e(writeln!(w, " scanline {y} DID runs: {:?}", runs))?; + } + for u in &b.unhandled { + e(writeln!(w, " not modelled: {u}"))?; + } + Ok(()) + } +} + +/// Save the display state for offline study: little-endian u32 sections, in +/// order: framebuffer (`WIDTH * HEIGHT`, row 0 at the bottom), overlay (same +/// size), colormap 0, the 32 main XMAP modes, then the video timing chip's +/// registers and SRAM as u32s. +fn dump_state(path: &str, b: &Board) -> std::io::Result<()> { + let mut f = std::io::BufWriter::new(std::fs::File::create(path)?); + let mut put = |v: u32| f.write_all(&v.to_le_bytes()); + for v in b.raster.fb.iter().chain(b.raster.overlay.iter()).chain(b.dcb.cmap[0].pal.iter()) { + put(*v)?; + } + for d in 0..32 { + put(b.dcb.xmap.main_mode(d))?; + } + for v in b.dcb.vc3.regs.iter().chain(b.dcb.vc3.sram.iter()) { + put(*v as u32)?; + } + Ok(()) +} + +/// Write a scanned-out frame (`0xFF_BB_GG_RR`, stride 2048) as a PNG. +fn save_png(path: &str, frame: &[u32]) -> Result<(), String> { + let file = std::fs::File::create(path).map_err(|e| e.to_string())?; + let mut enc = png::Encoder::new(std::io::BufWriter::new(file), raster::WIDTH as u32, raster::HEIGHT as u32); + enc.set_color(png::ColorType::Rgb); + enc.set_depth(png::BitDepth::Eight); + let mut out = enc.write_header().map_err(|e| e.to_string())?; + let mut rows = Vec::with_capacity(raster::WIDTH * raster::HEIGHT * 3); + for y in 0..raster::HEIGHT { + for px in &frame[y * 2048..y * 2048 + raster::WIDTH] { + rows.extend_from_slice(&[*px as u8, (px >> 8) as u8, (px >> 16) as u8]); + } + } + out.write_image_data(&rows).map_err(|e| e.to_string()) +} + +impl Saveable for Mgras { + // Bring-up: board state is not snapshotted yet. + fn save_state(&self) -> toml::Value { + toml::Value::Table(toml::map::Map::new()) + } + + fn load_state(&self, _v: &toml::Value) -> Result<(), String> { + Ok(()) + } +} + +impl BusDevice for Mgras { + fn read32(&self, addr: u32) -> BusRead32 { BusRead32::ok(self.do_read(addr, 32) as u32) } + fn write32(&self, addr: u32, val: u32) -> u32 { self.do_write(addr, 32, val as u64); BUS_OK } + fn read8(&self, addr: u32) -> BusRead8 { BusRead8::ok(self.do_read(addr, 8) as u8) } + fn write8(&self, addr: u32, val: u8) -> u32 { self.do_write(addr, 8, val as u64); BUS_OK } + fn read16(&self, addr: u32) -> BusRead16 { BusRead16::ok(self.do_read(addr, 16) as u16) } + fn write16(&self, addr: u32, val: u16) -> u32 { self.do_write(addr, 16, val as u64); BUS_OK } + fn read64(&self, addr: u32) -> BusRead64 { BusRead64::ok(self.do_read(addr, 64)) } + fn write64(&self, addr: u32, val: u64) -> u32 { self.do_write(addr, 64, val); BUS_OK } +} diff --git a/src/mgras/raster.rs b/src/mgras/raster.rs new file mode 100644 index 00000000..0d297160 --- /dev/null +++ b/src/mgras/raster.rs @@ -0,0 +1,823 @@ +//! The raster subsystem: the raster engine's register file, its indirect +//! device space, pixel transfers, and the framebuffer it draws into. +//! +//! Registers are numbered 0..0x3FF. A register write may carry an "execute" +//! flag, which runs the primitive held in the instruction register (IR) once +//! the write has landed. +//! +//! Coordinates: primitives give block corners in window coordinates. The +//! window origin (`xywin`: y in the high half, x in the low) is added, and +//! with Y-flip set in `config` y runs downward from it. The PROM draws with no +//! origin and no flip; the X server sets the origin to the top row and flips, +//! so it draws top-down. The framebuffer has row 0 at the bottom. +//! +//! A block runs one of several ways, chosen by the fill mode's block type: +//! fill it (fast fill uses the fill colour registers, others the red +//! iterator), stipple it with character data, or move pixels in or out of it +//! (by PIO through the character registers, or by DMA). + +use std::collections::{HashMap, VecDeque}; + +pub const WIDTH: usize = 1280; +pub const HEIGHT: usize = 1024; + +/// Raster registers. Those OpenBSD's impact(4) driver also uses carry its +/// names; the rest are named for what they do here. +pub mod reg { + /// The instruction register: the primitive a write with "execute" runs. + pub const IR: u32 = 0x013; + pub const LINE_START: u32 = 0x040; + pub const LINE_END: u32 = 0x041; + pub const IR_ALIAS: u32 = 0x045; + pub const BLOCKXYSTARTI: u32 = 0x046; + pub const BLOCKXYENDI: u32 = 0x047; + /// Packed RGB colour for character and line drawing in RGB modes. + pub const PACKEDCOLOR: u32 = 0x05B; + pub const RED: u32 = 0x05C; + pub const CHAR_H: u32 = 0x070; + pub const CHAR_L: u32 = 0x071; + pub const XFRCONTROL: u32 = 0x102; + pub const FILLMODE: u32 = 0x110; + pub const CONFIG: u32 = 0x112; + pub const XYWIN: u32 = 0x115; + /// The clip rectangle: x and y ranges, each `min << 16 | max`, and its + /// control (bit 0 enable, bit 4 keep the inside rather than the outside). + pub const CLIP_X: u32 = 0x147; + pub const CLIP_Y: u32 = 0x148; + pub const CLIP_MODE: u32 = 0x14F; + pub const XFRSIZE: u32 = 0x153; + pub const XFRMODE: u32 = 0x159; + pub const LINE_STIPPLE: u32 = 0x15A; + /// The indirect device space: an address, then its data. + pub const INDIRECT_ADDR: u32 = 0x15C; + pub const INDIRECT_DATA: u32 = 0x15D; + pub const STATUS: u32 = 0x15E; + pub const PP1FILLMODE: u32 = 0x161; + /// Plane write mask (low planes, buffer A). + pub const COLORMASKLSBSA: u32 = 0x163; + pub const DRBPOINTERS: u32 = 0x16D; + /// Fast-fill colour: one 12-bit component each, or the index in R. + pub const FILL_COLOR_R: u32 = 0x176; + pub const FILL_COLOR_G: u32 = 0x177; + pub const FILL_COLOR_B: u32 = 0x178; +} + +/// IR opcodes: a line between two points, and a block (rectangle). +const OP_LINE: u32 = 0x5; +const OP_BLOCK: u32 = 0x8; +/// Fill mode: lines follow the 32-bit line stipple pattern. +const FILL_LINE_STIPPLE: u32 = 1 << 5; +/// Status: command FIFO empty, engine and pixel processors idle, revision 1. +const STATUS_IDLE: u32 = 0x100 | (1 << 4); +/// Config: Y-flip. +const CONFIG_YFLIP: u32 = 1 << 3; +/// Fill mode: fast fill (solid, from the fill colour registers). +const FILL_FAST: u32 = 1 << 20; +/// Pixel processor fill mode used when drawing window IDs, which live in +/// their own planes, not the colour planes. +const PP1_DRAW_CID: u32 = 0x14_2600; +/// Scanout pointers: the low nine bits say which planes are drawn, and this +/// value means the overlay planes (the main planes read 0x240). +const DRB_PLANES: u32 = 0x1FF; +const DRB_OVERLAY: u32 = 0x1C0; + +/// Block types (fill mode bits 24:22). +mod block { + pub const NORMAL: u32 = 1; + pub const PIO_READ: u32 = 2; + pub const PIO_WRITE: u32 = 3; + pub const DMA_READ: u32 = 4; + pub const DMA_WRITE: u32 = 5; +} + +/// Whether the pixel processors' pixel type (fill mode bits 10:8) is an RGB +/// one; the others are colour index. +fn rgb_pixtype(pp1fillmode: u32) -> bool { + matches!((pp1fillmode >> 8) & 7, 0 | 1 | 4) +} + +/// Pixel processor logic op (fill mode bit 2 enables it; bits 29:26 hold +/// the X11 function number) applied to source `s` and destination `d`. +fn logic_op(op: u32, s: u32, d: u32) -> u32 { + match op & 0xF { + 0x0 => 0, + 0x1 => s & d, + 0x2 => s & !d, + 0x3 => s, + 0x4 => !s & d, + 0x5 => d, + 0x6 => s ^ d, + 0x7 => s | d, + 0x8 => !(s | d), + 0x9 => !(s ^ d), + 0xA => !d, + 0xB => s | !d, + 0xC => !s, + 0xD => !s | d, + 0xE => !(s & d), + _ => !0, + } +} +const PP1_LOGIC_OP_ENABLE: u32 = 1 << 2; + +/// Framebuffer pixels: colour indices as they are, RGB as `0x00BBGGRR` with +/// eight bits per component. +fn pack_rgb(r: u32, g: u32, b: u32) -> u32 { + (r & 0xFF) | (g & 0xFF) << 8 | (b & 0xFF) << 16 +} + +/// A host pixel of transfer format (PixelFormat, CompType) to a framebuffer +/// pixel, and back. Only the RGB formats convert. +fn from_host(format: (u32, u32), v: u32) -> u32 { + let c4 = |s: u32| ((v >> s) & 0xF) * 0x11; + let c5 = |s: u32| ((v >> s) & 0x1F) << 3 | ((v >> s) & 0x1F) >> 2; + match format { + (8, 8) => pack_rgb(c4(0), c4(4), c4(8)), + (8, 10) => pack_rgb(c5(0), c5(5), c5(10)), + (8, 0) => v & 0xFF_FFFF, + (0, 1) => v & 0xFFF, + _ => v, + } +} + +fn to_host(format: (u32, u32), v: u32) -> u32 { + let c = |s: u32| (v >> s) & 0xFF; + match format { + (8, 8) => c(0) / 0x11 | (c(8) / 0x11) << 4 | (c(16) / 0x11) << 8, + (8, 10) => c(0) >> 3 | (c(8) >> 3) << 5 | (c(16) >> 3) << 10, + _ => v, + } +} + +fn signed16(v: u32) -> i32 { + v as u16 as i16 as i32 +} + +/// A block, in window coordinates, with its colour. +#[derive(Clone, Copy)] +struct Block { + xs: i32, + ys: i32, + xe: i32, + ye: i32, + color: u32, +} + +impl Block { + fn dx(&self) -> i32 { + if self.xe < self.xs { -1 } else { 1 } + } + fn dy(&self) -> i32 { + if self.ye < self.ys { -1 } else { 1 } + } + fn rows(&self) -> i32 { + (self.ye - self.ys).abs() + 1 + } + fn cols(&self) -> i32 { + (self.xe - self.xs).abs() + 1 + } +} + +/// A character block being filled with stipple data, one row at a time. +#[derive(Clone, Copy)] +struct Stipple { + block: Block, + col: i32, + row: i32, +} + +/// An armed pixel transfer. +struct Xfer { + block: Block, + read: bool, + /// Pixels per line, bytes per pixel, and (PixelFormat, CompType). + width: u32, + bpp: u32, + format: (u32, u32), + begin_skip: u32, + stride_skip: u32, + /// PIO write stream: the line being assembled, its byte offset within + /// its first doubleword, the bytes collected, and filler still to skip. + line: u32, + line_begin: u32, + pending: Vec, + skip: u32, + /// PIO read: doublewords waiting to be read, and the low half of the one + /// the last high read took. + out: VecDeque, + out_lo: u32, +} + +impl Xfer { + fn line_bytes(&self) -> u32 { + self.width * self.bpp + } + + /// Where the line after one starting at `b` starts within its first + /// doubleword. + fn next_begin(&self, b: u32) -> u32 { + (b + self.line_bytes() + self.stride_skip) & 7 + } +} + +/// Bytes per pixel for an xfrmode (PixelFormat, PixelCompType) pair. +fn bytes_per_pixel(xfrmode: u32) -> u32 { + match ((xfrmode >> 4) & 0xF, xfrmode & 0xF) { + (0, 0) => 1, + (0, 1) | (8, 8) | (8, 10) => 2, + (8, 0) => 4, + (7, 1) => 6, + _ => 1, + } +} + +pub struct Raster { + regs: Vec, + device: HashMap, + /// Colour planes, one value per pixel, row 0 at the bottom. In + /// colour-index modes the low byte is the index. + pub fb: Vec, + /// Overlay planes (kept, not yet displayed). + pub overlay: Vec, + pending: Option, + /// Bumped for every new pending block, so the refresh thread can tell a + /// block that has waited a whole frame (see `flush_if_stale`). + pending_id: u64, + stipple: Option, + xfer: Option, + /// Fill modes seen with a block, for bring-up logging. + pub fillmodes_seen: Vec, +} + +impl Default for Raster { + fn default() -> Self { + Raster { + regs: vec![0; 0x400], + device: HashMap::new(), + fb: vec![0; WIDTH * HEIGHT], + overlay: vec![0; WIDTH * HEIGHT], + pending: None, + pending_id: 0, + stipple: None, + xfer: None, + fillmodes_seen: Vec::new(), + } + } +} + +impl Raster { + pub fn read(&self, r: u32) -> u32 { + match r & 0x3FF { + reg::STATUS => STATUS_IDLE, + reg::INDIRECT_DATA => { + self.device.get(&self.regs[reg::INDIRECT_ADDR as usize]).copied().unwrap_or(0) + } + r => self.regs[r as usize], + } + } + + fn reg(&self, r: u32) -> u32 { + self.regs[r as usize] + } + + /// Write register `r`; `exec` runs the primitive afterwards. Returns true + /// when the framebuffer changed. + pub fn write(&mut self, r: u32, val: u32, exec: bool) -> bool { + let r = r & 0x3FF; + self.regs[r as usize] = val; + match r { + reg::INDIRECT_DATA => { + self.device.insert(self.regs[reg::INDIRECT_ADDR as usize], val); + } + reg::IR_ALIAS => self.regs[reg::IR as usize] = val, + reg::XFRCONTROL if val == 0 => self.xfer = None, + _ => {} + } + if !exec { + return false; + } + let pio_write = self.xfer.as_ref().is_some_and(|x| !x.read); + match r { + reg::CHAR_L if pio_write => { + let dw = ((self.reg(reg::CHAR_H) as u64) << 32) | val as u64; + self.pio_write(dw) + } + reg::CHAR_H if pio_write => self.pio_write((val as u64) << 32), + reg::CHAR_H => self.stipple_bits((val as u64) << 32, 32), + reg::CHAR_L => { + let bits = ((self.reg(reg::CHAR_H) as u64) << 32) | val as u64; + self.stipple_bits(bits, 64) + } + _ => self.execute(), + } + } + + /// Whether pixels are RGB rather than colour indices: an RGB pixel type, + /// or a write mask of exactly the 24 RGB planes (24-bit windows draw + /// with other pixel types; colour-index drawing masks 8 or 12 planes, or + /// all 32 when the PROM and kernel draw). + fn rgb_mode(&self) -> bool { + rgb_pixtype(self.reg(reg::PP1FILLMODE)) || self.reg(reg::COLORMASKLSBSA) == 0xFF_FFFF + } + + /// Window coordinates to framebuffer coordinates. + fn to_fb(&self, x: i32, y: i32) -> (i32, i32) { + let win = self.reg(reg::XYWIN); + let (ox, oy) = (signed16(win), signed16(win >> 16)); + if self.reg(reg::CONFIG) & CONFIG_YFLIP != 0 { + (ox + x, oy - y) + } else { + (ox + x, oy + y) + } + } + + /// Whether a framebuffer pixel may be written: on screen, and inside + /// the clip rectangle when it is enabled (bit 4 chooses inside or outside). + fn visible(&self, x: i32, y: i32) -> bool { + if !(0..WIDTH as i32).contains(&x) || !(0..HEIGHT as i32).contains(&y) { + return false; + } + let clip = self.reg(reg::CLIP_MODE); + if clip & 1 != 0 { + let (mx, my) = (self.reg(reg::CLIP_X), self.reg(reg::CLIP_Y)); + let inside = (signed16(mx >> 16)..=signed16(mx)).contains(&x) + && (signed16(my >> 16)..=signed16(my)).contains(&y); + if inside != (clip & 0x10 != 0) { + return false; + } + } + true + } + + /// Store `v` at framebuffer `(x, y)` in the planes the pixel processors + /// are drawing to. Window-ID drawing is dropped. + fn put(&mut self, x: i32, y: i32, v: u32) { + if self.reg(reg::PP1FILLMODE) == PP1_DRAW_CID || !self.visible(x, y) { + return; + } + let i = y as usize * WIDTH + x as usize; + let pp1 = self.reg(reg::PP1FILLMODE); + // A write through all planes (window moves copy the screen that way, + // 24 bits a pixel) keeps the whole value; so does RGB. + let wide = self.rgb_mode() || self.reg(reg::COLORMASKLSBSA) == 0xFFFF_FFFF; + let plane = if self.reg(reg::DRBPOINTERS) & DRB_PLANES == DRB_OVERLAY { + &mut self.overlay + } else { + &mut self.fb + }; + plane[i] = if pp1 & PP1_LOGIC_OP_ENABLE != 0 { + let width = if wide { 0xFF_FFFF } else { 0xFFF }; + logic_op(pp1 >> 26, v, plane[i]) & width + } else { + v + }; + } + + fn get(&self, x: i32, y: i32) -> u32 { + if !(0..WIDTH as i32).contains(&x) || !(0..HEIGHT as i32).contains(&y) { + return 0; + } + let i = y as usize * WIDTH + x as usize; + if self.reg(reg::DRBPOINTERS) & DRB_PLANES == DRB_OVERLAY { self.overlay[i] } else { self.fb[i] } + } + + /// Block pixel `(col, row)` in framebuffer coordinates. + fn block_px(&self, b: &Block, col: i32, row: i32) -> (i32, i32) { + self.to_fb(b.xs + col * b.dx(), b.ys + row * b.dy()) + } + + fn current_block(&self) -> Block { + let s = self.reg(reg::BLOCKXYSTARTI); + let e = self.reg(reg::BLOCKXYENDI); + // Colour index modes: the index, from the fill colour register for + // fast fills and from the red iterator (12 fraction bits) otherwise. + // RGB modes: three 12-bit components for fast fills, a packed + // 8-8-8 colour otherwise. + let rgb = self.rgb_mode(); + let fast = self.reg(reg::FILLMODE) & FILL_FAST != 0; + let color = match (rgb, fast) { + (false, true) => self.reg(reg::FILL_COLOR_R), + (false, false) => (self.reg(reg::RED) >> 12) & 0xFFF, + (true, true) => pack_rgb( + self.reg(reg::FILL_COLOR_R) >> 4, + self.reg(reg::FILL_COLOR_G) >> 4, + self.reg(reg::FILL_COLOR_B) >> 4, + ), + (true, false) => self.reg(reg::PACKEDCOLOR) & 0xFF_FFFF, + }; + Block { xs: signed16(s >> 16), ys: signed16(s), xe: signed16(e >> 16), ye: signed16(e), color } + } + + /// Run the primitive in the IR. + fn execute(&mut self) -> bool { + let changed = self.flush_pending(); + match self.reg(reg::IR) & 0xF { + OP_BLOCK => {} + OP_LINE => return self.line() | changed, + _ => return changed, + } + let fm = self.reg(reg::FILLMODE); + if !self.fillmodes_seen.contains(&fm) && self.fillmodes_seen.len() < 32 { + self.fillmodes_seen.push(fm); + } + let b = self.current_block(); + // A new block ends whatever the last one was doing; a write transfer + // in particular is never disarmed explicitly. + self.stipple = None; + self.xfer = None; + let kind = (fm >> 22) & 7; + if fm & FILL_FAST != 0 { + self.fill(&b); + return true; + } + match kind { + block::NORMAL => { + // Character block: filled by the stipple data that follows. + self.stipple = Some(Stipple { block: b, col: 0, row: 0 }); + changed + } + block::PIO_READ | block::PIO_WRITE | block::DMA_READ | block::DMA_WRITE => { + let read = kind == block::PIO_READ || kind == block::DMA_READ; + self.arm_transfer(b, read, kind == block::PIO_READ); + changed + } + _ => { + // A block followed by character data is a stipple; anything + // else, a fill. + self.pending_id += 1; + self.pending = Some(b); + changed + } + } + } + + /// A line from the start to the end point, both included, in the current + /// colour; with line stipple on, pixel `k` is drawn only where bit + /// `31 - k % 32` of the pattern is set. + fn line(&mut self) -> bool { + let s = self.reg(reg::LINE_START); + let e = self.reg(reg::LINE_END); + let (mut x, mut y) = (signed16(s >> 16), signed16(s)); + let (x1, y1) = (signed16(e >> 16), signed16(e)); + let color = self.current_block().color; + let stipple = (self.reg(reg::FILLMODE) & FILL_LINE_STIPPLE != 0).then(|| self.reg(reg::LINE_STIPPLE)); + let (dx, dy) = ((x1 - x).abs(), -(y1 - y).abs()); + let (sx, sy) = (if x < x1 { 1 } else { -1 }, if y < y1 { 1 } else { -1 }); + let mut err = dx + dy; + let mut k = 0u32; + loop { + if stipple.map_or(true, |p| p & (1 << (31 - k % 32)) != 0) { + let (fx, fy) = self.to_fb(x, y); + self.put(fx, fy, color); + } + if x == x1 && y == y1 { + break; + } + let e2 = 2 * err; + if e2 >= dy { + err += dy; + x += sx; + } + if e2 <= dx { + err += dx; + y += sy; + } + k += 1; + } + true + } + + fn fill(&mut self, b: &Block) { + for row in 0..b.rows() { + for col in 0..b.cols() { + let (x, y) = self.block_px(b, col, row); + self.put(x, y, b.color); + } + } + } + + /// Called once per displayed frame: a block that was already pending at + /// the previous frame never got character data, so draw it as a fill. + pub fn flush_if_stale(&mut self, last_seen: &mut u64) -> bool { + if self.pending.is_none() { + return false; + } + if *last_seen == self.pending_id { + return self.flush_pending(); + } + *last_seen = self.pending_id; + false + } + + /// A pending block that got no character data is a solid fill. + fn flush_pending(&mut self) -> bool { + let Some(b) = self.pending.take() else { return false }; + self.fill(&b); + true + } + + /// Consume `n` stipple bits (most significant first) into the current + /// character block: 1 bits take the colour, 0 bits leave the pixel. A row + /// ends at the block's edge, discarding the rest of the chunk. + fn stipple_bits(&mut self, bits: u64, n: u32) -> bool { + if let Some(b) = self.pending.take() { + self.stipple = Some(Stipple { block: b, col: 0, row: 0 }); + } + let Some(mut s) = self.stipple else { return false }; + if s.row >= s.block.rows() { + return false; + } + for i in 0..n { + if bits & (1u64 << (63 - i)) != 0 { + let (x, y) = self.block_px(&s.block, s.col, s.row); + self.put(x, y, s.block.color); + } + s.col += 1; + if s.col >= s.block.cols() { + s.col = 0; + s.row += 1; + break; + } + } + self.stipple = Some(s); + true + } + + // ---- pixel transfers ---- + + fn arm_transfer(&mut self, block: Block, read: bool, pio_read: bool) { + let mode = self.reg(reg::XFRMODE); + let begin = (mode >> 8) & 7; + let mut x = Xfer { + block, + read, + width: self.reg(reg::XFRSIZE) & 0xFFFF, + bpp: bytes_per_pixel(mode), + format: ((mode >> 4) & 0xF, mode & 0xF), + begin_skip: begin, + stride_skip: (mode >> 14) & 0x1FF, + line: 0, + line_begin: begin, + pending: Vec::new(), + skip: begin, + out: VecDeque::new(), + out_lo: 0, + }; + if pio_read { + x.out = self.pio_read_stream(&x); + } + self.xfer = Some(x); + } + + /// `Some(read)` while a transfer is armed. + pub fn transfer_armed(&self) -> Option { + self.xfer.as_ref().map(|x| x.read) + } + + /// Lines and bytes per line of the armed transfer. + pub fn transfer_shape(&self) -> Option<(u32, u32)> { + self.xfer.as_ref().map(|x| (x.block.rows() as u32, x.line_bytes())) + } + + fn put_line(&mut self, x: &Xfer, line: u32, bytes: &[u8]) { + let bpp = x.bpp as usize; + for (k, px) in bytes.chunks(bpp).enumerate().take(x.width as usize) { + // Big-endian bytes; pixels wider than 32 bits keep their low word. + let v = px.iter().fold(0u64, |a, &b| (a << 8) | b as u64) as u32; + let (fx, fy) = self.block_px(&x.block, k as i32, line as i32); + self.put(fx, fy, from_host(x.format, v)); + } + } + + fn get_line(&self, x: &Xfer, line: u32) -> Vec { + let mut out = Vec::with_capacity(x.line_bytes() as usize); + for k in 0..x.width as i32 { + let (fx, fy) = self.block_px(&x.block, k, line as i32); + let v = to_host(x.format, self.get(fx, fy)) as u64; + for i in (0..x.bpp).rev() { + out.push((v >> (8 * i)) as u8); + } + } + out + } + + /// A PIO write doubleword. Each line starts in a fresh doubleword, after + /// its begin offset; the rest of a line's last doubleword is dropped. + fn pio_write(&mut self, dw: u64) -> bool { + let Some(mut x) = self.xfer.take() else { return false }; + let mut changed = false; + for i in 0..8 { + if x.line >= x.block.rows() as u32 { + break; + } + if x.skip > 0 { + x.skip -= 1; + continue; + } + x.pending.push((dw >> (56 - 8 * i)) as u8); + if x.pending.len() as u32 == x.line_bytes() { + let bytes = std::mem::take(&mut x.pending); + self.put_line(&x, x.line, &bytes); + changed = true; + x.line += 1; + x.line_begin = x.next_begin(x.line_begin); + x.skip = x.line_begin; + break; + } + } + self.xfer = Some(x); + changed + } + + /// The whole PIO read stream: per line, filler up to its begin offset, the + /// pixels, and padding to the doubleword. + fn pio_read_stream(&self, x: &Xfer) -> VecDeque { + let mut out = VecDeque::new(); + let mut b = x.begin_skip; + for line in 0..x.block.rows() as u32 { + let mut bytes = vec![0u8; b as usize]; + bytes.extend(self.get_line(x, line)); + while bytes.len() % 8 != 0 { + bytes.push(0); + } + for c in bytes.chunks(8) { + out.push_back(c.iter().fold(0u64, |a, &v| (a << 8) | v as u64)); + } + b = x.next_begin(b); + } + out + } + + /// PIO read, high half: takes the next doubleword. + pub fn pio_read_hi(&mut self) -> u32 { + let Some(x) = self.xfer.as_mut() else { return 0 }; + let dw = x.out.pop_front().unwrap_or(0); + x.out_lo = dw as u32; + (dw >> 32) as u32 + } + + /// PIO read, low half of the doubleword the last high read took. + pub fn pio_read_lo(&self) -> u32 { + self.xfer.as_ref().map(|x| x.out_lo).unwrap_or(0) + } + + /// DMA into the armed block: line `line`'s pixel bytes. + pub fn dma_write_line(&mut self, line: u32, bytes: &[u8]) { + if let Some(x) = self.xfer.take() { + self.put_line(&x, line, bytes); + self.xfer = Some(x); + } + } + + /// DMA out of the armed block: line `line`'s pixel bytes. + pub fn dma_read_line(&self, line: u32) -> Vec { + self.xfer.as_ref().map(|x| self.get_line(x, line)).unwrap_or_default() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn px(r: &Raster, x: usize, y_top: usize) -> u32 { + r.fb[(HEIGHT - 1 - y_top) * WIDTH + x] + } + + /// The X server's setup: window origin at the top row, Y-flip on, so + /// block coordinates are X's top-down ones. + fn x_server() -> Raster { + let mut r = Raster::default(); + r.write(reg::CONFIG, 0xCAC, false); + r.write(reg::XYWIN, 1023 << 16, false); + r.write(reg::PP1FILLMODE, 0x0C00_4504, false); + r + } + + fn block(r: &mut Raster, x0: u32, y0: u32, x1: u32, y1: u32) { + r.write(reg::IR_ALIAS, 0x18, false); + r.write(reg::BLOCKXYSTARTI, x0 << 16 | y0, false); + r.write(reg::BLOCKXYENDI, x1 << 16 | y1, true); + } + + #[test] + fn fast_fill_lands_top_down_with_yflip() { + let mut r = x_server(); + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0x13, false); + block(&mut r, 10, 20, 12, 21); + assert_eq!(px(&r, 10, 20), 0x13); + assert_eq!(px(&r, 12, 21), 0x13); + assert_eq!(px(&r, 13, 21), 0); + assert_eq!(px(&r, 10, 22), 0); + } + + #[test] + fn rgb_fast_fill_packs_components() { + let mut r = x_server(); + r.write(reg::PP1FILLMODE, 3 << 26 | 0x104, false); // RGB pixel type, copy + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0xF00, false); + r.write(reg::FILL_COLOR_G, 0x800, false); + r.write(reg::FILL_COLOR_B, 0x100, false); + block(&mut r, 0, 0, 0, 0); + assert_eq!(px(&r, 0, 0), 0x10_80F0); + } + + #[test] + fn pio_write_frames_each_line_in_a_fresh_doubleword() { + let mut r = x_server(); + // Three 1-byte pixels per line, two lines, begin skip 2: line 0 is + // bytes 2..5 of its doubleword; line 1 starts at (2 + 3) & 7 = 5. + r.write(reg::FILLMODE, 3 << 22, false); + r.write(reg::XFRMODE, 2 << 8, false); + r.write(reg::XFRSIZE, 2 << 16 | 3, false); + block(&mut r, 100, 50, 102, 51); + assert_eq!(r.transfer_armed(), Some(false)); + r.write(reg::CHAR_H, 0x0000_0102, false); + r.write(reg::CHAR_L, 0x03FF_FFFF, true); + r.write(reg::CHAR_H, 0x0000_0000, false); + r.write(reg::CHAR_L, 0x0004_0506, true); + assert_eq!([px(&r, 100, 50), px(&r, 101, 50), px(&r, 102, 50)], [1, 2, 3]); + assert_eq!([px(&r, 100, 51), px(&r, 101, 51), px(&r, 102, 51)], [4, 5, 6]); + } + + #[test] + fn a_new_block_disarms_a_finished_write_transfer() { + let mut r = x_server(); + r.write(reg::FILLMODE, 5 << 22, false); + r.write(reg::XFRSIZE, 1 << 16 | 1, false); + block(&mut r, 0, 0, 0, 0); + assert!(r.transfer_armed().is_some()); + // A glyph block: its char data must stipple, not feed the transfer. + r.write(reg::FILLMODE, 1 << 22, false); + r.write(reg::RED, 0x7 << 12, false); + block(&mut r, 200, 10, 203, 10); + assert_eq!(r.transfer_armed(), None); + r.write(reg::CHAR_H, 0xA000_0000, true); + assert_eq!([px(&r, 200, 10), px(&r, 201, 10), px(&r, 202, 10)], [7, 0, 7]); + } + + #[test] + fn rgb_host_formats_round_trip() { + for (fmt, v) in [((8, 8), 0x0ABCu32), ((8, 10), 0x7FFF), ((8, 0), 0x00C0_FFEE), ((0, 1), 0xFFF)] { + assert_eq!(to_host(fmt, from_host(fmt, v)), v, "{fmt:?}"); + } + assert_eq!(from_host((8, 8), 0x0F0), pack_rgb(0, 0xFF, 0)); + // 8-8-8 host pixels are X pixel values of the visuals, red in 7:0. + assert_eq!(from_host((8, 0), 0x00_00FF), pack_rgb(0xFF, 0, 0)); + } + + #[test] + fn a_24_plane_mask_means_rgb_fills() { + let mut r = x_server(); + r.write(reg::PP1FILLMODE, 0x0C00_6304, false); + r.write(reg::COLORMASKLSBSA, 0xFF_FFFF, false); + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0x380, false); + r.write(reg::FILL_COLOR_G, 0x8E0, false); + r.write(reg::FILL_COLOR_B, 0x8E0, false); + block(&mut r, 1, 1, 1, 1); + assert_eq!(px(&r, 1, 1), 0x8E_8E38); + } + + #[test] + fn an_all_planes_copy_keeps_24_bit_pixels() { + let mut r = x_server(); + // A window move's write-back: pixel type 2, every plane enabled, + // 4-byte host pixels. + r.write(reg::PP1FILLMODE, 0x0C00_6204, false); + r.write(reg::COLORMASKLSBSA, 0xFFFF_FFFF, false); + r.write(reg::FILLMODE, 5 << 22, false); + r.write(reg::XFRMODE, 0x80, false); + r.write(reg::XFRSIZE, 1 << 16 | 1, false); + block(&mut r, 7, 7, 7, 7); + r.dma_write_line(0, &[0x00, 0x50, 0x50, 0x50]); + assert_eq!(px(&r, 7, 7), 0x50_5050); + } + + #[test] + fn xor_fill_toggles_and_restores() { + let mut r = x_server(); + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0x13, false); + block(&mut r, 5, 5, 5, 5); + r.write(reg::PP1FILLMODE, 0x0C00_4504 & !(0xF << 26) | 6 << 26, false); + r.write(reg::FILL_COLOR_R, 0x0F, false); + block(&mut r, 5, 5, 5, 5); + assert_eq!(px(&r, 5, 5), 0x13 ^ 0x0F); + block(&mut r, 5, 5, 5, 5); + assert_eq!(px(&r, 5, 5), 0x13); + } + + #[test] + fn stippled_line_skips_zero_bits() { + let mut r = x_server(); + r.write(reg::FILLMODE, FILL_LINE_STIPPLE, false); + r.write(reg::RED, 0x9 << 12, false); + r.write(reg::LINE_STIPPLE, 0xAAAA_AAAA, false); + r.write(reg::IR_ALIAS, 0x15, false); + r.write(reg::LINE_START, 300 << 16 | 40, false); + r.write(reg::LINE_END, 303 << 16 | 40, true); + assert_eq!((300..304).map(|x| px(&r, x, 40)).collect::>(), [9, 0, 9, 0]); + } +} diff --git a/src/platform_profile_tests.rs b/src/platform_profile_tests.rs index bb3014a4..03e4eb16 100644 --- a/src/platform_profile_tests.rs +++ b/src/platform_profile_tests.rs @@ -15,7 +15,7 @@ mod tests { }; use crate::eeprom_93c56::Eeprom93c56; use crate::ioc::{Ioc, IOC_BASE, IOC_SYS_ID, l1_regs, IOC_INT3_L1_STAT}; - use crate::mgras::{self, reg as mgras_reg, Mgras, MGRAS_REG_OFF, MGRAS_SLOT_GFX_BASE}; + use crate::mgras::{Mgras, GIO_ID, MGRAS_SLOT_GFX_BASE}; use crate::traits::{BusDevice, Saveable}; use crate::dev::gr2::{Gr2, Gr2Stats, Gr2Variant, GR2_BASE}; @@ -185,35 +185,12 @@ mod tests { } #[test] - fn mgras_solid_impact_board_id() { - let cfg = ImpactSection { - gfx: ImpactSlot::Solid, - exp0: ImpactSlot::None, - exp1: ImpactSlot::None, - }; - let m = Mgras::new(&cfg); - let addr = MGRAS_SLOT_GFX_BASE + MGRAS_REG_OFF + mgras_reg::BOARD_ID; - let id = m.read32(addr).data; - assert_eq!(id, mgras_reg::board_id_for(ImpactSlot::Solid)); - } - - #[test] - fn mgras_maximum_three_slot_map() { - let cfg = ImpactSection { - gfx: ImpactSlot::Solid, - exp0: ImpactSlot::High, - exp1: ImpactSlot::Max, - }; - let m = Mgras::new(&cfg); - assert!(m.any_slot()); - for (slot, kind) in [ - (mgras::MGRAS_SLOT_GFX_BASE, ImpactSlot::Solid), - (mgras::MGRAS_SLOT_EXP0_BASE, ImpactSlot::High), - (mgras::MGRAS_SLOT_EXP1_BASE, ImpactSlot::Max), - ] { - let addr = slot + MGRAS_REG_OFF + mgras_reg::BOARD_ID; - assert_eq!(m.read32(addr).data, mgras_reg::board_id_for(kind)); - } + fn mgras_answers_the_gio_id_probe() { + let cfg = ImpactSection { gfx: ImpactSlot::Solid, exp0: ImpactSlot::None, exp1: ImpactSlot::None }; + let hb = Arc::new(std::sync::atomic::AtomicU64::new(0)); + let ioc = crate::ioc::Ioc::new(false); + let m = Mgras::new(&cfg, ioc, hb.clone(), hb); + assert_eq!(m.read32(MGRAS_SLOT_GFX_BASE).data, GIO_ID); } #[test] From 9084c4ff4d7f6fbf5f02614440166e7aaaea6701 Mon Sep 17 00:00:00 2001 From: atomchild411 <143453386+atomchild411@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:44:30 -0800 Subject: [PATCH 2/5] ip28: the Indigo2 IMPACT R10000 (IP28) machine and its R10000 CPU A machine profile, `indigo2_ip28`, with an R10000 CPU model, built only with `--features ip28` (which implies ppmem). Without the feature the profile and the model are refused at startup with a message naming it, and none of their code is compiled: `MachineProfile::ip28()` is a constant false, so the R4400/R5000 machines are unchanged. The R10000 (`CpuModel::R10000`, `R10000ShadowCache`): - A shadow cache, out of the data path: loads and stores go to memory, while tag and data arrays exist to answer CACHE ops and the PROM's diagnostics -- two ways per set, 64-bit tags with TagHi, the set's MRU bit, and the ECC bits Index_Store_Data carries. - Cache ops 5/6/7 take their R10000 meanings (CacheBarrier, Index_Load_Data, Index_Store_Data), and TagLo is a 64-bit register. - An R10000-format CP0 Config, a 64-entry JTLB (the arrays are sized to the largest model; R4400/R5000 still use 48), and 44 virtual address bits. The IP28 board: - The memory controller sizes banks the IP28 way (MEMCFG's size field counts units of the base shift, 256 MB banks, which only the IP28 accepts), and the low-memory alias follows where RAM actually is. - The HPC3 reports board revision 13, so IRIX does not take the machine for an early IP26 baseboard. - Count runs at 97.5 MHz and the MC clock at the CPU clock, the rates IRIX assumes from the PROM's `cpufreq`; the generic ones made UST run at a third of real time. - ppmem is always on. A MEMCFG write that moves no bank leaves the window alone, a real move bumps every generation counter, and cleared ranges are scrubbed to zero-filled memory rather than unmapped: IRIX rewrites MEMCFG during boot while the JIT's compile workers and DMA are running, and a stale reader used to fault. Tracing goes through devlog (`log mips mask cp0`, `log l2c`), with a few IRIS_IP28_* switches for the bring-up paths. `ip28.toml.example` is a starting config. The real IP28 PROM passes POST, including its secondary-cache diagnostic, and IRIX 6.5.22 (64-bit) boots to the 4Dwm desktop on the IMPACT from the previous commit, in about 27 s with jitv2. Co-Authored-By: Claude Opus 5.5 --- Cargo.toml | 6 + ip28.toml.example | 17 + src/config.rs | 78 ++++- src/devlog.rs | 6 +- src/ioc.rs | 23 +- src/lib.rs | 2 + src/machine.rs | 51 ++- src/mc.rs | 250 ++++++++++++-- src/mips_cache_shadow.rs | 680 +++++++++++++++++++++++++++++++++++++++ src/mips_cache_v2.rs | 235 ++++++++++++-- src/mips_core.rs | 144 ++++++++- src/mips_exec.rs | 285 +++++++++++++++- src/mips_tlb.rs | 93 +++--- src/mips_tlb_test.rs | 53 +++ src/physical.rs | 92 ++++-- src/ppmem/map_unix.rs | 39 +++ src/ppmem/map_windows.rs | 12 + src/ppmem/ppmem.rs | 40 ++- src/testdev.rs | 2 +- 19 files changed, 1958 insertions(+), 150 deletions(-) create mode 100644 ip28.toml.example create mode 100644 src/mips_cache_shadow.rs diff --git a/Cargo.toml b/Cargo.toml index 25b8e9ae..116ea07c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -191,6 +191,12 @@ jitv2 = ["cranelift-codegen", "cranelift-frontend", "cranelift-jit", "cranelift- # specifically; opt in with --features jitv2_opcodefusion once it's earned # more confidence. jitv2_opcodefusion = ["jitv2"] +# IP28: the Indigo2 IMPACT machine (IP28) and its R10000 CPU. Without it the +# IP28 profile and the R10000 model are refused at startup with a message +# naming this feature, and none of their code is built, so nothing behind it +# can reach the R4400/R5000 machines. It implies ppmem: IP28 guest RAM is +# always mapped through the host MMU. Add jitv2 for the JIT, as elsewhere. +ip28 = ["ppmem"] # jitv2 whole-page compile (§13, rules/jitv2/jit-v2-design.md): whole-page # compilation is now the standard and only JIT v2 design. Kept as a compatibility # alias for `jitv2` so downstream scripts and builds specifying `--features j2wp` diff --git a/ip28.toml.example b/ip28.toml.example new file mode 100644 index 00000000..3dcb3841 --- /dev/null +++ b/ip28.toml.example @@ -0,0 +1,17 @@ +# Indigo2 IMPACT R10000 (IP28) bring-up. +# Runs the real IP28 PROM against the existing fullhouse (IP22) machine. +# The CPU is still an R4400/R5000 model — the point of this run is to find +# out what the PROM objects to first, not to boot anything. +headless = true +no_audio = true +banks = [64, 64, 64, 64] +prom = "ip28/ip28prom.070-1477-002.bin" +nveeprom = "ip28/nveeprom-ip28.bin" +serial_log = "ip28/console-ip28.log" +[machine] +profile = "indigo2_ip22" +cpu = "r10000" +[scsi.1] +path = "ip28/blank.raw" +cdrom = false +overlay = true diff --git a/src/config.rs b/src/config.rs index b00fcbaa..4f794611 100644 --- a/src/config.rs +++ b/src/config.rs @@ -3,7 +3,7 @@ use serde::{Deserialize, Serialize}; use std::net::Ipv4Addr; /// Valid memory bank sizes in MB. -pub const VALID_BANK_SIZES: &[u32] = &[0, 8, 16, 32, 64, 128]; +pub const VALID_BANK_SIZES: &[u32] = &[0, 8, 16, 32, 64, 128, 256]; /// What sits at a SCSI id. `cdrom = true` remains the historical spelling for /// `kind = "cdrom"`; either works and they mean the same thing. @@ -384,28 +384,53 @@ pub enum MachineProfile { IndyIp24, /// SGI Indigo2 IP22 — fullhouse MC/IOC, Newport XL on GIO gfx slot. Indigo2Ip22, + /// SGI Indigo2 IMPACT IP28 — an R10000 CPU module in the Indigo2 chassis. + /// + /// Shares the fullhouse MC/IOC/HPC3 with IP22 and differs in the decodes + /// inside them: MEMCFG's base field is shifted by 24 rather than 22 (so + /// its size granule is 16 MB, not 4), RAM lives at 0x20000000 with the + /// low-memory alias following it there, and both the MC chip revision and + /// the HPC3 board revision have to read high enough for the kernel to + /// call the board an IP28. + /// + /// Graphics is IMPACT, which is a register stub — an IP28 kernel carries + /// no Newport driver, so REX3 is not an alternative here. + Indigo2Ip28, } impl MachineProfile { /// All selectable profiles, in display order. Single source of truth for the /// GUI dropdowns (Config tab + New Machine dialog) so they never drift. + #[cfg(feature = "ip28")] + pub const ALL: [Self; 3] = [Self::IndyIp24, Self::Indigo2Ip22, Self::Indigo2Ip28]; + #[cfg(not(feature = "ip28"))] pub const ALL: [Self; 2] = [Self::IndyIp24, Self::Indigo2Ip22]; pub fn label(self) -> &'static str { match self { Self::IndyIp24 => "SGI Indy (IP24)", Self::Indigo2Ip22 => "SGI Indigo2 (IP22)", + Self::Indigo2Ip28 => "SGI Indigo2 IMPACT (IP28)", } } pub fn supported(self) -> bool { matches!(self, Self::IndyIp24 | Self::Indigo2Ip22) + || (cfg!(feature = "ip28") && matches!(self, Self::Indigo2Ip28)) } /// MC/IOC/HPC3 Guinness vs Fullhouse layout. Indy IP24 is Guinness (`true`). pub fn guinness(self) -> bool { matches!(self, Self::IndyIp24) } + + /// The R10000 Indigo2. Selects the IP28 decodes inside the shared + /// fullhouse devices — see the variant's own documentation for the list. + /// + /// Always false without the `ip28` feature, so every IP28 decode folds away. + pub fn ip28(self) -> bool { + cfg!(feature = "ip28") && matches!(self, Self::Indigo2Ip28) + } } /// Indy / Indigo2 graphics board in the GIO gfx slot. @@ -525,12 +550,27 @@ pub enum CpuModel { R4400, /// MIPS R5000, 2-way 32K L1s, no secondary cache, MIPS IV. R5000, + /// MIPS R10000, 2-way 32K L1s, 1 MB secondary cache, MIPS IV. The CPU in + /// the Indigo2 IMPACT (IP28). Bring-up only — see docs/ip28-bringup.md. + R10000, } impl CpuModel { + #[cfg(feature = "ip28")] + pub const ALL: [Self; 3] = [Self::R4400, Self::R5000, Self::R10000]; + #[cfg(not(feature = "ip28"))] pub const ALL: [Self; 2] = [Self::R4400, Self::R5000]; + + /// Whether this build can run the model: the R10000 needs the `ip28` feature. + pub fn available(self) -> bool { + !matches!(self, Self::R10000) || cfg!(feature = "ip28") + } pub fn label(self) -> &'static str { - match self { Self::R4400 => "MIPS R4400", Self::R5000 => "MIPS R5000" } + match self { + Self::R4400 => "MIPS R4400", + Self::R5000 => "MIPS R5000", + Self::R10000 => "MIPS R10000", + } } } @@ -1205,6 +1245,11 @@ impl MachineConfig { /// Validate bank sizes, returns a description of any errors. pub fn validate(&self) -> Result<(), String> { + if (self.machine.profile == MachineProfile::Indigo2Ip28 && !cfg!(feature = "ip28")) + || !self.machine.cpu.available() + { + return Err("IP28 / R10000 support is not built into this binary; rebuild with --features ip28".to_string()); + } if !self.machine.profile.supported() { return Err(format!( "machine profile \"{}\" is not implemented; use {}", @@ -1235,9 +1280,11 @@ impl MachineConfig { return Err(format!("graphics.board \"{name}\" and [impact] both claim the GIO gfx slot")); } } - if self.impact.any_enabled() && self.machine.profile != MachineProfile::Indigo2Ip22 { + if self.impact.any_enabled() + && !matches!(self.machine.profile, MachineProfile::Indigo2Ip22 | MachineProfile::Indigo2Ip28) + { return Err( - "[impact] slots are preview-only on Indigo2 (machine.profile = indigo2_ip22)".into(), + "[impact] slots need an Indigo2 (machine.profile = indigo2_ip22 or indigo2_ip28)".into(), ); } self.impact.validate()?; @@ -1262,6 +1309,14 @@ impl MachineConfig { i, sz, VALID_BANK_SIZES )); } + // The IP22/IP24 MC cannot express a 256 MB bank at its base + // shift; only the IP28's can. + if sz == 256 && !self.machine.profile.ip28() { + return Err(format!( + "bank{} size 256 MB needs the IP28 (machine.profile = indigo2_ip28)", + i + )); + } } if let Some(ref s) = self.nat_subnet { if let Err(e) = parse_nat_subnet(s) { @@ -1810,6 +1865,21 @@ mod export_tests { assert!(cfg.machine.profile.supported()); assert!(!cfg.machine.profile.guinness()); } + + #[test] + fn a_256_mb_bank_is_an_ip28_bank() { + let mut cfg = MachineConfig::default(); + cfg.machine.profile = MachineProfile::Indigo2Ip22; + cfg.banks = [256, 128, 0, 0]; + let err = cfg.validate().expect_err("the IP22 MC cannot express a 256 MB bank"); + assert!(err.contains("256 MB"), "{err}"); + #[cfg(feature = "ip28")] + { + cfg.machine.profile = MachineProfile::Indigo2Ip28; + cfg.machine.cpu = CpuModel::R10000; + cfg.validate().expect("the IP28 MC can"); + } + } } #[cfg(test)] diff --git a/src/devlog.rs b/src/devlog.rs index fbda423a..e27dc116 100644 --- a/src/devlog.rs +++ b/src/devlog.rs @@ -155,7 +155,8 @@ impl LogModule { if mask & 0x0002 != 0 { parts.push("tlb"); } if mask & 0x0004 != 0 { parts.push("mem"); } if mask & 0x0008 != 0 { parts.push("fpu"); } - let rest = mask & !0x000F; + if mask & 0x0010 != 0 { parts.push("cp0"); } + let rest = mask & !0x001F; let mut s = parts.join("+"); if rest != 0 { s.push_str(&format!("+{:#010x}", rest)); } if s.is_empty() { format!("{:#010x}", mask) } else { s } @@ -192,6 +193,7 @@ impl LogModule { "tlb" => Some(0x0002), "mem" => Some(0x0004), "fpu" => Some(0x0008), + "cp0" => Some(0x0010), "on" | "all" => Some(0xFFFF_FFFF), "off" | "none" => Some(0x0000_0000), _ => u32::from_str_radix(s.trim_start_matches("0x"), 16).ok(), @@ -458,7 +460,7 @@ impl Device for DevLog { "[DEV] log | log mask | log file | log status\n\ \x20 modules: net hpc3 seeq hal2 mc rex3 mips ioc scsi pdma vino dcb vc2 cmap xmap bt445 scc ps2 rtc eeprom l1i l1d l2c\n\ \x20 pdma mask categories: hal enet scsi on/all off/none \n\ - \x20 mips mask categories: insn tlb mem fpu on/all off/none \n\ + \x20 mips mask categories: insn tlb mem fpu cp0 on/all off/none \n\ \x20 l1i/l1d/l2c mask categories: hit miss op on/all off/none \n\ \x20 [DEV] = requires --features developer build to produce output".to_string(), )] diff --git a/src/ioc.rs b/src/ioc.rs index 70549add..f46b71f0 100644 --- a/src/ioc.rs +++ b/src/ioc.rs @@ -444,8 +444,29 @@ impl Ioc { Self::new_inner(guinness, true) } + /// As `new`/`new_ci`, with the machine profile's IP28 flag: an IP28 + /// baseboard must report a high enough HPC3 board revision. + pub fn new_for_profile(guinness: bool, ci_mode: bool, ip28: bool) -> Self { + Self::new_inner_profile(guinness, ci_mode, ip28) + } + fn new_inner(guinness: bool, ci_mode: bool) -> Self { - let sys_id = if guinness { 0x26 } else { 0x11 }; // primarily prom looks at bit 1 to detect full house. + Self::new_inner_profile(guinness, ci_mode, false) + } + + fn new_inner_profile(guinness: bool, ci_mode: bool, ip28: bool) -> Self { + // HPC3 SYS_ID: [7:5] chip rev, [4:1] board rev, [0] 1 = fullhouse. + // The PROM looks at bit 0 to tell fullhouse from guinness. IRIX reads + // the board revision to tell an IP28 baseboard from an IP26 one and + // warns "CPU baseboard downrev (IP26 not IP28)" below 13, so an IP28 + // has to report at least that. + let sys_id: u8 = if guinness { + 0x26 + } else if ip28 { + 0x1B // board rev 13, fullhouse + } else { + 0x11 // board rev 8, fullhouse + }; let state = Arc::new(Mutex::new(IocState { sys_id, l0_stat: 0, diff --git a/src/lib.rs b/src/lib.rs index 3ab7c2ac..91d98e8b 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -172,6 +172,8 @@ pub mod mips_dis; pub mod mips_core; pub mod mips_tlb; pub mod mips_cache_v2; +#[cfg(feature = "ip28")] +pub mod mips_cache_shadow; pub mod mips_exec; pub mod mips_exec_test; pub mod mips_instr_stats; diff --git a/src/machine.rs b/src/machine.rs index 1ba79e46..a2a818f0 100644 --- a/src/machine.rs +++ b/src/machine.rs @@ -23,10 +23,16 @@ impl PhysPtr { use crate::physical::RamBank; use crate::prom::Prom; use crate::mc::MemoryController; + +/// Count on the IP28's 195 MHz R10000: half the pipeline clock, and what +/// IRIX assumes there (see where the MC is created). +const IP28_COUNT_HZ: u64 = 97_500_000; use crate::mips_tlb::MipsTlb; use crate::mips_exec::{MipsExecutor, MipsCpu, MipsCpuConfig, MipsCpuDebugAdapter}; use crate::gdb_stub::CpuDebug; use crate::mips_cache_v2::{MipsCache, R4400Cache, R5000Cache}; +#[cfg(feature = "ip28")] +use crate::mips_cache_shadow::R10000ShadowCache; use crate::hpc3::Hpc3; use crate::ioc::{Ioc, GioSlot, GIO_SLOT_MAP, profile_idx}; use crate::monitor::Monitor; @@ -257,6 +263,12 @@ impl Machine { let clock_fixed_mhz = cfg.clock.fixed_mhz; let cfg_cpu_model = cfg.machine.cpu; + if (cfg.machine.profile == MachineProfile::Indigo2Ip28 && !cfg!(feature = "ip28")) + || !cfg_cpu_model.available() + { + eprintln!("iris: IP28 / R10000 support is not built into this binary; rebuild with --features ip28"); + std::process::exit(1); + } if !cfg.machine.profile.supported() { eprintln!( "iris: machine profile \"{}\" is not implemented; use {}", @@ -284,6 +296,10 @@ impl Machine { let model_has_l2 = match cfg_cpu_model { crate::config::CpuModel::R4400 => ::L2_SIZE > 0, crate::config::CpuModel::R5000 => ::L2_SIZE > 0, + #[cfg(feature = "ip28")] + crate::config::CpuModel::R10000 => ::L2_SIZE > 0, + #[cfg(not(feature = "ip28"))] + crate::config::CpuModel::R10000 => unreachable!("refused above without the ip28 feature"), }; if !model_has_l2 { eeprom_mc.lock().set_cachsz(0); @@ -291,7 +307,24 @@ impl Machine { // 1. Create all devices first // Memory Controller - let mc = MemoryController::new(eeprom_mc.clone(), guinness, cfg.banks); + let mc = MemoryController::new_for_profile( + eeprom_mc.clone(), guinness, cfg.banks, cfg.machine.profile.ip28()); + // IP28: IRIX takes the CPU speed from the PROM's `cpufreq` (194, i.e. + // a 195 MHz R10000) instead of measuring it, and times with it: UST + // assumes Count at half that, and the RPSS cycle counter the MC clock + // at the CPU clock (Config.EC is 0, a 1:1 system-clock ratio) over the + // divider it programs. The generic 33 MHz Count and 50 MHz MC made + // UST run at a third of real time (Quake in slow motion) and the + // cycle counter at a quarter. + let ip28_count_hz: Option = if cfg.machine.profile.ip28() { + Some(clock_fixed_mhz.map_or(IP28_COUNT_HZ, |mhz| (mhz * 1_000_000.0) as u64)) + } else { + None + }; + if let Some(count_hz) = ip28_count_hz { + // The R10000's Count ticks at half the pipeline clock. + mc.set_clock_hz(2 * count_hz); + } // RAM banks sized per config. addr_mask is initialized to mem_size-1; // remap_banks() updates it via set_addr_mask() when MEMCFG0/1 are written during POST. @@ -325,7 +358,7 @@ impl Machine { // HPC3 (512KB at 0x1FB80000). CI mode skips the SCC TCP backend // bindings so multiple `--ci` instances can coexist. - let ioc = if ci_enabled { Ioc::new_ci(guinness) } else { Ioc::new(guinness) }; + let ioc = Ioc::new_for_profile(guinness, ci_enabled, cfg.machine.profile.ip28()); // CI mode replaces the default TCP backend on channel B (tty1, the // SGI serial console) with an in-process backend the control socket @@ -651,6 +684,7 @@ impl Machine { mc.clone(), hpc3.clone(), prom_port, + cfg.machine.profile.ip28(), ); // Wrap Physical in Arc @@ -733,7 +767,7 @@ impl Machine { // arm below monomorphises its own CPU — no per-model branch on the hot path. let sysad: Arc = phys.clone(); macro_rules! build_cpu { ($cache:ty) => {{ - let cfg = MipsCpuConfig::indy(); + let cfg = MipsCpuConfig::for_model::<$cache>(); let tlb = MipsTlb::new(cfg.tlb_entries); let mut executor: MipsExecutor = MipsExecutor::new(sysad.clone(), tlb, &cfg); @@ -751,7 +785,9 @@ impl Machine { // CP0 Count runs at a fixed frequency: DEFAULT_COUNT_HZ unless the // user overrode it via `[clock] fixed_mhz` or the CLI. Must happen // before the core starts executing. - if let Some(mhz) = clock_fixed_mhz { + if let Some(hz) = ip28_count_hz { + executor.core.set_count_hz(hz); + } else if let Some(mhz) = clock_fixed_mhz { executor.core.set_count_hz((mhz * 1_000_000.0) as u64); } @@ -778,6 +814,13 @@ impl Machine { let cpu: Arc = match cfg_cpu_model { crate::config::CpuModel::R4400 => build_cpu!(R4400Cache), crate::config::CpuModel::R5000 => build_cpu!(R5000Cache), + // IP28 uses the shadow cache: out of the data path entirely, with + // tag and data arrays that exist only to answer CACHE ops and the + // PROM's diagnostics. See mips_cache_shadow.rs. + #[cfg(feature = "ip28")] + crate::config::CpuModel::R10000 => build_cpu!(R10000ShadowCache), + #[cfg(not(feature = "ip28"))] + crate::config::CpuModel::R10000 => unreachable!("refused above without the ip28 feature"), }; // Share count_hz_atomic from MipsCore with Rex3 so the refresh thread can display it. diff --git a/src/mc.rs b/src/mc.rs index 0cdf61da..3de34e75 100644 --- a/src/mc.rs +++ b/src/mc.rs @@ -43,6 +43,9 @@ pub const REG_SYS_SEMAPHORE: u32 = 0x0100; pub const REG_LOCK_MEMORY: u32 = 0x0108; pub const REG_EISA_LOCK: u32 = 0x0110; pub const REG_RPSS_CTR: u32 = 0x1000; + +/// The MC clock of the IP22/IP24 models (see `MemoryController::set_clock_hz`). +pub const DEFAULT_CLOCK_HZ: u64 = 50_000_000; pub const REG_SEMAPHORE_0: u32 = 0x10000; // ... Semaphores 1-15 follow pattern +0x1000 @@ -56,6 +59,9 @@ struct MemoryControllerState { // Timer state last_host_ticks: u64, host_freq: u64, + /// The clock the refresh, watchdog and RPSS counters count (see + /// `set_clock_hz`). + clock_hz: u64, cpu_cycle_acc: u64, rpss_cycle_acc: u64, cpu: Option>, @@ -70,6 +76,11 @@ pub struct MemoryController { threads: Arc>>>, running: Arc, guinness: bool, + /// How far MEMCFG's base field is shifted to give a physical address: + /// 22 on IP22/IP24, 24 on IP28. The size field counts per-subbank units of + /// `1 << base_shift` bytes, so the granule follows it too — 4 MB against + /// 16 MB. Set from the machine profile, never from the environment. + base_shift: u32, /// Actual SIMM sizes in MB for each bank (index 0..3). Used by parse_memcfg to /// derive addr_mask (for aliasing) and limit (for device_map range). ram_sizes: Arc<[u32; 4]>, @@ -93,8 +104,21 @@ pub struct MemoryController { } impl MemoryController { + /// An IP22/IP24 memory controller. pub fn new(eeprom: Arc>, guinness: bool, ram_sizes: [u32; 4]) -> Self { - let regs = Self::init_registers(guinness); + Self::new_for_profile(eeprom, guinness, ram_sizes, false) + } + + /// `ip28` selects the IP28 decodes: a 24-bit MEMCFG base shift and an MC + /// chip revision the IP28 PROM accepts. + pub fn new_for_profile( + eeprom: Arc>, + guinness: bool, + ram_sizes: [u32; 4], + ip28: bool, + ) -> Self { + let base_shift = if ip28 { 24 } else { 22 }; + let regs = Self::init_registers_for(guinness, base_shift); let host_freq = crate::platform::get_host_tick_frequency(); let last_host_ticks = crate::platform::get_host_ticks(); @@ -106,6 +130,7 @@ impl MemoryController { user_semaphores: [false; 16], last_host_ticks, host_freq, + clock_hz: DEFAULT_CLOCK_HZ, cpu_cycle_acc: 0, rpss_cycle_acc: 0, cpu: None, @@ -116,6 +141,7 @@ impl MemoryController { threads: Arc::new(Mutex::new(Vec::new())), running: Arc::new(AtomicBool::new(false)), guinness, + base_shift, ram_sizes: Arc::new(ram_sizes), memcfg_callback: Arc::new(OnceLock::new()), event_tx: Arc::new(OnceLock::new()), @@ -127,6 +153,10 @@ impl MemoryController { } fn init_registers(guinness: bool) -> Vec { + Self::init_registers_for(guinness, 22) + } + + fn init_registers_for(guinness: bool, base_shift: u32) -> Vec { let mut regs = vec![0; (MC_SIZE / 4) as usize]; // Initialize CPUCTRL0: REFS=2, RFE=1, MUX_HWM=1 @@ -149,7 +179,20 @@ impl MemoryController { // unconditionally lets vino_init proceed; with it set, `vlinfo` // reports `vino 0` with 5 nodes (digital input = IndyCam, analog // input, two memory drains, controls). - regs[(REG_SYSID / 4) as usize] = 0x00000013; + // IP28 raises the bar: its power-on diagnostic reads the chip + // revision out of SYSID and rejects anything below 5 with "FATAL + // ERROR: Rev A/BC MC detected--Need rev D or greater." It needs a + // rev D part because that is what supports the high memory mapping + // IP28 uses. Report 5 there; every other machine keeps the rev C (3) + // it has always seen. Bit 4 stays set in both — that is the EISA / + // vino gate, not part of the revision. + regs[(REG_SYSID / 4) as usize] = if base_shift == 22 { + 0x00000013 + } else { + // IP28's chip revision lives in the low nibble; 5 is the lowest + // the PROM accepts. + 0x00000015 + }; // Initialize RPSS_DIVIDER: DIV=9, INC=3 (for 33MHz) // 33MHz: Divide by 10 (9+1), Increment by 3 -> 300ns per tick @@ -186,6 +229,16 @@ impl MemoryController { regs } + /// Set the clock the MC's counters count. IRIX derives the RPSS cycle + /// counter's period from its idea of this clock (on IP28: the CPU clock + /// times the R10000 Config.EC system-clock ratio, over the RPSS divider it + /// programs), and hands that period to programs that time with the + /// counter; the model has to count at the same rate or every such timing + /// is off by the difference. Call before the machine runs. + pub fn set_clock_hz(&self, hz: u64) { + self.state.lock().clock_hz = hz; + } + pub fn set_cpu(&self, cpu: Weak) { self.state.lock().cpu = Some(cpu); } @@ -233,19 +286,75 @@ impl MemoryController { /// inst_size_per_rank = size_mb_bytes >> inst_rank. /// Encode a MEMCFG half-word for a bank at `base` with `size_mb` installed. /// Inverse of [`memcfg_bank_info`] for the sizes IRIS supports. + /// How far the MEMCFG base field is shifted to give a physical address. + /// + /// IP22/IP24 use 22 (a 4 MB granule). IP28 appears to use 24: its PROM + /// writes base byte 0x60 and then probes 0x60000000, and base byte 0x20 + /// gives 0x20000000, which is where NetBSD loads IP28 kernels. Both fall + /// out of a 24-bit shift and neither does out of 22. + /// + /// Env-gated while IP28 has no machine profile of its own. It must not + /// change IP22/IP24, which this default preserves. + /// This controller's MEMCFG base shift. See the `base_shift` field. + fn memcfg_base_shift(&self) -> u32 { + self.base_shift + } + + /// A bank's installed size in MB → `(size_field, rank)` in register format. + /// + /// The size field counts per-subbank units of `1 << memcfg_base_shift()` + /// bytes, minus one, and the rank bit doubles the bank. The granule + /// therefore follows the base shift: 4 MB where the shift is 22, 16 MB + /// where it is 24. The same 64 MB bank is size field 15 on IP22 and size + /// field **3** on IP28 — which is what the IP28 PROM is observed to write + /// for its own banks (MEMCFG0 = 0x2320_2324, two 64 MB banks at + /// 0x20000000 and 0x24000000). + /// + /// The shift-22 rows are left exactly as they were, rank bits included: + /// they describe real IP22 SIMM topology, and more than one of them has + /// more than one arithmetically-equivalent encoding. + fn memcfg_size_rank(size_mb: u32) -> Option<(u32, u32)> { + Self::memcfg_size_rank_at(22, size_mb) + } + + /// `memcfg_size_rank` with the granule passed in, so both can be tested in + /// one process — the live shift is read from the environment once and + /// cached for the life of the program. + fn memcfg_size_rank_at(shift: u32, size_mb: u32) -> Option<(u32, u32)> { + if shift == 24 { + // 16 MB granule. + return match size_mb { + 16 => Some((0, 0)), + 32 => Some((1, 0)), + 64 => Some((3, 0)), + 128 => Some((7, 0)), + 256 => Some((15, 0)), + _ => None, + }; + } + // 4 MB granule. + match size_mb { + 8 => Some((0, 1)), + 16 => Some((3, 0)), + 32 => Some((3, 1)), + 64 => Some((15, 0)), + 128 => Some((15, 1)), + _ => None, + } + } + pub fn encode_memcfg_half(base: u32, size_mb: u32) -> Option { + Self::encode_memcfg_half_at(22, base, size_mb) + } + + /// `encode_memcfg_half` with the granule passed in. See + /// [`memcfg_size_rank_at`](Self::memcfg_size_rank_at). + fn encode_memcfg_half_at(shift: u32, base: u32, size_mb: u32) -> Option { if size_mb == 0 { return None; } - let (simm_size_field, simm_rank): (u32, u32) = match size_mb { - 8 => (0, 1), - 16 => (3, 0), - 32 => (3, 1), - 64 => (15, 0), - 128 => (15, 1), - _ => return None, - }; - let base_byte = (base >> 22) & 0xFF; + let (simm_size_field, simm_rank) = Self::memcfg_size_rank_at(shift, size_mb)?; + let base_byte = (base >> shift) & 0xFF; Some( (base_byte as u16) | (1 << 13) // VLD @@ -255,27 +364,27 @@ impl MemoryController { } pub fn memcfg_bank_info(half: u16, size_mb: u32) -> Option<(u32, u32, u32)> { + Self::memcfg_bank_info_at(22, half, size_mb) + } + + /// `memcfg_bank_info` with the granule passed in. See + /// [`memcfg_size_rank_at`](Self::memcfg_size_rank_at). + fn memcfg_bank_info_at(shift: u32, half: u16, size_mb: u32) -> Option<(u32, u32, u32)> { if size_mb == 0 { return None; } if (half >> 13) & 1 == 0 { return None; } - let base = ((half as u32) & 0xFF) << 22; + let base = ((half as u32) & 0xFF) << shift; let conf_rank = ((half >> 14) & 1) as u32; let conf_size_field = ((half >> 8) & 0x1F) as u32; - let conf_total = (conf_size_field + 1) << 22; - - // SIMM size → (size_field, rank) in register format (one unit = 4MB) - let (simm_size_field, simm_rank): (u32, u32) = match size_mb { - 8 => (0, 1), - 16 => (3, 0), - 32 => (3, 1), - 64 => (15, 0), - 128 => (15, 1), - _ => return None, - }; + let conf_total = (conf_size_field + 1) << shift; + + // SIMM size → (size_field, rank) in register format; the unit follows + // the base shift — see `memcfg_size_rank_at`. + let (simm_size_field, simm_rank) = Self::memcfg_size_rank_at(shift, size_mb)?; let conf_size = conf_total >> conf_rank; - let minus_size = (simm_size_field + 1) << (22 - simm_rank); - let plus_size = (simm_size_field + 1) << (22 + simm_rank); + let minus_size = (simm_size_field + 1) << (shift - simm_rank); + let plus_size = (simm_size_field + 1) << (shift + simm_rank); // BNK=0 (aliasing phase): wrap at inst_size so alias is detected at base+inst_size // BNK=1 (subbank/walkingbit): wrap at full bank size so both ranks are independent let addr_mask = if conf_rank == 0 { minus_size - 1 } else { plus_size - 1 }; @@ -294,7 +403,7 @@ impl MemoryController { return false; } let half = |i: usize, base: u32| { - Self::encode_memcfg_half(base, self.ram_sizes[i]).unwrap_or(0) as u32 + Self::encode_memcfg_half_at(self.base_shift, base, self.ram_sizes[i]).unwrap_or(0) as u32 }; let memcfg0 = (half(0, LOMEM_BASE) << 16) | half(1, LOMEM_BASE + BANK_SIZE); if memcfg0 == 0 { @@ -312,7 +421,7 @@ impl MemoryController { (memcfg1 >> 16) as u16, // bank 2: high half of MEMCFG1 (memcfg1 & 0xFFFF) as u16, // bank 3: low half of MEMCFG1 ]; - std::array::from_fn(|i| Self::memcfg_bank_info(halves[i], self.ram_sizes[i])) + std::array::from_fn(|i| Self::memcfg_bank_info_at(self.base_shift, halves[i], self.ram_sizes[i])) } /// If the embedded PROM POSTed lomem (banks 0–1) but skipped himem, synthesize @@ -320,6 +429,18 @@ impl MemoryController { fn synthesize_himem_banks(&self, state: &mut MemoryControllerState) -> bool { use crate::physical::{BANK_SIZE, HIMEM_BASE}; + // Only for a PROM that POSTs lomem and stops. The IP28 PROM sizes all + // four banks itself, walking them one at a time through a probe window + // at base byte 0x60; it finishes MEMCFG0 before it has finished with + // MEMCFG1. Synthesizing here fired on that intermediate state and + // overwrote the walk in progress, leaving banks 2 and 3 describing + // 256 MB apiece — bank 2 on top of bank 0 — and the PROM then never + // converged. HIMEM_BASE and BANK_SIZE are lomem/himem constants that + // mean nothing on a machine whose RAM starts at 0x20000000 anyway. + if self.base_shift != 22 { + return false; + } + let memcfg0 = state.regs[(REG_MEMCFG0 / 4) as usize]; let memcfg1 = state.regs[(REG_MEMCFG1 / 4) as usize]; let h0 = (memcfg0 >> 16) as u16; @@ -334,14 +455,14 @@ impl MemoryController { let mut changed = false; if self.ram_sizes[2] > 0 && (h2 >> 13) & 1 == 0 { - if let Some(enc) = Self::encode_memcfg_half(HIMEM_BASE, self.ram_sizes[2]) { + if let Some(enc) = Self::encode_memcfg_half_at(self.base_shift, HIMEM_BASE, self.ram_sizes[2]) { h2 = enc; changed = true; } } if self.ram_sizes[3] > 0 && (h3 >> 13) & 1 == 0 { if let Some(enc) = - Self::encode_memcfg_half(HIMEM_BASE + BANK_SIZE, self.ram_sizes[3]) + Self::encode_memcfg_half_at(self.base_shift, HIMEM_BASE + BANK_SIZE, self.ram_sizes[3]) { h3 = enc; changed = true; @@ -439,11 +560,11 @@ impl MemoryControllerState { let diff = now.wrapping_sub(self.last_host_ticks); self.last_host_ticks = now; - // Scale to 50MHz CPU cycles (20ns period) - // cycles = (diff * 50_000_000) / host_freq + // Scale to cycles of the MC clock (`clock_hz`). + // cycles = (diff * clock_hz) / host_freq // Use accumulator to maintain precision over many small updates - // Use u128 to prevent overflow during multiplication (diff * 50M can exceed u64) - let total_ticks = (diff as u128) * 50_000_000 + (self.cpu_cycle_acc as u128); + // Use u128 to prevent overflow during multiplication + let total_ticks = (diff as u128) * (self.clock_hz as u128) + (self.cpu_cycle_acc as u128); let cpu_cycles = (total_ticks / (self.host_freq as u128)) as u64; self.cpu_cycle_acc = (total_ticks % (self.host_freq as u128)) as u64; @@ -685,12 +806,18 @@ impl BusDevice for MemoryController { BUS_OK } REG_MEMCFG0 => { + if self.base_shift != 22 { + eprintln!("iris: ip28: guest writes MEMCFG0 = {val:#010x}"); + } dlog_dev!(LogModule::Mc, "MC: Write MEMCFG0 = {:08x}", val); state.regs[(REG_MEMCFG0 / 4) as usize] = val; self.on_memcfg_updated(&mut state); BUS_OK } REG_MEMCFG1 => { + if self.base_shift != 22 { + eprintln!("iris: ip28: guest writes MEMCFG1 = {val:#010x}"); + } dlog_dev!(LogModule::Mc, "MC: Write MEMCFG1 = {:08x}", val); state.regs[(REG_MEMCFG1 / 4) as usize] = val; self.on_memcfg_updated(&mut state); @@ -1351,4 +1478,61 @@ mod tests { assert!(addrs[3].is_some(), "bank 3 should be synthesized"); assert_eq!(addrs[2].unwrap().0, crate::physical::HIMEM_BASE); } + + /// The IP28 PROM's own MEMCFG0 write, copied from a POST trace: + /// `0x2320_2324` — two 64 MB banks at 0x20000000 and 0x24000000. + /// + /// Size field 3 means four units of 16 MB, because the IP28 granule is the + /// base shift (24), not IP22's 22. Decoding it with IP22's table called the + /// same bank 256 MB, which is how banks 2 and 3 came to claim memory that + /// was not there. + #[test] + fn ip28_memcfg_matches_what_the_prom_writes() { + const OBSERVED: u32 = 0x2320_2324; + let h0 = (OBSERVED >> 16) as u16; + let h1 = (OBSERVED & 0xFFFF) as u16; + + assert_eq!(MemoryController::encode_memcfg_half_at(24, 0x2000_0000, 64), Some(h0)); + assert_eq!(MemoryController::encode_memcfg_half_at(24, 0x2400_0000, 64), Some(h1)); + + let (base0, mask0, limit0) = MemoryController::memcfg_bank_info_at(24, h0, 64).unwrap(); + assert_eq!(base0, 0x2000_0000); + assert_eq!(limit0, 64 << 20, "a 64 MB bank must not claim more"); + assert_eq!(mask0, (64 << 20) - 1, "and must alias within its own 64 MB"); + + let (base1, _, limit1) = MemoryController::memcfg_bank_info_at(24, h1, 64).unwrap(); + assert_eq!(base1, 0x2400_0000); + assert_eq!(base1, base0 + limit0, "banks 0 and 1 are contiguous, not overlapping"); + assert_eq!(limit1, 64 << 20); + } + + /// Four 64 MB banks must tile 0x20000000..0x30000000 with no overlap and no + /// gap. The failure this pins is the one that stalled the IRIX install: + /// bank 2 landed on top of bank 0 and bank 3 claimed 256 MB, so the top of + /// "memory" was 0x38000000 and the miniroot was loaded into nothing. + #[test] + fn ip28_four_banks_tile_without_overlap() { + let mut next = 0x2000_0000u32; + for bank in 0..4 { + let half = MemoryController::encode_memcfg_half_at(24, next, 64) + .unwrap_or_else(|| panic!("bank {bank} did not encode")); + let (base, _, limit) = MemoryController::memcfg_bank_info_at(24, half, 64).unwrap(); + assert_eq!(base, next, "bank {bank} base"); + assert_eq!(limit, 64 << 20, "bank {bank} limit"); + next = base + limit; + } + assert_eq!(next, 0x3000_0000, "256 MB total, ending where RAM ends"); + } + + /// The IP22 encodings are unchanged by the granule work. + #[test] + fn ip22_memcfg_encodings_are_unchanged() { + for (size_mb, want) in [(8, (0, 1)), (16, (3, 0)), (32, (3, 1)), (64, (15, 0)), (128, (15, 1))] { + assert_eq!(MemoryController::memcfg_size_rank_at(22, size_mb), Some(want), "{size_mb} MB"); + } + let half = MemoryController::encode_memcfg_half_at(22, crate::physical::LOMEM_BASE, 64).unwrap(); + let (base, _, limit) = MemoryController::memcfg_bank_info_at(22, half, 64).unwrap(); + assert_eq!(base, crate::physical::LOMEM_BASE); + assert_eq!(limit, 64 << 20); + } } diff --git a/src/mips_cache_shadow.rs b/src/mips_cache_shadow.rs new file mode 100644 index 00000000..3ed40e0b --- /dev/null +++ b/src/mips_cache_shadow.rs @@ -0,0 +1,680 @@ +//! A cache that is visible to software but stays out of the data path. +//! +//! Every load, store and instruction fetch goes straight to memory, exactly as +//! in [`PassthroughCacheOf`](crate::mips_cache_v2::PassthroughCacheOf). What +//! this adds is a *shadow*: tag and data arrays that only the CACHE +//! instruction ever touches. +//! +//! The reasoning is that a cache's effect on a functional emulator is entirely +//! observational. Its timing is invisible to the guest, and its contents are +//! invisible too as long as they agree with memory — which here they trivially +//! do, because memory is the only store. What *is* visible is the CACHE +//! instruction, the geometry reported through CP0 Config, and the diagnostics +//! a PROM runs against both. So model those, and let the host CPU — which +//! already has real caches, real speculation and real out-of-order execution — +//! get on with it unimpeded. +//! +//! Two things fall out of this beyond speed: +//! +//! - **Coherency is exact and free.** There is no stale line to miss on a +//! self-modifying store, no L1/L2 inclusion policy to get wrong, and no way +//! for a missed writeback to lose data. +//! - **Diagnostics round-trip.** The shadow stores whatever bits are written +//! to it and returns them unchanged, so a walking-1s test over tag or data +//! SRAM passes without anyone having to know the hardware's field layout. +//! Where the emulator *must* know the layout — to decide whether a line is +//! valid, say — that is a separate question from storing the bits. +//! +//! The shadow is deliberately not consulted by `read`/`write`/`fetch`. If it +//! ever needs to be, this type is the wrong shape and should say so loudly +//! rather than growing a slow path. + +use std::cell::UnsafeCell; +use std::sync::Arc; + +use crate::mips_cache_v2::{ + cache_op_name, CpuModel, FetchInstrResult, MipsCache, C_ILT, C_IST, C_R10K_CBARRIER, + C_R10K_ILD, C_R10K_ISD, CACH_PD, CACH_PI, CACH_SD, CACH_SI, +}; +use crate::mips_exec::{DecodedInstr, FLAG_NOT_DECODED}; +use crate::traits::{BusDevice, BusRead64}; +use crate::devlog::{LogModule, CACHE_LOG_HIT, CACHE_LOG_OP, devlog_is_active, devlog_mask}; + +/// Shadow tag and data arrays for one cache. +/// +/// Indexed `set * WAYS + way`. The ways are here and nowhere else: a CACHE +/// index operation addresses one *way* of one set, so software can see them, +/// and the IP28 PROM's tag diagnostic depends on it — it writes different tags +/// to the two ways of a set and reads one back. Keeping them costs an array +/// dimension in a structure no load or store ever consults. +struct Shadow { + /// One raw tag per line, stored and returned verbatim. 64 bits: an + /// R10000 secondary tag carries a 40-bit physical address. + tags: Box<[u64]>, + /// Data array in u64 slots. Empty where the model has no data shadow. + data: Box<[u64]>, + /// The check bits stored with each data slot. On an R10000 these ride + /// with the data: `Index_Store_Data` takes them from CP0 ECC and + /// `Index_Load_Data` returns them there, so they are storage like the + /// data itself, not a computed value. + ecc: Box<[u32]>, + /// The most-recently-used bit for each set. Shared across the set's ways + /// — see `MRU_BIT`. Hardware state rather than tag storage, which is why + /// it is kept here and re-applied on read instead of living in `tags`. + mru: Box<[bool]>, +} + +impl Shadow { + fn new(lines: usize, data_words: usize) -> Self { + Self { + tags: vec![0u64; lines.max(1)].into_boxed_slice(), + data: vec![0u64; data_words].into_boxed_slice(), + ecc: vec![0u32; data_words].into_boxed_slice(), + mru: vec![false; (lines.max(1) / WAYS).max(1)].into_boxed_slice(), + } + } +} + +/// The most-recently-used bit: TagHi[31], i.e. bit 63 of the assembled tag. +/// +/// It is **one bit per set, shared between the ways** — not a per-way tag bit. +/// Writing a tag with it set through either way marks the set, and reading +/// the tag of *either* way reports it. The IP28 PROM's diagnostic tests +/// exactly that: it writes the bit through way 0, writes the next set to +/// clobber the signal line, then reads way 1 back and requires the bit to be +/// there. Storing "which way is MRU" and reporting it only on that way passes +/// nothing, because the way that is read is never the way that was written. +const MRU_BIT: u64 = 1 << 63; + +/// Ways per set, as CACHE index operations address them. Bit 0 of the index +/// selects the way on an R10000; the set starts above the line offset. +const WAYS: usize = 2; + +/// Bits a secondary-cache tag retains. +/// +/// All of them, now. A 36-bit mask was inferred here from the PROM's +/// walking-0s phase, which writes an all-ones TagLo and expects back +/// 0x0000000f_ffffcdfe — but that truncation comes from the tag being +/// assembled as `(TagHi << 32) | TagLo[31:0]`, not from the array dropping +/// bits. Once the executor carried TagHi properly the mask did nothing except +/// discard the MRU bit, which the PROM writes as TagHi[31] and reads back. +const L2_TAG_MASK: u64 = u64::MAX; + +/// Is the shadow cache's operation trace on, at this level of detail? +/// +/// `log l2c mask op` traces the tag operations a diagnostic actually cares +/// about; `log l2c mask op+hit` adds the data-array walk as well. That +/// distinction is load-bearing rather than tidy: the IP28 PROM issues 65k+ +/// `Index_Store_Data` ops walking the array, and tracing all of them once +/// slowed the guest so much a run never reached the test it was there to +/// observe. +/// +/// This used to be `std::env::var_os("IRIS_SHADOW_CACHEOPS")`, uncached, six +/// times per `cache_op` — a getenv on every CACHE instruction in every build. +/// `devlog` costs two relaxed atomic loads and can be turned on, masked and +/// redirected to a file while the machine runs. +#[inline(always)] +fn l2_op_log(bit: u32) -> bool { + devlog_is_active(LogModule::L2c) && (devlog_mask(LogModule::L2c) & bit) != 0 +} + +/// A `MipsCache` whose contents are only ever observed through CACHE ops. +/// +/// Const parameters carry the geometry that software can read back, and the +/// processor identity. Nothing here affects the speed of a load. +pub struct ShadowCache< + const IC_SIZE: usize, + const IC_LINE: usize, + const DC_SIZE: usize, + const DC_LINE: usize, + const L2_SIZE: usize, + const L2_LINE: usize, + const MIPS4: bool, + const PRID: u32, + const FIR: u32, + const TLB_ENTRIES: usize, + const R10K_OPS: bool, +> { + downstream: Arc, + llbit: UnsafeCell, + lladdr: UnsafeCell, + /// Somewhere to decode into. Not a cache line — there is no caching. + fetch_scratch: UnsafeCell, + /// CP0 ECC on the way in to a store, and on the way out of a load. + ecc_in: UnsafeCell, + ecc_out: UnsafeCell, + ic: UnsafeCell, + dc: UnsafeCell, + l2: UnsafeCell, +} + +// Safety: the CPU thread is the only accessor, as for every other cache model +// in this crate. +unsafe impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > Send + for ShadowCache +{ +} +unsafe impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > Sync + for ShadowCache +{ +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > ShadowCache +{ + const IC_LINES: usize = if IC_LINE == 0 { 0 } else { IC_SIZE / IC_LINE }; + const DC_LINES: usize = if DC_LINE == 0 { 0 } else { DC_SIZE / DC_LINE }; + const L2_LINES: usize = if L2_LINE == 0 { 0 } else { L2_SIZE / L2_LINE }; + + pub fn new(downstream: Arc) -> Self { + Self { + downstream, + llbit: UnsafeCell::new(false), + lladdr: UnsafeCell::new(0), + fetch_scratch: UnsafeCell::new(DecodedInstr::default()), + ecc_in: UnsafeCell::new(0), + ecc_out: UnsafeCell::new(0), + ic: UnsafeCell::new(Shadow::new(Self::IC_LINES, 0)), + dc: UnsafeCell::new(Shadow::new(Self::DC_LINES, 0)), + // Only the secondary keeps a data shadow: it is the one a PROM + // walks with Index_Store_Data, and a 1 MB array is cheap once. + l2: UnsafeCell::new(Shadow::new(Self::L2_LINES, L2_SIZE / 8)), + } + } + + #[allow(clippy::mut_from_ref)] + fn shadow(&self, sel: u32) -> &mut Shadow { + unsafe { + match sel { + CACH_PI => &mut *self.ic.get(), + CACH_PD => &mut *self.dc.get(), + _ => &mut *self.l2.get(), + } + } + } + + /// Tag slot for a CACHE index operation. + /// + /// Bit 0 of the index selects the way; the set number starts above the + /// line offset. Observed directly: the PROM initialises the secondary + /// cache at `…1000`, `…1001`, `…1080`, `…1081`, stepping by the 128-byte + /// line with the low bit alternating. + /// + /// Folding the way bit away instead — on the reasoning that it sits below + /// line granularity — made the two ways of a set alias onto one slot, so + /// a tag written to way 1 overwrote way 0 and the PROM read back the + /// wrong one. That is the whole of the "TAG walking 1s" failure. + fn tag_slot(&self, sel: u32, virt_addr: u64) -> usize { + let (line, lines) = match sel { + CACH_PI => (IC_LINE, Self::IC_LINES), + CACH_PD => (DC_LINE, Self::DC_LINES), + _ => (L2_LINE, Self::L2_LINES), + }; + if line == 0 || lines == 0 { + return 0; + } + let way = (virt_addr as usize) & (WAYS - 1); + let set = ((virt_addr as usize) / line) % (lines / WAYS).max(1); + (set * WAYS + way) % lines + } + + /// Data slot for an R10000 `Index_Load_Data` / `Index_Store_Data`. + /// + /// Same shape: way in bit 0, the rest addressing the array. The PROM walks + /// it at `…00`, `…01`, `…10`, `…11`, `…20` — one doubleword per operation + /// with the way bit shifted in beneath it. + fn data_slot(&self, len: usize, virt_addr: u64) -> usize { + if len == 0 { + return 0; + } + let way = (virt_addr as usize) & (WAYS - 1); + let word = (virt_addr as usize) >> 4; + (word * WAYS + way) % len + } +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > From> + for ShadowCache +{ + fn from(downstream: Arc) -> Self { + Self::new(downstream) + } +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > CpuModel + for ShadowCache +{ + const MIPS4: bool = MIPS4; + const PRID: u32 = PRID; + const FIR: u32 = FIR; + const TLB_ENTRIES: usize = TLB_ENTRIES; + // The R10000 implements 44 virtual address bits where the R4x00 implements + // 40; the shadow cache is only used for it, but key this off the same flag + // that selects its cache encodings rather than asserting it unconditionally. + const VA_BITS: u32 = if R10K_OPS { 44 } else { 40 }; + const NAME: &'static str = "shadow"; + const R10K_CACHE_OPS: bool = R10K_OPS; +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > MipsCache + for ShadowCache +{ + const IC_SIZE: usize = IC_SIZE; + const IC_LINE: usize = IC_LINE; + const IC_WAYS: usize = 1; + const DC_SIZE: usize = DC_SIZE; + const DC_LINE: usize = DC_LINE; + const DC_WAYS: usize = 1; + const L2_SIZE: usize = L2_SIZE; + const L2_LINE: usize = L2_LINE; + + fn fetch(&self, _virt_addr: u64, phys_addr: u64) -> FetchInstrResult { + let r = self.downstream.read32(phys_addr as u32); + if r.is_ok() { + let slot = unsafe { &mut *self.fetch_scratch.get() }; + slot.flags = FLAG_NOT_DECODED; + slot.raw = r.data; + FetchInstrResult::hit(slot as *const DecodedInstr) + } else { + FetchInstrResult::exception(r.status) + } + } + + fn read(&self, _virt_addr: u64, phys_addr: u64) -> BusRead64 { + const { + assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") + }; + let a = phys_addr as u32; + if SIZE == 1 { + let r = self.downstream.read8(a); + BusRead64 { status: r.status, data: r.data as u64 } + } else if SIZE == 2 { + let r = self.downstream.read16(a); + BusRead64 { status: r.status, data: r.data as u64 } + } else if SIZE == 4 { + let r = self.downstream.read32(a); + BusRead64 { status: r.status, data: r.data as u64 } + } else { + self.downstream.read64(a) + } + } + + fn write(&self, _virt_addr: u64, phys_addr: u64, val: u64) -> u32 { + const { + assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") + }; + let a = phys_addr as u32; + if SIZE == 1 { + self.downstream.write8(a, val as u8) + } else if SIZE == 2 { + self.downstream.write16(a, val as u16) + } else if SIZE == 4 { + self.downstream.write32(a, val as u32) + } else { + self.downstream.write64(a, val) + } + } + + fn write64_masked(&self, _virt_addr: u64, phys_addr: u64, val: u64, mask: u64) -> u32 { + let aligned = (phys_addr & !7) as u32; + let r = self.downstream.read64(aligned); + if !r.is_ok() { + return r.status; + } + self.downstream.write64(aligned, (r.data & !mask) | (val & mask)) + } + + /// The whole point of the type. + /// + /// Every operation that only *moves data between cache and memory* — + /// invalidate, writeback, fill — is a genuine no-op here, because the + /// cache and memory can never disagree. The operations that move data + /// between the cache and a register are the ones with observable effects, + /// and those are served from the shadow. + fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u64 { + let sel = cache_op & 3; + let op = cache_op & 0x1C; + + // Tag operations under `op`, the whole data-array walk only under + // `hit` as well — see `l2_op_log`. + if l2_op_log(if matches!(op, C_IST | C_ILT) { CACHE_LOG_OP } else { CACHE_LOG_HIT }) { + crate::dlog!(LogModule::L2c, + "shadow: {:<22} raw={cache_op:#04x} va={virt_addr:#018x} arg={phys_addr:#018x}", + cache_op_name(cache_op)); + } + + match op { + // Index_Store_Tag / Index_Load_Tag. Stored and returned verbatim: + // a tag test is a round trip, and round trips do not require + // knowing what the bits mean. + C_IST => { + let idx = self.tag_slot(sel, virt_addr); + let mask = if matches!(sel, CACH_SI | CACH_SD) { L2_TAG_MASK } else { u64::MAX }; + if l2_op_log(CACHE_LOG_OP) { + crate::dlog!(LogModule::L2c, "shadow: -> IST va={virt_addr:#012x} idx={idx} stores {:#018x}", + phys_addr & mask & !MRU_BIT); + } + let s = self.shadow(sel); + if idx < s.tags.len() { + // The MRU bit belongs to the set, not to this way, so it + // is lifted out and kept separately. Everything else is + // stored verbatim: bit 32 in particular is *not* spare — + // TagHi[3:0] are tag address bits 35:32, and clearing it + // here destroyed real tag bits. + let set = (idx / WAYS).min(s.mru.len() - 1); + s.mru[set] = phys_addr & MRU_BIT != 0; + s.tags[idx] = phys_addr & mask & !MRU_BIT; + } + 0 + } + C_ILT => { + let idx = self.tag_slot(sel, virt_addr); + let s = self.shadow(sel); + let set = (idx / WAYS).min(s.mru.len() - 1); + let mut v = if idx < s.tags.len() { s.tags[idx] } else { 0 }; + // The bit belongs to the set: whichever way is read reports + // it. + if s.mru[set] { + v |= MRU_BIT; + } + if l2_op_log(CACHE_LOG_OP) { + crate::dlog!(LogModule::L2c, "shadow: -> ILT va={virt_addr:#012x} idx={idx} set={set} mru={} tag={:#018x}", + s.mru[set], if idx < s.tags.len() { s.tags[idx] } else { 0 }); + crate::dlog!(LogModule::L2c, "shadow: -> ILT slot={idx} returns {v:#018x}"); + } + v + } + + // R10000 reassigns 5/6/7, scoped to particular cache selects. + C_R10K_CBARRIER if R10K_OPS && sel == CACH_PI => 0, + C_R10K_ILD if R10K_OPS && matches!(sel, CACH_PI | CACH_PD | CACH_SD) => { + let s = self.shadow(sel); + let slot = self.data_slot(s.data.len(), virt_addr); + if s.data.is_empty() { + 0 + } else { + unsafe { *self.ecc_out.get() = s.ecc[slot] }; + if l2_op_log(CACHE_LOG_HIT) { + crate::dlog!(LogModule::L2c, "shadow: -> ILD slot={slot} data={:#018x} ecc={:#x}", + s.data[slot], s.ecc[slot]); + } + s.data[slot] + } + } + C_R10K_ISD if R10K_OPS && matches!(sel, CACH_SI | CACH_SD) => { + let s = self.shadow(sel); + let slot = self.data_slot(s.data.len(), virt_addr); + if !s.data.is_empty() { + s.data[slot] = phys_addr; + s.ecc[slot] = unsafe { *self.ecc_in.get() }; + if l2_op_log(CACHE_LOG_HIT) { + crate::dlog!(LogModule::L2c, "shadow: -> ISD slot={slot} data={phys_addr:#018x} ecc={:#x}", + s.ecc[slot]); + } + } + 0 + } + + // Invalidate, writeback, fill, and the R4000 hit operations. All + // no-ops: there is nothing held that could be stale or dirty. + _ => 0, + } + } + + fn get_config(&self, cache_target: u32) -> (usize, usize) { + match cache_target { + CACH_PI => (IC_SIZE, IC_LINE), + CACH_PD => (DC_SIZE, DC_LINE), + _ => (L2_SIZE, L2_LINE), + } + } + + /// The check bits ride with cache data on an R10000: a store takes them + /// from CP0 ECC and a load returns them there. + fn set_cache_ecc(&self, v: u32) { + unsafe { *self.ecc_in.get() = v }; + } + + fn cache_op_ecc(&self) -> u32 { + unsafe { *self.ecc_out.get() } + } + + fn downstream(&self) -> Arc { + self.downstream.clone() + } + + fn check_and_clear_llbit(&self, _phys_addr: u64) { + unsafe { *self.llbit.get() = false }; + } + fn get_llbit(&self) -> bool { + unsafe { *self.llbit.get() } + } + fn set_llbit(&self, val: bool) { + unsafe { *self.llbit.get() = val }; + } + fn get_lladdr(&self) -> u32 { + unsafe { *self.lladdr.get() } + } + fn set_lladdr(&self, addr: u32) { + unsafe { *self.lladdr.get() = addr }; + } +} + +/// SGI Indigo2 IMPACT R10000 (IP28). +/// +/// Real geometry as software reads it — 32 KB primaries with 64-byte +/// instruction lines and 32-byte data lines, a 1 MB secondary with 128-byte +/// lines — 64 TLB entries, MIPS IV, and the R10000 cache operation encodings. +/// Nothing of the microarchitecture: no ways, no LRU, no out-of-order. +pub type R10000ShadowCache = + ShadowCache<32768, 64, 32768, 32, 1048576, 128, true, 0x0000_0900, 0x0000_0900, 64, true>; + +#[cfg(test)] +mod tests { + use super::*; + use crate::mem::Memory; + use crate::mips_cache_v2::{CACH_PD, CACH_SD}; + + fn cache() -> R10000ShadowCache { + let mem: Arc = Arc::new(Memory::new(1024 * 1024)); + R10000ShadowCache::from(mem) + } + + /// A tag test is a round trip, so the shadow must return exactly the bits + /// it was given — including any the real layout would not use. The cache + /// model that preceded this one decoded TagLo into ptag/state/pidx fields + /// and re-encoded on the way out, which silently dropped the low seven + /// bits and every state code it did not recognise. A walking-1s test then + /// failed on its very first bit. + #[test] + fn tags_round_trip_every_bit() { + let c = cache(); + for bit in 0..64 { + // 32 and 63 are MRU control/state, not tag storage. + if bit == 32 || bit == 63 { continue; } + let v = 1u64 << bit; + c.cache_op(C_IST | CACH_SD, 0, v); + assert_eq!( + c.cache_op(C_ILT | CACH_SD, 0, 0), + v, + "bit {bit} did not survive a store/load tag round trip" + ); + } + } + + /// Index_Store_Data / Index_Load_Data likewise. + #[test] + fn secondary_data_round_trips() { + let c = cache(); + c.cache_op(C_R10K_ISD | CACH_SD, 0x40, 0xdeadbeef); + assert_eq!(c.cache_op(C_R10K_ILD | CACH_SD, 0x40, 0), 0xdeadbeef); + } + + /// The MRU bit is **shared between the ways of a set**: written through + /// one way, it must read back through the other. The IP28 PROM's + /// diagnostic does exactly this and nothing else satisfies it — a model + /// that records which way was marked reports nothing on the way that is + /// actually read. + #[test] + fn the_mru_bit_is_shared_between_the_ways_of_a_set() { + let c = cache(); + // Way 0 of set 0, then way 1 of set 0. + c.cache_op(C_IST | CACH_SD, 0, MRU_BIT); + assert_ne!( + c.cache_op(C_ILT | CACH_SD, 1, 0) & MRU_BIT, + 0, + "written through way 0, must be visible through way 1" + ); + // And the other direction. + c.cache_op(C_IST | CACH_SD, 1, MRU_BIT); + assert_ne!(c.cache_op(C_ILT | CACH_SD, 0, 0) & MRU_BIT, 0); + } + + /// Writing a tag without the bit clears it, again for the whole set. + #[test] + fn writing_a_tag_without_the_mru_bit_clears_it() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0, MRU_BIT); + c.cache_op(C_IST | CACH_SD, 0, 0); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 1, 0) & MRU_BIT, 0); + } + + /// Marking one set must not mark its neighbour. The PROM writes the + /// following set specifically to clobber the signal line between the two + /// checks, so a model where the bit leaks across sets passes the first + /// check and fails the next. + #[test] + fn the_mru_bit_does_not_leak_between_sets() { + let c = cache(); + let next_set = 128; // one line + c.cache_op(C_IST | CACH_SD, 0, MRU_BIT); + c.cache_op(C_IST | CACH_SD, next_set, 0); + assert_ne!(c.cache_op(C_ILT | CACH_SD, 1, 0) & MRU_BIT, 0, "set 0 keeps its bit"); + assert_eq!(c.cache_op(C_ILT | CACH_SD, next_set, 0) & MRU_BIT, 0, "set 1 has none"); + } + + /// Check bits are stored with the data and returned on load. The PROM's + /// ECC walk writes a different value through each way and reads them + /// back, so both the round trip and the separation matter. + /// + /// This exists because the methods carrying ECC were once declared on the + /// trait with defaults and simply never implemented here: the data + /// round-tripped perfectly and every check bit read back as zero. + #[test] + fn check_bits_travel_with_the_data() { + let c = cache(); + c.set_cache_ecc(0x001); + c.cache_op(C_R10K_ISD | CACH_SD, 0, 0x1111_2222_3333_4444); + c.set_cache_ecc(0x3fe); + c.cache_op(C_R10K_ISD | CACH_SD, 1, 0xaaaa_bbbb_cccc_dddd); + + c.set_cache_ecc(0); + assert_eq!(c.cache_op(C_R10K_ILD | CACH_SD, 0, 0), 0x1111_2222_3333_4444); + assert_eq!(c.cache_op_ecc(), 0x001, "way 0's check bits"); + assert_eq!(c.cache_op(C_R10K_ILD | CACH_SD, 1, 0), 0xaaaa_bbbb_cccc_dddd); + assert_eq!(c.cache_op_ecc(), 0x3fe, "way 1's check bits"); + } + + /// An untouched set reports no MRU, so an ordinary tag read is not + /// contaminated by it. + #[test] + fn an_untouched_set_has_no_mru_bit() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0, 0xdead_beef); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0, 0), 0xdead_beef); + } + + /// Separate lines must not alias onto one another. + #[test] + fn distinct_lines_hold_distinct_tags() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0, 0x1111_1111); + c.cache_op(C_IST | CACH_SD, 128, 0x2222_2222); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0, 0), 0x1111_1111); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 128, 0), 0x2222_2222); + } + + /// Bit 0 of a CACHE index selects the **way**, and the two ways of a set + /// must not alias. + /// + /// This is exactly the IP28 PROM's secondary-cache tag test: it stores one + /// tag to way 0 and a different one to way 1 of the same set, then reads + /// way 0 back. Folding the way bit away let the second store clobber the + /// first, and the PROM reported + /// `Expected: 0x0000000000000001 ... TAG walking 1s`. + #[test] + fn the_two_ways_of_a_set_are_independent() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0x2000_0000, 0x0000_0001); + c.cache_op(C_IST | CACH_SD, 0x2000_0001, 0xffff_cdfe); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0x2000_0000, 0), 0x0000_0001, + "way 1's tag overwrote way 0's"); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0x2000_0001, 0), 0xffff_cdfe); + } + + /// Adjacent sets stay distinct once the way bit is accounted for. + #[test] + fn adjacent_sets_do_not_alias_through_the_way_bit() { + let c = cache(); + for set in 0..4u64 { + for way in 0..2u64 { + let va = set * 128 + way; + c.cache_op(C_IST | CACH_SD, va, 0x1000 + set * 16 + way); + } + } + for set in 0..4u64 { + for way in 0..2u64 { + let va = set * 128 + way; + assert_eq!(c.cache_op(C_ILT | CACH_SD, va, 0), + 0x1000 + set * 16 + way, + "set {set} way {way} aliased"); + } + } + } + + /// Nothing is ever held, so a load must see the store that preceded it + /// with no flush in between. This is the property that makes the design + /// safe, not merely fast. + #[test] + fn memory_is_the_only_store() { + let c = cache(); + c.write::<4>(0, 0x2000, 0x1234_5678); + assert_eq!(c.read::<4>(0, 0x2000).data, 0x1234_5678); + // An invalidate cannot lose it, and a writeback cannot be needed. + c.cache_op(crate::mips_cache_v2::C_IINV | CACH_PD, 0x2000, 0); + assert_eq!(c.read::<4>(0, 0x2000).data, 0x1234_5678); + } + + /// Geometry is what CP0 Config is built from, so it must be the real + /// part's even though the model holds nothing. + #[test] + fn reported_geometry_is_the_real_parts() { + let c = cache(); + assert_eq!(c.get_config(CACH_PI), (32768, 64)); + assert_eq!(c.get_config(CACH_PD), (32768, 32)); + assert_eq!(c.get_config(CACH_SD), (1048576, 128)); + } +} diff --git a/src/mips_cache_v2.rs b/src/mips_cache_v2.rs index 89e98ebf..560beed8 100644 --- a/src/mips_cache_v2.rs +++ b/src/mips_cache_v2.rs @@ -63,6 +63,16 @@ pub use crate::mips_isa::{ C_HINV, C_HWBINV, C_FILL, C_HWB, C_HSV, }; +// R10000 reassigns cache operations 5, 6 and 7. Where an R4000 has +// Hit_Invalidate, Hit_Writeback_Invalidate and Hit_Writeback, an R10000 has a +// cache barrier and index-addressed load/store of the cache *data* array. +// Confirmed against NetBSD's mips/include/cache_r10k.h, and against the IP28 +// PROM, whose secondary-cache SRAM test issues C_R10K_ISD(SD) — which this +// emulator was executing as a hit-writeback, so nothing was ever stored. +pub const C_R10K_CBARRIER: u32 = 5 << 2; +pub const C_R10K_ILD: u32 = 6 << 2; +pub const C_R10K_ISD: u32 = 7 << 2; + /// Decode a raw cache_op field (5-bit: op[4:2] | target[1:0]) to a human-readable name. /// Matches the disassembler mnemonic convention used by gas/objdump. pub fn cache_op_name(op: u32) -> &'static str { @@ -431,6 +441,11 @@ pub trait CpuModel: MipsCache { const VA_BITS: u32 = 40; /// Name as the guest and the benchmark report see it. const NAME: &'static str; + /// Cache ops 5/6/7 carry their R10000 meanings rather than their R4000 ones. + /// The executor needs this: `Index_Store_Data` takes its value from TagLo + /// and `Index_Load_Data` returns into it, neither of which is true of the + /// R4000 hit operations that share those encodings. + const R10K_CACHE_OPS: bool = false; } pub trait MipsCache: Send + Sync { @@ -633,7 +648,19 @@ pub trait MipsCache: Send + Sync { /// /// For Index_Load_Tag operations, returns the tag value in TagLo CP0 register format /// For other operations, returns 0 - fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u32; + /// Perform a CACHE operation. For `Index_Load_Tag` the return is the + /// tag as software sees it, full width — an R10000 secondary tag does not + /// fit in 32 bits. + fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u64; + + /// CP0 `ECC` ($26) on the way in to a cache-data store. + /// + /// On an R10000 the check bits ride with the data: `Index_Store_Data` + /// takes them from this register and `Index_Load_Data` returns them + /// there. Models that do not keep them ignore both of these. + fn set_cache_ecc(&self, _v: u32) {} + /// CP0 `ECC` after the last cache operation. + fn cache_op_ecc(&self) -> u32 { 0 } /// Write back dirty L1-D (and, if present, L2) lines covering /// `[phys_addr, phys_addr + size)` to memory, without invalidating them. @@ -800,7 +827,7 @@ impl MipsCache for PassthroughCacheOf { self.downstream.write64(aligned_addr, new_val) } - fn cache_op(&self, _cache_op: u32, _virt_addr: u64, _phys_addr: u64) -> u32 { + fn cache_op(&self, _cache_op: u32, _virt_addr: u64, _phys_addr: u64) -> u64 { // No-op for passthrough cache - just return 0 0 } @@ -1064,6 +1091,7 @@ pub struct CpuCache< const L2_CACHE_SIZE: usize, const L2_LINE: usize, const L2_TAGS: usize, const L2_DATA: usize, const L2_NINSTRS: usize, const HAS_L2: bool, const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, + const MODEL: u8, > { downstream: Arc, @@ -1141,12 +1169,12 @@ unsafe impl Send for CpuCache {} + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Send for CpuCache {} unsafe impl Sync for CpuCache {} + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Sync for CpuCache {} // Per-level cache types, parameterised so each CPU model monomorphises its own. type ICacheT = @@ -1156,24 +1184,91 @@ type DCacheT = Cache; +/// Which processor a monomorphisation models, as the `MODEL` const parameter. +/// +/// This exists because the thing that distinguishes these parts *in this file* +/// is not any one of the shape parameters. It was originally inferred from +/// `IC_WAYS == 2`, which worked only while "2-way" and "R5000" named the same +/// processor. They do not: the R10000 is also 2-way, and it shares neither the +/// R5000's TagLo layout nor its cache-op semantics. Inferring identity from +/// shape would have quietly given an R10000 every R5000 behaviour in the 40-odd +/// places that branch on it, each one individually plausible. +pub mod model { + pub const R4400: u8 = 0; + pub const R5000: u8 = 1; + pub const R10000: u8 = 2; +} + /// SGI Indy R4400: direct-mapped 16K L1s, 1 MB unified L2 owning the decode slots. pub type R4400Cache = CpuCache<16384, 16, 1, 1024, 16384, 16, 1, 1024, 2048, 1048576, 128, 8192, 131072, 262144, true, - false, 0x0000_0440, 0x0000_0500, 48>; + false, 0x0000_0440, 0x0000_0500, 48, { model::R4400 }>; /// SGI Indy R5000: 2-way 32K L1s, no secondary cache; L1I owns its decode slots. pub type R5000Cache = CpuCache<32768, 32, 2, 1024, 32768, 32, 2, 1024, 4096, 128, 128, 1, 16, 0, false, - true, 0x0000_2321, 0x0000_2300, 48>; + true, 0x0000_2321, 0x0000_2300, 48, { model::R5000 }>; +/// SGI Indigo2 IMPACT R10000 (IP28), modelled for speed rather than fidelity. +/// +/// The real part has two-way 32 KB L1s and a two-way secondary cache. This +/// models all three **direct-mapped**, keeping the real total sizes and line +/// sizes (64-byte L1I lines, 32-byte L1D, 128-byte L2). +/// +/// That is deliberate. Associativity reaches software only through the +/// way-select bits of `CACHE Index_*`, and the only thing an operating system +/// does by index is flush the whole cache — which comes out the same for any +/// geometry holding the same lines. Emulating ways costs a victim-selection +/// and LRU update on every access and buys nothing IRIX can observe. +/// +/// It also buys correctness here, not just speed. A two-way L1 *with* an L2 is +/// a combination this file has never had: `fetch()` selects the two-way tag +/// probe and the L1I-resident decode slots under one condition, and such a +/// part needs the first with the second's alternative. Direct-mapped plus L2 +/// is exactly the R4400's shape, so every associativity and decode-slot branch +/// already does the right thing for this model, unchanged. +/// +/// What is *not* faked is anything software reads back: the PRId, the TLB +/// size, MIPS IV decoding, and the cache tag layout the PROM's diagnostics +/// inspect directly. +#[cfg(feature = "ip28")] +pub type R10000Cache = CpuCache<32768, 64, 1, 512, + 32768, 32, 1, 1024, 4096, + 1048576, 128, 8192, 131072, 262144, true, + true, 0x0000_0900, 0x0000_0900, 64, { model::R10000 }>; impl CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuCache { // Model discriminator: folds to a literal, so it replaces #[cfg(feature = "r5k")]. - const IS_R5K: bool = IC_WAYS == 2; + // Keyed on MODEL, not on associativity — see `mod model`. + const IS_R5K: bool = MODEL == model::R5000; + + /// IP28 bring-up watch: physical window to trace, from IRIS_IP28_WATCH. + /// Folds to a constant `None` on every model but the R10000, so the + /// non-IP28 hot path keeps no trace of this. + #[inline(always)] + fn ip28_watch() -> Option { + if !Self::R10K_CACHE_OPS { return None; } + static W: std::sync::OnceLock> = std::sync::OnceLock::new(); + *W.get_or_init(|| { + std::env::var("IRIS_IP28_WATCH").ok().and_then(|v| { + u64::from_str_radix(v.trim_start_matches("0x"), 16).ok() + }) + }) + } + + #[inline(always)] + fn ip28_trace(&self, what: &str, addr: u64, val: u64) { + if let Some(w) = Self::ip28_watch() { + if (addr & !0x7f) == (w & !0x7f) { + eprintln!("ip28watch: {what:<18} addr={addr:#018x} val={val:#018x}"); + } + } + } + // Logical L2 size; 0 means the model has no secondary cache. pub const L2_SIZE: usize = if HAS_L2 { L2_CACHE_SIZE } else { 0 }; @@ -1298,7 +1393,7 @@ impl From> for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> From> for CpuCache { fn from(downstream: Arc) -> Self { Self::new(downstream) } @@ -1308,7 +1403,7 @@ impl CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuCache { /// Check if we're tracking this physical address (for debug purposes) #[cfg(feature = "debug_cache")] #[inline] @@ -2811,19 +2906,24 @@ impl CpuModel for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuModel for CpuCache { const MIPS4: bool = MIPS4; const PRID: u32 = PRID; const FIR: u32 = FIR; const TLB_ENTRIES: usize = TLB_ENTRIES; - const NAME: &'static str = if IC_WAYS == 2 { "R5000" } else { "R4400" }; + const R10K_CACHE_OPS: bool = MODEL == model::R10000; + const NAME: &'static str = match MODEL { + model::R5000 => "R5000", + model::R10000 => "R10000", + _ => "R4400", + }; } impl MipsCache for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> MipsCache for CpuCache { fn set_l1i_counters(&mut self, hit: Arc, fetch: Arc) { self.l1i_hit_count = hit; self.l1i_fetch_count = fetch; @@ -3124,6 +3224,7 @@ impl(&self, virt_addr: u64, phys_addr: u64) -> BusRead64 { + self.ip28_trace("read", phys_addr, 0); const { assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") }; #[cfg(feature = "debug_cache")] { @@ -3194,6 +3295,7 @@ impl(&self, virt_addr: u64, phys_addr: u64, val: u64) -> u32 { + self.ip28_trace("write", phys_addr, val); const { assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") }; #[cfg(feature = "debug_cache")] { @@ -3320,7 +3422,55 @@ impl u32 { + fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u64 { + if Self::R10K_CACHE_OPS && std::env::var_os("IRIS_IP28_CACHEOPS").is_some() { + use std::sync::atomic::{AtomicU32, Ordering}; + static SEEN: AtomicU32 = AtomicU32::new(0); + let bit = 1u32 << (cache_op & 0x1F); + let all = matches!(std::env::var("IRIS_IP28_CACHEOPS").as_deref(), Ok("all")); + if all || SEEN.fetch_or(bit, Ordering::Relaxed) & bit == 0 { + eprintln!("ip28: op {} raw={:#04x} va={:#018x} arg={:#018x}", + cache_op_name(cache_op), cache_op, virt_addr, phys_addr); + } + } + if Self::ip28_watch().is_some() { + self.ip28_trace(cache_op_name(cache_op), virt_addr, phys_addr); + } + // R10000 ops 5/6/7 are not the R4000 hit operations that share these + // encodings — see C_R10K_ISD. Handled before the shared decode below. + if Self::R10K_CACHE_OPS { + // The R10000 meanings are scoped to particular cache selects; + // outside those the R4000 operation on the same encoding still + // applies. cache_r10k.h annotates each one, and the IP28 PROM uses + // both readings of op 5: `Cache_Barrier` against the instruction + // cache, and R4000 `Hit_Invalidate` against the secondary. + let sel = cache_op & 3; + match cache_op & 0x1C { + // An ordering barrier. Nothing to do in a model with no + // speculative memory pipeline to hold back. + C_R10K_CBARRIER if sel == CACH_PI => return 0, + C_R10K_ILD if matches!(sel, CACH_PI | CACH_PD | CACH_SD) => { + let is_l2 = sel == CACH_SD; + // The index is a byte offset into the data array. Bit 0 + // selects the way on real silicon; this model is + // direct-mapped, and bit 0 falls below the u64 slot index, + // so it drops out without any special case. + let slot = (virt_addr as usize) >> 3; + if is_l2 && HAS_L2 { + return self.l2.data()[slot & (L2_DATA - 1)]; + } + return self.dc.data()[slot & (DC_DATA - 1)]; + } + C_R10K_ISD if matches!(sel, CACH_SI | CACH_SD) => { + let slot = (virt_addr as usize) >> 3; + if HAS_L2 { + self.l2.data_mut()[slot & (L2_DATA - 1)] = phys_addr; + } + return 0; + } + _ => {} + } + } // Decode cache operation let cache_target = cache_op & 0x3; // bits [17:16] let operation = cache_op & 0x1C; // bits [20:18] (shifted by 2) @@ -3410,6 +3560,10 @@ impl 0, L2_CS_CLEAN_EXCLUSIVE => 4, @@ -3418,7 +3572,7 @@ impl 7, _ => 0, }; - (tag.ptag() << 13) | (state << 10) | (tag.pidx() << 7) + ((tag.ptag() << 13) | (state << 10) | (tag.pidx() << 7)) as u64 } else if is_icache { let tag: L1ITag = self.ic.get_tag(idx); let raw_ptag = (tag.ptag >> L1_PTAG_SHIFT) as u32 & L1_PTAG_MASK; @@ -3426,11 +3580,11 @@ impl 3u32, _ => 0u32, }; - (raw_ptag << 8) | (pstate << 6) + ((raw_ptag << 8) | (pstate << 6)) as u64 } } } // Index Store Tag — write CP0 TagLo into internal tag C_IST => { - let tag_lo = phys_addr as u32; + let tag_lo = phys_addr as u32; // R4000-family tags are 32-bit if is_l2 { + if MODEL == model::R10000 { + eprintln!("ip28: C_IST(SD) idx={idx:#x} taglo={tag_lo:#010x} phys={phys_addr:#018x}"); + } // L2 TagLo format: [31:13] ptag [12:10] state [9:7] PIdx let ptag = (tag_lo >> 13) & L2_PTAG_MASK; let state = (tag_lo >> 10) & 0x7; @@ -4027,7 +4184,7 @@ impl Drop for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Drop for CpuCache { fn drop(&mut self) { self.ic.stop.store(true, Ordering::Relaxed); } @@ -4039,7 +4196,7 @@ impl Resettable for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Resettable for CpuCache { fn power_on(&self) { self.ic.tags_mut().fill(L1ITag::default()); self.dc.tags_mut().fill(L1DTag::default()); @@ -4070,7 +4227,7 @@ impl CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuCache { fn save_tags_as_u32>(tags: &[TAG]) -> Vec { tags.iter().map(|&t| t.into()).collect() } @@ -4219,6 +4376,32 @@ mod tests { R4400Cache::new(mem as Arc) } + /// Deliberately the R5000's exact shape, differing *only* in `MODEL`. If + /// model identity is ever inferred from a shape parameter again, this + /// aliases onto the R5000 and the test below fails. + type NotAnR5000 = CpuCache<32768, 32, 2, 1024, + 32768, 32, 2, 1024, 4096, + 128, 128, 1, 16, 0, false, + true, 0x0000_0900, 0x0000_0900, 64, { model::R10000 }>; + + /// Two-way associativity is not an identity. + /// + /// `IS_R5K` used to be `IC_WAYS == 2`, which was true of every 2-way part + /// the file knew about. The R10000 is also 2-way and shares neither the + /// R5000's TagLo layout nor its cache-op semantics, so that inference would + /// have handed it R5000 behaviour at all 40-odd sites that branch on it — + /// silently, and each one plausible on its own. + #[test] + fn model_identity_is_explicit_and_not_inferred_from_associativity() { + assert_eq!(::NAME, "R4400"); + assert_eq!(::NAME, "R5000"); + assert_eq!(::NAME, "R10000"); + + assert!(R5000Cache::IS_R5K, "the R5000 is the R5000"); + assert!(!NotAnR5000::IS_R5K, "a 2-way cache is not what makes an R5000"); + assert!(!R4400Cache::IS_R5K); + } + // Same helper for whichever CPU model a test wants to exercise. fn make_cache_of>>(mem: Arc) -> C { C::from(mem as Arc) @@ -4301,7 +4484,7 @@ mod tests { cache.cache_op(C_IST | CACH_PD, va, written as u64); let read_back = cache.cache_op(C_ILT | CACH_PD, va, idx); assert_eq!((read_back >> 6) & 0x3, 3, "R4400 DirtyExclusive must round-trip"); - assert_eq!(read_back >> 8, raw_ptag, "physical tag must round-trip"); + assert_eq!(read_back >> 8, raw_ptag as u64, "physical tag must round-trip"); } // R5000: valid+dirty is D=1,V=1 (0xC0). Under the old code this was @@ -4320,9 +4503,9 @@ mod tests { let read_back = cache.cache_op(C_ILT | CACH_PD, va, idx); assert_ne!(read_back & (1 << 6), 0, "R5000 {label} line must read back V=1"); - assert_eq!(read_back & (1 << 7), d_bit, + assert_eq!(read_back & (1 << 7), d_bit as u64, "R5000 {label} line must round-trip its D bit"); - assert_eq!(read_back >> 8, raw_ptag, "physical tag must round-trip"); + assert_eq!(read_back >> 8, raw_ptag as u64, "physical tag must round-trip"); } } } diff --git a/src/mips_core.rs b/src/mips_core.rs index dd0312e8..04e27715 100644 --- a/src/mips_core.rs +++ b/src/mips_core.rs @@ -28,6 +28,38 @@ pub const STATUS_CU1: u32 = 1 << 29; // Coprocessor 1 (FPU) Usable pub const STATUS_CU2: u32 = 1 << 30; // Coprocessor 2 Usable pub const STATUS_CU3: u32 = 1 << 31; // Coprocessor 3 Usable +/// Is CP0 register tracing on? `log mips mask cp0` on the monitor. +/// +/// Traces everything the cache-error machinery touches — CP0 ECC (26) and +/// CacheErr (27) accesses, Status.DE transitions, and CACHE ops. +/// +/// Built to explain why SGI's own IDE field diagnostic reports "Failure +/// detected on the CPU module" on an otherwise healthy emulated R10000. What +/// it found: the IDE ends in a bounded 3712-iteration loop reading CacheErr +/// and ECC, both of which return zero every pass, because nothing in this +/// emulator ever writes CacheErr — we do not detect or log a cache error at +/// all. Note it does this with Status.DE *set*: DE suppresses only the trap, +/// while real hardware still records the error, which is what a poll-based +/// parity test relies on. (An earlier guess that the IDE wanted a Cache Error +/// *exception* was refuted by this tracer — DE is never cleared outside the +/// PROM's own memory sizing.) +/// +/// Two relaxed atomic loads when off, and unlike the `IRIS_IP28_CACHEDIAG` +/// environment variable this replaced it can be turned on, masked and +/// redirected to a file part-way through a boot. +#[inline(always)] +pub fn cachediag_on() -> bool { + cp0_log() +} + +/// `log mips mask cp0` — CP0 register traffic. +#[inline(always)] +pub fn cp0_log() -> bool { + crate::devlog::devlog_is_active(crate::devlog::LogModule::Mips) + && (crate::devlog::devlog_mask(crate::devlog::LogModule::Mips) + & crate::mips_exec::MIPS_LOG_CP0) != 0 +} + // CP0 Cause Register bit definitions pub const CAUSE_EXCCODE_MASK: u32 = 0x1F << 2; // Exception Code mask pub const CAUSE_EXCCODE_SHIFT: u32 = 2; // Exception Code shift @@ -816,7 +848,14 @@ pub struct MipsCore { pub cp0_xcontext: u64, // 20: Extended Context (64-bit) pub cp0_ecc: u32, // 26: ECC Register pub cp0_cacheerr: u32, // 27: Cache Error - pub cp0_taglo: u32, // 28: Cache Tag Low + /// 28: Cache Tag Low. 64 bits, not 32. + /// + /// An R4000's TagLo fits in 32, but an R10000's secondary cache tag does + /// not — it carries a 40-bit physical address, and the IP28 PROM's tag + /// diagnostic writes and expects back values like 0x0000000f_ffffcdfe. + /// Truncating to 32 lost the top nibble and the PROM reported the + /// difference. + pub cp0_taglo: u64, // 28: Cache Tag Low pub cp0_taghi: u32, // 29: Cache Tag High pub cp0_errorepc: u64, // 30: Error Exception PC @@ -1466,6 +1505,24 @@ impl MipsCore { /// Write a GPR by index. Unconditionally re-zeros gpr[0] to avoid a branch. #[inline(always)] pub fn write_gpr(&mut self, reg: u32, value: u64) { + // IP28 bring-up: `IRIS_IP28_WATCHGPR=` reports every write to that + // GPR with the PC that did it. One cached bool and a compare against a + // register number already in hand; unarmed it is a predictable branch. + #[cfg(debug_assertions)] + let _ = (); + { + static WATCH: std::sync::OnceLock> = std::sync::OnceLock::new(); + let watch = *WATCH.get_or_init(|| { + std::env::var("IRIS_IP28_WATCHGPR").ok().and_then(|v| v.trim().parse().ok()) + }); + if watch == Some(reg) { + // The register number is a parameter, so it stays an + // environment variable; only the output moves, so it can be + // redirected with `log mips file `. + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28gpr: ${reg} = {value:#018x} at pc={:#018x}", self.pc); + } + } unsafe { *self.gpr.get_unchecked_mut(reg as usize) = value; } self.gpr[0] = 0; } @@ -1545,9 +1602,21 @@ impl MipsCore { eprintln!("[ip7] MFC0 PerfCnt (reg 25) read -> 0"); 0 } - 26 => self.cp0_ecc as u64, - 27 => self.cp0_cacheerr as u64, - 28 => self.cp0_taglo as u64, + 26 => { + if cachediag_on() { + eprintln!("ip28cd: MFC0 ECC -> {:#010x} pc={:#018x}", + self.cp0_ecc, self.pc); + } + self.cp0_ecc as u64 + } + 27 => { + if cachediag_on() { + eprintln!("ip28cd: MFC0 CacheErr -> {:#010x} pc={:#018x}", + self.cp0_cacheerr, self.pc); + } + self.cp0_cacheerr as u64 + } + 28 => self.cp0_taglo, 29 => self.cp0_taghi as u64, 30 => self.cp0_errorepc, _ => 0, // Unimplemented registers read as 0 @@ -1579,7 +1648,7 @@ impl MipsCore { 20 => self.cp0_xcontext, 26 => self.cp0_ecc as u64, 27 => self.cp0_cacheerr as u64, - 28 => self.cp0_taglo as u64, + 28 => self.cp0_taglo, 29 => self.cp0_taghi as u64, 30 => self.cp0_errorepc, _ => 0, @@ -1913,6 +1982,16 @@ impl MipsCore { self.reanchor_count_and_reschedule(); } 10 => { // always use 64bit mask because the entries need to be valid in 64 bit mode even when they were set from 32 bit mode + // `log mips mask cp0` shows what the guest wrote against + // what survives the mask. The mask is R4400's 40-bit virtual + // address; the R10000 implements 44. + if cp0_log() && (value & !0xC000_00FF_FFFF_E0FF) != 0 { + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28ehi: wrote {value:#018x} -> kept {:#018x} (lost {:#018x}) pc={:#018x}", + value & 0xC000_00FF_FFFF_E0FF, + value & !0xC000_00FF_FFFF_E0FF, + self.pc); + } self.cp0_entryhi = value & 0xC000_00FF_FFFF_E0FF; }, 11 => { @@ -2036,6 +2115,18 @@ impl MipsCore { 12 => { let old = self.cp0_status; self.cp0_status = value as u32; + // Status.DE gates cache error exceptions. A diagnostic that + // means to provoke one has to clear it first, so a DE + // transition is the guest announcing its intent. + if cachediag_on() && (old ^ self.cp0_status) & STATUS_DE != 0 { + eprintln!("ip28cd: Status.DE {} pc={:#018x}", + if self.cp0_status & STATUS_DE != 0 { + "SET (cache exceptions DISABLED)" + } else { + "CLEAR (cache exceptions ENABLED)" + }, + self.pc); + } // Trace every change to the IP7 mask bit (Status.IM7). Linux's // mips_cpu_irq_controller masks IM7 on interrupt entry // (irq_ack) and unmasks on EOI; if the unmask never comes, no @@ -2063,7 +2154,17 @@ impl MipsCore { let mask = CAUSE_IP0 | CAUSE_IP1; self.cp0_cause = (self.cp0_cause & !mask) | ((value as u32) & mask); } - 14 => self.cp0_epc = value, + 14 => { + // A guest write of a 32-bit value here is where a 64-bit + // return address would lose its top half, so the ERET that + // follows lands at a truncated PC. `log mips mask cp0`. + if cp0_log() { + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28epc: write EPC={value:#018x} (was {:#018x}) from pc={:#018x}", + self.cp0_epc, self.pc); + } + self.cp0_epc = value; + } 15 => { /* PRId is read-only */ } 16 => { // Bits 5:0 always writable (K0, CU, DB, IB). @@ -2087,14 +2188,37 @@ impl MipsCore { 17 => self.cp0_lladdr = value as u32, 18 => self.cp0_watchlo = value as u32, 19 => self.cp0_watchhi = value as u32, - 20 => self.cp0_xcontext = value, + 20 => { + // IP28 bring-up: IRIX's XTLB refill handler builds the page + // table base itself and expects XContext's PTEBase to be + // zero, so anything landing in bits [63:33] becomes a wild + // pointer inside the handler. `log mips mask cp0`. + if cp0_log() { + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28xctx: write {value:#018x} (ptebase={:#018x}) pc={:#018x}", + value & 0xFFFF_FFFE_0000_0000, self.pc); + } + self.cp0_xcontext = value; + } 25 => { #[cfg(feature = "developer_ip7")] eprintln!("[ip7] MTC0 PerfCnt (reg 25) write {:#018x} (ignored)", value); } - 26 => self.cp0_ecc = value as u32, - 27 => self.cp0_cacheerr = value as u32, - 28 => self.cp0_taglo = value as u32, + 26 => { + if cachediag_on() { + eprintln!("ip28cd: MTC0 ECC = {:#010x} (was {:#010x}) pc={:#018x}", + value as u32, self.cp0_ecc, self.pc); + } + self.cp0_ecc = value as u32; + } + 27 => { + if cachediag_on() { + eprintln!("ip28cd: MTC0 CacheErr = {:#010x} (was {:#010x}) pc={:#018x}", + value as u32, self.cp0_cacheerr, self.pc); + } + self.cp0_cacheerr = value as u32; + } + 28 => self.cp0_taglo = value, 29 => self.cp0_taghi = value as u32, 30 => self.cp0_errorepc = value, _ => {} // Writes to unimplemented registers are ignored diff --git a/src/mips_exec.rs b/src/mips_exec.rs index c1308662..9f61d999 100644 --- a/src/mips_exec.rs +++ b/src/mips_exec.rs @@ -24,6 +24,8 @@ pub const MIPS_LOG_INSN: u32 = 0x0001; // per-instruction disassembly trace pub const MIPS_LOG_TLB: u32 = 0x0002; // TLB read/write/probe pub const MIPS_LOG_MEM: u32 = 0x0004; // uncached memory accesses pub const MIPS_LOG_FPU: u32 = 0x0008; // FP compare/condmove/convert operand+result trace +pub const MIPS_LOG_CP0: u32 = 0x0010; // CP0 register traffic: EPC/EntryHi/XContext writes, + // ECC and CacheErr accesses, Status.DE transitions /// Opt-in gate for the `developerx` Coprocessor-Unusable break (see the three /// `handle_exception*` wrappers). Off unless `IRIS_BREAK_CPU=1`. @@ -44,6 +46,17 @@ fn mips_log(bit: u32) -> bool { devlog_is_active(LogModule::Mips) && (devlog_mask(LogModule::Mips) & bit) != 0 } +/// Like `mips_log`, but live in every build. +/// +/// The IP28 bring-up traces this gates were plain environment variables that +/// worked in release, and the machines they diagnose are booted in release — +/// so they keep that reach. The cost is two relaxed atomic loads, against the +/// `getenv` per call some of them used to do. +#[inline(always)] +fn mips_log_always(bit: u32) -> bool { + devlog_is_active(LogModule::Mips) && (devlog_mask(LogModule::Mips) & bit) != 0 +} + // Without `developer`, dlog_dev! is a no-op, so callers gate on this constant `false` // instead of paying an atomic load per call site to check a flag that can never fire. #[cfg(not(feature = "developer"))] @@ -333,7 +346,7 @@ struct CpuSnapshot { cp0_xcontext: u64, cp0_ecc: u32, cp0_cacheerr: u32, - cp0_taglo: u32, + cp0_taglo: u64, cp0_taghi: u32, cp0_errorepc: u64, @@ -1159,6 +1172,18 @@ impl MipsCpuConfig { pub const fn indy() -> Self { Self { tlb_entries: 48 } } + + /// The JTLB size the CPU model declares. + /// + /// `core.tlb_entries` is already taken from the model, so sizing the TLB + /// itself from anything else leaves the two disagreeing: Random cycles + /// over a range the array does not have, and a TLBWI to an index past the + /// end is silently dropped. The R10000 has 64 entries where the R4400 has + /// 48, and SGI's IP28 diagnostic writes index 48 on its first cache-alias + /// test — every one of its reported failures was that write going nowhere. + pub fn for_model() -> Self { + Self { tlb_entries: C::TLB_ENTRIES } + } } /// MIPS Execution Engine - combines CPU core with memory interface and TLB @@ -2704,6 +2729,37 @@ impl MipsExecutor { config |= ss << CONFIG_TR_SS; } + // R10000 lays Config out completely differently from an R4000, and the + // PROM sizes its cache diagnostics from it. Fields, per NetBSD's + // mips/include/cpuregs.h (MIPS4_CONFIG_*): + // [31:29] IC primary I-cache size, as 4096 << field + // [28:26] DC primary D-cache size, likewise + // [18:16] SS secondary cache size + // [15] BE big endian + // [13] SB secondary block size, 0 = 64B, 1 = 128B + // [2:0] K0 kseg0 cacheability, which the PROM sets for itself + // Presenting an R4000 Config here told an R10000 PROM that its + // secondary cache size field was zero. + if C::R10K_CACHE_OPS { + let log2 = |n: usize| (n / 4096).trailing_zeros(); + let mut c = 0u32; + c |= log2(32768) << 29; // 32 KB L1I + c |= log2(32768) << 26; // 32 KB L1D + c |= 1 << 15; // big endian + if C::L2_LINE == 128 { c |= 1 << 13; } + // Secondary cache size. The encoding is not in anything to hand, + // so it was swept against the PROM rather than guessed. + // + // 1 is the value to keep: IRIX reports "Secondary unified + // instruction/data cache size: 1 Mbyte", which agrees with this + // model's own `L2_SIZE`, and the PROM's power-on diagnostics pass + // and IRIX boots with it. 4 also passes POST but has IRIX report + // 8 MB, contradicting the model. + c |= 1 << 16; + c |= 2; // K0 = uncached at reset + config = c; + } + core.cp0_config = config; core.tlb_entries = C::TLB_ENTRIES as u32; // MipsCore::new already ran reset_registers, so set both the reset value @@ -4252,7 +4308,91 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", /// Handle an exception: update CP0 registers and jump to handler vector. /// Takes an ExecStatus with EXEC_IS_EXCEPTION set; extracts code and TLB-refill flag. + /// IP28 bring-up: dump everything about the one exception whose BadVAddr + /// is `IRIS_IP28_EXC_VADDR`. + /// + /// A first-N trace is useless for a fault that lands after thousands of + /// ordinary TLB misses, and reading the register file from the monitor + /// afterwards shows the guest's panic handler, not the fault. Costs one + /// relaxed load per exception on the R10000 model and folds away on every + /// other; exceptions are not a hot path. + #[inline] + fn ip28_trace_exception(&self, status: ExecStatus) { + if !C::R10K_CACHE_OPS { + return; + } + static WANT: std::sync::OnceLock> = std::sync::OnceLock::new(); + let want = *WANT.get_or_init(|| { + std::env::var("IRIS_IP28_EXC_VADDR").ok().and_then(|v| { + let v = v.trim(); + if v.is_empty() { return None; } + u64::from_str_radix(v.trim_start_matches("0x"), 16).ok() + }) + }); + let Some(want) = want else { return }; + + // Keep the run-up. The exception that panics the guest is the second + // one: the first is an ordinary miss, and its handler then faults. + // Only the run-up says what the handler was handed, and BadVAddr has + // already been overwritten by the time the match fires. + static RING: std::sync::Mutex> = std::sync::Mutex::new(Vec::new()); + const KEEP: usize = 12; + let line = format!( + "code={:<2} pc={:#018x} bd={} badvaddr={:#018x} xcontext={:#018x} \ + context={:#018x} entryhi={:#018x} ra={:#018x}", + (status & CAUSE_EXCCODE_MASK) >> 2, + self.core.pc, + self.core.in_delay_slot as u8, + self.core.cp0_badvaddr, + self.core.cp0_xcontext, + self.core.cp0_context, + self.core.cp0_entryhi, + self.core.read_gpr(31), + ); + if let Ok(mut ring) = RING.lock() { + if ring.len() == KEEP { + ring.remove(0); + } + ring.push(line); + if want == self.core.cp0_badvaddr { + eprintln!("ip28exc: --- last {} exceptions, oldest first ---", ring.len()); + for (i, l) in ring.iter().enumerate() { + eprintln!("ip28exc: [{i}] {l}"); + } + } + } + + if want != self.core.cp0_badvaddr { + return; + } + const NAMES: [&str; 32] = [ + "zero", "at", "v0", "v1", "a0", "a1", "a2", "a3", + "t0", "t1", "t2", "t3", "t4", "t5", "t6", "t7", + "s0", "s1", "s2", "s3", "s4", "s5", "s6", "s7", + "t8", "t9", "k0", "k1", "gp", "sp", "fp", "ra", + ]; + let code = (status & CAUSE_EXCCODE_MASK) >> 2; + eprintln!( + "ip28exc: code={code} pc={:#018x} delay_slot={} badvaddr={:#018x} status={:#010x}", + self.core.pc, self.core.in_delay_slot, self.core.cp0_badvaddr, + self.core.cp0_status, + ); + for i in 0..32u32 { + if self.core.read_gpr(i) == self.core.cp0_badvaddr { + eprintln!("ip28exc: {} (${i}) holds the bad address", NAMES[i as usize]); + } + } + for chunk in (0..32u32).collect::>().chunks(4) { + let mut line = String::from("ip28exc: "); + for &i in chunk { + line.push_str(&format!(" {:>4}={:#018x}", NAMES[i as usize], self.core.read_gpr(i))); + } + eprintln!("{line}"); + } + } + fn handle_exception(&mut self, status: ExecStatus) -> ExecStatus { + self.ip28_trace_exception(status); // In developer builds, bus/address error exceptions break into the // monitor at the fault site rather than dispatching to the MIPS // vector — must be decided before deliver_exception runs, since that @@ -6663,7 +6803,19 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", let op = cache_op & 0x1C; // Determine if this is a Hit operation that needs address translation - let needs_translation = matches!(op, C_CDX | C_HINV | C_HWBINV | C_HWB | C_HSV); + // On an R10000 the encodings for 5/6/7 are index operations, not the + // R4000 hit operations — translating them would fault on an index that + // is not a valid virtual address. See C_R10K_ISD in mips_cache_v2. + let sel = cache_op & 3; + let r10k_index_op = C::R10K_CACHE_OPS + && match op { + C_R10K_CBARRIER => sel == CACH_PI, + C_R10K_ILD => matches!(sel, CACH_PI | CACH_PD | CACH_SD), + C_R10K_ISD => matches!(sel, CACH_SI | CACH_SD), + _ => false, + }; + let needs_translation = + !r10k_index_op && matches!(op, C_CDX | C_HINV | C_HWBINV | C_HWB | C_HSV); let phys_addr = if needs_translation { // Hit operations need address translation @@ -6677,19 +6829,73 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", // For Index_Store_Tag, pass TagLo via phys_addr let op = cache_op & 0x1C; - let phys_addr_or_taglo = if op == C_IST { - self.core.cp0_taglo as u64 + // A cache tag is TagHi:TagLo, not TagLo alone. The IP28 PROM writes + // the low 32 bits to $28 and the bits above to $29 — a 36-bit + // secondary tag arrives as TagLo=0xffffcdfe, TagHi=0xf — so a model + // that reads only TagLo sees a different tag from the one written. + // Harmless for R4400/R5000, whose tags fit in 32 bits and who leave + // TagHi zero: the shift-in contributes nothing there. + let stores_taglo = op == C_IST || (r10k_index_op && op == C_R10K_ISD); + let phys_addr_or_taglo = if stores_taglo { + ((self.core.cp0_taghi as u64) << 32) | (self.core.cp0_taglo as u32 as u64) } else { phys_addr }; + // On an R10000 the check bits travel with cache data through CP0 ECC. + if C::R10K_CACHE_OPS { + self.cache.set_cache_ecc(self.core.cp0_ecc); + } + // IP28 cache-error investigation (`log mips mask cp0`). Two kinds + // of op are worth seeing: one carrying non-zero check bits, which is a + // guest staging a parity error on purpose, and any op from outside the + // PROM — i.e. from a loaded diagnostic, which runs out of XKPHYS. + // + // Both halves of that gate were learned the hard way. Gating on ECC + // alone hid every op the IDE issued, because the IDE's ECC reads back + // zero: the exact symptom under investigation was also blinding the + // instrument to its cause. The cap then has to be generous, because a + // cap of 600 silently truncated a run at precisely the boundary and + // made a partial picture look like the whole one. It exists only so a + // hot loop cannot rewrite the timing it is measuring. + if crate::mips_core::cachediag_on() { + let from_prom = (self.core.pc >> 32) == 0xFFFF_FFFF; + // CBARRIER is pure ordering: it names no line and carries no check + // bits, and it outnumbers everything else ~8:1 (83534 of 94322 in + // one IDE run). Tracing it swamped the log and slowed the guest + // enough that the run no longer reached the loop being studied — + // so drop it unless it is carrying staged check bits. + let noise = op == C_R10K_CBARRIER && self.core.cp0_ecc == 0; + if (self.core.cp0_ecc != 0 || !from_prom) && !noise { + const CACHE_TRACE_CAP: u32 = 200_000; + static SEEN: std::sync::atomic::AtomicU32 = + std::sync::atomic::AtomicU32::new(0); + let n = SEEN.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + if n < CACHE_TRACE_CAP { + crate::dlog!(LogModule::Mips, + "ip28cd: CACHE op={:#04x} sel={} vaddr={:#018x} ECC={:#010x} \ + TagHi:Lo={:#010x}:{:#018x} pc={:#018x}", + op, sel, virt_addr, self.core.cp0_ecc, + self.core.cp0_taghi, self.core.cp0_taglo, self.core.pc); + } else if n == CACHE_TRACE_CAP { + crate::dlog!(LogModule::Mips, + "ip28cd: CACHE trace capped at {CACHE_TRACE_CAP} ops \ + — output past this point is incomplete"); + } + } + } // Call unified cache interface let result = self.cache.cache_op(cache_op, virt_addr, phys_addr_or_taglo); + if C::R10K_CACHE_OPS && op == C_R10K_ILD { + self.core.cp0_ecc = self.cache.cache_op_ecc(); + } // For Index_Load_Tag, update CP0 TagLo from result - if op == C_ILT { - self.core.cp0_taglo = result; - self.core.cp0_taghi = 0; + if op == C_ILT || (r10k_index_op && op == C_R10K_ILD) { + // Split back the way it arrived. Zeroing TagHi unconditionally + // threw away the top of every tag wider than 32 bits. + self.core.cp0_taglo = result & 0xFFFF_FFFF; + self.core.cp0_taghi = (result >> 32) as u32; } self.handle_exec_complete() @@ -6917,6 +7123,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", let rt_reg = d.rt as u32; let rd_val = d.rd as u32; let value = self.core.read_cp0(rd_val); + self.ip28_cp0_trace("mfc0", rd_val, value); // Sign-extend 32-bit value to 64 bits self.core.write_gpr(rt_reg, value as u32 as i32 as i64 as u64); self.handle_exec_complete() @@ -6927,6 +7134,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", let rt_reg = d.rt as u32; let rd_val = d.rd as u32; let value = self.core.read_cp0(rd_val); + self.ip28_cp0_trace("dmfc0", rd_val, value); self.core.write_gpr(rt_reg, value); self.handle_exec_complete() } @@ -6935,13 +7143,16 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", /// The CP0 registers that are 64 bits wide. /// /// EntryLo0/1, Context, BadVAddr, EntryHi, EPC, XContext and ErrorEPC are - /// 64-bit in MIPS III and stay so in MIPS IV. + /// 64-bit in MIPS III and stay so in MIPS IV. TagLo is 32-bit on R4x00 but + /// 64-bit on the R10000, which is why it is asked of the model rather than + /// listed flat. fn cp0_is_64bit(reg: u32) -> bool { - matches!(reg, 2 | 3 | 4 | 8 | 10 | 14 | 20 | 30) + matches!(reg, 2 | 3 | 4 | 8 | 10 | 14 | 20 | 30) || (C::R10K_CACHE_OPS && reg == 28) } fn exec_mtc0(&mut self, d: &DecodedInstr) -> ExecStatus { let rt_val = self.core.read_gpr(d.rt as u32); + self.ip28_cp0_trace("mtc0", d.rd as u32, rt_val); let rd_val = d.rd as u32; // MTC0 moves the whole register. DMTC0 differs in what the @@ -6976,8 +7187,22 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", } // DMTC0 - Doubleword Move To CP0 (MIPS III) + /// IP28 bring-up: watch the cache-tag CP0 registers (26 ECC, 28 TagLo, + /// 29 TagHi) under `log mips mask cp0`. Only the R10000 model compiles + /// this in. + /// + /// The env-var form this replaced did an uncached `getenv` on every + /// DMTC0/MTC0, armed or not. + #[inline(always)] + fn ip28_cp0_trace(&self, what: &str, reg: u32, val: u64) { + if C::R10K_CACHE_OPS && matches!(reg, 26 | 28 | 29) && mips_log_always(MIPS_LOG_CP0) { + crate::dlog!(LogModule::Mips, "ip28cp0: {what} ${reg} = {val:#018x}"); + } + } + fn exec_dmtc0(&mut self, d: &DecodedInstr) -> ExecStatus { let rt_val = self.core.read_gpr(d.rt as u32); + self.ip28_cp0_trace("dmtc0", d.rd as u32, rt_val); let rd_val = d.rd as u32; self.core.write_cp0(rd_val, rt_val); self.handle_cp0_side_effects(rd_val); @@ -7143,6 +7368,31 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", // TLBWI - Write Indexed TLB Entry // Writes CP0.EntryHi, CP0.EntryLo0, CP0.EntryLo1, and CP0.PageMask to the TLB entry indexed by CP0.Index + /// Report every TLB write, under `log mips mask tlb`. + /// + /// `IRIS_IP28_TLBW=wired` additionally narrows it to writes below CP0 + /// Wired — the mappings a kernel means never to be evicted. That stays an + /// environment variable because it selects a *subset*, which the module + /// mask has no way to express; the on/off is devlog's. + #[inline] + fn ip28_trace_tlb_write(&self, op: &str, index: usize, entry: &crate::mips_tlb::TlbEntry) { + if !mips_log_always(MIPS_LOG_TLB) { + return; + } + static WIRED_ONLY: std::sync::OnceLock = std::sync::OnceLock::new(); + let wired_only = *WIRED_ONLY.get_or_init(|| { + matches!(std::env::var("IRIS_IP28_TLBW").as_deref(), Ok("wired")) + }); + if wired_only && index >= self.core.cp0_wired as usize { + return; + } + crate::dlog!(LogModule::Mips, + "ip28tlbw: {op} idx={index:<2} wired={} entryhi={:#018x} lo0={:#018x} lo1={:#018x} \ + mask={:#x} pc={:#018x}", + self.core.cp0_wired, entry.entry_hi, entry.entry_lo[0], entry.entry_lo[1], + entry.page_mask, self.core.pc); + } + fn exec_tlbwi(&mut self) -> ExecStatus { // The slot is Index[5:0]. Masking rather than `%` matters twice: // @@ -7168,7 +7418,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", return self.handle_exec_complete(); } let entry = self.create_tlb_entry_from_cp0(); - //eprintln!("TLBWI idx={} entryhi={:#018x} lo0={:#018x} lo1={:#018x} pc={:#018x}", index, entry.entry_hi, entry.entry_lo[0], entry.entry_lo[1], self.core.pc); + self.ip28_trace_tlb_write("tlbwi", index, &entry); self.tlb.write(index, entry); // Flushes the nutlb too — the TLB it caches just changed. self.nanotlb_invalidate(); @@ -7192,6 +7442,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", self.core.update_random(); let index = (self.core.cp0_random as usize) % self.tlb.num_entries(); let entry = self.create_tlb_entry_from_cp0(); + self.ip28_trace_tlb_write("tlbwr", index, &entry); self.tlb.write(index, entry); self.nanotlb_invalidate(); @@ -7361,6 +7612,15 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", // *guarantees* the next access misses and therefore calls `translate_fn`. self.resync_privilege_state(); + // `log mips mask cp0`. If the target's top half is already gone by + // the time we get here, the guest wrote a truncated EPC; if it is + // intact and the next fetch is truncated anyway, the loss is + // downstream of this. + if C::R10K_CACHE_OPS && mips_log_always(MIPS_LOG_CP0) { + crate::dlog!(LogModule::Mips, + "ip28epc: eret -> {target:#018x} (epc={:#018x} errorepc={:#018x} status={:#010x})", + self.core.cp0_epc, self.core.cp0_errorepc, self.core.cp0_status); + } // ERET jumps immediately without delay slot self.core.pc = target; @@ -14143,7 +14403,7 @@ impl Saveable for MipsCpu cp0u32!(cp0_status); cp0u32!(cp0_cause); cp0u32!(cp0_prid); cp0u32!(cp0_config); cp0u32!(cp0_lladdr); cp0u32!(cp0_watchlo); cp0u32!(cp0_watchhi); cp0u32!(cp0_ecc); cp0u32!(cp0_cacheerr); - cp0u32!(cp0_taglo); cp0u32!(cp0_taghi); + cp0u64!(cp0_taglo); cp0u32!(cp0_taghi); cp0u64!(cp0_badvaddr); cp0u64!(cp0_epc); cp0u64!(cp0_errorepc); cp0u64!(cp0_entrylo0); cp0u64!(cp0_entrylo1); cp0u64!(cp0_context); cp0u64!(cp0_pagemask); cp0u64!(cp0_entryhi); cp0u64!(cp0_xcontext); @@ -14208,7 +14468,7 @@ impl Saveable for MipsCpu c.count_hz_atomic.store(c.count_hz, std::sync::atomic::Ordering::Relaxed); ld32!(cp0_status); ld32!(cp0_cause); ld32!(cp0_prid); ld32!(cp0_config); ld32!(cp0_lladdr); ld32!(cp0_watchlo); ld32!(cp0_watchhi); - ld32!(cp0_ecc); ld32!(cp0_cacheerr); ld32!(cp0_taglo); ld32!(cp0_taghi); + ld32!(cp0_ecc); ld32!(cp0_cacheerr); ld64!(cp0_taglo); ld32!(cp0_taghi); ld64!(cp0_entrylo0); ld64!(cp0_entrylo1); ld64!(cp0_context); ld64!(cp0_pagemask); ld64!(cp0_badvaddr); ld64!(cp0_entryhi); ld64!(cp0_xcontext); ld64!(cp0_epc); ld64!(cp0_errorepc); @@ -14657,6 +14917,7 @@ mod xcontext_layout_tests { } } + #[cfg(test)] mod round_to_int_mode_tests { use super::*; diff --git a/src/mips_tlb.rs b/src/mips_tlb.rs index d52861ec..1c387bd7 100644 --- a/src/mips_tlb.rs +++ b/src/mips_tlb.rs @@ -4,9 +4,21 @@ use crate::mips_exec::CacheAttr; use std::fmt::Write; use crate::snapshot::{u64_slice_to_toml, load_u64_slice}; -/// Number of TLB entries in R4000 (48 dual-entries = 96 pages) +/// Default JTLB size: the R4000/R4400/R5000 count (48 dual-entries = 96 pages). +/// +/// The *live* size is per-CPU-model and lives in `MipsTlb::num_entries`; this +/// is only the default for `Default::default()` and for callers that do not +/// name a model. pub const TLB_NUM_ENTRIES: usize = 48; +/// Array capacity: the largest JTLB any modelled CPU has. +/// +/// The R10000 has 64 entries where the R4400 has 48. Sizing the arrays to the +/// maximum and carrying the live count separately keeps one concrete `MipsTlb` +/// type — the lookup walks an MRU list that only ever contains live slots, so +/// the unused tail costs nothing on the hot path. +pub const TLB_MAX_ENTRIES: usize = 64; + // ── TLB statistics (feature = "tlbstats") ──────────────────────────────────── #[cfg(feature = "tlbstats")] @@ -508,12 +520,13 @@ impl ShadowEntry { /// Real R4000 TLB implementation /// -/// Implements a fully associative JTLB (Joint TLB) with 48 dual-entries. +/// Implements a fully associative JTLB (Joint TLB); `num_entries` dual-entries, +/// 48 on R4x00/R5000 and 64 on the R10000. /// /// **32-bit mode (and 64-bit sign-extended ±2GB) fast path**: a 512KB `vmap` /// array indexed by VA[31:13] gives O(1) lookup. Each slot holds the TLB /// entry index (0-47) or VMAP_MISS. After the index is found we still verify -/// ASID/Global and the valid/dirty bits — but the linear scan over 48 entries +/// ASID/Global and the valid/dirty bits — but the linear scan over the entries /// is eliminated. /// /// For 64-bit VAs that are sign-extended 32-bit values (upper 32 bits all-zero @@ -528,14 +541,17 @@ impl ShadowEntry { #[derive(Clone)] pub struct MipsTlb { /// Architectural TLB entries (read/written by TLBR/TLBWI/TLBWR/TLBP). - entries: [TlbEntry; TLB_NUM_ENTRIES], + entries: [TlbEntry; TLB_MAX_ENTRIES], /// Cache-friendly shadow used by translate() and probe(). Kept in sync /// with `entries` — rebuilt whenever an entry is written. - shadow: [ShadowEntry; TLB_NUM_ENTRIES], + shadow: [ShadowEntry; TLB_MAX_ENTRIES], /// Head of each MRU list (slot index, or MRU_NONE). mru_head: [u8; MRU_LISTS], /// `mru_next[list][slot]` — next slot in that list, or MRU_NONE. - mru_next: [[u8; TLB_NUM_ENTRIES]; MRU_LISTS], + mru_next: [[u8; TLB_MAX_ENTRIES]; MRU_LISTS], + + /// Live JTLB entries for the CPU model in use (<= TLB_MAX_ENTRIES). + num_entries: usize, /// O(1) lookup for 32-bit (and sign-extended 64-bit) VAs. /// Indexed by VA[31:13] (19 bits). Value = entry index or VMAP_MISS. vmap: [u8; VMAP_SIZE], @@ -555,13 +571,14 @@ pub struct MipsTlb { impl MipsTlb { pub fn new(num_entries: usize) -> Self { - assert_eq!(num_entries, TLB_NUM_ENTRIES, - "MipsTlb currently requires exactly {} entries", TLB_NUM_ENTRIES); + assert!(num_entries > 1 && num_entries <= TLB_MAX_ENTRIES, + "MipsTlb supports 2..={} entries, got {}", TLB_MAX_ENTRIES, num_entries); let mut tlb = Self { - entries: [TlbEntry::new(); TLB_NUM_ENTRIES], - shadow: [ShadowEntry::invalid(); TLB_NUM_ENTRIES], + entries: [TlbEntry::new(); TLB_MAX_ENTRIES], + shadow: [ShadowEntry::invalid(); TLB_MAX_ENTRIES], mru_head: [0u8; MRU_LISTS], - mru_next: [[MRU_NONE; TLB_NUM_ENTRIES]; MRU_LISTS], + mru_next: [[MRU_NONE; TLB_MAX_ENTRIES]; MRU_LISTS], + num_entries, vmap: [VMAP_MISS; VMAP_SIZE], #[cfg(feature = "tlbcheck")] vmap_touched: std::collections::HashSet::new(), @@ -570,10 +587,12 @@ impl MipsTlb { }; for list in 0..MRU_LISTS { tlb.mru_head[list] = 0; - for i in 0..TLB_NUM_ENTRIES - 1 { + for i in 0..num_entries - 1 { tlb.mru_next[list][i] = (i + 1) as u8; } - // slot 47 already MRU_NONE from array initialisation + // The last live slot is already MRU_NONE from array + // initialisation, and so is every slot past it — the unused tail + // of a 48-entry model never joins a list, so lookups never see it. } tlb } @@ -667,7 +686,7 @@ impl MipsTlb { let (old_start, old_count) = old_range; self.vmap_clear(old_start, old_count); let old_end = old_start + old_count; - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { if i == index { continue; } @@ -707,12 +726,12 @@ impl MipsTlb { // Two entries conflict if their VPN2 ranges intersect (after masking // by the wider of the two page masks) and they'd both be visible to // the same lookup: either entry is global, or both share an ASID. - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let a = &self.entries[i]; if !a.is_valid_even() && !a.is_valid_odd() { continue; // fully-invalid entries can't conflict } - for j in (i + 1)..TLB_NUM_ENTRIES { + for j in (i + 1)..self.num_entries { let b = &self.entries[j]; if !b.is_valid_even() && !b.is_valid_odd() { continue; @@ -741,7 +760,7 @@ impl MipsTlb { // (an all-zero architectural entry) derives 0 for the same fields. // Both are correctly unmatchable — the bit patterns just differ by // construction — so comparing them here would be a false positive. - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let expected = ShadowEntry::from_entry(&self.entries[i]); let s = &self.shadow[i]; if s.valid_dirty != expected.valid_dirty || s.asid != expected.asid @@ -766,12 +785,12 @@ impl MipsTlb { // 3. MRU list integrity: each list must visit every slot exactly once. for list in 0..MRU_LISTS { - let mut seen = [false; TLB_NUM_ENTRIES]; + let mut seen = [false; TLB_MAX_ENTRIES]; let mut cur = self.mru_head[list]; let mut count = 0usize; while cur != MRU_NONE { let slot = cur as usize; - if slot >= TLB_NUM_ENTRIES { + if slot >= self.num_entries { report(format!("MRU list {} contains out-of-range slot {}", list, slot)); break; } @@ -783,9 +802,9 @@ impl MipsTlb { count += 1; cur = self.mru_next[list][slot]; } - let missing: Vec = (0..TLB_NUM_ENTRIES).filter(|&i| !seen[i]).collect(); + let missing: Vec = (0..self.num_entries).filter(|&i| !seen[i]).collect(); if !missing.is_empty() { - report(format!("MRU list {} is missing slot(s) {:?} (visited {} of {})", list, missing, count, TLB_NUM_ENTRIES)); + report(format!("MRU list {} is missing slot(s) {:?} (visited {} of {})", list, missing, count, self.num_entries)); } } @@ -817,7 +836,7 @@ impl MipsTlb { { use std::collections::HashMap; let mut owners: HashMap = HashMap::new(); // vpn2 -> (most_recent_entry_idx, multiple) - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let (start, count) = Self::vmap_range(&self.entries[i]); for k in 0..count { let vpn2 = start.wrapping_add(k); @@ -882,7 +901,7 @@ impl MipsTlb { eprintln!("TLBCHECK VIOLATION [{}]: {}", context, msg); } eprintln!("TLBCHECK: dumping full TLB state for [{}]:", context); - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { eprintln!("{}", self.format_entry(i)); } true @@ -1055,7 +1074,7 @@ impl Tlb for MipsTlb { } fn write(&mut self, index: usize, mut entry: TlbEntry) { - if index < self.entries.len() { + if index < self.num_entries { let mask = entry.page_mask | 0x1FFF; entry.selector_bit_shift = (mask.trailing_ones() - 1) as u8; entry.vcmp32 = !mask & 0x0000_0000_FFFF_E000; @@ -1076,7 +1095,7 @@ impl Tlb for MipsTlb { } fn read(&self, index: usize) -> TlbEntry { - if index < self.entries.len() { + if index < self.num_entries { self.entries[index] } else { TlbEntry::new() @@ -1117,7 +1136,7 @@ impl Tlb for MipsTlb { } fn format_entry(&self, index: usize) -> String { - if index >= self.entries.len() { + if index >= self.num_entries { return format!("Index {} out of bounds", index); } let e = &self.entries[index]; @@ -1188,14 +1207,14 @@ impl Tlb for MipsTlb { } fn power_on(&mut self) { - self.entries = [TlbEntry::new(); TLB_NUM_ENTRIES]; - self.shadow = [ShadowEntry::invalid(); TLB_NUM_ENTRIES]; + self.entries = [TlbEntry::new(); TLB_MAX_ENTRIES]; + self.shadow = [ShadowEntry::invalid(); TLB_MAX_ENTRIES]; for list in 0..MRU_LISTS { self.mru_head[list] = 0; - for i in 0..TLB_NUM_ENTRIES - 1 { + for i in 0..self.num_entries - 1 { self.mru_next[list][i] = (i + 1) as u8; } - self.mru_next[list][TLB_NUM_ENTRIES - 1] = MRU_NONE; + self.mru_next[list][self.num_entries - 1] = MRU_NONE; } self.vmap.fill(VMAP_MISS); #[cfg(feature = "tlbcheck")] @@ -1204,7 +1223,9 @@ impl Tlb for MipsTlb { fn save_state(&self) -> toml::Value { // Each entry stored as [page_mask, entry_hi, entry_lo0, entry_lo1] - let arr: Vec = self.entries.iter().map(|e| { + // Only the live entries: a 48-entry model must still produce a + // 48-entry snapshot even though the array has room for 64. + let arr: Vec = self.entries[..self.num_entries].iter().map(|e| { let words = [e.page_mask, e.entry_hi, e.entry_lo[0], e.entry_lo[1]]; u64_slice_to_toml(&words) }).collect(); @@ -1214,7 +1235,7 @@ impl Tlb for MipsTlb { fn load_state(&mut self, v: &toml::Value) -> Result<(), String> { if let toml::Value::Array(arr) = v { for (i, item) in arr.iter().enumerate() { - if i >= TLB_NUM_ENTRIES { break; } + if i >= self.num_entries { break; } let mut words = [0u64; 4]; load_u64_slice(item, &mut words); let page_mask = words[0]; @@ -1240,19 +1261,19 @@ impl Tlb for MipsTlb { }; } } - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { self.shadow[i] = ShadowEntry::from_entry(&self.entries[i]); } // Reset MRU lists to canonical order so snapshot restores are deterministic. for list in 0..MRU_LISTS { self.mru_head[list] = 0; - for i in 0..TLB_NUM_ENTRIES - 1 { + for i in 0..self.num_entries - 1 { self.mru_next[list][i] = (i + 1) as u8; } - self.mru_next[list][TLB_NUM_ENTRIES - 1] = MRU_NONE; + self.mru_next[list][self.num_entries - 1] = MRU_NONE; } self.vmap.fill(VMAP_MISS); - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let (start, count) = Self::vmap_range(&self.entries[i]); self.vmap_fill_range(i, start, count); } diff --git a/src/mips_tlb_test.rs b/src/mips_tlb_test.rs index 57077e6d..c5a82467 100644 --- a/src/mips_tlb_test.rs +++ b/src/mips_tlb_test.rs @@ -440,4 +440,57 @@ mod tests { .join() .expect("thread panicked"); } + +/// The JTLB is as big as the CPU model says, and no bigger. +/// +/// The R10000 has 64 entries where the R4400 has 48. The count used to be a +/// compile-time 48 for every model while `core.tlb_entries` was taken from the +/// model, so on an R10000 the two disagreed: Random cycled over a range the +/// array did not have, and a TLBWI to index 48..63 went nowhere. SGI's IP28 +/// diagnostic writes index 48 on its first cache-alias test, and every failure +/// it reported was that write being dropped. +#[test] +fn tlb_size_follows_the_cpu_model() { + #[cfg(feature = "ip28")] + use crate::mips_cache_shadow::R10000ShadowCache; + use crate::mips_cache_v2::{R4400Cache, R5000Cache}; + use crate::mips_exec::MipsCpuConfig; + + assert_eq!(MipsCpuConfig::for_model::().tlb_entries, 48); + assert_eq!(MipsCpuConfig::for_model::().tlb_entries, 48); + #[cfg(feature = "ip28")] + assert_eq!(MipsCpuConfig::for_model::().tlb_entries, 64); +} + +/// An index past the live count is dropped; one inside it is kept. +#[test] +fn tlb_write_past_the_live_entry_count_is_dropped() { + let mut entry = TlbEntry::new(); + entry.entry_hi = 0xC000_0000_0004_0000; + + let mut r4400 = MipsTlb::new(48); + r4400.write(48, entry); + assert_eq!(r4400.read(48).entry_hi, 0, "48-entry TLB must drop index 48"); +} + +/// The other half of the pair, deliberately in its own `#[test]`. +/// +/// `MipsTlb` carries its arrays inline and is large enough that **two** of +/// them alive at once overflows a test thread's stack in a debug build — the +/// single test these two replace aborted the whole binary with SIGABRT. It +/// went unnoticed because every test run in this project passes `--release`, +/// where the temporaries collapse; `cargo test` on its own is what an +/// upstream reviewer would run. One TLB per test, as every neighbouring test +/// already does. +#[test] +fn tlb_write_inside_a_larger_entry_count_is_kept() { + let mut entry = TlbEntry::new(); + entry.entry_hi = 0xC000_0000_0004_0000; + + let mut r10000 = MipsTlb::new(64); + r10000.write(48, entry); + assert_eq!(r10000.read(48).entry_hi, entry.entry_hi, "64-entry TLB must keep it"); + r10000.write(63, entry); + assert_eq!(r10000.read(63).entry_hi, entry.entry_hi, "...and its last slot"); +} } diff --git a/src/physical.rs b/src/physical.rs index b5f90d2b..b27bd5e8 100644 --- a/src/physical.rs +++ b/src/physical.rs @@ -211,7 +211,24 @@ const PROM_END: u32 = 0x1FD00000; // accesses go through the normal lomem device_map entries — no direct bank pointer needed. const ALIAS_BASE: u32 = 0x00000000; const ALIAS_END: u32 = 0x00080000; -const ALIAS_OFFSET: u32 = LOMEM_BASE; + +/// Where the 512 KB alias at physical 0 points. +/// +/// The MC mirrors the bottom 512 KB of *memory*, and which physical address +/// that is depends on the machine: LOMEM_BASE on IP22/IP24, 0x20000000 on +/// IP28, whose RAM starts there and has nothing at lomem at all. With the +/// offset fixed at LOMEM_BASE the whole window read back as zero on IP28. +/// +/// That window is not spare space. ARCS builds its system parameter block at +/// physical 0x1000 and its firmware vector table at 0x1800, and a 64-bit sash +/// loads its firmware pointer straight out of 0x1018 — so the PROM's writes +/// were being discarded and sash then dereferenced the null it read back. +/// +/// Taken from the machine profile, never from the environment. +fn alias_offset_for(ip28: bool) -> u32 { + if ip28 { HIMEM_BASE } else { LOMEM_BASE } +} + /// What one 64 KB `device_map` slot should point at after a MEMCFG write. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -308,6 +325,10 @@ pub struct Physical { /// `u64` before that — never null, so the test needs no guard. #[cfg(feature = "ppmem")] ppmem_bitmap: *const u64, + /// IP28: the bank placement ppmem's window was last built for, so a + /// MEMCFG write that moves nothing leaves the window alone. + #[cfg(all(feature = "ppmem", feature = "ip28"))] + ppmem_last_placement: Option<([Option<(u32, u32, u32)>; 4], [usize; 4])>, pub rex3: Option>, /// Second Newport head (dual-head Indigo2 / `graphics.heads = 2`). @@ -343,6 +364,10 @@ pub struct Physical { /// two, and a bank parked out there has to be unmapped again when it /// moves — see `plan_bank_slots`. banks_outside_windows: Vec, + /// Where the 512 KB alias at physical 0 points — see `alias_offset_for`. + alias_offset: u32, + /// This machine is an IP28. Only used to label the bank-map trace. + is_ip28: bool, trace: AtomicBool, start_tick: u64, @@ -399,6 +424,8 @@ impl Physical { mc: MemoryController, hpc3: Hpc3, prom: PromPort, + // IP28: RAM starts at 0x20000000, so the low-memory alias follows it. + ip28: bool, ) -> Self { let host_freq = crate::platform::get_host_tick_frequency(); let start_tick = crate::platform::get_host_ticks(); @@ -409,7 +436,7 @@ impl Physical { let gio_bus_error = GioBusErrorDevice { mc: mc.clone() }; // Alias targets will be set in build_device_map once Physical is in final location let unmapped_ram = UnmappedRam; - let alias_bus = AliasBus::new(std::ptr::null::(), ALIAS_OFFSET); + let alias_bus = AliasBus::new(std::ptr::null::(), alias_offset_for(ip28)); // VINO GIO alias: 0x1F08xxxx → 0x0008xxxx (subtract 0x1F000000 = add 0xFF000000) // GIO64 VINO aperture sits at 0x1F080000; the chip's primary registers // live at physical 0x00080000 (VINO_BASE). To map 0x1F080000 → 0x00080000 @@ -467,6 +494,8 @@ impl Physical { ppmem_gen_base, #[cfg(feature = "ppmem")] ppmem_bitmap, + #[cfg(all(feature = "ppmem", feature = "ip28"))] + ppmem_last_placement: None, rex3, rex3_head1, gr2, @@ -487,6 +516,8 @@ impl Physical { black_hole, device_map, banks_outside_windows: Vec::new(), + alias_offset: alias_offset_for(ip28), + is_ip28: ip28, trace: AtomicBool::new(false), start_tick, host_freq, @@ -571,16 +602,11 @@ impl Physical { self.device_map[i as usize] = gr2_ptr; } } else if let Some(mgras_ptr) = mgras_ptr { - // Indigo2 IMPACT preview: MGRAS stub spans gfx + populated expansion slots. + // IMPACT in the graphics slot. The expansion slots stay unmapped so + // their probes bus-error, as empty slots do. for i in (NEWPORT_BASE >> 16)..((NEWPORT_END - 1) >> 16) + 1 { self.device_map[i as usize] = mgras_ptr; } - for i in (GIO_SLOT0_BASE >> 16)..((GIO_SLOT0_END - 1) >> 16) + 1 { - self.device_map[i as usize] = mgras_ptr; - } - for i in (GIO_SLOT1_BASE >> 16)..((GIO_SLOT1_END - 1) >> 16) + 1 { - self.device_map[i as usize] = mgras_ptr; - } } // else: GIO timeout from layer 2 already covers the Newport slot @@ -631,7 +657,7 @@ impl Physical { self.device_map[i as usize] = prom_ptr; } - // Alias: points back into Physical itself with ALIAS_OFFSET added. + // Alias: points back into Physical itself with `alias_offset()` added. // So alias accesses go: AliasBus → Physical::read/write(addr + LOMEM_BASE) // → device_map lookup → whichever bank is mapped at LOMEM_BASE. // This way alias automatically tracks whatever MEMCFG maps at LOMEM_BASE. @@ -647,6 +673,7 @@ impl Physical { self.vino_gio_alias.target = self as *const Physical as *const dyn BusDevice; let vino_gio_alias_ptr: *const dyn BusDevice = &self.vino_gio_alias; self.device_map[(0x1F080000u32 >> 16) as usize] = vino_gio_alias_ptr; + } /// Remap memory banks in device_map. @@ -669,12 +696,38 @@ impl Physical { &self.banks[3], ]; - // ppmem: drop every mapping before re-placing the banks. Safe to leave - // the window briefly unmapped — remapping runs inside the CPU's MEMCFG - // store during PROM POST, before DMA is running (design doc §5.1). + // IP28: a MEMCFG write that leaves every bank where it was (IRIX's + // kernel rewrites MEMCFG1 during boot to fix the refresh bits) must + // not tear down and rebuild ppmem's window. JIT compile workers and + // the DMA thread are running by then, and the rebuild is exactly the + // window in which they used to fault. + #[cfg(all(feature = "ppmem", feature = "ip28"))] + let ppmem_placement_changed = { + let sizes = [0, 1, 2, 3].map(|i| self.banks[i].size()); + let key = (bank_addrs, sizes); + let changed = self.ppmem_last_placement != Some(key); + self.ppmem_last_placement = Some(key); + changed + }; + #[cfg(all(feature = "ppmem", not(feature = "ip28")))] + let ppmem_placement_changed = true; + + // ppmem: drop every mapping before re-placing the banks. The comment + // this replaced said the window is safe to leave unmapped because + // remapping only runs during PROM POST, before DMA; that is not true + // on IP28 (see above), which is why ppmem scrubs rather than unmaps (see `AddrSpace::scrub`). #[cfg(feature = "ppmem")] - if let Some(sp) = &self.ppmem_space { - sp.clear_mappings(); + if ppmem_placement_changed { + if let Some(sp) = &self.ppmem_space { + sp.clear_mappings(); + } + // A bank now answering at an address another bank answered at + // must look changed to the JIT even if the two counters happen to + // be equal, so move every counter. + #[cfg(all(feature = "ip28", feature = "jitv2"))] + for b in &self.banks { + b.bump_gen_all(); + } } for (bank_idx, maybe_bank) in bank_addrs.iter().enumerate() { @@ -683,6 +736,9 @@ impl Physical { continue; }; + if self.is_ip28 { + eprintln!("iris: IP28 experiment: bank {bank_idx} -> base {conf_base:#010x} mask {addr_mask:#010x} limit {limit:#010x}"); + } dlog_dev!(LogModule::Mc, "[MEMCFG] bank {} mapped at 0x{:08x}..0x{:08x} addr_mask={:08x} limit={:08x} ({}MB visible, {}MB per rank)", bank_idx, conf_base, conf_base + limit, addr_mask, limit, limit >> 20, (addr_mask + 1) >> 20); @@ -693,7 +749,7 @@ impl Physical { // undersized bank repeats to fill `limit`, which is exactly the // SIMM mirroring `addr_mask` encodes — see docs/ppmem-design.md §5. #[cfg(feature = "ppmem")] - if let Some(sp) = &self.ppmem_space { + if let (true, Some(sp)) = (ppmem_placement_changed, &self.ppmem_space) { // `addr_mask + 1` is the SIMM's mirror period, and `limit` the // configured slot. They are independent: a dual-rank SIMM has a // slot half the size of the bank (each rank placed separately), @@ -740,8 +796,8 @@ impl Physical { // re-dispatch. AliasBus stays installed in device_map as the bus-path // equivalent; both see identical memory. #[cfg(feature = "ppmem")] - if let Some(sp) = &self.ppmem_space { - let bank0_mapped = bank_addrs[0].is_some_and(|(base, _, _)| base == LOMEM_BASE); + if let (true, Some(sp)) = (ppmem_placement_changed, &self.ppmem_space) { + let bank0_mapped = bank_addrs[0].is_some_and(|(base, _, _)| base == self.alias_offset); if bank0_mapped { let alias_len = (ALIAS_END - ALIAS_BASE) as u64; if (self.banks[0].size() as u64) >= alias_len { diff --git a/src/ppmem/map_unix.rs b/src/ppmem/map_unix.rs index 98076dd7..2e3e380d 100644 --- a/src/ppmem/map_unix.rs +++ b/src/ppmem/map_unix.rs @@ -276,6 +276,45 @@ impl AddrSpace { } } +impl AddrSpace { + /// Replace `[at, at+len)` with private, zero-filled, read-write memory, + /// **retaining the claim** like `unmap` does. + /// + /// For ranges something may still point into. ppmem's generation window + /// is read by JIT compile workers through pointers taken while a bank was + /// mapped, and the data window by the DMA thread after a bitmap check; a + /// remap runs on the CPU thread meanwhile. With `PROT_NONE` there, a + /// reader in that window took the whole emulator down (SIGBUS in + /// `PhysicalCodePage::current_gen`, twice on IP28, where IRIX's kernel + /// rewrites MEMCFG1 during boot). Here it reads zeros, which the + /// generation check sees as a change and the bitmap never lets reach + /// guest-visible state. + /// + /// # Safety + /// + /// As `map`: whatever was mapped in the range is gone. + pub unsafe fn scrub(&self, at: usize, len: usize) -> io::Result<()> { + self.check_range(at, len); + let want = self.base.add(at); + let got = libc::mmap( + want as *mut libc::c_void, + len, + libc::PROT_READ | libc::PROT_WRITE, + libc::MAP_PRIVATE | libc::MAP_ANONYMOUS | libc::MAP_FIXED | libc::MAP_NORESERVE, + -1, + 0, + ); + if got == libc::MAP_FAILED { + return Err(oserr("mmap(scrub)")); + } + assert_eq!( + got as usize, want as usize, + "ppmem: scrub landed at {got:p}, wanted {want:p}" + ); + Ok(()) + } +} + impl Drop for AddrSpace { fn drop(&mut self) { // One munmap releases the whole reservation including every view diff --git a/src/ppmem/map_windows.rs b/src/ppmem/map_windows.rs index 8acdfffa..20aad16e 100644 --- a/src/ppmem/map_windows.rs +++ b/src/ppmem/map_windows.rs @@ -272,6 +272,18 @@ impl AddrSpace { } Ok(()) } + + /// The Unix backend's `scrub`. Not done here: a placeholder cannot be + /// made readable without committing a view, and IP28 is not built for + /// Windows. Falls back to `unmap`, which keeps the old fault-on-access + /// behaviour. + /// + /// # Safety + /// + /// Same as `unmap`. + pub unsafe fn scrub(&self, at: usize, len: usize) -> io::Result<()> { + unsafe { self.unmap(at, len) } + } } impl Drop for AddrSpace { diff --git a/src/ppmem/ppmem.rs b/src/ppmem/ppmem.rs index a5ccff12..085335e1 100644 --- a/src/ppmem/ppmem.rs +++ b/src/ppmem/ppmem.rs @@ -319,7 +319,7 @@ impl PpMemory { } #[cfg(feature = "jitv2")] - fn bump_gen_all(&self) { + pub fn bump_gen_all(&self) { for i in 0..self.gen_count() { unsafe { (*self.gen_base.add(i)).fetch_add(1, Ordering::Relaxed) }; } @@ -768,7 +768,8 @@ impl MappedMemory for PpMemSpace { for m in std::mem::take(&mut st.mappings) { // Best-effort: a failure here leaves the range mapped, which is // still safe — it just isn't reverted. - let _ = unsafe { self.space.unmap(m.at as usize, m.len as usize) }; + // `scrub`, not `unmap`: see `AddrSpace::scrub`. + let _ = unsafe { self.space.scrub(m.at as usize, m.len as usize) }; #[cfg(feature = "jitv2")] { let gen_off = m.at / GEN_RATIO; @@ -776,7 +777,7 @@ impl MappedMemory for PpMemSpace { let gran = super::map::granularity() as u64; if gen_len >= gran && gen_off % gran == 0 && gen_len % gran == 0 { let _ = unsafe { - self.gen_space.unmap(gen_off as usize, gen_len as usize) + self.gen_space.scrub(gen_off as usize, gen_len as usize) }; } } @@ -1105,6 +1106,39 @@ mod tests { } } + /// Clearing the mappings must leave the window readable. JIT compile + /// workers hold generation-counter pointers taken while a bank was mapped, + /// and read them from other threads while the CPU thread remaps; with + /// `PROT_NONE` there, that read was a SIGBUS that killed the emulator + /// (twice in IP28 boots). Without the fix this test dies the same way. + #[cfg(all(unix, feature = "jitv2"))] + #[test] + fn a_cleared_window_is_still_readable_through_old_pointers() { + let (space, _banks) = PpMemSpace::with_bank_sizes(&[8]).unwrap(); + let base = 0x0800_0000u64; + space.map_bank(0, base, 8 * MB as u64, 8 * MB as u64).unwrap(); + let data = unsafe { space.window_base().add(base as usize + 0x100) as *const u32 }; + let gen = unsafe { space.gen_window_base().add((base as usize + 0x100) >> 12) }; + unsafe { + *(data as *mut u32) = 0xFEED_FACE; + (*gen).fetch_add(7, std::sync::atomic::Ordering::Relaxed); + } + + space.clear_mappings(); + + unsafe { + assert_eq!(*data, 0, "a cleared data range reads as zeros"); + assert_eq!((*gen).load(std::sync::atomic::Ordering::Relaxed), 0, + "a cleared generation range reads as zero, which differs from the live counter"); + } + // Mapped back, the bank's own contents and counter are there again. + space.map_bank(0, base, 8 * MB as u64, 8 * MB as u64).unwrap(); + unsafe { + assert_eq!(*data, 0xFEED_FACE); + assert_eq!((*gen).load(std::sync::atomic::Ordering::Relaxed), 7); + } + } + /// A bank mapped into the window and the same bank accessed through /// `BusDevice` must be the same memory — that is what makes ppmem /// pluggable into the existing bus rather than a parallel universe. diff --git a/src/testdev.rs b/src/testdev.rs index 54503472..3aff30e2 100644 --- a/src/testdev.rs +++ b/src/testdev.rs @@ -348,7 +348,7 @@ pub fn dump_json(core: &MipsCore, tag: u32) -> String { ("XContext", hex64(core.cp0_xcontext)), ("ECC", hex32(core.cp0_ecc)), ("CacheErr", hex32(core.cp0_cacheerr)), - ("TagLo", hex32(core.cp0_taglo)), + ("TagLo", hex32(core.cp0_taglo as u32)), ("TagHi", hex32(core.cp0_taghi)), ]; let body: Vec = cp0.iter().map(|(n, v)| format!(" \"{}\": {}", n, v)).collect(); From f6394629f4d2c88ecffb6781710becdbfbb4f50b Mon Sep 17 00:00:00 2001 From: atomchild411 <143453386+atomchild411@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:47:13 -0800 Subject: [PATCH 3/5] jitv2: tcache serves the R10000 through the window alone The R10000's shadow cache keeps neither line data nor tags on the load/store path: every access goes to memory. Under tcache, jitv2's inline path for it is therefore tcache's window half alone. `JitDcGeometry` gains `tagless`. The shadow cache implements the tcache hooks (window base, inline bitmap, generation window) and, once both windows are published, reports a tagless geometry; without tcache it reports none and every access calls out, as before. For a tagless geometry `emit_inline_mem_guard` skips the L1-D tag match, LRU update and dirty bit and goes straight to tcache's gate, now `emit_tc_mapped` and shared with the tagged path, then to `jit_tc_base + phys` with tcache's byte lanes; a store's generation bump is tcache's own. `phys` is cut to 32 bits first, as the callout does. The R4400 and R5000 emit the same code as before. On the IP28 (IRIX 6.5.22), the same tree built with and without tcache: a workload over ssh (awk, a floating-point loop, tar of /usr/include, an awk array loop) 17.5-18.5 s with, 26.4 s without, with identical output. Co-Authored-By: Claude Opus 5.5 --- src/jitv2/codegen.rs | 120 ++++++++++++++++++++++++++++----------- src/jitv2/mod.rs | 1 + src/mips_cache_shadow.rs | 70 +++++++++++++++++++++++ src/mips_cache_v2.rs | 8 ++- 4 files changed, 166 insertions(+), 33 deletions(-) diff --git a/src/jitv2/codegen.rs b/src/jitv2/codegen.rs index 10a4de31..a98b967c 100644 --- a/src/jitv2/codegen.rs +++ b/src/jitv2/codegen.rs @@ -3834,8 +3834,9 @@ fn l1d_tag_dirty_off() -> i32 { std::mem::offset_of!(crate::mips_cache_v2::L1DTa struct InlineMemPath { fast_data_ptr: Value, /// Pointer to the matched L1-D tag, live in the fast block. Stores use it - /// to set the dirty flag, mirroring `mark_l1d_dirty`. - fast_tag_ptr: Value, + /// to set the dirty flag, mirroring `mark_l1d_dirty`. `None` on a tagless + /// cache, which has no L1-D model to update. + fast_tag_ptr: Option, /// Physical address of the access, live in the fast block. Needed by the /// tcache store path to index the jitv2 generation array. fast_phys: Value, @@ -3866,6 +3867,11 @@ fn emit_inline_mem_guard( if !geom.supported { return None; } + // A tagless cache has an inline path only through tcache's window. + #[cfg(not(feature = "tcache"))] + if geom.tagless { + return None; + } let mem = MemFlagsData::trusted(); let ptr_ty = ctx.module.target_config().pointer_type(); @@ -3956,6 +3962,11 @@ fn emit_inline_mem_guard( let va_off = ctx.builder.ins().band_imm_s(vaddr, 0xFFF); let phys = ctx.builder.ins().bor(phys_page, va_off); + #[cfg(feature = "tcache")] + if geom.tagless { + return Some(emit_tagless_tail(ctx, phys, size, fast_block, slow_block, join_block)); + } + // ---- 2. L1D tag match -------------------------------------------- // // Mirrors `ensure_l1d_line`'s hit path exactly (mips_cache_v2.rs). On a @@ -4024,32 +4035,7 @@ fn emit_inline_mem_guard( // paper over a startup ordering problem would be paid forever. #[cfg(feature = "tcache")] let proceed = { - let bm_ptr = ctx.builder.ins().load(ptr_ty, mem, ctx.core_ptr, - ir::immediates::Offset32::new(core_offset_of_jit_tc_bitmap())); - let bits = ctx.builder.ins().load(i64t, mem, bm_ptr, ir::immediates::Offset32::new(0)); - // Tested `(bits >> region) & 1` here as a "two fewer IR instructions" - // simplification. It is not one: measured over 500 corpus pages with - // the inline path emitting, it produced *more* code (5,901,554 -> - // 5,966,525 bytes, +1.1%). Cranelift lowers the `1 << region` form - // better — the mask feeds a `test` directly, while the shift form - // needs the variable shift's result materialized first. Left as is. - // Test the region's bit as `(bits >> region) & 1` instead of building - // `1 << region` and masking with it — same predicate, one fewer IR - // instruction (no materialized `1`). - // - // Static code size says this is slightly worse (+1.1% over 500 corpus - // pages with the inline path emitting: 5,901,554 -> 5,966,525), but - // this sits on the hot path of every guest memory access and byte - // count has repeatedly mispredicted real throughput in this codebase. - // Under evaluation on a live workload. - // - // Shift counts are masked to 6 bits by both x86 and Cranelift's `ushr` - // definition, exactly as the old `ishl` relied on, so an out-of-range - // `region` behaves identically. - let region = ctx.builder.ins().ushr_imm_s(phys, crate::ppmem::BITMAP_SHIFT as i64); - let shifted = ctx.builder.ins().ushr(bits, region); - let mapped = ctx.builder.ins().band_imm_s(shifted, 1); - let mapped = ctx.builder.ins().icmp_imm_s(IntCC::NotEqual, mapped, 0); + let mapped = emit_tc_mapped(ctx, phys); ctx.builder.ins().band(proceed, mapped) }; @@ -4129,7 +4115,75 @@ fn emit_inline_mem_guard( let swizzled = emit_swizzle_index(ctx, index, size); let fast_data_ptr = ctx.builder.ins().iadd(base, swizzled); - Some(InlineMemPath { fast_data_ptr, fast_tag_ptr: tag_ptr, fast_phys: phys, slow_block, join_block }) + Some(InlineMemPath { fast_data_ptr, fast_tag_ptr: Some(tag_ptr), fast_phys: phys, slow_block, join_block }) +} + +/// tcache's gate, step 4 of `emit_inline_mem_guard`: is `phys` in a region +/// ppmem maps whole? Shared with the tagless path. +#[cfg(feature = "tcache")] +fn emit_tc_mapped(ctx: &mut EmitCtx, phys: Value) -> Value { + let mem = MemFlagsData::trusted(); + let ptr_ty = ctx.module.target_config().pointer_type(); + let i64t = ir::types::I64; + let bm_ptr = ctx.builder.ins().load(ptr_ty, mem, ctx.core_ptr, + ir::immediates::Offset32::new(core_offset_of_jit_tc_bitmap())); + let bits = ctx.builder.ins().load(i64t, mem, bm_ptr, ir::immediates::Offset32::new(0)); + // Tested `(bits >> region) & 1` here as a "two fewer IR instructions" + // simplification. It is not one: measured over 500 corpus pages with + // the inline path emitting, it produced *more* code (5,901,554 -> + // 5,966,525 bytes, +1.1%). Cranelift lowers the `1 << region` form + // better — the mask feeds a `test` directly, while the shift form + // needs the variable shift's result materialized first. Left as is. + // Test the region's bit as `(bits >> region) & 1` instead of building + // `1 << region` and masking with it — same predicate, one fewer IR + // instruction (no materialized `1`). + // + // Static code size says this is slightly worse (+1.1% over 500 corpus + // pages with the inline path emitting: 5,901,554 -> 5,966,525), but + // this sits on the hot path of every guest memory access and byte + // count has repeatedly mispredicted real throughput in this codebase. + // Under evaluation on a live workload. + // + // Shift counts are masked to 6 bits by both x86 and Cranelift's `ushr` + // definition, exactly as the old `ishl` relied on, so an out-of-range + // `region` behaves identically. + let region = ctx.builder.ins().ushr_imm_s(phys, crate::ppmem::BITMAP_SHIFT as i64); + let shifted = ctx.builder.ins().ushr(bits, region); + let mapped = ctx.builder.ins().band_imm_s(shifted, 1); + let mapped = ctx.builder.ins().icmp_imm_s(IntCC::NotEqual, mapped, 0); + mapped +} + +/// The tcache inline path for a tagless cache (`JitDcGeometry::tagless`, the +/// R10000's shadow cache): there is no L1-D model to probe, so once the +/// address is translated and cacheable the only question is tcache's gate, +/// and a hit is a plain access at `jit_tc_base + phys` — the same window, +/// byte lanes and (for stores, in `emit_mem_write_split`) generation bump as +/// the tagged path, less the tag match, LRU and dirty bit. +/// +/// `phys` is cut to 32 bits first, as the callout does: the shadow cache +/// hands `phys as u32` to the bus, so an address above 4GB aliases here +/// exactly as it does there. +#[cfg(feature = "tcache")] +fn emit_tagless_tail( + ctx: &mut EmitCtx, phys: Value, size: MemSize, + fast_block: ir::Block, slow_block: ir::Block, join_block: ir::Block, +) -> InlineMemPath { + let mem = MemFlagsData::trusted(); + let ptr_ty = ctx.module.target_config().pointer_type(); + + let phys = ctx.builder.ins().band_imm_s(phys, 0xFFFF_FFFF); + let mapped = emit_tc_mapped(ctx, phys); + ctx.builder.ins().brif(mapped, fast_block, &[], slow_block, &[]); + + ctx.builder.switch_to_block(fast_block); + ctx.builder.seal_block(fast_block); + let base = ctx.builder.ins().load(ptr_ty, mem, ctx.core_ptr, + ir::immediates::Offset32::new(core_offset_of_jit_tc_base())); + let swizzled = emit_swizzle_index(ctx, phys, size); + let fast_data_ptr = ctx.builder.ins().iadd(base, swizzled); + + InlineMemPath { fast_data_ptr, fast_tag_ptr: None, fast_phys: phys, slow_block, join_block } } /// Byte index within the data array/window for `size`, applying the @@ -4618,9 +4672,11 @@ fn emit_mem_write_split( // mark_l1d_dirty: one byte store into the tag we already located. Set // unconditionally rather than read-modify-write — it is idempotent, and a // load+branch to skip an already-dirty line costs more than the store. - let one = ctx.builder.ins().iconst(ir::types::I8, 1); - ctx.builder.ins().store(mem, one, path.fast_tag_ptr, - ir::immediates::Offset32::new(l1d_tag_dirty_off())); + if let Some(tag_ptr) = path.fast_tag_ptr { + let one = ctx.builder.ins().iconst(ir::types::I8, 1); + ctx.builder.ins().store(mem, one, tag_ptr, + ir::immediates::Offset32::new(l1d_tag_dirty_off())); + } // This mirrors `MipsCache::write`'s hit path exactly — that path is the // specification, and the job here is to reproduce it, not to re-derive diff --git a/src/jitv2/mod.rs b/src/jitv2/mod.rs index f67dda36..07ad74df 100644 --- a/src/jitv2/mod.rs +++ b/src/jitv2/mod.rs @@ -113,6 +113,7 @@ mod zz_corpus { // never participates. `num_lines_shift` is unread when ways == 1. ways: 1, num_lines_shift: 0, + tagless: false, }; let mut total: u64 = 0; let mut n_ok = 0u64; diff --git a/src/mips_cache_shadow.rs b/src/mips_cache_shadow.rs index 3ed40e0b..68b95b6f 100644 --- a/src/mips_cache_shadow.rs +++ b/src/mips_cache_shadow.rs @@ -146,6 +146,17 @@ pub struct ShadowCache< ic: UnsafeCell, dc: UnsafeCell, l2: UnsafeCell, + /// tcache: ppmem's window base, its mapped-region bitmap (published into + /// directly on every remap) and the jitv2 generation window. This cache + /// reads nothing through them — `read`/`write` go to `downstream`, whose + /// ppmem path uses the same window — they are here only so jitv2's inline + /// tcache path can serve this model too (`jit_dc_geometry`). + #[cfg(feature = "tcache")] + tc_base: UnsafeCell<*mut u8>, + #[cfg(feature = "tcache")] + tc_bitmap: UnsafeCell, + #[cfg(all(feature = "tcache", feature = "jitv2"))] + tc_gen: UnsafeCell<*mut std::sync::atomic::AtomicU64>, } // Safety: the CPU thread is the only accessor, as for every other cache model @@ -190,6 +201,12 @@ impl< // Only the secondary keeps a data shadow: it is the one a PROM // walks with Index_Store_Data, and a 1 MB array is cheap once. l2: UnsafeCell::new(Shadow::new(Self::L2_LINES, L2_SIZE / 8)), + #[cfg(feature = "tcache")] + tc_base: UnsafeCell::new(std::ptr::null_mut()), + #[cfg(feature = "tcache")] + tc_bitmap: UnsafeCell::new(0), + #[cfg(all(feature = "tcache", feature = "jitv2"))] + tc_gen: UnsafeCell::new(std::ptr::null_mut()), } } @@ -291,6 +308,41 @@ impl< const L2_SIZE: usize = L2_SIZE; const L2_LINE: usize = L2_LINE; + #[cfg(feature = "tcache")] + unsafe fn set_tcache_window(&self, base: *mut u8) { + unsafe { *self.tc_base.get() = base }; + } + + #[cfg(feature = "tcache")] + fn tcache_bitmap_ptr(&self) -> *mut u64 { self.tc_bitmap.get() } + + #[cfg(feature = "tcache")] + fn tcache_base_ptr(&self) -> *mut u8 { unsafe { *self.tc_base.get() } } + + #[cfg(all(feature = "tcache", feature = "jitv2"))] + fn tcache_gen_ptr(&self) -> *mut u8 { unsafe { *self.tc_gen.get() as *mut u8 } } + + #[cfg(all(feature = "tcache", feature = "jitv2"))] + unsafe fn set_tcache_gen_window(&self, gen_base: *mut std::sync::atomic::AtomicU64) { + unsafe { *self.tc_gen.get() = gen_base }; + } + + /// Under tcache, a tagless geometry: the inline path is the window alone. + /// Declined until both windows are published, as `CpuCache` does, so the + /// emitted code can assume them. Without tcache there is no inline path: + /// every access calls out, exactly as for `PassthroughCache`. + fn jit_dc_geometry(&self) -> crate::mips_cache_v2::JitDcGeometry { + #[cfg(all(feature = "tcache", feature = "jitv2"))] + if !unsafe { *self.tc_base.get() }.is_null() && !unsafe { *self.tc_gen.get() }.is_null() { + return crate::mips_cache_v2::JitDcGeometry { + supported: true, + tagless: true, + ..crate::mips_cache_v2::JitDcGeometry::unsupported() + }; + } + crate::mips_cache_v2::JitDcGeometry::unsupported() + } + fn fetch(&self, _virt_addr: u64, phys_addr: u64) -> FetchInstrResult { let r = self.downstream.read32(phys_addr as u32); if r.is_ok() { @@ -677,4 +729,22 @@ mod tests { assert_eq!(c.get_config(CACH_PD), (32768, 32)); assert_eq!(c.get_config(CACH_SD), (1048576, 128)); } + + /// Under tcache the JIT serves this model through the window alone, and + /// only once both windows are published: the emitted code dereferences + /// them without a null check. + #[test] + #[cfg(all(feature = "tcache", feature = "jitv2"))] + fn tcache_offers_a_tagless_path_once_both_windows_exist() { + let c = cache(); + assert!(!c.jit_dc_geometry().supported, "no window yet"); + let mut window = [0u8; 8]; + unsafe { c.set_tcache_window(window.as_mut_ptr()) }; + assert!(!c.jit_dc_geometry().supported, "no generation window yet"); + let mut gens = [std::sync::atomic::AtomicU64::new(0)]; + unsafe { c.set_tcache_gen_window(gens.as_mut_ptr()) }; + let g = c.jit_dc_geometry(); + assert!(g.supported && g.tagless && !g.has_l2, + "tagless, and no L2 decode slots for a store to invalidate: {g:?}"); + } } diff --git a/src/mips_cache_v2.rs b/src/mips_cache_v2.rs index 560beed8..8552c7d6 100644 --- a/src/mips_cache_v2.rs +++ b/src/mips_cache_v2.rs @@ -212,6 +212,11 @@ pub struct JitDcGeometry { /// `set | (way << num_lines_shift)` = extended tag index. Only meaningful /// when `ways > 1`. pub num_lines_shift: u32, + /// No L1-D model at all: loads and stores go straight to memory, so under + /// tcache the inline path is the ppmem-window half alone, with no tag + /// probe, no LRU and no dirty bit (the R10000's shadow cache). The other + /// fields are then unused. + pub tagless: bool, } impl JitDcGeometry { @@ -219,7 +224,7 @@ impl JitDcGeometry { Self { supported: false, line_shift: 0, num_lines_mask: 0, data_mask: 0, has_l2: false, l2_line_shift: 0, l2_num_lines_mask: 0, - ways: 1, num_lines_shift: 0, + ways: 1, num_lines_shift: 0, tagless: false, } } } @@ -3009,6 +3014,7 @@ impl Date: Tue, 29 Sep 2026 22:59:08 -0800 Subject: [PATCH 4/5] jitv2: tcache's gate tests the bitmap bit as a mask, in one load Two changes to the mapped-region test every inline tcache load and store makes: - The bitmap is read from `MipsCore::ppmem_bitmap`, where ppmem publishes it in the same statement as the cache's own copy, instead of through `MipsCore::jit_tc_bitmap`, a pointer to that copy: one load instead of two dependent ones. `jit_tc_bitmap` is gone. - The bit is tested as `bits & (1 << region)` rather than `(bits >> region) & 1`. The code-size measurement already in the comment favoured this form; on a live workload it is faster too. IP28, IRIX 6.5.22, an awk array loop, three runs each: 4.57-4.91 s before, 4.57-4.79 s with the one-load change alone, 3.99-4.02 s with both; the whole workload 17.5-18.5 s -> 16.7-16.9 s. Measured on an aarch64 host; the R4400/R5000 tagged path takes the same gate. Co-Authored-By: Claude Opus 5.5 --- src/jitv2/codegen.rs | 57 +++++++++++++++++++------------------------- src/mips_core.rs | 6 ----- src/mips_exec.rs | 1 - 3 files changed, 24 insertions(+), 40 deletions(-) diff --git a/src/jitv2/codegen.rs b/src/jitv2/codegen.rs index a98b967c..1ba475c8 100644 --- a/src/jitv2/codegen.rs +++ b/src/jitv2/codegen.rs @@ -3810,8 +3810,6 @@ fn core_offset_of_jit_dc_data() -> i32 { std::mem::offset_of!(MipsCore, jit_dc_d #[cfg(feature = "tcache")] fn core_offset_of_jit_tc_base() -> i32 { std::mem::offset_of!(MipsCore, jit_tc_base) as i32 } #[cfg(feature = "tcache")] -fn core_offset_of_jit_tc_bitmap() -> i32 { std::mem::offset_of!(MipsCore, jit_tc_bitmap) as i32 } -#[cfg(feature = "tcache")] fn core_offset_of_jit_tc_gen() -> i32 { std::mem::offset_of!(MipsCore, jit_tc_gen) as i32 } #[cfg(feature = "tcache")] fn core_offset_of_jit_l2_tags() -> i32 { std::mem::offset_of!(MipsCore, jit_l2_tags) as i32 } @@ -4028,11 +4026,10 @@ fn emit_inline_mem_guard( // ---- 4. tcache: region must be mapped RAM ------------------------ // - // `jit_tc_bitmap` is never null here: `install_jit_mem_ptrs` points it at - // the cache's own inline bitmap word, which exists from construction, and - // ppmem later publishes into that same word rather than replacing the - // pointer. No null check is emitted — one on every guest memory access to - // paper over a startup ordering problem would be paid forever. + // The bitmap is `MipsCore::ppmem_bitmap`, read at a fixed offset from the + // core: ppmem publishes the same bits there and into the cache's own + // copy in one statement on every remap, so it needs no pointer, no null + // check and no second load. #[cfg(feature = "tcache")] let proceed = { let mapped = emit_tc_mapped(ctx, phys); @@ -4120,38 +4117,32 @@ fn emit_inline_mem_guard( /// tcache's gate, step 4 of `emit_inline_mem_guard`: is `phys` in a region /// ppmem maps whole? Shared with the tagless path. +/// +/// One load: the bitmap is the core's own `ppmem_bitmap` field. Reaching it +/// through a pointer to the cache's copy put a second, dependent load on +/// every guest load and store. #[cfg(feature = "tcache")] fn emit_tc_mapped(ctx: &mut EmitCtx, phys: Value) -> Value { let mem = MemFlagsData::trusted(); - let ptr_ty = ctx.module.target_config().pointer_type(); let i64t = ir::types::I64; - let bm_ptr = ctx.builder.ins().load(ptr_ty, mem, ctx.core_ptr, - ir::immediates::Offset32::new(core_offset_of_jit_tc_bitmap())); - let bits = ctx.builder.ins().load(i64t, mem, bm_ptr, ir::immediates::Offset32::new(0)); - // Tested `(bits >> region) & 1` here as a "two fewer IR instructions" - // simplification. It is not one: measured over 500 corpus pages with - // the inline path emitting, it produced *more* code (5,901,554 -> - // 5,966,525 bytes, +1.1%). Cranelift lowers the `1 << region` form - // better — the mask feeds a `test` directly, while the shift form - // needs the variable shift's result materialized first. Left as is. - // Test the region's bit as `(bits >> region) & 1` instead of building - // `1 << region` and masking with it — same predicate, one fewer IR - // instruction (no materialized `1`). - // - // Static code size says this is slightly worse (+1.1% over 500 corpus - // pages with the inline path emitting: 5,901,554 -> 5,966,525), but - // this sits on the hot path of every guest memory access and byte - // count has repeatedly mispredicted real throughput in this codebase. - // Under evaluation on a live workload. + let bits = ctx.builder.ins().load(i64t, mem, ctx.core_ptr, + ir::immediates::Offset32::new(std::mem::offset_of!(MipsCore, ppmem_bitmap) as i32)); + // `bits & (1 << region)`, not `(bits >> region) & 1`. The shift form + // is one IR instruction shorter but not less code: over 500 corpus pages + // with the inline path emitting it produced more (5,901,554 -> 5,966,525 + // bytes, +1.1%), because Cranelift feeds the mask straight into a `test` + // while the shift form needs the variable shift's result materialized + // first. On a live workload it was also slower: an awk array loop in + // IRIX 6.5.22 on the IP28 (aarch64 host) ran 4.6-4.9 s with the shift + // form and 4.0 s with this one. // - // Shift counts are masked to 6 bits by both x86 and Cranelift's `ushr` - // definition, exactly as the old `ishl` relied on, so an out-of-range - // `region` behaves identically. + // Cranelift's `ishl` masks the shift count to 6 bits, as x86 and aarch64 + // do, so an out-of-range `region` wraps rather than being undefined. let region = ctx.builder.ins().ushr_imm_s(phys, crate::ppmem::BITMAP_SHIFT as i64); - let shifted = ctx.builder.ins().ushr(bits, region); - let mapped = ctx.builder.ins().band_imm_s(shifted, 1); - let mapped = ctx.builder.ins().icmp_imm_s(IntCC::NotEqual, mapped, 0); - mapped + let one = ctx.builder.ins().iconst(i64t, 1); + let bit = ctx.builder.ins().ishl(one, region); + let hit = ctx.builder.ins().band(bits, bit); + ctx.builder.ins().icmp_imm_s(IntCC::NotEqual, hit, 0) } /// The tcache inline path for a tagless cache (`JitDcGeometry::tagless`, the diff --git a/src/mips_core.rs b/src/mips_core.rs index 04e27715..571fb0bd 100644 --- a/src/mips_core.rs +++ b/src/mips_core.rs @@ -431,10 +431,6 @@ pub struct MipsCore { /// ppmem window base — tcache's data source (`tc_base + phys`). #[cfg(all(feature = "jitv2", feature = "tcache"))] pub jit_tc_base: *mut u8, - /// 64MB-granularity mapped-region bitmap; bit `phys >> 26` set = the - /// region is fully mapped RAM and reachable through the window. - #[cfg(all(feature = "jitv2", feature = "tcache"))] - pub jit_tc_bitmap: *const u64, /// Base of the L2 tag array (`[L2Tag]`, `u32` each). The inline tcache /// store clears `has_code` on the written line so L1-I refills cannot /// reuse stale decoded instructions — mirroring `tc_invalidate_l2_code`. @@ -1292,8 +1288,6 @@ impl MipsCore { #[cfg(all(feature = "jitv2", feature = "tcache"))] jit_tc_base: std::ptr::null_mut(), #[cfg(all(feature = "jitv2", feature = "tcache"))] - jit_tc_bitmap: std::ptr::null(), - #[cfg(all(feature = "jitv2", feature = "tcache"))] jit_l2_tags: std::ptr::null_mut(), #[cfg(all(feature = "jitv2", feature = "tcache"))] jit_tc_gen: std::ptr::null_mut(), diff --git a/src/mips_exec.rs b/src/mips_exec.rs index 9f61d999..f70a8471 100644 --- a/src/mips_exec.rs +++ b/src/mips_exec.rs @@ -3094,7 +3094,6 @@ impl MipsExecutor { #[cfg(feature = "tcache")] { self.core.jit_tc_base = self.cache.tcache_base_ptr(); - self.core.jit_tc_bitmap = self.cache.tcache_bitmap_ptr() as *const u64; self.core.jit_tc_gen = self.cache.tcache_gen_ptr(); self.core.jit_l2_tags = self.cache.jit_l2_tags_ptr(); } From ed6562d625b443504d4a9b98503476d180d22103 Mon Sep 17 00:00:00 2001 From: atomchild411 <143453386+atomchild411@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:48:57 -0800 Subject: [PATCH 5/5] monitor: inject mouse input; status bar: count the kernel's clock ticks `ps2 mouse [buttons]` pushes one mouse packet (buttons: 1 left, 2 right, 4 middle), so a desktop can be driven from the monitor or a script without a window to click in. The status bar's Hz now counts 8254 timer 0/1 interrupts as well. IRIX keeps time with the 8254 unless the IOC's is known broken, in which case it uses CP0 Compare, which was already counted; either way the number is the kernel's tick rate, and the two never both run. Co-Authored-By: Claude Opus 5.5 --- src/ioc.rs | 20 ++++++++++++++++++++ src/machine.rs | 1 + src/ps2.rs | 14 ++++++++++++-- 3 files changed, 33 insertions(+), 2 deletions(-) diff --git a/src/ioc.rs b/src/ioc.rs index f46b71f0..3019d8ce 100644 --- a/src/ioc.rs +++ b/src/ioc.rs @@ -384,10 +384,15 @@ struct IocIrqLine { struct IocTimerCallback { state: Arc>, source: IocInterrupt, + /// The status bar's clock-tick counter (see `Ioc::set_clock_ticks`). + ticks: Arc>>, } impl TimerCallback for IocTimerCallback { fn callback(&self) { + if let Some(t) = self.ticks.get() { + t.fetch_add(1, Ordering::Relaxed); + } let mut state = self.state.lock(); match self.source { IocInterrupt::Mappable0 => state.map_stat |= 1 << 0, @@ -427,6 +432,8 @@ pub struct Ioc { event_tx: Arc>>, /// Shared heartbeat — IOC sets/clears HB_LED_RED/GREEN bits directly. heartbeat: Arc>>, + /// Counter bumped on every 8254 timer 0/1 interrupt (see `set_clock_ticks`). + clock_ticks: Arc>>, /// Shared timer manager for PIT channels. timer_manager: Arc>>, } @@ -496,14 +503,17 @@ impl Ioc { source: IocInterrupt::Serial, }); + let clock_ticks = Arc::new(std::sync::OnceLock::new()); let timer0_cb = Arc::new(IocTimerCallback { state: state.clone(), source: IocInterrupt::Mappable0, + ticks: clock_ticks.clone(), }); let timer1_cb = Arc::new(IocTimerCallback { state: state.clone(), source: IocInterrupt::Mappable1, + ticks: clock_ticks.clone(), }); let ps2_cb = Arc::new(IocIrqLine { @@ -525,6 +535,7 @@ impl Ioc { guinness, event_tx: Arc::new(std::sync::OnceLock::new()), heartbeat: Arc::new(std::sync::OnceLock::new()), + clock_ticks, timer_manager: Arc::new(std::sync::OnceLock::new()), } } @@ -538,6 +549,15 @@ impl Ioc { let _ = self.event_tx.set(tx); } + /// Count the kernel's clock ticks into the status bar's Hz counter. IRIX + /// keeps time with the 8254 (timer 0 is the system clock, timer 1 the + /// profiling clock) unless the IOC's 8254 is known broken, in which case + /// it uses CP0 Compare, which the CPU already counts. Either way the + /// counter shows the kernel's tick rate, and never both at once. + pub fn set_clock_ticks(&self, ticks: Arc) { + let _ = self.clock_ticks.set(ticks); + } + pub fn set_heartbeat(&self, heartbeat: Arc) { let _ = self.heartbeat.set(heartbeat); } diff --git a/src/machine.rs b/src/machine.rs index a2a818f0..2da8a8d4 100644 --- a/src/machine.rs +++ b/src/machine.rs @@ -398,6 +398,7 @@ impl Machine { let timer_manager = Arc::new(TimerManager::new()); ioc.set_timer_manager(timer_manager.clone()); ioc.set_heartbeat(heartbeat.clone()); + ioc.set_clock_ticks(fasttick_count.clone()); let hpc3 = Hpc3::with_net(eeprom_hpc3.clone(), ioc.clone(), guinness, heartbeat.clone(), cfg.network(), cfg.no_audio, cfg.audio.clone(), cfg.nvram.clone(), cfg.rtc_offset, cfg.scsi_deferred_int); hpc3.set_timer_manager(timer_manager.clone()); diff --git a/src/ps2.rs b/src/ps2.rs index c321403b..e707c66e 100644 --- a/src/ps2.rs +++ b/src/ps2.rs @@ -917,7 +917,7 @@ impl Device for Ps2Controller { fn get_clock(&self) -> u64 { 0 } fn register_commands(&self) -> Vec<(String, String)> { - vec![("ps2".to_string(), "PS/2 commands: ps2 debug | ps2 type | ps2 enter | ps2 status".to_string())] + vec![("ps2".to_string(), "PS/2 commands: ps2 debug | ps2 type | ps2 enter | ps2 mouse [buttons] | ps2 status".to_string())] } fn execute_command(&self, cmd: &str, args: &[&str], mut writer: Box) -> Result<(), String> { @@ -953,6 +953,16 @@ impl Device for Ps2Controller { writeln!(writer, "PS/2: pressed Enter").unwrap(); return Ok(()); } + if !args.is_empty() && args[0] == "mouse" { + let n = |i: usize| args.get(i).and_then(|v| v.parse::().ok()); + let (Some(dx), Some(dy)) = (n(1), n(2)) else { + return Err("Usage: ps2 mouse [buttons: 1 left, 2 right, 4 middle]".to_string()); + }; + let buttons = n(3).unwrap_or(0) as u8; + self.push_mouse_input(buttons, dx, dy, 0); + writeln!(writer, "PS/2: mouse {} {} buttons {}", dx, dy, buttons).unwrap(); + return Ok(()); + } if !args.is_empty() && args[0] == "status" { let s = self.state.lock(); writeln!(writer, "PS/2 state: running={} scanning_enabled={} mouse_enabled={} mouse_id={} rx_queue_len={} mouse_queue_bytes={} scancode_set={} config={:02x} last_read={:02x}", @@ -961,7 +971,7 @@ impl Device for Ps2Controller { s.mouse_queue_bytes, s.scancode_set, s.config, s.last_read).unwrap(); return Ok(()); } - return Err("Usage: ps2 debug | ps2 type | ps2 enter | ps2 status".to_string()); + return Err("Usage: ps2 debug | ps2 type | ps2 enter | ps2 mouse [buttons] | ps2 status".to_string()); } Err("Command not found".to_string()) }