diff --git a/Cargo.toml b/Cargo.toml index 25b8e9ae..116ea07c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -191,6 +191,12 @@ jitv2 = ["cranelift-codegen", "cranelift-frontend", "cranelift-jit", "cranelift- # specifically; opt in with --features jitv2_opcodefusion once it's earned # more confidence. jitv2_opcodefusion = ["jitv2"] +# IP28: the Indigo2 IMPACT machine (IP28) and its R10000 CPU. Without it the +# IP28 profile and the R10000 model are refused at startup with a message +# naming this feature, and none of their code is built, so nothing behind it +# can reach the R4400/R5000 machines. It implies ppmem: IP28 guest RAM is +# always mapped through the host MMU. Add jitv2 for the JIT, as elsewhere. +ip28 = ["ppmem"] # jitv2 whole-page compile (§13, rules/jitv2/jit-v2-design.md): whole-page # compilation is now the standard and only JIT v2 design. Kept as a compatibility # alias for `jitv2` so downstream scripts and builds specifying `--features j2wp` diff --git a/ip28.toml.example b/ip28.toml.example new file mode 100644 index 00000000..3dcb3841 --- /dev/null +++ b/ip28.toml.example @@ -0,0 +1,17 @@ +# Indigo2 IMPACT R10000 (IP28) bring-up. +# Runs the real IP28 PROM against the existing fullhouse (IP22) machine. +# The CPU is still an R4400/R5000 model — the point of this run is to find +# out what the PROM objects to first, not to boot anything. +headless = true +no_audio = true +banks = [64, 64, 64, 64] +prom = "ip28/ip28prom.070-1477-002.bin" +nveeprom = "ip28/nveeprom-ip28.bin" +serial_log = "ip28/console-ip28.log" +[machine] +profile = "indigo2_ip22" +cpu = "r10000" +[scsi.1] +path = "ip28/blank.raw" +cdrom = false +overlay = true diff --git a/src/config.rs b/src/config.rs index f1c7fa78..4f794611 100644 --- a/src/config.rs +++ b/src/config.rs @@ -3,7 +3,7 @@ use serde::{Deserialize, Serialize}; use std::net::Ipv4Addr; /// Valid memory bank sizes in MB. -pub const VALID_BANK_SIZES: &[u32] = &[0, 8, 16, 32, 64, 128]; +pub const VALID_BANK_SIZES: &[u32] = &[0, 8, 16, 32, 64, 128, 256]; /// What sits at a SCSI id. `cdrom = true` remains the historical spelling for /// `kind = "cdrom"`; either works and they mean the same thing. @@ -384,28 +384,53 @@ pub enum MachineProfile { IndyIp24, /// SGI Indigo2 IP22 — fullhouse MC/IOC, Newport XL on GIO gfx slot. Indigo2Ip22, + /// SGI Indigo2 IMPACT IP28 — an R10000 CPU module in the Indigo2 chassis. + /// + /// Shares the fullhouse MC/IOC/HPC3 with IP22 and differs in the decodes + /// inside them: MEMCFG's base field is shifted by 24 rather than 22 (so + /// its size granule is 16 MB, not 4), RAM lives at 0x20000000 with the + /// low-memory alias following it there, and both the MC chip revision and + /// the HPC3 board revision have to read high enough for the kernel to + /// call the board an IP28. + /// + /// Graphics is IMPACT, which is a register stub — an IP28 kernel carries + /// no Newport driver, so REX3 is not an alternative here. + Indigo2Ip28, } impl MachineProfile { /// All selectable profiles, in display order. Single source of truth for the /// GUI dropdowns (Config tab + New Machine dialog) so they never drift. + #[cfg(feature = "ip28")] + pub const ALL: [Self; 3] = [Self::IndyIp24, Self::Indigo2Ip22, Self::Indigo2Ip28]; + #[cfg(not(feature = "ip28"))] pub const ALL: [Self; 2] = [Self::IndyIp24, Self::Indigo2Ip22]; pub fn label(self) -> &'static str { match self { Self::IndyIp24 => "SGI Indy (IP24)", Self::Indigo2Ip22 => "SGI Indigo2 (IP22)", + Self::Indigo2Ip28 => "SGI Indigo2 IMPACT (IP28)", } } pub fn supported(self) -> bool { matches!(self, Self::IndyIp24 | Self::Indigo2Ip22) + || (cfg!(feature = "ip28") && matches!(self, Self::Indigo2Ip28)) } /// MC/IOC/HPC3 Guinness vs Fullhouse layout. Indy IP24 is Guinness (`true`). pub fn guinness(self) -> bool { matches!(self, Self::IndyIp24) } + + /// The R10000 Indigo2. Selects the IP28 decodes inside the shared + /// fullhouse devices — see the variant's own documentation for the list. + /// + /// Always false without the `ip28` feature, so every IP28 decode folds away. + pub fn ip28(self) -> bool { + cfg!(feature = "ip28") && matches!(self, Self::Indigo2Ip28) + } } /// Indy / Indigo2 graphics board in the GIO gfx slot. @@ -469,23 +494,10 @@ impl ImpactSection { || self.exp1 != ImpactSlot::None } - /// Hardware-valid slot population (rejects High+High and orphan expansion boards). + /// One IMPACT board, in the graphics slot; a second head is not modelled yet. pub fn validate(&self) -> Result<(), String> { - let slots = [self.gfx, self.exp0, self.exp1]; - let high_count = slots.iter().filter(|&&s| s == ImpactSlot::High).count(); - if high_count >= 2 { - return Err( - "[impact] High+High is invalid — at most one High IMPACT board per system".into(), - ); - } - if self.exp0 != ImpactSlot::None && self.gfx == ImpactSlot::None { - return Err("[impact] exp0 requires gfx slot populated".into()); - } - if self.exp1 != ImpactSlot::None && self.exp0 == ImpactSlot::None { - return Err("[impact] exp1 requires exp0 populated (Maximum IMPACT uses all three slots)".into()); - } - if self.exp1 == ImpactSlot::Max && self.exp0 != ImpactSlot::High { - return Err("[impact] Maximum IMPACT expects exp0=high when exp1=max".into()); + if self.exp0 != ImpactSlot::None || self.exp1 != ImpactSlot::None { + return Err("[impact] only the graphics slot (gfx) is supported so far".into()); } Ok(()) } @@ -538,12 +550,27 @@ pub enum CpuModel { R4400, /// MIPS R5000, 2-way 32K L1s, no secondary cache, MIPS IV. R5000, + /// MIPS R10000, 2-way 32K L1s, 1 MB secondary cache, MIPS IV. The CPU in + /// the Indigo2 IMPACT (IP28). Bring-up only — see docs/ip28-bringup.md. + R10000, } impl CpuModel { + #[cfg(feature = "ip28")] + pub const ALL: [Self; 3] = [Self::R4400, Self::R5000, Self::R10000]; + #[cfg(not(feature = "ip28"))] pub const ALL: [Self; 2] = [Self::R4400, Self::R5000]; + + /// Whether this build can run the model: the R10000 needs the `ip28` feature. + pub fn available(self) -> bool { + !matches!(self, Self::R10000) || cfg!(feature = "ip28") + } pub fn label(self) -> &'static str { - match self { Self::R4400 => "MIPS R4400", Self::R5000 => "MIPS R5000" } + match self { + Self::R4400 => "MIPS R4400", + Self::R5000 => "MIPS R5000", + Self::R10000 => "MIPS R10000", + } } } @@ -1218,6 +1245,11 @@ impl MachineConfig { /// Validate bank sizes, returns a description of any errors. pub fn validate(&self) -> Result<(), String> { + if (self.machine.profile == MachineProfile::Indigo2Ip28 && !cfg!(feature = "ip28")) + || !self.machine.cpu.available() + { + return Err("IP28 / R10000 support is not built into this binary; rebuild with --features ip28".to_string()); + } if !self.machine.profile.supported() { return Err(format!( "machine profile \"{}\" is not implemented; use {}", @@ -1248,9 +1280,11 @@ impl MachineConfig { return Err(format!("graphics.board \"{name}\" and [impact] both claim the GIO gfx slot")); } } - if self.impact.any_enabled() && self.machine.profile != MachineProfile::Indigo2Ip22 { + if self.impact.any_enabled() + && !matches!(self.machine.profile, MachineProfile::Indigo2Ip22 | MachineProfile::Indigo2Ip28) + { return Err( - "[impact] slots are preview-only on Indigo2 (machine.profile = indigo2_ip22)".into(), + "[impact] slots need an Indigo2 (machine.profile = indigo2_ip22 or indigo2_ip28)".into(), ); } self.impact.validate()?; @@ -1275,6 +1309,14 @@ impl MachineConfig { i, sz, VALID_BANK_SIZES )); } + // The IP22/IP24 MC cannot express a 256 MB bank at its base + // shift; only the IP28's can. + if sz == 256 && !self.machine.profile.ip28() { + return Err(format!( + "bank{} size 256 MB needs the IP28 (machine.profile = indigo2_ip28)", + i + )); + } } if let Some(ref s) = self.nat_subnet { if let Err(e) = parse_nat_subnet(s) { @@ -1823,6 +1865,21 @@ mod export_tests { assert!(cfg.machine.profile.supported()); assert!(!cfg.machine.profile.guinness()); } + + #[test] + fn a_256_mb_bank_is_an_ip28_bank() { + let mut cfg = MachineConfig::default(); + cfg.machine.profile = MachineProfile::Indigo2Ip22; + cfg.banks = [256, 128, 0, 0]; + let err = cfg.validate().expect_err("the IP22 MC cannot express a 256 MB bank"); + assert!(err.contains("256 MB"), "{err}"); + #[cfg(feature = "ip28")] + { + cfg.machine.profile = MachineProfile::Indigo2Ip28; + cfg.machine.cpu = CpuModel::R10000; + cfg.validate().expect("the IP28 MC can"); + } + } } #[cfg(test)] diff --git a/src/devlog.rs b/src/devlog.rs index fbda423a..e27dc116 100644 --- a/src/devlog.rs +++ b/src/devlog.rs @@ -155,7 +155,8 @@ impl LogModule { if mask & 0x0002 != 0 { parts.push("tlb"); } if mask & 0x0004 != 0 { parts.push("mem"); } if mask & 0x0008 != 0 { parts.push("fpu"); } - let rest = mask & !0x000F; + if mask & 0x0010 != 0 { parts.push("cp0"); } + let rest = mask & !0x001F; let mut s = parts.join("+"); if rest != 0 { s.push_str(&format!("+{:#010x}", rest)); } if s.is_empty() { format!("{:#010x}", mask) } else { s } @@ -192,6 +193,7 @@ impl LogModule { "tlb" => Some(0x0002), "mem" => Some(0x0004), "fpu" => Some(0x0008), + "cp0" => Some(0x0010), "on" | "all" => Some(0xFFFF_FFFF), "off" | "none" => Some(0x0000_0000), _ => u32::from_str_radix(s.trim_start_matches("0x"), 16).ok(), @@ -458,7 +460,7 @@ impl Device for DevLog { "[DEV] log | log mask | log file | log status\n\ \x20 modules: net hpc3 seeq hal2 mc rex3 mips ioc scsi pdma vino dcb vc2 cmap xmap bt445 scc ps2 rtc eeprom l1i l1d l2c\n\ \x20 pdma mask categories: hal enet scsi on/all off/none \n\ - \x20 mips mask categories: insn tlb mem fpu on/all off/none \n\ + \x20 mips mask categories: insn tlb mem fpu cp0 on/all off/none \n\ \x20 l1i/l1d/l2c mask categories: hit miss op on/all off/none \n\ \x20 [DEV] = requires --features developer build to produce output".to_string(), )] diff --git a/src/ioc.rs b/src/ioc.rs index 70549add..3019d8ce 100644 --- a/src/ioc.rs +++ b/src/ioc.rs @@ -384,10 +384,15 @@ struct IocIrqLine { struct IocTimerCallback { state: Arc>, source: IocInterrupt, + /// The status bar's clock-tick counter (see `Ioc::set_clock_ticks`). + ticks: Arc>>, } impl TimerCallback for IocTimerCallback { fn callback(&self) { + if let Some(t) = self.ticks.get() { + t.fetch_add(1, Ordering::Relaxed); + } let mut state = self.state.lock(); match self.source { IocInterrupt::Mappable0 => state.map_stat |= 1 << 0, @@ -427,6 +432,8 @@ pub struct Ioc { event_tx: Arc>>, /// Shared heartbeat — IOC sets/clears HB_LED_RED/GREEN bits directly. heartbeat: Arc>>, + /// Counter bumped on every 8254 timer 0/1 interrupt (see `set_clock_ticks`). + clock_ticks: Arc>>, /// Shared timer manager for PIT channels. timer_manager: Arc>>, } @@ -444,8 +451,29 @@ impl Ioc { Self::new_inner(guinness, true) } + /// As `new`/`new_ci`, with the machine profile's IP28 flag: an IP28 + /// baseboard must report a high enough HPC3 board revision. + pub fn new_for_profile(guinness: bool, ci_mode: bool, ip28: bool) -> Self { + Self::new_inner_profile(guinness, ci_mode, ip28) + } + fn new_inner(guinness: bool, ci_mode: bool) -> Self { - let sys_id = if guinness { 0x26 } else { 0x11 }; // primarily prom looks at bit 1 to detect full house. + Self::new_inner_profile(guinness, ci_mode, false) + } + + fn new_inner_profile(guinness: bool, ci_mode: bool, ip28: bool) -> Self { + // HPC3 SYS_ID: [7:5] chip rev, [4:1] board rev, [0] 1 = fullhouse. + // The PROM looks at bit 0 to tell fullhouse from guinness. IRIX reads + // the board revision to tell an IP28 baseboard from an IP26 one and + // warns "CPU baseboard downrev (IP26 not IP28)" below 13, so an IP28 + // has to report at least that. + let sys_id: u8 = if guinness { + 0x26 + } else if ip28 { + 0x1B // board rev 13, fullhouse + } else { + 0x11 // board rev 8, fullhouse + }; let state = Arc::new(Mutex::new(IocState { sys_id, l0_stat: 0, @@ -475,14 +503,17 @@ impl Ioc { source: IocInterrupt::Serial, }); + let clock_ticks = Arc::new(std::sync::OnceLock::new()); let timer0_cb = Arc::new(IocTimerCallback { state: state.clone(), source: IocInterrupt::Mappable0, + ticks: clock_ticks.clone(), }); let timer1_cb = Arc::new(IocTimerCallback { state: state.clone(), source: IocInterrupt::Mappable1, + ticks: clock_ticks.clone(), }); let ps2_cb = Arc::new(IocIrqLine { @@ -504,6 +535,7 @@ impl Ioc { guinness, event_tx: Arc::new(std::sync::OnceLock::new()), heartbeat: Arc::new(std::sync::OnceLock::new()), + clock_ticks, timer_manager: Arc::new(std::sync::OnceLock::new()), } } @@ -517,6 +549,15 @@ impl Ioc { let _ = self.event_tx.set(tx); } + /// Count the kernel's clock ticks into the status bar's Hz counter. IRIX + /// keeps time with the 8254 (timer 0 is the system clock, timer 1 the + /// profiling clock) unless the IOC's 8254 is known broken, in which case + /// it uses CP0 Compare, which the CPU already counts. Either way the + /// counter shows the kernel's tick rate, and never both at once. + pub fn set_clock_ticks(&self, ticks: Arc) { + let _ = self.clock_ticks.set(ticks); + } + pub fn set_heartbeat(&self, heartbeat: Arc) { let _ = self.heartbeat.set(heartbeat); } diff --git a/src/jitv2/codegen.rs b/src/jitv2/codegen.rs index 10a4de31..1ba475c8 100644 --- a/src/jitv2/codegen.rs +++ b/src/jitv2/codegen.rs @@ -3810,8 +3810,6 @@ fn core_offset_of_jit_dc_data() -> i32 { std::mem::offset_of!(MipsCore, jit_dc_d #[cfg(feature = "tcache")] fn core_offset_of_jit_tc_base() -> i32 { std::mem::offset_of!(MipsCore, jit_tc_base) as i32 } #[cfg(feature = "tcache")] -fn core_offset_of_jit_tc_bitmap() -> i32 { std::mem::offset_of!(MipsCore, jit_tc_bitmap) as i32 } -#[cfg(feature = "tcache")] fn core_offset_of_jit_tc_gen() -> i32 { std::mem::offset_of!(MipsCore, jit_tc_gen) as i32 } #[cfg(feature = "tcache")] fn core_offset_of_jit_l2_tags() -> i32 { std::mem::offset_of!(MipsCore, jit_l2_tags) as i32 } @@ -3834,8 +3832,9 @@ fn l1d_tag_dirty_off() -> i32 { std::mem::offset_of!(crate::mips_cache_v2::L1DTa struct InlineMemPath { fast_data_ptr: Value, /// Pointer to the matched L1-D tag, live in the fast block. Stores use it - /// to set the dirty flag, mirroring `mark_l1d_dirty`. - fast_tag_ptr: Value, + /// to set the dirty flag, mirroring `mark_l1d_dirty`. `None` on a tagless + /// cache, which has no L1-D model to update. + fast_tag_ptr: Option, /// Physical address of the access, live in the fast block. Needed by the /// tcache store path to index the jitv2 generation array. fast_phys: Value, @@ -3866,6 +3865,11 @@ fn emit_inline_mem_guard( if !geom.supported { return None; } + // A tagless cache has an inline path only through tcache's window. + #[cfg(not(feature = "tcache"))] + if geom.tagless { + return None; + } let mem = MemFlagsData::trusted(); let ptr_ty = ctx.module.target_config().pointer_type(); @@ -3956,6 +3960,11 @@ fn emit_inline_mem_guard( let va_off = ctx.builder.ins().band_imm_s(vaddr, 0xFFF); let phys = ctx.builder.ins().bor(phys_page, va_off); + #[cfg(feature = "tcache")] + if geom.tagless { + return Some(emit_tagless_tail(ctx, phys, size, fast_block, slow_block, join_block)); + } + // ---- 2. L1D tag match -------------------------------------------- // // Mirrors `ensure_l1d_line`'s hit path exactly (mips_cache_v2.rs). On a @@ -4017,39 +4026,13 @@ fn emit_inline_mem_guard( // ---- 4. tcache: region must be mapped RAM ------------------------ // - // `jit_tc_bitmap` is never null here: `install_jit_mem_ptrs` points it at - // the cache's own inline bitmap word, which exists from construction, and - // ppmem later publishes into that same word rather than replacing the - // pointer. No null check is emitted — one on every guest memory access to - // paper over a startup ordering problem would be paid forever. + // The bitmap is `MipsCore::ppmem_bitmap`, read at a fixed offset from the + // core: ppmem publishes the same bits there and into the cache's own + // copy in one statement on every remap, so it needs no pointer, no null + // check and no second load. #[cfg(feature = "tcache")] let proceed = { - let bm_ptr = ctx.builder.ins().load(ptr_ty, mem, ctx.core_ptr, - ir::immediates::Offset32::new(core_offset_of_jit_tc_bitmap())); - let bits = ctx.builder.ins().load(i64t, mem, bm_ptr, ir::immediates::Offset32::new(0)); - // Tested `(bits >> region) & 1` here as a "two fewer IR instructions" - // simplification. It is not one: measured over 500 corpus pages with - // the inline path emitting, it produced *more* code (5,901,554 -> - // 5,966,525 bytes, +1.1%). Cranelift lowers the `1 << region` form - // better — the mask feeds a `test` directly, while the shift form - // needs the variable shift's result materialized first. Left as is. - // Test the region's bit as `(bits >> region) & 1` instead of building - // `1 << region` and masking with it — same predicate, one fewer IR - // instruction (no materialized `1`). - // - // Static code size says this is slightly worse (+1.1% over 500 corpus - // pages with the inline path emitting: 5,901,554 -> 5,966,525), but - // this sits on the hot path of every guest memory access and byte - // count has repeatedly mispredicted real throughput in this codebase. - // Under evaluation on a live workload. - // - // Shift counts are masked to 6 bits by both x86 and Cranelift's `ushr` - // definition, exactly as the old `ishl` relied on, so an out-of-range - // `region` behaves identically. - let region = ctx.builder.ins().ushr_imm_s(phys, crate::ppmem::BITMAP_SHIFT as i64); - let shifted = ctx.builder.ins().ushr(bits, region); - let mapped = ctx.builder.ins().band_imm_s(shifted, 1); - let mapped = ctx.builder.ins().icmp_imm_s(IntCC::NotEqual, mapped, 0); + let mapped = emit_tc_mapped(ctx, phys); ctx.builder.ins().band(proceed, mapped) }; @@ -4129,7 +4112,69 @@ fn emit_inline_mem_guard( let swizzled = emit_swizzle_index(ctx, index, size); let fast_data_ptr = ctx.builder.ins().iadd(base, swizzled); - Some(InlineMemPath { fast_data_ptr, fast_tag_ptr: tag_ptr, fast_phys: phys, slow_block, join_block }) + Some(InlineMemPath { fast_data_ptr, fast_tag_ptr: Some(tag_ptr), fast_phys: phys, slow_block, join_block }) +} + +/// tcache's gate, step 4 of `emit_inline_mem_guard`: is `phys` in a region +/// ppmem maps whole? Shared with the tagless path. +/// +/// One load: the bitmap is the core's own `ppmem_bitmap` field. Reaching it +/// through a pointer to the cache's copy put a second, dependent load on +/// every guest load and store. +#[cfg(feature = "tcache")] +fn emit_tc_mapped(ctx: &mut EmitCtx, phys: Value) -> Value { + let mem = MemFlagsData::trusted(); + let i64t = ir::types::I64; + let bits = ctx.builder.ins().load(i64t, mem, ctx.core_ptr, + ir::immediates::Offset32::new(std::mem::offset_of!(MipsCore, ppmem_bitmap) as i32)); + // `bits & (1 << region)`, not `(bits >> region) & 1`. The shift form + // is one IR instruction shorter but not less code: over 500 corpus pages + // with the inline path emitting it produced more (5,901,554 -> 5,966,525 + // bytes, +1.1%), because Cranelift feeds the mask straight into a `test` + // while the shift form needs the variable shift's result materialized + // first. On a live workload it was also slower: an awk array loop in + // IRIX 6.5.22 on the IP28 (aarch64 host) ran 4.6-4.9 s with the shift + // form and 4.0 s with this one. + // + // Cranelift's `ishl` masks the shift count to 6 bits, as x86 and aarch64 + // do, so an out-of-range `region` wraps rather than being undefined. + let region = ctx.builder.ins().ushr_imm_s(phys, crate::ppmem::BITMAP_SHIFT as i64); + let one = ctx.builder.ins().iconst(i64t, 1); + let bit = ctx.builder.ins().ishl(one, region); + let hit = ctx.builder.ins().band(bits, bit); + ctx.builder.ins().icmp_imm_s(IntCC::NotEqual, hit, 0) +} + +/// The tcache inline path for a tagless cache (`JitDcGeometry::tagless`, the +/// R10000's shadow cache): there is no L1-D model to probe, so once the +/// address is translated and cacheable the only question is tcache's gate, +/// and a hit is a plain access at `jit_tc_base + phys` — the same window, +/// byte lanes and (for stores, in `emit_mem_write_split`) generation bump as +/// the tagged path, less the tag match, LRU and dirty bit. +/// +/// `phys` is cut to 32 bits first, as the callout does: the shadow cache +/// hands `phys as u32` to the bus, so an address above 4GB aliases here +/// exactly as it does there. +#[cfg(feature = "tcache")] +fn emit_tagless_tail( + ctx: &mut EmitCtx, phys: Value, size: MemSize, + fast_block: ir::Block, slow_block: ir::Block, join_block: ir::Block, +) -> InlineMemPath { + let mem = MemFlagsData::trusted(); + let ptr_ty = ctx.module.target_config().pointer_type(); + + let phys = ctx.builder.ins().band_imm_s(phys, 0xFFFF_FFFF); + let mapped = emit_tc_mapped(ctx, phys); + ctx.builder.ins().brif(mapped, fast_block, &[], slow_block, &[]); + + ctx.builder.switch_to_block(fast_block); + ctx.builder.seal_block(fast_block); + let base = ctx.builder.ins().load(ptr_ty, mem, ctx.core_ptr, + ir::immediates::Offset32::new(core_offset_of_jit_tc_base())); + let swizzled = emit_swizzle_index(ctx, phys, size); + let fast_data_ptr = ctx.builder.ins().iadd(base, swizzled); + + InlineMemPath { fast_data_ptr, fast_tag_ptr: None, fast_phys: phys, slow_block, join_block } } /// Byte index within the data array/window for `size`, applying the @@ -4618,9 +4663,11 @@ fn emit_mem_write_split( // mark_l1d_dirty: one byte store into the tag we already located. Set // unconditionally rather than read-modify-write — it is idempotent, and a // load+branch to skip an already-dirty line costs more than the store. - let one = ctx.builder.ins().iconst(ir::types::I8, 1); - ctx.builder.ins().store(mem, one, path.fast_tag_ptr, - ir::immediates::Offset32::new(l1d_tag_dirty_off())); + if let Some(tag_ptr) = path.fast_tag_ptr { + let one = ctx.builder.ins().iconst(ir::types::I8, 1); + ctx.builder.ins().store(mem, one, tag_ptr, + ir::immediates::Offset32::new(l1d_tag_dirty_off())); + } // This mirrors `MipsCache::write`'s hit path exactly — that path is the // specification, and the job here is to reproduce it, not to re-derive diff --git a/src/jitv2/mod.rs b/src/jitv2/mod.rs index f67dda36..07ad74df 100644 --- a/src/jitv2/mod.rs +++ b/src/jitv2/mod.rs @@ -113,6 +113,7 @@ mod zz_corpus { // never participates. `num_lines_shift` is unread when ways == 1. ways: 1, num_lines_shift: 0, + tagless: false, }; let mut total: u64 = 0; let mut n_ok = 0u64; diff --git a/src/lib.rs b/src/lib.rs index 3ab7c2ac..91d98e8b 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -172,6 +172,8 @@ pub mod mips_dis; pub mod mips_core; pub mod mips_tlb; pub mod mips_cache_v2; +#[cfg(feature = "ip28")] +pub mod mips_cache_shadow; pub mod mips_exec; pub mod mips_exec_test; pub mod mips_instr_stats; diff --git a/src/machine.rs b/src/machine.rs index dfde50b7..2da8a8d4 100644 --- a/src/machine.rs +++ b/src/machine.rs @@ -23,10 +23,16 @@ impl PhysPtr { use crate::physical::RamBank; use crate::prom::Prom; use crate::mc::MemoryController; + +/// Count on the IP28's 195 MHz R10000: half the pipeline clock, and what +/// IRIX assumes there (see where the MC is created). +const IP28_COUNT_HZ: u64 = 97_500_000; use crate::mips_tlb::MipsTlb; use crate::mips_exec::{MipsExecutor, MipsCpu, MipsCpuConfig, MipsCpuDebugAdapter}; use crate::gdb_stub::CpuDebug; use crate::mips_cache_v2::{MipsCache, R4400Cache, R5000Cache}; +#[cfg(feature = "ip28")] +use crate::mips_cache_shadow::R10000ShadowCache; use crate::hpc3::Hpc3; use crate::ioc::{Ioc, GioSlot, GIO_SLOT_MAP, profile_idx}; use crate::monitor::Monitor; @@ -257,6 +263,12 @@ impl Machine { let clock_fixed_mhz = cfg.clock.fixed_mhz; let cfg_cpu_model = cfg.machine.cpu; + if (cfg.machine.profile == MachineProfile::Indigo2Ip28 && !cfg!(feature = "ip28")) + || !cfg_cpu_model.available() + { + eprintln!("iris: IP28 / R10000 support is not built into this binary; rebuild with --features ip28"); + std::process::exit(1); + } if !cfg.machine.profile.supported() { eprintln!( "iris: machine profile \"{}\" is not implemented; use {}", @@ -284,6 +296,10 @@ impl Machine { let model_has_l2 = match cfg_cpu_model { crate::config::CpuModel::R4400 => ::L2_SIZE > 0, crate::config::CpuModel::R5000 => ::L2_SIZE > 0, + #[cfg(feature = "ip28")] + crate::config::CpuModel::R10000 => ::L2_SIZE > 0, + #[cfg(not(feature = "ip28"))] + crate::config::CpuModel::R10000 => unreachable!("refused above without the ip28 feature"), }; if !model_has_l2 { eeprom_mc.lock().set_cachsz(0); @@ -291,7 +307,24 @@ impl Machine { // 1. Create all devices first // Memory Controller - let mc = MemoryController::new(eeprom_mc.clone(), guinness, cfg.banks); + let mc = MemoryController::new_for_profile( + eeprom_mc.clone(), guinness, cfg.banks, cfg.machine.profile.ip28()); + // IP28: IRIX takes the CPU speed from the PROM's `cpufreq` (194, i.e. + // a 195 MHz R10000) instead of measuring it, and times with it: UST + // assumes Count at half that, and the RPSS cycle counter the MC clock + // at the CPU clock (Config.EC is 0, a 1:1 system-clock ratio) over the + // divider it programs. The generic 33 MHz Count and 50 MHz MC made + // UST run at a third of real time (Quake in slow motion) and the + // cycle counter at a quarter. + let ip28_count_hz: Option = if cfg.machine.profile.ip28() { + Some(clock_fixed_mhz.map_or(IP28_COUNT_HZ, |mhz| (mhz * 1_000_000.0) as u64)) + } else { + None + }; + if let Some(count_hz) = ip28_count_hz { + // The R10000's Count ticks at half the pipeline clock. + mc.set_clock_hz(2 * count_hz); + } // RAM banks sized per config. addr_mask is initialized to mem_size-1; // remap_banks() updates it via set_addr_mask() when MEMCFG0/1 are written during POST. @@ -325,7 +358,7 @@ impl Machine { // HPC3 (512KB at 0x1FB80000). CI mode skips the SCC TCP backend // bindings so multiple `--ci` instances can coexist. - let ioc = if ci_enabled { Ioc::new_ci(guinness) } else { Ioc::new(guinness) }; + let ioc = Ioc::new_for_profile(guinness, ci_enabled, cfg.machine.profile.ip28()); // CI mode replaces the default TCP backend on channel B (tty1, the // SGI serial console) with an in-process backend the control socket @@ -365,6 +398,7 @@ impl Machine { let timer_manager = Arc::new(TimerManager::new()); ioc.set_timer_manager(timer_manager.clone()); ioc.set_heartbeat(heartbeat.clone()); + ioc.set_clock_ticks(fasttick_count.clone()); let hpc3 = Hpc3::with_net(eeprom_hpc3.clone(), ioc.clone(), guinness, heartbeat.clone(), cfg.network(), cfg.no_audio, cfg.audio.clone(), cfg.nvram.clone(), cfg.rtc_offset, cfg.scsi_deferred_int); hpc3.set_timer_manager(timer_manager.clone()); @@ -520,7 +554,8 @@ impl Machine { let nvram_provenance = cfg.nvram.clone(); // REX3 Graphics — Newport only; skipped in headless mode or when XZ board selected - let rex3: Option> = if cfg.headless || cfg.graphics.board != crate::config::GraphicsBoard::Newport { + // An IMPACT board takes the graphics slot and the window; no Newport then. + let rex3: Option> = if cfg.headless || cfg.graphics.board != crate::config::GraphicsBoard::Newport || cfg.impact.any_enabled() { None } else { let r = Arc::new(Rex3::new(heartbeat.clone(), fasttick_count.clone(), decoded_count.clone(), Arc::clone(&l1i_hit_count), Arc::clone(&l1i_fetch_count), Arc::clone(&uncached_fetch_count))); @@ -578,9 +613,9 @@ impl Machine { } }; - // Indigo2 IMPACT/MGRAS preview — multi-slot GIO stub. + // Indigo2 IMPACT graphics in the GIO graphics slot. let mgras: Option> = if !guinness && cfg.impact.any_enabled() { - Some(Arc::new(crate::mgras::Mgras::new(&cfg.impact))) + Some(Arc::new(crate::mgras::Mgras::new(&cfg.impact, ioc.clone(), heartbeat.clone(), fasttick_count.clone()))) } else { None }; @@ -650,6 +685,7 @@ impl Machine { mc.clone(), hpc3.clone(), prom_port, + cfg.machine.profile.ip28(), ); // Wrap Physical in Arc @@ -690,6 +726,7 @@ impl Machine { // Connect VINO to System Memory, install a video source, start DMA. // Source kind + broadcast standard come from `[vino]` in iris.toml. phys.vino.set_phys(phys.clone()); + if let Some(mgras) = &phys.mgras { mgras.set_phys(phys.clone()); } let standard = match cfg.vino.standard { crate::config::VinoStandard::Ntsc => crate::video_source::VideoStandard::Ntsc, crate::config::VinoStandard::Pal => crate::video_source::VideoStandard::Pal, @@ -731,7 +768,7 @@ impl Machine { // arm below monomorphises its own CPU — no per-model branch on the hot path. let sysad: Arc = phys.clone(); macro_rules! build_cpu { ($cache:ty) => {{ - let cfg = MipsCpuConfig::indy(); + let cfg = MipsCpuConfig::for_model::<$cache>(); let tlb = MipsTlb::new(cfg.tlb_entries); let mut executor: MipsExecutor = MipsExecutor::new(sysad.clone(), tlb, &cfg); @@ -749,7 +786,9 @@ impl Machine { // CP0 Count runs at a fixed frequency: DEFAULT_COUNT_HZ unless the // user overrode it via `[clock] fixed_mhz` or the CLI. Must happen // before the core starts executing. - if let Some(mhz) = clock_fixed_mhz { + if let Some(hz) = ip28_count_hz { + executor.core.set_count_hz(hz); + } else if let Some(mhz) = clock_fixed_mhz { executor.core.set_count_hz((mhz * 1_000_000.0) as u64); } @@ -776,6 +815,13 @@ impl Machine { let cpu: Arc = match cfg_cpu_model { crate::config::CpuModel::R4400 => build_cpu!(R4400Cache), crate::config::CpuModel::R5000 => build_cpu!(R5000Cache), + // IP28 uses the shadow cache: out of the data path entirely, with + // tag and data arrays that exist only to answer CACHE ops and the + // PROM's diagnostics. See mips_cache_shadow.rs. + #[cfg(feature = "ip28")] + crate::config::CpuModel::R10000 => build_cpu!(R10000ShadowCache), + #[cfg(not(feature = "ip28"))] + crate::config::CpuModel::R10000 => unreachable!("refused above without the ip28 feature"), }; // Share count_hz_atomic from MipsCore with Rex3 so the refresh thread can display it. @@ -827,6 +873,7 @@ impl Machine { // happen before hpc3.scsi().start() (called later, from // Machine::start) actually spawns the worker thread that reads it. if let Some(rex3) = &phys.rex3 { rex3.set_cpu_cycles(cpu.cycles_ptr()); } + if let Some(mgras) = &phys.mgras { mgras.set_cpu_cycles(cpu.cycles_ptr()); } if let Some(rex3) = &phys.rex3_head1 { rex3.set_cpu_cycles(cpu.cycles_ptr()); } if let Some(gr2) = &phys.gr2 { gr2.set_cpu_cycles(cpu.cycles_ptr()); } if let Some(td) = &phys.testdev { td.attach_core(cpu.core_ptr()); } @@ -1014,6 +1061,7 @@ impl Machine { // Program VC2 before the refresh thread runs so the first frame has size. self.apply_host_display_resolution(); if let Some(rex3) = &self._phys.rex3 { rex3.start(); } + if let Some(mgras) = &self._phys.mgras { mgras.start_display(); } if let Some(rex3) = &self._phys.rex3_head1 { rex3.start(); } if let Some(gr2) = &self._phys.gr2 { gr2.start(); } #[cfg(feature = "ultra64")] @@ -1102,6 +1150,7 @@ impl Machine { pub fn stop(&mut self) { self.cpu.stop(); if let Some(rex3) = &self._phys.rex3 { rex3.stop(); } + if let Some(mgras) = &self._phys.mgras { mgras.stop_display(); } if let Some(rex3) = &self._phys.rex3_head1 { rex3.stop(); } if let Some(gr2) = &self._phys.gr2 { gr2.stop(); } self.hpc3.stop(); @@ -1176,6 +1225,9 @@ impl Machine { if let Some(g) = &self._phys.gr2 { return Some(g.clone() as Arc); } + if let Some(m) = &self._phys.mgras { + return Some(m.clone() as Arc); + } self._phys.rex3.clone().map(|r| r as Arc) } diff --git a/src/mc.rs b/src/mc.rs index 0cdf61da..3de34e75 100644 --- a/src/mc.rs +++ b/src/mc.rs @@ -43,6 +43,9 @@ pub const REG_SYS_SEMAPHORE: u32 = 0x0100; pub const REG_LOCK_MEMORY: u32 = 0x0108; pub const REG_EISA_LOCK: u32 = 0x0110; pub const REG_RPSS_CTR: u32 = 0x1000; + +/// The MC clock of the IP22/IP24 models (see `MemoryController::set_clock_hz`). +pub const DEFAULT_CLOCK_HZ: u64 = 50_000_000; pub const REG_SEMAPHORE_0: u32 = 0x10000; // ... Semaphores 1-15 follow pattern +0x1000 @@ -56,6 +59,9 @@ struct MemoryControllerState { // Timer state last_host_ticks: u64, host_freq: u64, + /// The clock the refresh, watchdog and RPSS counters count (see + /// `set_clock_hz`). + clock_hz: u64, cpu_cycle_acc: u64, rpss_cycle_acc: u64, cpu: Option>, @@ -70,6 +76,11 @@ pub struct MemoryController { threads: Arc>>>, running: Arc, guinness: bool, + /// How far MEMCFG's base field is shifted to give a physical address: + /// 22 on IP22/IP24, 24 on IP28. The size field counts per-subbank units of + /// `1 << base_shift` bytes, so the granule follows it too — 4 MB against + /// 16 MB. Set from the machine profile, never from the environment. + base_shift: u32, /// Actual SIMM sizes in MB for each bank (index 0..3). Used by parse_memcfg to /// derive addr_mask (for aliasing) and limit (for device_map range). ram_sizes: Arc<[u32; 4]>, @@ -93,8 +104,21 @@ pub struct MemoryController { } impl MemoryController { + /// An IP22/IP24 memory controller. pub fn new(eeprom: Arc>, guinness: bool, ram_sizes: [u32; 4]) -> Self { - let regs = Self::init_registers(guinness); + Self::new_for_profile(eeprom, guinness, ram_sizes, false) + } + + /// `ip28` selects the IP28 decodes: a 24-bit MEMCFG base shift and an MC + /// chip revision the IP28 PROM accepts. + pub fn new_for_profile( + eeprom: Arc>, + guinness: bool, + ram_sizes: [u32; 4], + ip28: bool, + ) -> Self { + let base_shift = if ip28 { 24 } else { 22 }; + let regs = Self::init_registers_for(guinness, base_shift); let host_freq = crate::platform::get_host_tick_frequency(); let last_host_ticks = crate::platform::get_host_ticks(); @@ -106,6 +130,7 @@ impl MemoryController { user_semaphores: [false; 16], last_host_ticks, host_freq, + clock_hz: DEFAULT_CLOCK_HZ, cpu_cycle_acc: 0, rpss_cycle_acc: 0, cpu: None, @@ -116,6 +141,7 @@ impl MemoryController { threads: Arc::new(Mutex::new(Vec::new())), running: Arc::new(AtomicBool::new(false)), guinness, + base_shift, ram_sizes: Arc::new(ram_sizes), memcfg_callback: Arc::new(OnceLock::new()), event_tx: Arc::new(OnceLock::new()), @@ -127,6 +153,10 @@ impl MemoryController { } fn init_registers(guinness: bool) -> Vec { + Self::init_registers_for(guinness, 22) + } + + fn init_registers_for(guinness: bool, base_shift: u32) -> Vec { let mut regs = vec![0; (MC_SIZE / 4) as usize]; // Initialize CPUCTRL0: REFS=2, RFE=1, MUX_HWM=1 @@ -149,7 +179,20 @@ impl MemoryController { // unconditionally lets vino_init proceed; with it set, `vlinfo` // reports `vino 0` with 5 nodes (digital input = IndyCam, analog // input, two memory drains, controls). - regs[(REG_SYSID / 4) as usize] = 0x00000013; + // IP28 raises the bar: its power-on diagnostic reads the chip + // revision out of SYSID and rejects anything below 5 with "FATAL + // ERROR: Rev A/BC MC detected--Need rev D or greater." It needs a + // rev D part because that is what supports the high memory mapping + // IP28 uses. Report 5 there; every other machine keeps the rev C (3) + // it has always seen. Bit 4 stays set in both — that is the EISA / + // vino gate, not part of the revision. + regs[(REG_SYSID / 4) as usize] = if base_shift == 22 { + 0x00000013 + } else { + // IP28's chip revision lives in the low nibble; 5 is the lowest + // the PROM accepts. + 0x00000015 + }; // Initialize RPSS_DIVIDER: DIV=9, INC=3 (for 33MHz) // 33MHz: Divide by 10 (9+1), Increment by 3 -> 300ns per tick @@ -186,6 +229,16 @@ impl MemoryController { regs } + /// Set the clock the MC's counters count. IRIX derives the RPSS cycle + /// counter's period from its idea of this clock (on IP28: the CPU clock + /// times the R10000 Config.EC system-clock ratio, over the RPSS divider it + /// programs), and hands that period to programs that time with the + /// counter; the model has to count at the same rate or every such timing + /// is off by the difference. Call before the machine runs. + pub fn set_clock_hz(&self, hz: u64) { + self.state.lock().clock_hz = hz; + } + pub fn set_cpu(&self, cpu: Weak) { self.state.lock().cpu = Some(cpu); } @@ -233,19 +286,75 @@ impl MemoryController { /// inst_size_per_rank = size_mb_bytes >> inst_rank. /// Encode a MEMCFG half-word for a bank at `base` with `size_mb` installed. /// Inverse of [`memcfg_bank_info`] for the sizes IRIS supports. + /// How far the MEMCFG base field is shifted to give a physical address. + /// + /// IP22/IP24 use 22 (a 4 MB granule). IP28 appears to use 24: its PROM + /// writes base byte 0x60 and then probes 0x60000000, and base byte 0x20 + /// gives 0x20000000, which is where NetBSD loads IP28 kernels. Both fall + /// out of a 24-bit shift and neither does out of 22. + /// + /// Env-gated while IP28 has no machine profile of its own. It must not + /// change IP22/IP24, which this default preserves. + /// This controller's MEMCFG base shift. See the `base_shift` field. + fn memcfg_base_shift(&self) -> u32 { + self.base_shift + } + + /// A bank's installed size in MB → `(size_field, rank)` in register format. + /// + /// The size field counts per-subbank units of `1 << memcfg_base_shift()` + /// bytes, minus one, and the rank bit doubles the bank. The granule + /// therefore follows the base shift: 4 MB where the shift is 22, 16 MB + /// where it is 24. The same 64 MB bank is size field 15 on IP22 and size + /// field **3** on IP28 — which is what the IP28 PROM is observed to write + /// for its own banks (MEMCFG0 = 0x2320_2324, two 64 MB banks at + /// 0x20000000 and 0x24000000). + /// + /// The shift-22 rows are left exactly as they were, rank bits included: + /// they describe real IP22 SIMM topology, and more than one of them has + /// more than one arithmetically-equivalent encoding. + fn memcfg_size_rank(size_mb: u32) -> Option<(u32, u32)> { + Self::memcfg_size_rank_at(22, size_mb) + } + + /// `memcfg_size_rank` with the granule passed in, so both can be tested in + /// one process — the live shift is read from the environment once and + /// cached for the life of the program. + fn memcfg_size_rank_at(shift: u32, size_mb: u32) -> Option<(u32, u32)> { + if shift == 24 { + // 16 MB granule. + return match size_mb { + 16 => Some((0, 0)), + 32 => Some((1, 0)), + 64 => Some((3, 0)), + 128 => Some((7, 0)), + 256 => Some((15, 0)), + _ => None, + }; + } + // 4 MB granule. + match size_mb { + 8 => Some((0, 1)), + 16 => Some((3, 0)), + 32 => Some((3, 1)), + 64 => Some((15, 0)), + 128 => Some((15, 1)), + _ => None, + } + } + pub fn encode_memcfg_half(base: u32, size_mb: u32) -> Option { + Self::encode_memcfg_half_at(22, base, size_mb) + } + + /// `encode_memcfg_half` with the granule passed in. See + /// [`memcfg_size_rank_at`](Self::memcfg_size_rank_at). + fn encode_memcfg_half_at(shift: u32, base: u32, size_mb: u32) -> Option { if size_mb == 0 { return None; } - let (simm_size_field, simm_rank): (u32, u32) = match size_mb { - 8 => (0, 1), - 16 => (3, 0), - 32 => (3, 1), - 64 => (15, 0), - 128 => (15, 1), - _ => return None, - }; - let base_byte = (base >> 22) & 0xFF; + let (simm_size_field, simm_rank) = Self::memcfg_size_rank_at(shift, size_mb)?; + let base_byte = (base >> shift) & 0xFF; Some( (base_byte as u16) | (1 << 13) // VLD @@ -255,27 +364,27 @@ impl MemoryController { } pub fn memcfg_bank_info(half: u16, size_mb: u32) -> Option<(u32, u32, u32)> { + Self::memcfg_bank_info_at(22, half, size_mb) + } + + /// `memcfg_bank_info` with the granule passed in. See + /// [`memcfg_size_rank_at`](Self::memcfg_size_rank_at). + fn memcfg_bank_info_at(shift: u32, half: u16, size_mb: u32) -> Option<(u32, u32, u32)> { if size_mb == 0 { return None; } if (half >> 13) & 1 == 0 { return None; } - let base = ((half as u32) & 0xFF) << 22; + let base = ((half as u32) & 0xFF) << shift; let conf_rank = ((half >> 14) & 1) as u32; let conf_size_field = ((half >> 8) & 0x1F) as u32; - let conf_total = (conf_size_field + 1) << 22; - - // SIMM size → (size_field, rank) in register format (one unit = 4MB) - let (simm_size_field, simm_rank): (u32, u32) = match size_mb { - 8 => (0, 1), - 16 => (3, 0), - 32 => (3, 1), - 64 => (15, 0), - 128 => (15, 1), - _ => return None, - }; + let conf_total = (conf_size_field + 1) << shift; + + // SIMM size → (size_field, rank) in register format; the unit follows + // the base shift — see `memcfg_size_rank_at`. + let (simm_size_field, simm_rank) = Self::memcfg_size_rank_at(shift, size_mb)?; let conf_size = conf_total >> conf_rank; - let minus_size = (simm_size_field + 1) << (22 - simm_rank); - let plus_size = (simm_size_field + 1) << (22 + simm_rank); + let minus_size = (simm_size_field + 1) << (shift - simm_rank); + let plus_size = (simm_size_field + 1) << (shift + simm_rank); // BNK=0 (aliasing phase): wrap at inst_size so alias is detected at base+inst_size // BNK=1 (subbank/walkingbit): wrap at full bank size so both ranks are independent let addr_mask = if conf_rank == 0 { minus_size - 1 } else { plus_size - 1 }; @@ -294,7 +403,7 @@ impl MemoryController { return false; } let half = |i: usize, base: u32| { - Self::encode_memcfg_half(base, self.ram_sizes[i]).unwrap_or(0) as u32 + Self::encode_memcfg_half_at(self.base_shift, base, self.ram_sizes[i]).unwrap_or(0) as u32 }; let memcfg0 = (half(0, LOMEM_BASE) << 16) | half(1, LOMEM_BASE + BANK_SIZE); if memcfg0 == 0 { @@ -312,7 +421,7 @@ impl MemoryController { (memcfg1 >> 16) as u16, // bank 2: high half of MEMCFG1 (memcfg1 & 0xFFFF) as u16, // bank 3: low half of MEMCFG1 ]; - std::array::from_fn(|i| Self::memcfg_bank_info(halves[i], self.ram_sizes[i])) + std::array::from_fn(|i| Self::memcfg_bank_info_at(self.base_shift, halves[i], self.ram_sizes[i])) } /// If the embedded PROM POSTed lomem (banks 0–1) but skipped himem, synthesize @@ -320,6 +429,18 @@ impl MemoryController { fn synthesize_himem_banks(&self, state: &mut MemoryControllerState) -> bool { use crate::physical::{BANK_SIZE, HIMEM_BASE}; + // Only for a PROM that POSTs lomem and stops. The IP28 PROM sizes all + // four banks itself, walking them one at a time through a probe window + // at base byte 0x60; it finishes MEMCFG0 before it has finished with + // MEMCFG1. Synthesizing here fired on that intermediate state and + // overwrote the walk in progress, leaving banks 2 and 3 describing + // 256 MB apiece — bank 2 on top of bank 0 — and the PROM then never + // converged. HIMEM_BASE and BANK_SIZE are lomem/himem constants that + // mean nothing on a machine whose RAM starts at 0x20000000 anyway. + if self.base_shift != 22 { + return false; + } + let memcfg0 = state.regs[(REG_MEMCFG0 / 4) as usize]; let memcfg1 = state.regs[(REG_MEMCFG1 / 4) as usize]; let h0 = (memcfg0 >> 16) as u16; @@ -334,14 +455,14 @@ impl MemoryController { let mut changed = false; if self.ram_sizes[2] > 0 && (h2 >> 13) & 1 == 0 { - if let Some(enc) = Self::encode_memcfg_half(HIMEM_BASE, self.ram_sizes[2]) { + if let Some(enc) = Self::encode_memcfg_half_at(self.base_shift, HIMEM_BASE, self.ram_sizes[2]) { h2 = enc; changed = true; } } if self.ram_sizes[3] > 0 && (h3 >> 13) & 1 == 0 { if let Some(enc) = - Self::encode_memcfg_half(HIMEM_BASE + BANK_SIZE, self.ram_sizes[3]) + Self::encode_memcfg_half_at(self.base_shift, HIMEM_BASE + BANK_SIZE, self.ram_sizes[3]) { h3 = enc; changed = true; @@ -439,11 +560,11 @@ impl MemoryControllerState { let diff = now.wrapping_sub(self.last_host_ticks); self.last_host_ticks = now; - // Scale to 50MHz CPU cycles (20ns period) - // cycles = (diff * 50_000_000) / host_freq + // Scale to cycles of the MC clock (`clock_hz`). + // cycles = (diff * clock_hz) / host_freq // Use accumulator to maintain precision over many small updates - // Use u128 to prevent overflow during multiplication (diff * 50M can exceed u64) - let total_ticks = (diff as u128) * 50_000_000 + (self.cpu_cycle_acc as u128); + // Use u128 to prevent overflow during multiplication + let total_ticks = (diff as u128) * (self.clock_hz as u128) + (self.cpu_cycle_acc as u128); let cpu_cycles = (total_ticks / (self.host_freq as u128)) as u64; self.cpu_cycle_acc = (total_ticks % (self.host_freq as u128)) as u64; @@ -685,12 +806,18 @@ impl BusDevice for MemoryController { BUS_OK } REG_MEMCFG0 => { + if self.base_shift != 22 { + eprintln!("iris: ip28: guest writes MEMCFG0 = {val:#010x}"); + } dlog_dev!(LogModule::Mc, "MC: Write MEMCFG0 = {:08x}", val); state.regs[(REG_MEMCFG0 / 4) as usize] = val; self.on_memcfg_updated(&mut state); BUS_OK } REG_MEMCFG1 => { + if self.base_shift != 22 { + eprintln!("iris: ip28: guest writes MEMCFG1 = {val:#010x}"); + } dlog_dev!(LogModule::Mc, "MC: Write MEMCFG1 = {:08x}", val); state.regs[(REG_MEMCFG1 / 4) as usize] = val; self.on_memcfg_updated(&mut state); @@ -1351,4 +1478,61 @@ mod tests { assert!(addrs[3].is_some(), "bank 3 should be synthesized"); assert_eq!(addrs[2].unwrap().0, crate::physical::HIMEM_BASE); } + + /// The IP28 PROM's own MEMCFG0 write, copied from a POST trace: + /// `0x2320_2324` — two 64 MB banks at 0x20000000 and 0x24000000. + /// + /// Size field 3 means four units of 16 MB, because the IP28 granule is the + /// base shift (24), not IP22's 22. Decoding it with IP22's table called the + /// same bank 256 MB, which is how banks 2 and 3 came to claim memory that + /// was not there. + #[test] + fn ip28_memcfg_matches_what_the_prom_writes() { + const OBSERVED: u32 = 0x2320_2324; + let h0 = (OBSERVED >> 16) as u16; + let h1 = (OBSERVED & 0xFFFF) as u16; + + assert_eq!(MemoryController::encode_memcfg_half_at(24, 0x2000_0000, 64), Some(h0)); + assert_eq!(MemoryController::encode_memcfg_half_at(24, 0x2400_0000, 64), Some(h1)); + + let (base0, mask0, limit0) = MemoryController::memcfg_bank_info_at(24, h0, 64).unwrap(); + assert_eq!(base0, 0x2000_0000); + assert_eq!(limit0, 64 << 20, "a 64 MB bank must not claim more"); + assert_eq!(mask0, (64 << 20) - 1, "and must alias within its own 64 MB"); + + let (base1, _, limit1) = MemoryController::memcfg_bank_info_at(24, h1, 64).unwrap(); + assert_eq!(base1, 0x2400_0000); + assert_eq!(base1, base0 + limit0, "banks 0 and 1 are contiguous, not overlapping"); + assert_eq!(limit1, 64 << 20); + } + + /// Four 64 MB banks must tile 0x20000000..0x30000000 with no overlap and no + /// gap. The failure this pins is the one that stalled the IRIX install: + /// bank 2 landed on top of bank 0 and bank 3 claimed 256 MB, so the top of + /// "memory" was 0x38000000 and the miniroot was loaded into nothing. + #[test] + fn ip28_four_banks_tile_without_overlap() { + let mut next = 0x2000_0000u32; + for bank in 0..4 { + let half = MemoryController::encode_memcfg_half_at(24, next, 64) + .unwrap_or_else(|| panic!("bank {bank} did not encode")); + let (base, _, limit) = MemoryController::memcfg_bank_info_at(24, half, 64).unwrap(); + assert_eq!(base, next, "bank {bank} base"); + assert_eq!(limit, 64 << 20, "bank {bank} limit"); + next = base + limit; + } + assert_eq!(next, 0x3000_0000, "256 MB total, ending where RAM ends"); + } + + /// The IP22 encodings are unchanged by the granule work. + #[test] + fn ip22_memcfg_encodings_are_unchanged() { + for (size_mb, want) in [(8, (0, 1)), (16, (3, 0)), (32, (3, 1)), (64, (15, 0)), (128, (15, 1))] { + assert_eq!(MemoryController::memcfg_size_rank_at(22, size_mb), Some(want), "{size_mb} MB"); + } + let half = MemoryController::encode_memcfg_half_at(22, crate::physical::LOMEM_BASE, 64).unwrap(); + let (base, _, limit) = MemoryController::memcfg_bank_info_at(22, half, 64).unwrap(); + assert_eq!(base, crate::physical::LOMEM_BASE); + assert_eq!(limit, 64 << 20); + } } diff --git a/src/mgras.rs b/src/mgras.rs deleted file mode 100644 index feb5e2b7..00000000 --- a/src/mgras.rs +++ /dev/null @@ -1,347 +0,0 @@ -/// IMPACT / MGRAS graphics — preview stub (Indigo2 IP22) -/// -/// Post-1995 IMPACT boards use the MGRAS ASIC set (geometry engine, raster -/// engine, TRAM controllers) spread across one to three GIO64 slots depending on -/// the option (Solid / High / Maximum). -/// -/// **Preview only:** per-slot register files with probe-friendly board IDs and -/// idle status. No TRAM, no DMA, no GL/command processing. -/// -/// GIO slot bases (physical, IP22): -/// gfx — 0x1F000000 (4 MB) -/// exp0 — 0x1F400000 (2 MB) -/// exp1 — 0x1F600000 (4 MB) -/// -/// See `docs/impact-mgras-research.md` for hardware notes and implementation status. - -use parking_lot::Mutex; -use std::io::Write as IoWrite; - -use crate::config::{ImpactSection, ImpactSlot}; -use crate::devlog::LogModule; -use crate::snapshot::{get_field, hex_u32, toml_u32}; -use crate::traits::{BusDevice, BusRead8, BusRead16, BusRead32, BusRead64, BUS_OK, Device, Saveable}; - -// ─── GIO slot geometry ─────────────────────────────────────────────────────── - -pub const MGRAS_SLOT_GFX_BASE: u32 = 0x1F00_0000; -pub const MGRAS_SLOT_GFX_SIZE: u32 = 0x0040_0000; -pub const MGRAS_SLOT_EXP0_BASE: u32 = 0x1F40_0000; -pub const MGRAS_SLOT_EXP0_SIZE: u32 = 0x0020_0000; -pub const MGRAS_SLOT_EXP1_BASE: u32 = 0x1F60_0000; -pub const MGRAS_SLOT_EXP1_SIZE: u32 = 0x0040_0000; - -/// MGRAS CPU register window within each populated slot (research placeholder; -/// mirrors the Newport/XZ `+0x0F0000` pattern until a verified map lands). -pub const MGRAS_REG_OFF: u32 = 0x000F_0000; -pub const MGRAS_REG_SIZE: u32 = 0x2000; - -// ─── Register offsets (relative to slot base + MGRAS_REG_OFF) ─────────────── - -pub mod reg { - use crate::config::ImpactSlot; - - pub const BOARD_ID: u32 = 0x0000; - pub const REVISION: u32 = 0x0004; - pub const STATUS: u32 = 0x0008; - pub const INTR_STATUS: u32 = 0x000C; - pub const INTR_ENABLE: u32 = 0x0010; - pub const FIFO_WRITE: u32 = 0x0018; - pub const SLOT_ROLE: u32 = 0x0024; // which GE/TRAM slice (multi-slot boards) - - pub const STATUS_IDLE: u32 = 0x0000_0007; // FIFO empty + engines idle (preview) - - pub fn board_id_for(slot: ImpactSlot) -> u32 { - match slot { - ImpactSlot::None => 0, - ImpactSlot::Solid => 0x004D_4752, // "MGR" + Solid class tag (ASCII-ish) - ImpactSlot::High => 0x004D_4748, // High IMPACT - ImpactSlot::Max => 0x004D_474D, // Maximum IMPACT - } - } - - pub fn revision_for(slot: ImpactSlot) -> u32 { - match slot { - ImpactSlot::None => 0, - ImpactSlot::Solid => 0x0000_0100, - ImpactSlot::High => 0x0000_0200, - ImpactSlot::Max => 0x0000_0300, - } - } -} - -#[derive(Clone, Copy)] -struct SlotMap { - kind: ImpactSlot, - base: u32, - size: u32, -} - -struct SlotState { - intr_status: u32, - intr_enable: u32, - fifo_depth: u32, -} - -struct MgrasState { - slots: [Option; 3], -} - -#[derive(Clone)] -pub struct Mgras { - map: [SlotMap; 3], - state: std::sync::Arc>, -} - -impl Mgras { - pub fn new(cfg: &ImpactSection) -> Self { - let kinds = [cfg.gfx, cfg.exp0, cfg.exp1]; - let bases = [MGRAS_SLOT_GFX_BASE, MGRAS_SLOT_EXP0_BASE, MGRAS_SLOT_EXP1_BASE]; - let sizes = [MGRAS_SLOT_GFX_SIZE, MGRAS_SLOT_EXP0_SIZE, MGRAS_SLOT_EXP1_SIZE]; - let mut map = [SlotMap { kind: ImpactSlot::None, base: 0, size: 0 }; 3]; - for i in 0..3 { - map[i] = SlotMap { kind: kinds[i], base: bases[i], size: sizes[i] }; - } - let mut slots = [None, None, None]; - for (i, m) in map.iter().enumerate() { - if m.kind != ImpactSlot::None { - slots[i] = Some(SlotState { intr_status: 0, intr_enable: 0, fifo_depth: 0 }); - } - } - Self { - map, - state: std::sync::Arc::new(Mutex::new(MgrasState { slots })), - } - } - - pub fn power_on(&self) { - let mut st = self.state.lock(); - for (i, m) in self.map.iter().enumerate() { - st.slots[i] = if m.kind != ImpactSlot::None { - Some(SlotState { intr_status: 0, intr_enable: 0, fifo_depth: 0 }) - } else { - None - }; - } - } - - pub fn any_slot(&self) -> bool { - self.map.iter().any(|m| m.kind != ImpactSlot::None) - } - - fn locate(&self, addr: u32) -> Option<(usize, u32)> { - for (i, m) in self.map.iter().enumerate() { - if m.kind == ImpactSlot::None { - continue; - } - let reg_base = m.base.wrapping_add(MGRAS_REG_OFF); - if (addr & 0xFFFF_E000) == reg_base { - return Some((i, addr & (MGRAS_REG_SIZE - 1))); - } - // Rest of the slot aperture: reads as 0, writes ignored. - if addr >= m.base && addr < m.base.wrapping_add(m.size) { - return None; - } - } - None - } - - fn read_reg(&self, slot_idx: usize, off: u32) -> u32 { - let kind = self.map[slot_idx].kind; - let st = self.state.lock(); - let Some(s) = &st.slots[slot_idx] else { return 0 }; - match off { - reg::BOARD_ID => reg::board_id_for(kind), - reg::REVISION => reg::revision_for(kind), - reg::STATUS => reg::STATUS_IDLE, - reg::INTR_STATUS => s.intr_status, - reg::INTR_ENABLE => s.intr_enable, - reg::SLOT_ROLE => slot_idx as u32, - _ => 0, - } - } - - fn write_reg(&self, slot_idx: usize, off: u32, val: u32) { - let mut st = self.state.lock(); - let Some(s) = &mut st.slots[slot_idx] else { return }; - match off { - reg::INTR_STATUS => s.intr_status &= !val, - reg::INTR_ENABLE => s.intr_enable = val, - reg::FIFO_WRITE => { - s.fifo_depth = s.fifo_depth.saturating_add(4); - dlog_dev!( - LogModule::Rex3, - "MGRAS slot{}: FIFO write {:08x} (stub, depth={})", - slot_idx, - val, - s.fifo_depth - ); - } - _ => { - dlog_dev!( - LogModule::Rex3, - "MGRAS slot{}: write reg {:04x} = {:08x} (ignored)", - slot_idx, - off, - val - ); - } - } - } -} - -impl Device for Mgras { - fn step(&self, _cycles: u64) {} - fn stop(&self) {} - fn start(&self) {} - fn is_running(&self) -> bool { false } - fn get_clock(&self) -> u64 { 0 } - - fn register_commands(&self) -> Vec<(String, String)> { - vec![ - ("mgras".into(), "IMPACT/MGRAS preview stub (status)".into()), - ("impact".into(), "IMPACT inventory summary (hinv-style preview)".into()), - ] - } - - fn execute_command(&self, cmd: &str, args: &[&str], mut writer: Box) -> Result<(), String> { - if cmd != "mgras" && cmd != "impact" { - return Err(format!("unknown mgras command: {cmd}")); - } - if !args.is_empty() { - return Err(format!("usage: {cmd}")); - } - let st = self.state.lock(); - let names = ["gfx", "exp0", "exp1"]; - if cmd == "impact" { - writeln!(writer, "Graphics inventory (IMPACT preview — driver attach not implemented):").map_err(|e| e.to_string())?; - } - for (i, m) in self.map.iter().enumerate() { - if m.kind == ImpactSlot::None { - if cmd == "impact" { - continue; - } - writeln!(writer, " {}: (empty)", names[i]).map_err(|e| e.to_string())?; - continue; - } - let depth = st.slots[i].as_ref().map(|s| s.fifo_depth).unwrap_or(0); - let label = match m.kind { - ImpactSlot::Solid => "IMPACT Solid", - ImpactSlot::High => "IMPACT High", - ImpactSlot::Max => "IMPACT Maximum", - ImpactSlot::None => unreachable!(), - }; - if cmd == "impact" { - writeln!( - writer, - " Graphics board {}: {} (GIO @ {:#010x})", - i, - label, - m.base.wrapping_add(MGRAS_REG_OFF), - ) - .map_err(|e| e.to_string())?; - } else { - writeln!( - writer, - " {}: {:?} @ {:#010x} fifo_bytes={}", - names[i], - m.kind, - m.base.wrapping_add(MGRAS_REG_OFF), - depth, - ) - .map_err(|e| e.to_string())?; - } - } - if cmd == "impact" && !self.map.iter().any(|m| m.kind != ImpactSlot::None) { - writeln!(writer, " (no IMPACT boards configured in [impact])").map_err(|e| e.to_string())?; - } - Ok(()) - } -} - -impl BusDevice for Mgras { - fn read32(&self, addr: u32) -> BusRead32 { - match self.locate(addr) { - Some((idx, off)) => BusRead32::ok(self.read_reg(idx, off)), - None => BusRead32::ok(0), - } - } - - fn write32(&self, addr: u32, val: u32) -> u32 { - if let Some((idx, off)) = self.locate(addr) { - self.write_reg(idx, off, val); - } - BUS_OK - } - - fn read8(&self, addr: u32) -> BusRead8 { - BusRead8::ok(self.read32(addr & !3).data as u8) - } - fn write8(&self, addr: u32, val: u8) -> u32 { - let shift = (addr & 3) * 8; - let cur = self.read32(addr & !3).data; - self.write32(addr & !3, (cur & !(0xFF << shift)) | ((val as u32) << shift)); - BUS_OK - } - fn read16(&self, addr: u32) -> BusRead16 { - BusRead16::ok(self.read32(addr & !3).data as u16) - } - fn write16(&self, addr: u32, val: u16) -> u32 { - let shift = if (addr & 2) != 0 { 16 } else { 0 }; - let cur = self.read32(addr & !3).data; - self.write32(addr & !3, (cur & !(0xFFFF << shift)) | ((val as u32) << shift)); - BUS_OK - } - fn read64(&self, addr: u32) -> BusRead64 { - let lo = self.read32(addr).data as u64; - let hi = self.read32(addr.wrapping_add(4)).data as u64; - BusRead64::ok((hi << 32) | lo) - } - fn write64(&self, addr: u32, val: u64) -> u32 { - self.write32(addr, val as u32); - self.write32(addr.wrapping_add(4), (val >> 32) as u32); - BUS_OK - } -} - -impl Saveable for Mgras { - fn save_state(&self) -> toml::Value { - let st = self.state.lock(); - let mut slots = toml::map::Map::new(); - let names = ["gfx", "exp0", "exp1"]; - for (i, name) in names.iter().enumerate() { - if self.map[i].kind == ImpactSlot::None { - continue; - } - let Some(s) = st.slots[i].as_ref() else { continue }; - let mut slot_tbl = toml::map::Map::new(); - slot_tbl.insert("intr_status".into(), hex_u32(s.intr_status)); - slot_tbl.insert("intr_enable".into(), hex_u32(s.intr_enable)); - slot_tbl.insert("fifo_depth".into(), toml::Value::Integer(s.fifo_depth as i64)); - slots.insert((*name).into(), toml::Value::Table(slot_tbl)); - } - toml::Value::Table(slots) - } - - fn load_state(&self, v: &toml::Value) -> Result<(), String> { - let Some(tbl) = v.as_table() else { return Ok(()) }; - let names = ["gfx", "exp0", "exp1"]; - let mut st = self.state.lock(); - for (i, name) in names.iter().enumerate() { - let Some(slot_v) = tbl.get(*name) else { continue }; - let Some(s) = st.slots[i].as_mut() else { continue }; - if let Some(x) = get_field(slot_v, "intr_status") { - s.intr_status = toml_u32(x).unwrap_or(s.intr_status); - } - if let Some(x) = get_field(slot_v, "intr_enable") { - s.intr_enable = toml_u32(x).unwrap_or(s.intr_enable); - } - if let Some(x) = get_field(slot_v, "fifo_depth") { - if let Some(n) = x.as_integer() { - s.fifo_depth = n as u32; - } - } - } - Ok(()) - } -} diff --git a/src/mgras/dcb.rs b/src/mgras/dcb.rs new file mode 100644 index 00000000..42b88470 --- /dev/null +++ b/src/mgras/dcb.rs @@ -0,0 +1,422 @@ +//! The display control bus (DCB): the board's slow side bus to its video +//! chips, reached through a 32 KB window at slot offset `0x60000`. +//! +//! Each device owns a 1 KB window (`0x60000 + dev * 0x400`), and the address +//! within it encodes the transaction: bits 9:7 select the chip's register +//! (its "CRS" line), bits 4:3 the transfer width in bytes (0 = 4), bit 5 asks +//! the chip to increment its register select, bit 6 packs data. Data rides in +//! the most significant bytes of a 32-bit bus access, so a one-byte register +//! read with a word load comes back in bits 31:24. + +use std::collections::HashMap; + +/// Device numbers on the bus. +pub const DEV_CMAP_ALL: u32 = 3; +pub const DEV_CMAP0: u32 = 4; +pub const DEV_CMAP1: u32 = 5; +pub const DEV_DAC: u32 = 6; +pub const DEV_XMAP: u32 = 7; +pub const DEV_VC3: u32 = 8; +pub const DEV_BDVERS: u32 = 9; +pub const DEV_I2C: u32 = 11; + +/// One decoded bus transaction. +#[derive(Clone, Copy, Debug)] +pub struct Txn { + pub dev: u32, + pub crs: u32, + /// Transfer width in bytes, 1-4. + pub width: u32, +} + +impl Txn { + /// `off` is the offset within the slot, in `0x60000..0x68000`. + pub fn decode(off: u32) -> Self { + let w = off & 0x3FF; + let width = match (w >> 3) & 3 { 0 => 4, n => n }; + Txn { dev: (off - 0x60000) >> 10, crs: (w >> 7) & 7, width } + } + + /// The data a CPU store of `bits` carries for this transaction. + pub fn store_data(&self, bits: u32, val: u64) -> u32 { + match bits { + 8 => val as u32 & 0xFF, + 16 => val as u32 & 0xFFFF, + 32 => (val as u32) >> (8 * (4 - self.width)), + _ => ((val >> 32) as u32) >> (8 * (4 - self.width)), + } + } + + /// Place `data` where a CPU load of `bits` expects it. + pub fn load_value(&self, bits: u32, data: u32) -> u64 { + match bits { + 8 => (data & 0xFF) as u64, + 16 => (data & 0xFFFF) as u64, + 32 => (data << (8 * (4 - self.width))) as u64 & 0xFFFF_FFFF, + _ => ((data << (8 * (4 - self.width))) as u64) << 32, + } + } +} + +/// A colormap chip: 8192 entries of 24-bit RGB. +pub struct Cmap { + pub pal: Vec, + addr: u32, + rev: u32, + cmd: u32, +} + +impl Cmap { + fn new(rev: u32) -> Self { + Cmap { pal: vec![0; 8192], addr: 0, rev, cmd: 0 } + } + + fn write(&mut self, t: Txn, d: u32) { + match (t.crs, t.width) { + // Address: one byte at a time (low, then high), or 16 bits sent low + // byte first. + (0, 1) => self.addr = (self.addr & 0x1F00) | d, + (0, _) => self.addr = (((d & 0xFF) << 8) | (d >> 8 & 0xFF)) & 0x1FFF, + (1, _) => self.addr = ((d & 0x1F) << 8) | (self.addr & 0xFF), + // Palette entry: red, green, blue; the address advances. + (2, _) => { + self.pal[self.addr as usize] = d & 0xFF_FFFF; + self.addr = (self.addr + 1) & 0x1FFF; + } + (3, _) => self.cmd = d, + _ => {} + } + } + + fn read(&self, t: Txn) -> u32 { + match t.crs { + 0 => self.addr & 0xFF, + 1 => self.addr >> 8, + 2 => self.pal[self.addr as usize], + 3 => self.cmd, + 4 => 0x08, // status: ready for writes + 6 => self.rev, + _ => 0, + } + } +} + +/// The RAMDAC: an address register selecting internal registers, and a +/// 256-entry, 10-bit gamma table written as red, green, blue in turn. +pub struct Dac { + addr: u32, + regs: HashMap, + pub gamma: Vec<[u16; 3]>, + gamma_comp: usize, + mode: u32, +} + +/// DAC register: the pixel read mask; zero blanks the screen. +pub const DAC_PIXMASK: u32 = 4; + +impl Dac { + fn new() -> Self { + let gamma = (0..256).map(|i| { let v = (i << 2) as u16; [v, v, v] }).collect(); + Dac { addr: 0, regs: HashMap::new(), gamma, gamma_comp: 0, mode: 0 } + } + + fn write(&mut self, t: Txn, d: u32) { + match t.crs { + // 16 bits, low byte first; a single byte sets the low half. + 0 if t.width >= 2 => { self.addr = ((d & 0xFF) << 8) | (d >> 8 & 0xFF); self.gamma_comp = 0; } + 0 => { self.addr = d & 0xFF; self.gamma_comp = 0; } + 1 => { + let i = (self.addr & 0xFF) as usize; + // Ten-bit entries: a byte write gives the top eight bits; a + // 16-bit write gives bits 9:2 in its first byte and 1:0 in + // the second. + let v = if t.width >= 2 { (((d >> 8 & 0xFF) << 2) | (d & 3)) as u16 } else { (d as u16) << 2 }; + self.gamma[i][self.gamma_comp] = v & 0x3FF; + self.gamma_comp += 1; + if self.gamma_comp == 3 { + self.gamma_comp = 0; + self.addr = self.addr.wrapping_add(1); + } + } + 2 => { self.regs.insert(self.addr, d & 0xFF); } + 3 => self.mode = d, + _ => {} + } + } + + fn read(&self, t: Txn) -> u32 { + match t.crs { + 0 => self.addr, + 2 => self.regs.get(&self.addr).copied().unwrap_or(0), + 3 => self.mode, + _ => 0, + } + } + + pub fn pixmask(&self) -> u32 { + self.regs.get(&DAC_PIXMASK).copied().unwrap_or(0xFF) + } +} + +/// The XMAP: the pixel processors' display side (display modes per window ID, +/// buffer selects, scanout pointers), reached as an index register (`INDEX`) +/// plus register files on the selects after it. +pub struct Xmap { + index: u32, + regs: HashMap<(u32, u32), u32>, +} + +/// XMAP registers, by select (named as in OpenBSD's impact(4)). +mod xmap { + /// Which pixel processor hears writes. + pub const PP1SELECT: u32 = 0; + pub const INDEX: u32 = 1; + pub const CONFIG: u32 = 2; + pub const BUF_SELECT: u32 = 3; + /// Display mode per window ID, at index `did * 4`. + pub const MAIN_MODE: u32 = 4; + pub const OVERLAY_MODE: u32 = 5; + /// The raster engine to pixel processor link. + pub const RE_RAC: u32 = 7; +} + +/// `CONFIG` index of the byte whose bit 3 makes the index auto-increment. +const XMAP_CONFIG_BYTE: u32 = 1; +const XMAP_AUTOINC: u32 = 0x08; +/// `CONFIG` index 4: the pixel processor revision. +const XMAP_REV_INDEX: u32 = 4; +/// `CONFIG` index read before each display-mode write. +const XMAP_MODE_ROOM_INDEX: u32 = 8; + +impl Xmap { + fn new() -> Self { + Xmap { index: 0, regs: HashMap::new() } + } + + fn autoinc(&mut self, t: Txn) { + let cfg = self.regs.get(&(xmap::CONFIG, XMAP_CONFIG_BYTE)).copied().unwrap_or(0); + if t.crs >= xmap::BUF_SELECT && cfg & XMAP_AUTOINC != 0 { + self.index = self.index.wrapping_add(t.width); + } + } + + fn write(&mut self, t: Txn, d: u32) { + match t.crs { + xmap::PP1SELECT => { self.regs.insert((xmap::PP1SELECT, 0), d); } + xmap::INDEX => self.index = d, + crs => { + self.regs.insert((crs, self.index), d); + self.autoinc(t); + } + } + } + + fn read(&mut self, t: Txn) -> u32 { + let v = match t.crs { + xmap::INDEX => self.index, + // The raster engine to pixel processor link reports its sync + // signature here; 1 in each nibble lane means "in sync". + xmap::RE_RAC => 0x0001_0101, + xmap::CONFIG if self.index == XMAP_REV_INDEX => 1, + // Polled nonzero before every display-mode write: room for a + // mode update. + xmap::CONFIG if self.index == XMAP_MODE_ROOM_INDEX => 0x10, + crs => self.regs.get(&(crs, self.index)).copied().unwrap_or(0), + }; + if t.crs >= xmap::CONFIG { self.autoinc(t); } + v + } + + /// Display mode of overlay window ID `did` (`OVERLAY_MODE`, index `did * 4`); + /// 0 is "overlay off". + pub fn overlay_mode(&self, did: u32) -> u32 { + self.regs.get(&(xmap::OVERLAY_MODE, (did & 0x1F) << 2)).copied().unwrap_or(0) + } + + /// Colormap address of cursor colour 0: the config register (`CONFIG`, + /// index 0) holds it divided by four. + pub fn cursor_cmap_base(&self) -> usize { + (self.regs.get(&(xmap::CONFIG, 0)).copied().unwrap_or(0) as usize & 0x7FF) << 2 + } + + /// Display mode of window ID `did` (`MAIN_MODE`, index `did * 4`). + pub fn main_mode(&self, did: u32) -> u32 { + self.regs.get(&(xmap::MAIN_MODE, did << 2)).copied().unwrap_or(0) + } +} + +/// The video timing chip: indexed 16-bit registers and a 32K x 16 SRAM holding +/// line and frame tables, the cursor glyph and window-ID tables. +/// VC3 cursor registers: glyph address, position, and control (bit 0 +/// enable, bit 1 display, bit 3 64x64 glyph). +const CURSOR_GLYPH_REG: usize = 1; +const CURSOR_X_REG: usize = 2; +const CURSOR_Y_REG: usize = 3; +const CURSOR_CONTROL_REG: usize = 0x1D; +/// VC3 register: SRAM address of the main planes' window-ID frame table. +const MAIN_WID_FRAME_REG: usize = 4; +const OVERLAY_WID_FRAME_REG: usize = 5; + +pub struct Vc3 { + index: u32, + pub regs: [u16; 32], + pub sram: Vec, + line_counter: u16, +} + +pub const VC3_SRAM_POINTER: usize = 7; +const VC3_CURRENT_LINE: usize = 0xB; + +impl Vc3 { + fn new() -> Self { + Vc3 { index: 0, regs: [0; 32], sram: vec![0; 0x8000], line_counter: 0 } + } + + /// The hardware cursor, when shown: the screen position of its top-left + /// corner, its size (32 or 64), and the SRAM address of its glyph. The + /// glyph is two bit planes, one after the other, a row at a time, most + /// significant bit leftmost; the planes make a two-bit colour, 0 being + /// transparent. The position registers are offset by 31. + pub fn cursor(&self) -> Option<(i32, i32, usize, usize)> { + let ctl = self.regs[CURSOR_CONTROL_REG]; + if ctl & 3 != 3 { + return None; + } + let size = if ctl & 8 != 0 { 64 } else { 32 }; + Some(( + self.regs[CURSOR_X_REG] as i32 - 31, + self.regs[CURSOR_Y_REG] as i32 - 31, + size, + self.regs[CURSOR_GLYPH_REG] as usize, + )) + } + + /// Window-ID runs of scanline `y` (0 is the top) in the main planes, as + /// `(first x, did)`. Register 4 points at the frame table, one line-table + /// address per scanline; a line table lists `(x << 5) | did` entries, + /// each starting a run, and ends at x = 0x7FF. + pub fn main_did_runs(&self, y: usize, out: &mut Vec<(u16, u8)>) { + self.did_runs(MAIN_WID_FRAME_REG, y, out) + } + + /// The same for the overlay planes, from register 5's frame table. + pub fn overlay_did_runs(&self, y: usize, out: &mut Vec<(u16, u8)>) { + self.did_runs(OVERLAY_WID_FRAME_REG, y, out) + } + + fn did_runs(&self, reg: usize, y: usize, out: &mut Vec<(u16, u8)>) { + out.clear(); + let frame = self.regs[reg] as usize; + let line = self.sram[(frame + y) & 0x7FFF]; + if line == 0xFFFF { + return; + } + for k in 0..64 { + let e = self.sram[(line as usize + k) & 0x7FFF]; + let x = e >> 5; + if x == 0x7FF { + break; + } + out.push((x, (e & 0x1F) as u8)); + } + } + + fn sram_write(&mut self, v: u16) { + let a = self.regs[VC3_SRAM_POINTER] as usize & 0x7FFF; + self.sram[a] = v; + self.regs[VC3_SRAM_POINTER] = self.regs[VC3_SRAM_POINTER].wrapping_add(1); + } + + fn sram_read(&mut self) -> u16 { + let a = self.regs[VC3_SRAM_POINTER] as usize & 0x7FFF; + self.regs[VC3_SRAM_POINTER] = self.regs[VC3_SRAM_POINTER].wrapping_add(1); + self.sram[a] + } + + fn write(&mut self, t: Txn, d: u32) { + match (t.crs, t.width) { + (0, 1) => self.index = d & 0x1F, + // Index and 16-bit value in one three-byte transfer. + (0, _) => { + self.index = (d >> 16) & 0x1F; + self.regs[self.index as usize] = d as u16; + } + (1, _) => self.regs[self.index as usize] = d as u16, + (3, 4) => { self.sram_write((d >> 16) as u16); self.sram_write(d as u16); } + (3, _) => self.sram_write(d as u16), + _ => {} + } + } + + fn read(&mut self, t: Txn) -> u32 { + match (t.crs, t.width) { + (0, _) => self.index, + (1, _) if self.index as usize == VC3_CURRENT_LINE => { + // Keep moving so vertical-blank polls terminate. + self.line_counter = (self.line_counter + 1) % 1066; + self.line_counter as u32 + } + (1, _) => self.regs[self.index as usize] as u32, + (3, 4) => { let hi = self.sram_read() as u32; (hi << 16) | self.sram_read() as u32 } + (3, _) => self.sram_read() as u32, + _ => 0, + } + } +} + +/// The whole bus: all devices of one board. +pub struct Dcb { + pub cmap: [Cmap; 2], + pub dac: Dac, + pub xmap: Xmap, + pub vc3: Vc3, + bdvers: [u32; 2], + bc1: u32, + other: HashMap<(u32, u32), u32>, +} + +impl Dcb { + /// `bdvers` are the two board-version bytes; `cmap_rev` the colormaps' + /// revision registers. + pub fn new(bdvers: [u32; 2], cmap_rev: [u32; 2]) -> Self { + Dcb { + cmap: [Cmap::new(cmap_rev[0]), Cmap::new(cmap_rev[1])], + dac: Dac::new(), + xmap: Xmap::new(), + vc3: Vc3::new(), + bdvers, + bc1: 0, + other: HashMap::new(), + } + } + + pub fn write(&mut self, t: Txn, d: u32) { + match t.dev { + DEV_CMAP_ALL => { self.cmap[0].write(t, d); self.cmap[1].write(t, d); } + DEV_CMAP0 => self.cmap[0].write(t, d), + DEV_CMAP1 => self.cmap[1].write(t, d), + DEV_DAC => self.dac.write(t, d), + DEV_XMAP => self.xmap.write(t, d), + DEV_VC3 => self.vc3.write(t, d), + DEV_BDVERS => { + if t.crs == 1 { self.bc1 = d; } + } + DEV_I2C => {} + dev => { self.other.insert((dev, t.crs), d); } + } + } + + pub fn read(&mut self, t: Txn) -> u32 { + match t.dev { + DEV_CMAP_ALL | DEV_CMAP0 => self.cmap[0].read(t), + DEV_CMAP1 => self.cmap[1].read(t), + DEV_DAC => self.dac.read(t), + DEV_XMAP => self.xmap.read(t), + DEV_VC3 => self.vc3.read(t), + DEV_BDVERS => self.bdvers.get(t.crs as usize).copied().unwrap_or(0), + // No flat panel: the I2C controller reads back nothing. + DEV_I2C => 0, + dev => self.other.get(&(dev, t.crs)).copied().unwrap_or(0), + } + } +} diff --git a/src/mgras/mod.rs b/src/mgras/mod.rs new file mode 100644 index 00000000..b062389a --- /dev/null +++ b/src/mgras/mod.rs @@ -0,0 +1,1222 @@ +//! IMPACT (MGRAS) graphics for the Indigo2. +//! +//! One board occupies the GIO graphics slot (`0x1F000000`) and decodes a 1 MB +//! window there. The host talks to one chip on it, the host interface, +//! which offers: +//! +//! | offset | what | +//! |-------------------|---------------------------------------------------| +//! | `0x00000` | GIO ID | +//! | `0x40000-0x45FFF` | command-processor microcode RAM (24-bit words) | +//! | `0x50000-0x5FFFF` | privileged registers, privileged command FIFO | +//! | `0x60000-0x67FFF` | display control bus devices (see `dcb`) | +//! | `0x68000-0x6FFFF` | per-device bus protocol registers | +//! | `0x70000-0x7BFFF` | user registers: status, flags, command FIFO | +//! | `0x7C000-0x7FFFF` | raster registers, direct access (see `raster`); | +//! | | the alias at `+0x1000` also executes the IR | +//! +//! Drawing arrives through the command FIFO as (command, data) pairs. Commands +//! at `0x1000` and up write raster registers directly, with bit `0x400` +//! meaning "execute"; lower numbers go to the command processor's microcode, +//! which this model does not run (the PROM, the kernel's console and the X +//! server's 2D paths never need it). The FIFO is drained as it is written, so +//! it always reads as empty. +//! +//! The model scans its framebuffer out through the colormap and DAC gamma into +//! the same window and status bar Newport uses (`GfxDisplay`). +//! +//! `IRIS_MGRAS_TRACE=` logs every access to the board. + +mod dcb; +mod raster; + +use parking_lot::Mutex; +use std::collections::{HashMap, HashSet}; +use std::io::Write as IoWrite; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::Arc; + +use crate::config::{ImpactSection, ImpactSlot}; +use crate::rex3::Renderer; +use crate::traits::{BusDevice, BusRead8, BusRead16, BusRead32, BusRead64, BUS_OK, Device, Saveable}; + +/// The graphics slot the board decodes. +pub const MGRAS_SLOT_GFX_BASE: u32 = 0x1F00_0000; +pub const MGRAS_SLOT_GFX_SIZE: u32 = 0x0040_0000; +/// The board's register window within the slot. +const MAP_SIZE: u32 = 0x10_0000; + +/// GIO ID: product 0x10, 32-bit ID, revision 1, GIO64, no ROM, manufacturer 1. +pub const GIO_ID: u32 = 0x0005_0190; + +/// Host interface register offsets. +mod host { + pub const UCODE: u32 = 0x40000; + pub const UCODE_END: u32 = 0x46000; + pub const SET_FLAGS_PRIVILEGED: u32 = 0x50008; + pub const CLEAR_FLAGS_PRIVILEGED: u32 = 0x5000C; + pub const CFIFO_PRIVILEGED: u32 = 0x50080; + pub const STATUS: u32 = 0x70000; + pub const FIFOSTATUS: u32 = 0x70004; + pub const SET_FLAGS: u32 = 0x70008; + pub const CLEAR_FLAGS: u32 = 0x7000C; + pub const GE_READBACK_HI: u32 = 0x70010; + pub const GE_READBACK_LO: u32 = 0x70014; + pub const CFIFO: u32 = 0x70080; + pub const GIOSTATUS: u32 = 0x70100; + pub const DMABUSY: u32 = 0x70104; + pub const RASTER: u32 = 0x7C000; + pub const RASTER_END: u32 = 0x80000; + + /// Status: raster idle and host idle, command and data FIFOs at or below + /// their low-water marks. + pub const STATUS_IDLE: u32 = 0x01 | 0x02 | 0x10 | 0x40; + pub const STATUS_VBLANK: u32 = 0x04; + + /// Flag set by the "set flag" command (0xE04): DMA/sync completion. + pub const FLAG_DONE: u32 = 1 << 16; + /// Flag set when the geometry engine has data waiting to be read back. + pub const FLAG_GE_DATA: u32 = 1 << 17; + /// Flag set while a geometry engine diagnostic readback has data waiting. + pub const FLAG_GE_DIAG: u32 = 1 << 18; + /// Context switch: outgoing context saved (phase 1) and incoming context + /// loaded (phase 2). + pub const FLAG_CONTEXT_SAVED: u32 = 1 << 19; + pub const FLAG_CONTEXT_LOADED: u32 = 1 << 6; + /// Command-processor flag 0: a scheduled buffer swap has happened. + pub const FLAG_CP0: u32 = 1 << 10; + /// Flags that can raise the general interrupt. + pub const INTR_CAUSES: u32 = 0x7F_FFFF; + + /// Flag and interrupt enables: read at the first address, write-1-to-set + /// there, write-1-to-clear at the second. + pub const FLAG_ENABLE_SET: u32 = 0x50010; + pub const FLAG_ENABLE_CLEAR: u32 = 0x50014; + pub const INTERRUPT_ENABLE_SET: u32 = 0x50018; + pub const INTERRUPT_ENABLE_CLEAR: u32 = 0x5001C; + /// Context switch request: starts the switch routine at the written + /// microcode address. + pub const CONTEXT_SWITCH: u32 = 0x50050; + /// Words of incoming context the host pushes after the save phase. + pub const CONTEXT_SWITCH_WORDS: u32 = 63; + + /// Geometry engine diagnostic ports, per engine: data, then address. + pub const GE_DIAG: [(u32, u32); 2] = [(0x50040, 0x50044), (0x50048, 0x5004C)]; + /// Diagnostic readback words: a discarded word, then the data word. + pub const GE_DIAG_READ_PAD: u32 = 0x50230; + pub const GE_DIAG_READ: u32 = 0x5022C; + + /// Host DMA engine and raster-interface context, read back per register. + pub const DMA_CONTEXT: u32 = 0x50300; + pub const RASTER_IF_CONTEXT: u32 = 0x50200; + /// PIO pixel reads: the raster char registers, high word at the execute + /// alias (which takes the next doubleword), low word at the plain one. + pub const PIO_READ_HI: u32 = 0x7D1C0; + pub const PIO_READ_LO: u32 = 0x7C1C4; +} + +/// A geometry engine as its diagnostic port sees it. The engine itself does +/// not run; it only has to hold downloaded microcode, read it back, and +/// answer "started". +#[derive(Default)] +struct GeDiag { + /// Current diagnostic address: a microcode line (from `UCODE_BASE`) or an + /// internal register. + addr: u32, + /// Which 32-bit word of the current 72-bit microcode line comes next. + word: usize, + ucode: HashMap, +} + +impl GeDiag { + const UCODE_BASE: u32 = 0x20_0000; + /// Internal register: execution control; bit 0 starts the engine. + const EXEC_CONTROL: u32 = 0x4_0000; + + /// Words a readback starting at `addr` delivers, in order: for each pair + /// of lines, word 0 and the top byte of the first, then word 1 of the + /// second. A readback starting on an odd line is preceded by two words + /// that carry nothing. (The driver's verifier reads exactly this; the + /// order is inferred from it.) + fn readback(&self, lines: u32) -> Vec { + let w = |l: u32, i: usize| self.ucode.get(&l).map(|x| x[i]).unwrap_or(0); + let mut out = Vec::new(); + let mut l = self.addr; + if l.wrapping_sub(Self::UCODE_BASE) & 1 == 1 { + out.extend([0, 0]); + } + for _ in 0..lines / 2 { + out.extend([w(l, 0), w(l, 2) & 0xFF, w(l + 1, 1)]); + l += 2; + } + out + } +} + +/// Command FIFO command numbers. +mod cmd { + /// Command-processor token: schedule a buffer swap for the next retrace. + pub const CP_SCHEDULE_SWAP: u32 = 0x37; + pub const SET_DONE_FLAG: u32 = 0xE04; + pub const RASTER_BASE: u32 = 0x1000; + pub const RASTER_EXECUTE: u32 = 0x400; + pub const DMA_BASE: u32 = 0x800; + pub const RASTER_IF_BASE: u32 = 0xA00; + pub const FORMATTER: u32 = 0xC00; + /// Below this, commands are command-processor microcode tokens. + pub const CP_LIMIT: u32 = 0x200; +} + +/// The command FIFO's word-stream parser: a command word, then the data words +/// its byte count announces. +#[derive(Default)] +struct Cfifo { + cmd: u32, + pixel: bool, + need: u32, + data: Vec, +} + +/// Host DMA engine registers. +mod dma { + pub const PAGE_LIST: usize = 0x00; + pub const STRIDE: usize = 0x04; + pub const ROW_OFFSET: usize = 0x05; + pub const ROW_START: usize = 0x06; + pub const LINES: usize = 0x07; + pub const LINE_BYTES: usize = 0x08; + /// The start word: bit 0 run, bits 2:1 pool, bit 3 board to host. + pub const START: usize = 0x0B; + /// Page-table base of pool `p`: eight bytes at `TABLE_BASE + 2p`, the + /// address in the low word. + pub const TABLE_BASE: usize = 0x20; +} + +/// Board state behind one lock. +struct Board { + regs: HashMap, + ucode: Vec, + flags: u32, + flag_enable: u32, + interrupt_enable: u32, + /// Context-switch packet words still to swallow from the FIFO. + context_words: u32, + ge_readback: [u32; 2], + ge: [GeDiag; 2], + /// Pending diagnostic readback words, oldest first. + ge_out: std::collections::VecDeque, + /// One parser per FIFO port (user, privileged): the two may interleave. + cfifo: [Cfifo; 2], + /// Host DMA engine registers as 32-bit words; an eight-byte register + /// takes two, high word first. + dma_regs: [u32; 0x80], + raster_if_regs: [u32; 0x10], + formatter: u32, + /// A board-to-host DMA started on the host side, waiting for the raster + /// engine to be started (xfrcontrol = 9). + dma_read_pending: Option, + /// System memory, for DMA. + mem: Option>, + /// DMA transfers logged so far (bring-up). + dma_logged: u32, + dcb: dcb::Dcb, + raster: raster::Raster, + /// Commands and registers not modelled yet, reported once each. + unhandled: HashSet, +} + +impl Board { + fn new(kind: ImpactSlot) -> Self { + // Board version bytes: [RA/RB boards + TRAMs, product + GE count]. + let bdvers = match kind { + ImpactSlot::Solid => [0x70, 0x21], + ImpactSlot::High => [0x70, 0x01], + ImpactSlot::Max => [0x00, 0x02], + ImpactSlot::None => [0, 0], + }; + Board { + regs: HashMap::new(), + ucode: vec![0; ((host::UCODE_END - host::UCODE) / 4) as usize], + flags: 0, + flag_enable: 0, + interrupt_enable: 0, + context_words: 0, + ge_readback: [0; 2], + ge: [GeDiag::default(), GeDiag::default()], + ge_out: std::collections::VecDeque::new(), + cfifo: [Cfifo::default(), Cfifo::default()], + dma_regs: [0; 0x80], + raster_if_regs: [0; 0x10], + formatter: 0, + dma_read_pending: None, + mem: None, + dma_logged: 0, + dcb: dcb::Dcb::new(bdvers, [0xFB, 0xFB]), + raster: raster::Raster::default(), + unhandled: HashSet::new(), + } + } + + fn note(&mut self, what: String) { + if self.unhandled.len() < 256 && self.unhandled.insert(what.clone()) { + eprintln!("mgras: not modelled yet: {what}"); + } + } + + /// Push one 32-bit word into the command FIFO. Returns true when the + /// framebuffer changed. + fn cfifo_word(&mut self, port: usize, w: u32) -> bool { + // A context switch's incoming state follows the save phase as raw + // words; the command processor would load it. Consume it here, and + // report the load done after the last word. + if self.context_words > 0 { + self.context_words -= 1; + if self.context_words == 0 { + self.flags |= host::FLAG_CONTEXT_LOADED; + } + return false; + } + let f = &mut self.cfifo[port]; + if f.need == 0 { + if w & 0x8000_0000 != 0 { + // Pixel data: byte count in bits 19:0, sent as doublewords. + f.pixel = true; + f.cmd = w; + f.need = (((w & 0xF_FFFF) + 7) / 8) * 2; + } else { + f.pixel = false; + f.cmd = (w >> 8) & 0x1FFF; + f.need = ((w & 0xFF) + 3) / 4; + } + f.data.clear(); + if f.need == 0 { + return self.dispatch(port); + } + return false; + } + f.need -= 1; + if f.data.len() < 64 { + f.data.push(w); + } + if f.need == 0 { self.dispatch(port) } else { false } + } + + fn dispatch(&mut self, port: usize) -> bool { + let cmd = self.cfifo[port].cmd; + let data = std::mem::take(&mut self.cfifo[port].data); + if self.cfifo[port].pixel { + self.note(format!("pixel data command {cmd:#010x}")); + return false; + } + let changed = if cmd >= cmd::RASTER_BASE { + let r = cmd & 0x3FF; + let exec = cmd & cmd::RASTER_EXECUTE != 0; + let changed = match data.as_slice() { + [] => self.raster.write(r, 0, exec), + [d] => self.raster.write(r, *d, exec), + [hi, lo, ..] => { + self.raster.write(r, *hi, false); + self.raster.write(r + 1, *lo, exec) + } + }; + if r == raster::reg::XFRCONTROL && data.first() == Some(&9) { + changed | self.start_read_dma() + } else { + changed + } + } else if cmd == cmd::SET_DONE_FLAG { + self.flags |= host::FLAG_DONE; + false + } else if cmd >= cmd::DMA_BASE { + let n = (cmd & 0x1FF) as usize; + let v = data.first().copied().unwrap_or(0); + if cmd >= cmd::FORMATTER { + self.formatter = v; + false + } else if cmd >= cmd::RASTER_IF_BASE { + self.raster_if_regs[n & 0xF] = v; + false + } else { + // Eight-byte registers arrive as a high word, then a low word, + // and fill two register slots. + let n = n & 0x7F; + for (i, d) in data.iter().take(2).enumerate() { + self.dma_regs[(n + i) & 0x7F] = *d; + } + if data.is_empty() { + self.dma_regs[n] = 0; + } + if n == dma::START { self.dma_start(v) } else { false } + } + } else if cmd == cmd::CP_SCHEDULE_SWAP { + // No retrace wait: the swap is reported done at once. + self.flags |= host::FLAG_CP0; + false + } else if cmd < cmd::CP_LIMIT { + self.note(format!("command-processor token {cmd:#x} ({} data words)", data.len())); + false + } else { + self.note(format!("command {cmd:#x}")); + false + }; + self.cfifo[port].data = data; + changed + } + + /// The host DMA engine's start word. Host to board runs now: the raster + /// engine was armed first. Board to host waits for the raster engine. + fn dma_start(&mut self, word: u32) -> bool { + if word & 1 == 0 { + return false; + } + if word & 8 != 0 { + self.dma_read_pending = Some(word); + return false; + } + match self.raster.transfer_armed() { + Some(false) => self.dma(word, false), + _ => { + self.note(format!("host DMA start {word:#x} with no write transfer armed")); + false + } + } + } + + fn start_read_dma(&mut self) -> bool { + let Some(word) = self.dma_read_pending.take() else { + self.note("raster DMA read started with no host DMA pending".into()); + return false; + }; + if self.raster.transfer_armed() != Some(true) { + self.note(format!("host DMA read {word:#x} with no read transfer armed")); + return false; + } + self.dma(word, true) + } + + /// Run a DMA between host memory and the armed raster transfer, a line at + /// a time. Host addresses are logical within the pool and translate + /// through its page table: one 32-bit frame number per 4 KB page. + fn dma(&mut self, word: u32, read: bool) -> bool { + let Some(mem) = self.mem.clone() else { return false }; + let pool = ((word >> 1) & 3) as usize; + let table = self.dma_regs[dma::TABLE_BASE + 2 * pool + 1] & !3; + let base = self.dma_regs[dma::ROW_START].wrapping_add(self.dma_regs[dma::ROW_OFFSET]); + let stride = self.dma_regs[dma::STRIDE]; + let lines = self.dma_regs[dma::LINES]; + let len = self.dma_regs[dma::LINE_BYTES]; + if self.dma_logged < 16 { + self.dma_logged += 1; + eprintln!( + "mgras: DMA {} pool {pool} table {table:#x} base {base:#x} stride {stride} lines {lines} bytes {len} pglist {:#x} shape {:?}", + if read { "read" } else { "write" }, + self.dma_regs[dma::PAGE_LIST], + self.raster.transfer_shape() + ); + } + let frame_of = |page: u32| -> Option { + let r = mem.read32(table.wrapping_add(4 * page)); + r.is_ok().then_some(r.data << 12) + }; + let mut cached: Option<(u32, u32)> = None; + let mut phys = |l: u32| -> Option { + let page = l >> 12; + let f = match cached { + Some((p, f)) if p == page => f, + _ => { + let f = frame_of(page)?; + cached = Some((page, f)); + f + } + }; + Some(f | (l & 0xFFF)) + }; + let mut changed = false; + for i in 0..lines { + let a = base.wrapping_add(i.wrapping_mul(stride)); + if read { + let bytes = self.raster.dma_read_line(i); + for (k, b) in bytes.iter().take(len as usize).enumerate() { + let Some(pa) = phys(a + k as u32) else { return changed }; + mem.write8(pa, *b); + } + } else { + let mut bytes = Vec::with_capacity(len as usize); + for k in 0..len { + let Some(pa) = phys(a + k) else { return changed }; + let r = mem.read8(pa); + bytes.push(if r.is_ok() { r.data } else { 0 }); + } + self.raster.dma_write_line(i, &bytes); + changed = true; + } + } + changed + } + + /// Flags as the host reads them, including those derived from state. + fn all_flags(&self) -> u32 { + self.flags | if self.ge_out.is_empty() { 0 } else { host::FLAG_GE_DIAG } + } + + /// Whether the general interrupt (GIO line 1) is asserted: some enabled + /// flag is set. It is a level, held until the handler clears the flag or + /// its enable. + fn general_irq(&self) -> bool { + self.all_flags() & self.interrupt_enable & host::INTR_CAUSES != 0 + } + + /// Video timing chip display control bit 0: vertical retrace interrupts + /// enabled. + fn retrace_enabled(&self) -> bool { + self.dcb.vc3.regs[0x1E] & 1 != 0 + } + + /// A write to a geometry engine's diagnostic data or address port. + fn ge_diag_write(&mut self, off: u32, val: u32) { + let n = host::GE_DIAG.iter().position(|&(d, a)| off == d || off == a).unwrap(); + let (data_port, _) = host::GE_DIAG[n]; + let ge = &mut self.ge[n]; + if off != data_port { + if val & 0x8000_0000 != 0 { + // Read request for `val & 0x7FFF_FFFF` lines from `addr`; it + // replaces anything a previous request left unread. + let words = ge.readback(val & 0x7FFF_FFFF); + self.ge_out.clear(); + self.ge_out.extend(words); + } else { + ge.addr = val; + ge.word = 0; + } + return; + } + if ge.addr >= GeDiag::UCODE_BASE { + let line = ge.ucode.entry(ge.addr).or_insert([0; 3]); + line[ge.word] = val; + ge.word += 1; + if ge.word == 3 { + ge.word = 0; + ge.addr += 1; + } + } else if ge.addr == GeDiag::EXEC_CONTROL && val & 1 != 0 { + // Started: the version program's answer is waiting (revision 1). + self.ge_readback = [0, 1]; + self.flags |= host::FLAG_GE_DATA; + } + } + + fn status(&self) -> u32 { + // A vertical blank of about 1 ms in every 60 Hz frame. + let us = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_micros() as u64 % 16_667) + .unwrap_or(0); + host::STATUS_IDLE | if us < 1000 { host::STATUS_VBLANK } else { 0 } + } + + fn read(&mut self, off: u32, bits: u32) -> u64 { + match off { + 0 => GIO_ID as u64, + host::UCODE..=0x45FFF => self.ucode[((off - host::UCODE) / 4) as usize] as u64, + 0x60000..=0x67FFF => { + let t = dcb::Txn::decode(off); + let d = self.dcb.read(t); + t.load_value(bits, d) + } + host::STATUS => self.status() as u64, + host::FIFOSTATUS | host::GIOSTATUS | host::DMABUSY => 0, + host::SET_FLAGS | host::CLEAR_FLAGS | host::SET_FLAGS_PRIVILEGED | host::CLEAR_FLAGS_PRIVILEGED => self.all_flags() as u64, + host::FLAG_ENABLE_SET | host::FLAG_ENABLE_CLEAR => self.flag_enable as u64, + host::INTERRUPT_ENABLE_SET | host::INTERRUPT_ENABLE_CLEAR => self.interrupt_enable as u64, + host::GE_DIAG_READ => self.ge_out.pop_front().unwrap_or(0) as u64, + host::RASTER_IF_CONTEXT..=0x50228 => self.raster_if_regs[((off - host::RASTER_IF_CONTEXT) / 4) as usize] as u64, + host::DMA_CONTEXT..=0x504FF => self.dma_regs[((off - host::DMA_CONTEXT) / 4) as usize] as u64, + host::PIO_READ_HI => self.raster.pio_read_hi() as u64, + host::PIO_READ_LO => self.raster.pio_read_lo() as u64, + host::GE_READBACK_HI => self.ge_readback[0] as u64, + host::GE_READBACK_LO => self.ge_readback[1] as u64, + host::RASTER..=0x7FFFF => { + let r = (off & 0xFFC) >> 2; + if bits == 64 { + ((self.raster.read(r) as u64) << 32) | self.raster.read(r + 1) as u64 + } else { + self.raster.read(r) as u64 + } + } + _ => self.regs.get(&off).copied().unwrap_or(0) as u64, + } + } + + /// Returns true when the framebuffer changed. + fn write(&mut self, off: u32, bits: u32, val: u64) -> bool { + match off { + host::UCODE..=0x45FFF => { + self.ucode[((off - host::UCODE) / 4) as usize] = val as u32 & 0xFF_FFFF; + false + } + 0x60000..=0x67FFF => { + let t = dcb::Txn::decode(off); + self.dcb.write(t, t.store_data(bits, val)); + // Display-side state (colormaps, modes, the cursor) changes + // what is shown without touching the framebuffer. Index + // writes (select 1 and below) change nothing by themselves. + t.crs >= 2 || t.dev == dcb::DEV_VC3 + } + off if host::GE_DIAG.iter().any(|&(d, a)| off == d || off == a) => { + self.ge_diag_write(off, val as u32); + false + } + host::FLAG_ENABLE_SET => { self.flag_enable |= val as u32; false } + host::FLAG_ENABLE_CLEAR => { self.flag_enable &= !(val as u32); false } + host::INTERRUPT_ENABLE_SET => { self.interrupt_enable |= val as u32; false } + host::INTERRUPT_ENABLE_CLEAR => { self.interrupt_enable &= !(val as u32); false } + host::CONTEXT_SWITCH => { + // The save phase completes at once; the incoming context + // follows through the FIFO (see `cfifo_word`). + self.flags |= host::FLAG_CONTEXT_SAVED; + self.context_words = host::CONTEXT_SWITCH_WORDS; + false + } + host::SET_FLAGS | host::SET_FLAGS_PRIVILEGED => { self.flags |= val as u32; false } + host::CLEAR_FLAGS | host::CLEAR_FLAGS_PRIVILEGED => { self.flags &= !(val as u32); false } + host::CFIFO | 0x70084 | host::CFIFO_PRIVILEGED | 0x50084 => { + let port = if off >= host::STATUS { 0 } else { 1 }; + if bits == 64 { + let a = self.cfifo_word(port, (val >> 32) as u32); + let b = self.cfifo_word(port, val as u32); + a | b + } else { + self.cfifo_word(port, val as u32) + } + } + host::RASTER..=0x7FFFF => { + let r = (off & 0xFFC) >> 2; + // Registers at 0x7C000 + 4r; the same register at +0x1000 + // also executes the primitive in the IR after the write. + let exec = off & 0x1000 != 0; + if bits == 64 { + self.raster.write(r, (val >> 32) as u32, false); + self.raster.write(r + 1, val as u32, exec) + } else { + self.raster.write(r, val as u32, exec) + } + } + 0x80000..=0xFFFFF => { + self.note(format!("fast-path command window write at {off:#x}")); + false + } + _ => { + self.regs.insert(off, val as u32); + false + } + } + } + + /// The window ID a frame for the window at (`x`, `y`), `w` x `h` (screen + /// coordinates, top-down) should be painted through: the commonest ID + /// over a grid of samples inside it that has an RGB display mode. None + /// if no sampled pixel is in an RGB window. + fn window_did(&self, x: i32, y: i32, w: usize, h: usize) -> Option { + let mut counts = [0u32; 32]; + let mut runs = Vec::new(); + for sy in 0..16 { + let py = y + (h as i32 * (2 * sy + 1)) / 32; + if !(0..raster::HEIGHT as i32).contains(&py) { + continue; + } + self.dcb.vc3.main_did_runs(py as usize, &mut runs); + for sx in 0..16 { + let px = x + (w as i32 * (2 * sx + 1)) / 32; + if !(0..raster::WIDTH as i32).contains(&px) { + continue; + } + let did = runs.iter().rev().find(|r| r.0 as i32 <= px).map_or(0, |r| r.1); + if self.dcb.xmap.main_mode(did as u32) & 0x1F >= 4 { + counts[did as usize & 31] += 1; + } + } + } + let (did, n) = counts.iter().enumerate().max_by_key(|&(_, n)| *n)?; + (*n > 0).then_some(did as u8) + } + + /// Paint a host GL frame (`bgra`: `h` rows, top first, `stride` bytes + /// each) into the framebuffer at screen position (`x`, `y`), wherever the + /// pixel belongs to the frame's window ID, so windows over it stay over it. + /// The pixels become part of the framebuffer, as GL's would on the board, + /// for anything that reads them back. False when no window ID fits. + fn composite(&mut self, x: i32, y: i32, bgra: &[u8], stride: usize, w: usize, h: usize) -> bool { + let Some(target) = self.window_did(x, y, w, h) else { return false }; + let mut runs = Vec::new(); + for row in 0..h { + let sy = y + row as i32; + if !(0..raster::HEIGHT as i32).contains(&sy) { + continue; + } + self.dcb.vc3.main_did_runs(sy as usize, &mut runs); + if runs.is_empty() || runs[0].0 != 0 { + runs.insert(0, (0, 0)); + } + let fb_row = (raster::HEIGHT - 1 - sy as usize) * raster::WIDTH; + for (k, &(x0, did)) in runs.iter().enumerate() { + if did != target { + continue; + } + let x1 = runs.get(k + 1).map_or(raster::WIDTH as i32, |r| r.0 as i32); + let lo = (x0 as i32).max(x).max(0); + let hi = x1.min(x + w as i32).min(raster::WIDTH as i32); + for sx in lo..hi { + let i = row * stride + (sx - x) as usize * 4; + let Some(p) = bgra.get(i..i + 4) else { break }; + self.raster.fb[fb_row + sx as usize] = p[2] as u32 | (p[1] as u32) << 8 | (p[0] as u32) << 16; + } + } + } + true + } + + /// Scan the framebuffer out to `0xFF_BB_GG_RR` (the compositor's order, + /// red in the low byte), stride 2048, top row first. Each pixel's window + /// ID (from the video timing chip's tables) picks its display mode. Modes + /// with a pixel format (bits 4:0) of 4 and up are RGB; the others are + /// colour index, into the colormap block that bits 9:5 choose. Both go + /// through the DAC gamma. + fn scanout(&self, out: &mut [u32]) { + if self.dcb.dac.pixmask() == 0 { + out.iter_mut().for_each(|p| *p = 0xFF00_0000); + return; + } + let pal = &self.dcb.cmap[0].pal; + let gamma = &self.dcb.dac.gamma; + let gamma_rgb = |r: u32, g: u32, b: u32| -> u32 { + let g1 = |v: u32, comp: usize| (gamma[(v & 0xFF) as usize][comp] >> 2) as u32; + 0xFF00_0000 | (g1(b, 2) << 16) | (g1(g, 1) << 8) | g1(r, 0) + }; + // Per window ID: None for an RGB mode, else the colormap block's + // entries through the gamma tables. + let luts: Vec>> = (0..32u32) + .map(|did| { + let mode = self.dcb.xmap.main_mode(did); + if mode & 0x1F >= 4 { + return None; + } + let base = ((mode >> 5) & 0x1F) as usize * 256; + Some( + (0..4096usize) + .map(|i| { + let c = pal.get((base + i) % pal.len().max(1)).copied().unwrap_or(0); + gamma_rgb(c >> 16, c >> 8, c) + }) + .collect(), + ) + }) + .collect(); + let mut runs = Vec::new(); + for row in 0..raster::HEIGHT { + let y = raster::HEIGHT - 1 - row; + let src = &self.raster.fb[y * raster::WIDTH..(y + 1) * raster::WIDTH]; + let dst = &mut out[row * 2048..row * 2048 + raster::WIDTH]; + self.dcb.vc3.main_did_runs(row, &mut runs); + if runs.is_empty() || runs[0].0 != 0 { + runs.insert(0, (0, 0)); + } + for (k, &(x0, did)) in runs.iter().enumerate() { + let x0 = (x0 as usize).min(raster::WIDTH); + let x1 = runs.get(k + 1).map(|r| (r.0 as usize).min(raster::WIDTH)).unwrap_or(raster::WIDTH); + if x1 <= x0 { + continue; + } + match &luts[did as usize & 31] { + Some(lut) => { + for x in x0..x1 { + dst[x] = lut[(src[x] & 0xFFF) as usize]; + } + } + None => { + for x in x0..x1 { + let v = src[x]; + dst[x] = gamma_rgb(v, v >> 8, v >> 16); + } + } + } + } + } + self.draw_overlay(out, &gamma_rgb); + self.draw_cursor(out, &gamma_rgb); + } + + /// Overlay planes over the main scanout: a nonzero pixel is a colour + /// index into the block its overlay mode names (bits 7:3); zero, or a + /// window ID whose overlay is off, shows the main planes. + fn draw_overlay(&self, out: &mut [u32], gamma_rgb: &dyn Fn(u32, u32, u32) -> u32) { + let pal = &self.dcb.cmap[0].pal; + let bases: Vec> = (0..32u32) + .map(|did| { + let mode = self.dcb.xmap.overlay_mode(did); + (mode != 0).then(|| ((mode >> 3) & 0x1F) as usize * 256) + }) + .collect(); + if bases.iter().all(Option::is_none) { + return; + } + let mut runs = Vec::new(); + for row in 0..raster::HEIGHT { + let y = raster::HEIGHT - 1 - row; + let src = &self.raster.overlay[y * raster::WIDTH..(y + 1) * raster::WIDTH]; + let dst = &mut out[row * 2048..row * 2048 + raster::WIDTH]; + self.dcb.vc3.overlay_did_runs(row, &mut runs); + if runs.is_empty() || runs[0].0 != 0 { + runs.insert(0, (0, 0)); + } + for (k, &(x0, did)) in runs.iter().enumerate() { + let Some(base) = bases[did as usize & 31] else { continue }; + let x0 = (x0 as usize).min(raster::WIDTH); + let x1 = runs.get(k + 1).map(|r| (r.0 as usize).min(raster::WIDTH)).unwrap_or(raster::WIDTH); + for x in x0..x1 { + let v = src[x] & 0xFF; + if v != 0 { + let c = pal.get((base + v as usize) % pal.len().max(1)).copied().unwrap_or(0); + dst[x] = gamma_rgb(c >> 16, c >> 8, c); + } + } + } + } + } + + /// Overlay the hardware cursor, its colours from the cursor colormap. + fn draw_cursor(&self, out: &mut [u32], gamma_rgb: &dyn Fn(u32, u32, u32) -> u32) { + let Some((cx, cy, size, glyph)) = self.dcb.vc3.cursor() else { return }; + let sram = &self.dcb.vc3.sram; + let pal = &self.dcb.cmap[0].pal; + let base = self.dcb.xmap.cursor_cmap_base(); + let words_per_row = size / 16; + let plane_words = size * words_per_row; + let bit = |plane: usize, row: usize, col: usize| -> u32 { + let w = sram[(glyph + plane * plane_words + row * words_per_row + col / 16) & 0x7FFF]; + (w >> (15 - col % 16)) as u32 & 1 + }; + for row in 0..size { + let y = cy + row as i32; + if !(0..raster::HEIGHT as i32).contains(&y) { + continue; + } + for col in 0..size { + let x = cx + col as i32; + if !(0..raster::WIDTH as i32).contains(&x) { + continue; + } + let c = bit(0, row, col) | bit(1, row, col) << 1; + if c != 0 { + let rgb = pal.get(base + c as usize).copied().unwrap_or(0xFF_FFFF); + out[y as usize * 2048 + x as usize] = gamma_rgb(rgb >> 16, rgb >> 8, rgb); + } + } + } + } +} + +/// Access trace for bring-up: every access to the board, one line each +/// (`R`/`W`, width in bits, physical address, value). Started from +/// `IRIS_MGRAS_TRACE=` or the monitor (`mgras trace `, `mgras +/// trace off`). Off, it costs one relaxed load per access. +mod trace { + use std::io::Write; + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::OnceLock; + use parking_lot::Mutex; + + static ON: AtomicBool = AtomicBool::new(false); + + fn sink() -> &'static Mutex>> { + static SINK: OnceLock>>> = OnceLock::new(); + SINK.get_or_init(|| { + let w = std::env::var_os("IRIS_MGRAS_TRACE").and_then(|p| open(&p).ok()); + ON.store(w.is_some(), Ordering::Relaxed); + Mutex::new(w) + }) + } + + fn open(path: &std::ffi::OsStr) -> std::io::Result> { + let f = std::fs::OpenOptions::new().create(true).append(true).open(path)?; + Ok(std::io::BufWriter::new(f)) + } + + /// Start tracing to `path`, or stop with `None`. + pub fn set(path: Option<&str>) -> std::io::Result<()> { + let mut s = sink().lock(); + if let Some(mut w) = s.take() { + let _ = w.flush(); + } + if let Some(p) = path { + *s = Some(open(std::ffi::OsStr::new(p))?); + } + ON.store(s.is_some(), Ordering::Relaxed); + Ok(()) + } + + pub fn init() { + let _ = sink(); + } + + pub fn note(dir: char, bits: u32, addr: u32, val: u64) { + if !ON.load(Ordering::Relaxed) { + return; + } + if let Some(w) = sink().lock().as_mut() { + let _ = writeln!(w, "{dir}{bits} {addr:08x} {val:x}"); + } + } +} + +/// The board's three interrupt lines to the GIO slot. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Line { + /// GIO interrupt 0: command FIFO high/low water. + Fifo, + /// GIO interrupt 1: the general graphics interrupt. + General, + /// GIO interrupt 2: vertical retrace. + Retrace, +} + +pub struct Mgras { + kind: ImpactSlot, + ioc: crate::ioc::Ioc, + /// System memory, for the board's DMA (it is a GIO bus master). + sys_mem: Mutex>>, + /// Current level of the general interrupt line. + general: AtomicBool, + board: Mutex, + dirty: AtomicBool, + running: AtomicBool, + refresh: Mutex>>, + renderer: Mutex>>, + /// The finished frame, handed to the renderer as a prebuilt picture + /// (`Rex3Screen::prebuilt`), as GR2 does. + screen: Mutex, + screenshot_pending: AtomicBool, + heartbeat: Arc, + fasttick: Arc, + cycles: Mutex, +} + +impl Mgras { + pub fn new(cfg: &ImpactSection, ioc: crate::ioc::Ioc, heartbeat: Arc, fasttick: Arc) -> Self { + trace::init(); + Mgras { + kind: cfg.gfx, + ioc, + sys_mem: Mutex::new(None), + general: AtomicBool::new(false), + board: Mutex::new(Board::new(cfg.gfx)), + dirty: AtomicBool::new(true), + running: AtomicBool::new(false), + refresh: Mutex::new(None), + renderer: Mutex::new(None), + screen: Mutex::new(crate::disp::Rex3Screen::new()), + screenshot_pending: AtomicBool::new(false), + heartbeat, + fasttick, + cycles: Mutex::new(crate::mips_core::CyclesPtr::dangling()), + } + } + + /// Give the board its path to system memory, for DMA. + pub fn set_phys(&self, mem: Arc) { + self.board.lock().mem = Some(mem.clone()); + *self.sys_mem.lock() = Some(mem); + } + + /// Drive one of the board's interrupt lines (graphics slot wiring). + fn set_line(&self, line: Line, active: bool) { + use crate::ioc::IocInterrupt; + let src = match line { + Line::Fifo => IocInterrupt::GioSgFifo, + Line::General => IocInterrupt::GioSgGraphics, + Line::Retrace => IocInterrupt::GioSgRetrace, + }; + self.ioc.set_interrupt(src, active); + } + + /// Wire up the CPU cycle counter for the status bar's MIPS figure. + pub fn set_cpu_cycles(&self, ptr: crate::mips_core::CyclesPtr) { + *self.cycles.lock() = ptr; + } + + fn offset(&self, addr: u32) -> Option { + let off = addr.wrapping_sub(MGRAS_SLOT_GFX_BASE); + (off < MAP_SIZE).then_some(off) + } + + /// Bring the general interrupt line in line with the board's flags. + fn update_general(&self, level: bool) { + if self.general.swap(level, Ordering::AcqRel) != level { + self.set_line(Line::General, level); + } + } + + fn do_read(&self, addr: u32, bits: u32) -> u64 { + let v = match self.offset(addr) { + Some(off) => { + let mut b = self.board.lock(); + let v = b.read(off, bits); + let irq = b.general_irq(); + drop(b); + self.update_general(irq); + v + } + None => 0, + }; + trace::note('R', bits, addr, v); + v + } + + fn do_write(&self, addr: u32, bits: u32, val: u64) { + trace::note('W', bits, addr, val); + if let Some(off) = self.offset(addr) { + let mut b = self.board.lock(); + let changed = b.write(off, bits, val); + let irq = b.general_irq(); + drop(b); + if changed { + self.dirty.store(true, Ordering::Release); + } + self.update_general(irq); + } + } + + fn refresh_loop(self: &Arc) { + let frame = std::time::Duration::from_micros(16_667); + { + let mut screen = self.screen.lock(); + screen.width = raster::WIDTH; + screen.height = raster::HEIGHT; + screen.fb_rgb.fill(0xFF00_0000); + screen.prebuilt = true; + } + let mut overlay = crate::debug_overlay::DebugOverlay::new(); + let mut status_bar = crate::disp::StatusBar::new(); + let mut sbtex = crate::disp::StatusBarTexture::new(); + let mut sized = false; + let mut last_pending = 0u64; + let mut idle_frames = 0u32; + const PERSISTENT: u64 = crate::rex3::Rex3::HB_LED_RED | crate::rex3::Rex3::HB_LED_GREEN; + + while self.running.load(Ordering::Relaxed) { + let start = std::time::Instant::now(); + let stats = crate::disp::BarStats { + now: start, + hb: self.heartbeat.fetch_and(PERSISTENT, Ordering::Relaxed), + cycles: self.cycles.lock().get(), + fasttick: self.fasttick.load(Ordering::Relaxed), + decoded_delta: 0, + l1i_hits: 0, + l1i_fetches: 0, + uncached: 0, + count_hz: 0, + gfifo_pending: 0, + }; + let retrace = { + let mut b = self.board.lock(); + if b.raster.flush_if_stale(&mut last_pending) { + self.dirty.store(true, Ordering::Release); + } + b.retrace_enabled() + }; + // Vertical retrace: a pulse per frame. The handler acknowledges + // nothing on the board, so the line must drop again by itself. + if retrace { + self.set_line(Line::Retrace, true); + std::thread::sleep(std::time::Duration::from_micros(500)); + self.set_line(Line::Retrace, false); + } + let dirty = self.dirty.swap(false, Ordering::AcqRel); + let shot = self.screenshot_pending.swap(false, Ordering::Relaxed); + idle_frames += 1; + if dirty || shot || idle_frames >= 6 { + idle_frames = 0; + let mut screen = self.screen.lock(); + if dirty || shot { + let screen = &mut *screen; + self.board.lock().scanout(&mut screen.fb_rgb); + // The frame is already final RGB, so keep `rgba` (what CI + // screenshots read) current without a renderer readback, + // as GR2 does. This also makes screenshots work headless. + for y in 0..raster::HEIGHT { + let row = y * 2048; + screen.rgba[row..row + raster::WIDTH].copy_from_slice(&screen.fb_rgb[row..row + raster::WIDTH]); + } + screen.status_bar_only = false; + } else { + screen.status_bar_only = true; + } + if let Some(r) = self.renderer.lock().as_mut() { + if !sized { + r.resize(raster::WIDTH, raster::HEIGHT); + sized = true; + } + r.present(&mut screen, &mut overlay, &mut status_bar, &mut sbtex, &stats, shot, None, None); + } + } + if let Some(rest) = frame.checked_sub(start.elapsed()) { + std::thread::sleep(rest); + } + } + // The renderer's GL state belongs to this thread (its context is + // current here), so it is torn down here and nowhere else. + if let Some(r) = self.renderer.lock().as_mut() { + r.stop(); + } + } + + /// Start the display refresh thread. + pub fn start_display(self: &Arc) { + if self.running.swap(true, Ordering::AcqRel) { + return; + } + let me = Arc::clone(self); + *self.refresh.lock() = Some( + std::thread::Builder::new() + .name("MGRAS-Refresh".into()) + .spawn(move || me.refresh_loop()) + .expect("spawn MGRAS refresh thread"), + ); + } + + pub fn stop_display(&self) { + self.running.store(false, Ordering::Release); + if let Some(h) = self.refresh.lock().take() { + let _ = h.join(); + } + } +} + +impl crate::gfx_display::GfxDisplay for Mgras { + fn renderer_slot(&self) -> &Mutex>> { &self.renderer } + fn screen(&self) -> &Mutex { &self.screen } + fn request_screenshot(&self) { self.screenshot_pending.store(true, Ordering::Relaxed); } + fn cycles(&self) -> crate::mips_core::CyclesPtr { *self.cycles.lock() } +} + +impl Device for Mgras { + fn step(&self, _cycles: u64) {} + fn stop(&self) {} + fn start(&self) {} + fn is_running(&self) -> bool { self.running.load(Ordering::Relaxed) } + fn get_clock(&self) -> u64 { 0 } + + fn register_commands(&self) -> Vec<(String, String)> { + vec![("mgras".into(), "IMPACT graphics: mgras (board state) | mgras shot (save the displayed frame) | mgras dump (raw display state) | mgras trace |off".into())] + } + + fn execute_command(&self, cmd: &str, args: &[&str], mut w: Box) -> Result<(), String> { + if cmd != "mgras" { + return Err(format!("unknown command: {cmd}")); + } + if let ["trace", what] = args { + let r = if *what == "off" { trace::set(None) } else { trace::set(Some(what)) }; + r.map_err(|e| format!("mgras trace: {e}"))?; + return writeln!(w, "trace {what}").map_err(|e| e.to_string()); + } + if let ["dump", path] = args { + let b = self.board.lock(); + dump_state(path, &b).map_err(|e| format!("mgras dump: {e}"))?; + return writeln!(w, "dumped {path}").map_err(|e| e.to_string()); + } + if let ["shot", path] = args { + let mut frame = vec![0u32; 2048 * raster::HEIGHT]; + self.board.lock().scanout(&mut frame); + save_png(path, &frame).map_err(|e| format!("mgras shot: {e}"))?; + return writeln!(w, "saved {path}").map_err(|e| e.to_string()); + } + let b = self.board.lock(); + let e = |r: std::io::Result<()>| r.map_err(|e| e.to_string()); + e(writeln!(w, "IMPACT {:?} at {:#010x}", self.kind, MGRAS_SLOT_GFX_BASE))?; + e(writeln!(w, " flags {:#010x} DAC pixmask {:#04x} XMAP DID0 mode {:#x}", + b.flags, b.dcb.dac.pixmask(), b.dcb.xmap.main_mode(0)))?; + e(writeln!(w, " fill modes seen: {:x?}", b.raster.fillmodes_seen))?; + let mut hist: HashMap = HashMap::new(); + for v in &b.raster.fb { + *hist.entry(*v).or_default() += 1; + } + let mut top: Vec<_> = hist.into_iter().collect(); + top.sort_by(|a, b| b.1.cmp(&a.1)); + e(writeln!(w, " framebuffer values (value, pixels): {:x?}", &top[..top.len().min(12)]))?; + let pal = &b.dcb.cmap[0].pal; + let blocks: Vec = (0..pal.len() / 256) + .filter(|k| pal[k * 256..(k + 1) * 256].iter().any(|c| *c != 0)) + .collect(); + e(writeln!(w, " colormap blocks in use (of {}): {:?}", pal.len() / 256, blocks))?; + for k in blocks.iter().take(4) { + e(writeln!(w, " block {k}: {:06x?}", &pal[k * 256..k * 256 + 8]))?; + } + let modes: Vec<(u32, u32)> = (0..32).map(|d| (d, b.dcb.xmap.main_mode(d))).filter(|m| m.1 != 0).collect(); + e(writeln!(w, " XMAP main modes (did, mode): {:x?}", modes))?; + let mut runs = Vec::new(); + for y in [0usize, 100, 400, 700, 1023] { + b.dcb.vc3.main_did_runs(y, &mut runs); + e(writeln!(w, " scanline {y} DID runs: {:?}", runs))?; + } + for u in &b.unhandled { + e(writeln!(w, " not modelled: {u}"))?; + } + Ok(()) + } +} + +/// Save the display state for offline study: little-endian u32 sections, in +/// order: framebuffer (`WIDTH * HEIGHT`, row 0 at the bottom), overlay (same +/// size), colormap 0, the 32 main XMAP modes, then the video timing chip's +/// registers and SRAM as u32s. +fn dump_state(path: &str, b: &Board) -> std::io::Result<()> { + let mut f = std::io::BufWriter::new(std::fs::File::create(path)?); + let mut put = |v: u32| f.write_all(&v.to_le_bytes()); + for v in b.raster.fb.iter().chain(b.raster.overlay.iter()).chain(b.dcb.cmap[0].pal.iter()) { + put(*v)?; + } + for d in 0..32 { + put(b.dcb.xmap.main_mode(d))?; + } + for v in b.dcb.vc3.regs.iter().chain(b.dcb.vc3.sram.iter()) { + put(*v as u32)?; + } + Ok(()) +} + +/// Write a scanned-out frame (`0xFF_BB_GG_RR`, stride 2048) as a PNG. +fn save_png(path: &str, frame: &[u32]) -> Result<(), String> { + let file = std::fs::File::create(path).map_err(|e| e.to_string())?; + let mut enc = png::Encoder::new(std::io::BufWriter::new(file), raster::WIDTH as u32, raster::HEIGHT as u32); + enc.set_color(png::ColorType::Rgb); + enc.set_depth(png::BitDepth::Eight); + let mut out = enc.write_header().map_err(|e| e.to_string())?; + let mut rows = Vec::with_capacity(raster::WIDTH * raster::HEIGHT * 3); + for y in 0..raster::HEIGHT { + for px in &frame[y * 2048..y * 2048 + raster::WIDTH] { + rows.extend_from_slice(&[*px as u8, (px >> 8) as u8, (px >> 16) as u8]); + } + } + out.write_image_data(&rows).map_err(|e| e.to_string()) +} + +impl Saveable for Mgras { + // Bring-up: board state is not snapshotted yet. + fn save_state(&self) -> toml::Value { + toml::Value::Table(toml::map::Map::new()) + } + + fn load_state(&self, _v: &toml::Value) -> Result<(), String> { + Ok(()) + } +} + +impl BusDevice for Mgras { + fn read32(&self, addr: u32) -> BusRead32 { BusRead32::ok(self.do_read(addr, 32) as u32) } + fn write32(&self, addr: u32, val: u32) -> u32 { self.do_write(addr, 32, val as u64); BUS_OK } + fn read8(&self, addr: u32) -> BusRead8 { BusRead8::ok(self.do_read(addr, 8) as u8) } + fn write8(&self, addr: u32, val: u8) -> u32 { self.do_write(addr, 8, val as u64); BUS_OK } + fn read16(&self, addr: u32) -> BusRead16 { BusRead16::ok(self.do_read(addr, 16) as u16) } + fn write16(&self, addr: u32, val: u16) -> u32 { self.do_write(addr, 16, val as u64); BUS_OK } + fn read64(&self, addr: u32) -> BusRead64 { BusRead64::ok(self.do_read(addr, 64)) } + fn write64(&self, addr: u32, val: u64) -> u32 { self.do_write(addr, 64, val); BUS_OK } +} diff --git a/src/mgras/raster.rs b/src/mgras/raster.rs new file mode 100644 index 00000000..0d297160 --- /dev/null +++ b/src/mgras/raster.rs @@ -0,0 +1,823 @@ +//! The raster subsystem: the raster engine's register file, its indirect +//! device space, pixel transfers, and the framebuffer it draws into. +//! +//! Registers are numbered 0..0x3FF. A register write may carry an "execute" +//! flag, which runs the primitive held in the instruction register (IR) once +//! the write has landed. +//! +//! Coordinates: primitives give block corners in window coordinates. The +//! window origin (`xywin`: y in the high half, x in the low) is added, and +//! with Y-flip set in `config` y runs downward from it. The PROM draws with no +//! origin and no flip; the X server sets the origin to the top row and flips, +//! so it draws top-down. The framebuffer has row 0 at the bottom. +//! +//! A block runs one of several ways, chosen by the fill mode's block type: +//! fill it (fast fill uses the fill colour registers, others the red +//! iterator), stipple it with character data, or move pixels in or out of it +//! (by PIO through the character registers, or by DMA). + +use std::collections::{HashMap, VecDeque}; + +pub const WIDTH: usize = 1280; +pub const HEIGHT: usize = 1024; + +/// Raster registers. Those OpenBSD's impact(4) driver also uses carry its +/// names; the rest are named for what they do here. +pub mod reg { + /// The instruction register: the primitive a write with "execute" runs. + pub const IR: u32 = 0x013; + pub const LINE_START: u32 = 0x040; + pub const LINE_END: u32 = 0x041; + pub const IR_ALIAS: u32 = 0x045; + pub const BLOCKXYSTARTI: u32 = 0x046; + pub const BLOCKXYENDI: u32 = 0x047; + /// Packed RGB colour for character and line drawing in RGB modes. + pub const PACKEDCOLOR: u32 = 0x05B; + pub const RED: u32 = 0x05C; + pub const CHAR_H: u32 = 0x070; + pub const CHAR_L: u32 = 0x071; + pub const XFRCONTROL: u32 = 0x102; + pub const FILLMODE: u32 = 0x110; + pub const CONFIG: u32 = 0x112; + pub const XYWIN: u32 = 0x115; + /// The clip rectangle: x and y ranges, each `min << 16 | max`, and its + /// control (bit 0 enable, bit 4 keep the inside rather than the outside). + pub const CLIP_X: u32 = 0x147; + pub const CLIP_Y: u32 = 0x148; + pub const CLIP_MODE: u32 = 0x14F; + pub const XFRSIZE: u32 = 0x153; + pub const XFRMODE: u32 = 0x159; + pub const LINE_STIPPLE: u32 = 0x15A; + /// The indirect device space: an address, then its data. + pub const INDIRECT_ADDR: u32 = 0x15C; + pub const INDIRECT_DATA: u32 = 0x15D; + pub const STATUS: u32 = 0x15E; + pub const PP1FILLMODE: u32 = 0x161; + /// Plane write mask (low planes, buffer A). + pub const COLORMASKLSBSA: u32 = 0x163; + pub const DRBPOINTERS: u32 = 0x16D; + /// Fast-fill colour: one 12-bit component each, or the index in R. + pub const FILL_COLOR_R: u32 = 0x176; + pub const FILL_COLOR_G: u32 = 0x177; + pub const FILL_COLOR_B: u32 = 0x178; +} + +/// IR opcodes: a line between two points, and a block (rectangle). +const OP_LINE: u32 = 0x5; +const OP_BLOCK: u32 = 0x8; +/// Fill mode: lines follow the 32-bit line stipple pattern. +const FILL_LINE_STIPPLE: u32 = 1 << 5; +/// Status: command FIFO empty, engine and pixel processors idle, revision 1. +const STATUS_IDLE: u32 = 0x100 | (1 << 4); +/// Config: Y-flip. +const CONFIG_YFLIP: u32 = 1 << 3; +/// Fill mode: fast fill (solid, from the fill colour registers). +const FILL_FAST: u32 = 1 << 20; +/// Pixel processor fill mode used when drawing window IDs, which live in +/// their own planes, not the colour planes. +const PP1_DRAW_CID: u32 = 0x14_2600; +/// Scanout pointers: the low nine bits say which planes are drawn, and this +/// value means the overlay planes (the main planes read 0x240). +const DRB_PLANES: u32 = 0x1FF; +const DRB_OVERLAY: u32 = 0x1C0; + +/// Block types (fill mode bits 24:22). +mod block { + pub const NORMAL: u32 = 1; + pub const PIO_READ: u32 = 2; + pub const PIO_WRITE: u32 = 3; + pub const DMA_READ: u32 = 4; + pub const DMA_WRITE: u32 = 5; +} + +/// Whether the pixel processors' pixel type (fill mode bits 10:8) is an RGB +/// one; the others are colour index. +fn rgb_pixtype(pp1fillmode: u32) -> bool { + matches!((pp1fillmode >> 8) & 7, 0 | 1 | 4) +} + +/// Pixel processor logic op (fill mode bit 2 enables it; bits 29:26 hold +/// the X11 function number) applied to source `s` and destination `d`. +fn logic_op(op: u32, s: u32, d: u32) -> u32 { + match op & 0xF { + 0x0 => 0, + 0x1 => s & d, + 0x2 => s & !d, + 0x3 => s, + 0x4 => !s & d, + 0x5 => d, + 0x6 => s ^ d, + 0x7 => s | d, + 0x8 => !(s | d), + 0x9 => !(s ^ d), + 0xA => !d, + 0xB => s | !d, + 0xC => !s, + 0xD => !s | d, + 0xE => !(s & d), + _ => !0, + } +} +const PP1_LOGIC_OP_ENABLE: u32 = 1 << 2; + +/// Framebuffer pixels: colour indices as they are, RGB as `0x00BBGGRR` with +/// eight bits per component. +fn pack_rgb(r: u32, g: u32, b: u32) -> u32 { + (r & 0xFF) | (g & 0xFF) << 8 | (b & 0xFF) << 16 +} + +/// A host pixel of transfer format (PixelFormat, CompType) to a framebuffer +/// pixel, and back. Only the RGB formats convert. +fn from_host(format: (u32, u32), v: u32) -> u32 { + let c4 = |s: u32| ((v >> s) & 0xF) * 0x11; + let c5 = |s: u32| ((v >> s) & 0x1F) << 3 | ((v >> s) & 0x1F) >> 2; + match format { + (8, 8) => pack_rgb(c4(0), c4(4), c4(8)), + (8, 10) => pack_rgb(c5(0), c5(5), c5(10)), + (8, 0) => v & 0xFF_FFFF, + (0, 1) => v & 0xFFF, + _ => v, + } +} + +fn to_host(format: (u32, u32), v: u32) -> u32 { + let c = |s: u32| (v >> s) & 0xFF; + match format { + (8, 8) => c(0) / 0x11 | (c(8) / 0x11) << 4 | (c(16) / 0x11) << 8, + (8, 10) => c(0) >> 3 | (c(8) >> 3) << 5 | (c(16) >> 3) << 10, + _ => v, + } +} + +fn signed16(v: u32) -> i32 { + v as u16 as i16 as i32 +} + +/// A block, in window coordinates, with its colour. +#[derive(Clone, Copy)] +struct Block { + xs: i32, + ys: i32, + xe: i32, + ye: i32, + color: u32, +} + +impl Block { + fn dx(&self) -> i32 { + if self.xe < self.xs { -1 } else { 1 } + } + fn dy(&self) -> i32 { + if self.ye < self.ys { -1 } else { 1 } + } + fn rows(&self) -> i32 { + (self.ye - self.ys).abs() + 1 + } + fn cols(&self) -> i32 { + (self.xe - self.xs).abs() + 1 + } +} + +/// A character block being filled with stipple data, one row at a time. +#[derive(Clone, Copy)] +struct Stipple { + block: Block, + col: i32, + row: i32, +} + +/// An armed pixel transfer. +struct Xfer { + block: Block, + read: bool, + /// Pixels per line, bytes per pixel, and (PixelFormat, CompType). + width: u32, + bpp: u32, + format: (u32, u32), + begin_skip: u32, + stride_skip: u32, + /// PIO write stream: the line being assembled, its byte offset within + /// its first doubleword, the bytes collected, and filler still to skip. + line: u32, + line_begin: u32, + pending: Vec, + skip: u32, + /// PIO read: doublewords waiting to be read, and the low half of the one + /// the last high read took. + out: VecDeque, + out_lo: u32, +} + +impl Xfer { + fn line_bytes(&self) -> u32 { + self.width * self.bpp + } + + /// Where the line after one starting at `b` starts within its first + /// doubleword. + fn next_begin(&self, b: u32) -> u32 { + (b + self.line_bytes() + self.stride_skip) & 7 + } +} + +/// Bytes per pixel for an xfrmode (PixelFormat, PixelCompType) pair. +fn bytes_per_pixel(xfrmode: u32) -> u32 { + match ((xfrmode >> 4) & 0xF, xfrmode & 0xF) { + (0, 0) => 1, + (0, 1) | (8, 8) | (8, 10) => 2, + (8, 0) => 4, + (7, 1) => 6, + _ => 1, + } +} + +pub struct Raster { + regs: Vec, + device: HashMap, + /// Colour planes, one value per pixel, row 0 at the bottom. In + /// colour-index modes the low byte is the index. + pub fb: Vec, + /// Overlay planes (kept, not yet displayed). + pub overlay: Vec, + pending: Option, + /// Bumped for every new pending block, so the refresh thread can tell a + /// block that has waited a whole frame (see `flush_if_stale`). + pending_id: u64, + stipple: Option, + xfer: Option, + /// Fill modes seen with a block, for bring-up logging. + pub fillmodes_seen: Vec, +} + +impl Default for Raster { + fn default() -> Self { + Raster { + regs: vec![0; 0x400], + device: HashMap::new(), + fb: vec![0; WIDTH * HEIGHT], + overlay: vec![0; WIDTH * HEIGHT], + pending: None, + pending_id: 0, + stipple: None, + xfer: None, + fillmodes_seen: Vec::new(), + } + } +} + +impl Raster { + pub fn read(&self, r: u32) -> u32 { + match r & 0x3FF { + reg::STATUS => STATUS_IDLE, + reg::INDIRECT_DATA => { + self.device.get(&self.regs[reg::INDIRECT_ADDR as usize]).copied().unwrap_or(0) + } + r => self.regs[r as usize], + } + } + + fn reg(&self, r: u32) -> u32 { + self.regs[r as usize] + } + + /// Write register `r`; `exec` runs the primitive afterwards. Returns true + /// when the framebuffer changed. + pub fn write(&mut self, r: u32, val: u32, exec: bool) -> bool { + let r = r & 0x3FF; + self.regs[r as usize] = val; + match r { + reg::INDIRECT_DATA => { + self.device.insert(self.regs[reg::INDIRECT_ADDR as usize], val); + } + reg::IR_ALIAS => self.regs[reg::IR as usize] = val, + reg::XFRCONTROL if val == 0 => self.xfer = None, + _ => {} + } + if !exec { + return false; + } + let pio_write = self.xfer.as_ref().is_some_and(|x| !x.read); + match r { + reg::CHAR_L if pio_write => { + let dw = ((self.reg(reg::CHAR_H) as u64) << 32) | val as u64; + self.pio_write(dw) + } + reg::CHAR_H if pio_write => self.pio_write((val as u64) << 32), + reg::CHAR_H => self.stipple_bits((val as u64) << 32, 32), + reg::CHAR_L => { + let bits = ((self.reg(reg::CHAR_H) as u64) << 32) | val as u64; + self.stipple_bits(bits, 64) + } + _ => self.execute(), + } + } + + /// Whether pixels are RGB rather than colour indices: an RGB pixel type, + /// or a write mask of exactly the 24 RGB planes (24-bit windows draw + /// with other pixel types; colour-index drawing masks 8 or 12 planes, or + /// all 32 when the PROM and kernel draw). + fn rgb_mode(&self) -> bool { + rgb_pixtype(self.reg(reg::PP1FILLMODE)) || self.reg(reg::COLORMASKLSBSA) == 0xFF_FFFF + } + + /// Window coordinates to framebuffer coordinates. + fn to_fb(&self, x: i32, y: i32) -> (i32, i32) { + let win = self.reg(reg::XYWIN); + let (ox, oy) = (signed16(win), signed16(win >> 16)); + if self.reg(reg::CONFIG) & CONFIG_YFLIP != 0 { + (ox + x, oy - y) + } else { + (ox + x, oy + y) + } + } + + /// Whether a framebuffer pixel may be written: on screen, and inside + /// the clip rectangle when it is enabled (bit 4 chooses inside or outside). + fn visible(&self, x: i32, y: i32) -> bool { + if !(0..WIDTH as i32).contains(&x) || !(0..HEIGHT as i32).contains(&y) { + return false; + } + let clip = self.reg(reg::CLIP_MODE); + if clip & 1 != 0 { + let (mx, my) = (self.reg(reg::CLIP_X), self.reg(reg::CLIP_Y)); + let inside = (signed16(mx >> 16)..=signed16(mx)).contains(&x) + && (signed16(my >> 16)..=signed16(my)).contains(&y); + if inside != (clip & 0x10 != 0) { + return false; + } + } + true + } + + /// Store `v` at framebuffer `(x, y)` in the planes the pixel processors + /// are drawing to. Window-ID drawing is dropped. + fn put(&mut self, x: i32, y: i32, v: u32) { + if self.reg(reg::PP1FILLMODE) == PP1_DRAW_CID || !self.visible(x, y) { + return; + } + let i = y as usize * WIDTH + x as usize; + let pp1 = self.reg(reg::PP1FILLMODE); + // A write through all planes (window moves copy the screen that way, + // 24 bits a pixel) keeps the whole value; so does RGB. + let wide = self.rgb_mode() || self.reg(reg::COLORMASKLSBSA) == 0xFFFF_FFFF; + let plane = if self.reg(reg::DRBPOINTERS) & DRB_PLANES == DRB_OVERLAY { + &mut self.overlay + } else { + &mut self.fb + }; + plane[i] = if pp1 & PP1_LOGIC_OP_ENABLE != 0 { + let width = if wide { 0xFF_FFFF } else { 0xFFF }; + logic_op(pp1 >> 26, v, plane[i]) & width + } else { + v + }; + } + + fn get(&self, x: i32, y: i32) -> u32 { + if !(0..WIDTH as i32).contains(&x) || !(0..HEIGHT as i32).contains(&y) { + return 0; + } + let i = y as usize * WIDTH + x as usize; + if self.reg(reg::DRBPOINTERS) & DRB_PLANES == DRB_OVERLAY { self.overlay[i] } else { self.fb[i] } + } + + /// Block pixel `(col, row)` in framebuffer coordinates. + fn block_px(&self, b: &Block, col: i32, row: i32) -> (i32, i32) { + self.to_fb(b.xs + col * b.dx(), b.ys + row * b.dy()) + } + + fn current_block(&self) -> Block { + let s = self.reg(reg::BLOCKXYSTARTI); + let e = self.reg(reg::BLOCKXYENDI); + // Colour index modes: the index, from the fill colour register for + // fast fills and from the red iterator (12 fraction bits) otherwise. + // RGB modes: three 12-bit components for fast fills, a packed + // 8-8-8 colour otherwise. + let rgb = self.rgb_mode(); + let fast = self.reg(reg::FILLMODE) & FILL_FAST != 0; + let color = match (rgb, fast) { + (false, true) => self.reg(reg::FILL_COLOR_R), + (false, false) => (self.reg(reg::RED) >> 12) & 0xFFF, + (true, true) => pack_rgb( + self.reg(reg::FILL_COLOR_R) >> 4, + self.reg(reg::FILL_COLOR_G) >> 4, + self.reg(reg::FILL_COLOR_B) >> 4, + ), + (true, false) => self.reg(reg::PACKEDCOLOR) & 0xFF_FFFF, + }; + Block { xs: signed16(s >> 16), ys: signed16(s), xe: signed16(e >> 16), ye: signed16(e), color } + } + + /// Run the primitive in the IR. + fn execute(&mut self) -> bool { + let changed = self.flush_pending(); + match self.reg(reg::IR) & 0xF { + OP_BLOCK => {} + OP_LINE => return self.line() | changed, + _ => return changed, + } + let fm = self.reg(reg::FILLMODE); + if !self.fillmodes_seen.contains(&fm) && self.fillmodes_seen.len() < 32 { + self.fillmodes_seen.push(fm); + } + let b = self.current_block(); + // A new block ends whatever the last one was doing; a write transfer + // in particular is never disarmed explicitly. + self.stipple = None; + self.xfer = None; + let kind = (fm >> 22) & 7; + if fm & FILL_FAST != 0 { + self.fill(&b); + return true; + } + match kind { + block::NORMAL => { + // Character block: filled by the stipple data that follows. + self.stipple = Some(Stipple { block: b, col: 0, row: 0 }); + changed + } + block::PIO_READ | block::PIO_WRITE | block::DMA_READ | block::DMA_WRITE => { + let read = kind == block::PIO_READ || kind == block::DMA_READ; + self.arm_transfer(b, read, kind == block::PIO_READ); + changed + } + _ => { + // A block followed by character data is a stipple; anything + // else, a fill. + self.pending_id += 1; + self.pending = Some(b); + changed + } + } + } + + /// A line from the start to the end point, both included, in the current + /// colour; with line stipple on, pixel `k` is drawn only where bit + /// `31 - k % 32` of the pattern is set. + fn line(&mut self) -> bool { + let s = self.reg(reg::LINE_START); + let e = self.reg(reg::LINE_END); + let (mut x, mut y) = (signed16(s >> 16), signed16(s)); + let (x1, y1) = (signed16(e >> 16), signed16(e)); + let color = self.current_block().color; + let stipple = (self.reg(reg::FILLMODE) & FILL_LINE_STIPPLE != 0).then(|| self.reg(reg::LINE_STIPPLE)); + let (dx, dy) = ((x1 - x).abs(), -(y1 - y).abs()); + let (sx, sy) = (if x < x1 { 1 } else { -1 }, if y < y1 { 1 } else { -1 }); + let mut err = dx + dy; + let mut k = 0u32; + loop { + if stipple.map_or(true, |p| p & (1 << (31 - k % 32)) != 0) { + let (fx, fy) = self.to_fb(x, y); + self.put(fx, fy, color); + } + if x == x1 && y == y1 { + break; + } + let e2 = 2 * err; + if e2 >= dy { + err += dy; + x += sx; + } + if e2 <= dx { + err += dx; + y += sy; + } + k += 1; + } + true + } + + fn fill(&mut self, b: &Block) { + for row in 0..b.rows() { + for col in 0..b.cols() { + let (x, y) = self.block_px(b, col, row); + self.put(x, y, b.color); + } + } + } + + /// Called once per displayed frame: a block that was already pending at + /// the previous frame never got character data, so draw it as a fill. + pub fn flush_if_stale(&mut self, last_seen: &mut u64) -> bool { + if self.pending.is_none() { + return false; + } + if *last_seen == self.pending_id { + return self.flush_pending(); + } + *last_seen = self.pending_id; + false + } + + /// A pending block that got no character data is a solid fill. + fn flush_pending(&mut self) -> bool { + let Some(b) = self.pending.take() else { return false }; + self.fill(&b); + true + } + + /// Consume `n` stipple bits (most significant first) into the current + /// character block: 1 bits take the colour, 0 bits leave the pixel. A row + /// ends at the block's edge, discarding the rest of the chunk. + fn stipple_bits(&mut self, bits: u64, n: u32) -> bool { + if let Some(b) = self.pending.take() { + self.stipple = Some(Stipple { block: b, col: 0, row: 0 }); + } + let Some(mut s) = self.stipple else { return false }; + if s.row >= s.block.rows() { + return false; + } + for i in 0..n { + if bits & (1u64 << (63 - i)) != 0 { + let (x, y) = self.block_px(&s.block, s.col, s.row); + self.put(x, y, s.block.color); + } + s.col += 1; + if s.col >= s.block.cols() { + s.col = 0; + s.row += 1; + break; + } + } + self.stipple = Some(s); + true + } + + // ---- pixel transfers ---- + + fn arm_transfer(&mut self, block: Block, read: bool, pio_read: bool) { + let mode = self.reg(reg::XFRMODE); + let begin = (mode >> 8) & 7; + let mut x = Xfer { + block, + read, + width: self.reg(reg::XFRSIZE) & 0xFFFF, + bpp: bytes_per_pixel(mode), + format: ((mode >> 4) & 0xF, mode & 0xF), + begin_skip: begin, + stride_skip: (mode >> 14) & 0x1FF, + line: 0, + line_begin: begin, + pending: Vec::new(), + skip: begin, + out: VecDeque::new(), + out_lo: 0, + }; + if pio_read { + x.out = self.pio_read_stream(&x); + } + self.xfer = Some(x); + } + + /// `Some(read)` while a transfer is armed. + pub fn transfer_armed(&self) -> Option { + self.xfer.as_ref().map(|x| x.read) + } + + /// Lines and bytes per line of the armed transfer. + pub fn transfer_shape(&self) -> Option<(u32, u32)> { + self.xfer.as_ref().map(|x| (x.block.rows() as u32, x.line_bytes())) + } + + fn put_line(&mut self, x: &Xfer, line: u32, bytes: &[u8]) { + let bpp = x.bpp as usize; + for (k, px) in bytes.chunks(bpp).enumerate().take(x.width as usize) { + // Big-endian bytes; pixels wider than 32 bits keep their low word. + let v = px.iter().fold(0u64, |a, &b| (a << 8) | b as u64) as u32; + let (fx, fy) = self.block_px(&x.block, k as i32, line as i32); + self.put(fx, fy, from_host(x.format, v)); + } + } + + fn get_line(&self, x: &Xfer, line: u32) -> Vec { + let mut out = Vec::with_capacity(x.line_bytes() as usize); + for k in 0..x.width as i32 { + let (fx, fy) = self.block_px(&x.block, k, line as i32); + let v = to_host(x.format, self.get(fx, fy)) as u64; + for i in (0..x.bpp).rev() { + out.push((v >> (8 * i)) as u8); + } + } + out + } + + /// A PIO write doubleword. Each line starts in a fresh doubleword, after + /// its begin offset; the rest of a line's last doubleword is dropped. + fn pio_write(&mut self, dw: u64) -> bool { + let Some(mut x) = self.xfer.take() else { return false }; + let mut changed = false; + for i in 0..8 { + if x.line >= x.block.rows() as u32 { + break; + } + if x.skip > 0 { + x.skip -= 1; + continue; + } + x.pending.push((dw >> (56 - 8 * i)) as u8); + if x.pending.len() as u32 == x.line_bytes() { + let bytes = std::mem::take(&mut x.pending); + self.put_line(&x, x.line, &bytes); + changed = true; + x.line += 1; + x.line_begin = x.next_begin(x.line_begin); + x.skip = x.line_begin; + break; + } + } + self.xfer = Some(x); + changed + } + + /// The whole PIO read stream: per line, filler up to its begin offset, the + /// pixels, and padding to the doubleword. + fn pio_read_stream(&self, x: &Xfer) -> VecDeque { + let mut out = VecDeque::new(); + let mut b = x.begin_skip; + for line in 0..x.block.rows() as u32 { + let mut bytes = vec![0u8; b as usize]; + bytes.extend(self.get_line(x, line)); + while bytes.len() % 8 != 0 { + bytes.push(0); + } + for c in bytes.chunks(8) { + out.push_back(c.iter().fold(0u64, |a, &v| (a << 8) | v as u64)); + } + b = x.next_begin(b); + } + out + } + + /// PIO read, high half: takes the next doubleword. + pub fn pio_read_hi(&mut self) -> u32 { + let Some(x) = self.xfer.as_mut() else { return 0 }; + let dw = x.out.pop_front().unwrap_or(0); + x.out_lo = dw as u32; + (dw >> 32) as u32 + } + + /// PIO read, low half of the doubleword the last high read took. + pub fn pio_read_lo(&self) -> u32 { + self.xfer.as_ref().map(|x| x.out_lo).unwrap_or(0) + } + + /// DMA into the armed block: line `line`'s pixel bytes. + pub fn dma_write_line(&mut self, line: u32, bytes: &[u8]) { + if let Some(x) = self.xfer.take() { + self.put_line(&x, line, bytes); + self.xfer = Some(x); + } + } + + /// DMA out of the armed block: line `line`'s pixel bytes. + pub fn dma_read_line(&self, line: u32) -> Vec { + self.xfer.as_ref().map(|x| self.get_line(x, line)).unwrap_or_default() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn px(r: &Raster, x: usize, y_top: usize) -> u32 { + r.fb[(HEIGHT - 1 - y_top) * WIDTH + x] + } + + /// The X server's setup: window origin at the top row, Y-flip on, so + /// block coordinates are X's top-down ones. + fn x_server() -> Raster { + let mut r = Raster::default(); + r.write(reg::CONFIG, 0xCAC, false); + r.write(reg::XYWIN, 1023 << 16, false); + r.write(reg::PP1FILLMODE, 0x0C00_4504, false); + r + } + + fn block(r: &mut Raster, x0: u32, y0: u32, x1: u32, y1: u32) { + r.write(reg::IR_ALIAS, 0x18, false); + r.write(reg::BLOCKXYSTARTI, x0 << 16 | y0, false); + r.write(reg::BLOCKXYENDI, x1 << 16 | y1, true); + } + + #[test] + fn fast_fill_lands_top_down_with_yflip() { + let mut r = x_server(); + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0x13, false); + block(&mut r, 10, 20, 12, 21); + assert_eq!(px(&r, 10, 20), 0x13); + assert_eq!(px(&r, 12, 21), 0x13); + assert_eq!(px(&r, 13, 21), 0); + assert_eq!(px(&r, 10, 22), 0); + } + + #[test] + fn rgb_fast_fill_packs_components() { + let mut r = x_server(); + r.write(reg::PP1FILLMODE, 3 << 26 | 0x104, false); // RGB pixel type, copy + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0xF00, false); + r.write(reg::FILL_COLOR_G, 0x800, false); + r.write(reg::FILL_COLOR_B, 0x100, false); + block(&mut r, 0, 0, 0, 0); + assert_eq!(px(&r, 0, 0), 0x10_80F0); + } + + #[test] + fn pio_write_frames_each_line_in_a_fresh_doubleword() { + let mut r = x_server(); + // Three 1-byte pixels per line, two lines, begin skip 2: line 0 is + // bytes 2..5 of its doubleword; line 1 starts at (2 + 3) & 7 = 5. + r.write(reg::FILLMODE, 3 << 22, false); + r.write(reg::XFRMODE, 2 << 8, false); + r.write(reg::XFRSIZE, 2 << 16 | 3, false); + block(&mut r, 100, 50, 102, 51); + assert_eq!(r.transfer_armed(), Some(false)); + r.write(reg::CHAR_H, 0x0000_0102, false); + r.write(reg::CHAR_L, 0x03FF_FFFF, true); + r.write(reg::CHAR_H, 0x0000_0000, false); + r.write(reg::CHAR_L, 0x0004_0506, true); + assert_eq!([px(&r, 100, 50), px(&r, 101, 50), px(&r, 102, 50)], [1, 2, 3]); + assert_eq!([px(&r, 100, 51), px(&r, 101, 51), px(&r, 102, 51)], [4, 5, 6]); + } + + #[test] + fn a_new_block_disarms_a_finished_write_transfer() { + let mut r = x_server(); + r.write(reg::FILLMODE, 5 << 22, false); + r.write(reg::XFRSIZE, 1 << 16 | 1, false); + block(&mut r, 0, 0, 0, 0); + assert!(r.transfer_armed().is_some()); + // A glyph block: its char data must stipple, not feed the transfer. + r.write(reg::FILLMODE, 1 << 22, false); + r.write(reg::RED, 0x7 << 12, false); + block(&mut r, 200, 10, 203, 10); + assert_eq!(r.transfer_armed(), None); + r.write(reg::CHAR_H, 0xA000_0000, true); + assert_eq!([px(&r, 200, 10), px(&r, 201, 10), px(&r, 202, 10)], [7, 0, 7]); + } + + #[test] + fn rgb_host_formats_round_trip() { + for (fmt, v) in [((8, 8), 0x0ABCu32), ((8, 10), 0x7FFF), ((8, 0), 0x00C0_FFEE), ((0, 1), 0xFFF)] { + assert_eq!(to_host(fmt, from_host(fmt, v)), v, "{fmt:?}"); + } + assert_eq!(from_host((8, 8), 0x0F0), pack_rgb(0, 0xFF, 0)); + // 8-8-8 host pixels are X pixel values of the visuals, red in 7:0. + assert_eq!(from_host((8, 0), 0x00_00FF), pack_rgb(0xFF, 0, 0)); + } + + #[test] + fn a_24_plane_mask_means_rgb_fills() { + let mut r = x_server(); + r.write(reg::PP1FILLMODE, 0x0C00_6304, false); + r.write(reg::COLORMASKLSBSA, 0xFF_FFFF, false); + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0x380, false); + r.write(reg::FILL_COLOR_G, 0x8E0, false); + r.write(reg::FILL_COLOR_B, 0x8E0, false); + block(&mut r, 1, 1, 1, 1); + assert_eq!(px(&r, 1, 1), 0x8E_8E38); + } + + #[test] + fn an_all_planes_copy_keeps_24_bit_pixels() { + let mut r = x_server(); + // A window move's write-back: pixel type 2, every plane enabled, + // 4-byte host pixels. + r.write(reg::PP1FILLMODE, 0x0C00_6204, false); + r.write(reg::COLORMASKLSBSA, 0xFFFF_FFFF, false); + r.write(reg::FILLMODE, 5 << 22, false); + r.write(reg::XFRMODE, 0x80, false); + r.write(reg::XFRSIZE, 1 << 16 | 1, false); + block(&mut r, 7, 7, 7, 7); + r.dma_write_line(0, &[0x00, 0x50, 0x50, 0x50]); + assert_eq!(px(&r, 7, 7), 0x50_5050); + } + + #[test] + fn xor_fill_toggles_and_restores() { + let mut r = x_server(); + r.write(reg::FILLMODE, FILL_FAST, false); + r.write(reg::FILL_COLOR_R, 0x13, false); + block(&mut r, 5, 5, 5, 5); + r.write(reg::PP1FILLMODE, 0x0C00_4504 & !(0xF << 26) | 6 << 26, false); + r.write(reg::FILL_COLOR_R, 0x0F, false); + block(&mut r, 5, 5, 5, 5); + assert_eq!(px(&r, 5, 5), 0x13 ^ 0x0F); + block(&mut r, 5, 5, 5, 5); + assert_eq!(px(&r, 5, 5), 0x13); + } + + #[test] + fn stippled_line_skips_zero_bits() { + let mut r = x_server(); + r.write(reg::FILLMODE, FILL_LINE_STIPPLE, false); + r.write(reg::RED, 0x9 << 12, false); + r.write(reg::LINE_STIPPLE, 0xAAAA_AAAA, false); + r.write(reg::IR_ALIAS, 0x15, false); + r.write(reg::LINE_START, 300 << 16 | 40, false); + r.write(reg::LINE_END, 303 << 16 | 40, true); + assert_eq!((300..304).map(|x| px(&r, x, 40)).collect::>(), [9, 0, 9, 0]); + } +} diff --git a/src/mips_cache_shadow.rs b/src/mips_cache_shadow.rs new file mode 100644 index 00000000..68b95b6f --- /dev/null +++ b/src/mips_cache_shadow.rs @@ -0,0 +1,750 @@ +//! A cache that is visible to software but stays out of the data path. +//! +//! Every load, store and instruction fetch goes straight to memory, exactly as +//! in [`PassthroughCacheOf`](crate::mips_cache_v2::PassthroughCacheOf). What +//! this adds is a *shadow*: tag and data arrays that only the CACHE +//! instruction ever touches. +//! +//! The reasoning is that a cache's effect on a functional emulator is entirely +//! observational. Its timing is invisible to the guest, and its contents are +//! invisible too as long as they agree with memory — which here they trivially +//! do, because memory is the only store. What *is* visible is the CACHE +//! instruction, the geometry reported through CP0 Config, and the diagnostics +//! a PROM runs against both. So model those, and let the host CPU — which +//! already has real caches, real speculation and real out-of-order execution — +//! get on with it unimpeded. +//! +//! Two things fall out of this beyond speed: +//! +//! - **Coherency is exact and free.** There is no stale line to miss on a +//! self-modifying store, no L1/L2 inclusion policy to get wrong, and no way +//! for a missed writeback to lose data. +//! - **Diagnostics round-trip.** The shadow stores whatever bits are written +//! to it and returns them unchanged, so a walking-1s test over tag or data +//! SRAM passes without anyone having to know the hardware's field layout. +//! Where the emulator *must* know the layout — to decide whether a line is +//! valid, say — that is a separate question from storing the bits. +//! +//! The shadow is deliberately not consulted by `read`/`write`/`fetch`. If it +//! ever needs to be, this type is the wrong shape and should say so loudly +//! rather than growing a slow path. + +use std::cell::UnsafeCell; +use std::sync::Arc; + +use crate::mips_cache_v2::{ + cache_op_name, CpuModel, FetchInstrResult, MipsCache, C_ILT, C_IST, C_R10K_CBARRIER, + C_R10K_ILD, C_R10K_ISD, CACH_PD, CACH_PI, CACH_SD, CACH_SI, +}; +use crate::mips_exec::{DecodedInstr, FLAG_NOT_DECODED}; +use crate::traits::{BusDevice, BusRead64}; +use crate::devlog::{LogModule, CACHE_LOG_HIT, CACHE_LOG_OP, devlog_is_active, devlog_mask}; + +/// Shadow tag and data arrays for one cache. +/// +/// Indexed `set * WAYS + way`. The ways are here and nowhere else: a CACHE +/// index operation addresses one *way* of one set, so software can see them, +/// and the IP28 PROM's tag diagnostic depends on it — it writes different tags +/// to the two ways of a set and reads one back. Keeping them costs an array +/// dimension in a structure no load or store ever consults. +struct Shadow { + /// One raw tag per line, stored and returned verbatim. 64 bits: an + /// R10000 secondary tag carries a 40-bit physical address. + tags: Box<[u64]>, + /// Data array in u64 slots. Empty where the model has no data shadow. + data: Box<[u64]>, + /// The check bits stored with each data slot. On an R10000 these ride + /// with the data: `Index_Store_Data` takes them from CP0 ECC and + /// `Index_Load_Data` returns them there, so they are storage like the + /// data itself, not a computed value. + ecc: Box<[u32]>, + /// The most-recently-used bit for each set. Shared across the set's ways + /// — see `MRU_BIT`. Hardware state rather than tag storage, which is why + /// it is kept here and re-applied on read instead of living in `tags`. + mru: Box<[bool]>, +} + +impl Shadow { + fn new(lines: usize, data_words: usize) -> Self { + Self { + tags: vec![0u64; lines.max(1)].into_boxed_slice(), + data: vec![0u64; data_words].into_boxed_slice(), + ecc: vec![0u32; data_words].into_boxed_slice(), + mru: vec![false; (lines.max(1) / WAYS).max(1)].into_boxed_slice(), + } + } +} + +/// The most-recently-used bit: TagHi[31], i.e. bit 63 of the assembled tag. +/// +/// It is **one bit per set, shared between the ways** — not a per-way tag bit. +/// Writing a tag with it set through either way marks the set, and reading +/// the tag of *either* way reports it. The IP28 PROM's diagnostic tests +/// exactly that: it writes the bit through way 0, writes the next set to +/// clobber the signal line, then reads way 1 back and requires the bit to be +/// there. Storing "which way is MRU" and reporting it only on that way passes +/// nothing, because the way that is read is never the way that was written. +const MRU_BIT: u64 = 1 << 63; + +/// Ways per set, as CACHE index operations address them. Bit 0 of the index +/// selects the way on an R10000; the set starts above the line offset. +const WAYS: usize = 2; + +/// Bits a secondary-cache tag retains. +/// +/// All of them, now. A 36-bit mask was inferred here from the PROM's +/// walking-0s phase, which writes an all-ones TagLo and expects back +/// 0x0000000f_ffffcdfe — but that truncation comes from the tag being +/// assembled as `(TagHi << 32) | TagLo[31:0]`, not from the array dropping +/// bits. Once the executor carried TagHi properly the mask did nothing except +/// discard the MRU bit, which the PROM writes as TagHi[31] and reads back. +const L2_TAG_MASK: u64 = u64::MAX; + +/// Is the shadow cache's operation trace on, at this level of detail? +/// +/// `log l2c mask op` traces the tag operations a diagnostic actually cares +/// about; `log l2c mask op+hit` adds the data-array walk as well. That +/// distinction is load-bearing rather than tidy: the IP28 PROM issues 65k+ +/// `Index_Store_Data` ops walking the array, and tracing all of them once +/// slowed the guest so much a run never reached the test it was there to +/// observe. +/// +/// This used to be `std::env::var_os("IRIS_SHADOW_CACHEOPS")`, uncached, six +/// times per `cache_op` — a getenv on every CACHE instruction in every build. +/// `devlog` costs two relaxed atomic loads and can be turned on, masked and +/// redirected to a file while the machine runs. +#[inline(always)] +fn l2_op_log(bit: u32) -> bool { + devlog_is_active(LogModule::L2c) && (devlog_mask(LogModule::L2c) & bit) != 0 +} + +/// A `MipsCache` whose contents are only ever observed through CACHE ops. +/// +/// Const parameters carry the geometry that software can read back, and the +/// processor identity. Nothing here affects the speed of a load. +pub struct ShadowCache< + const IC_SIZE: usize, + const IC_LINE: usize, + const DC_SIZE: usize, + const DC_LINE: usize, + const L2_SIZE: usize, + const L2_LINE: usize, + const MIPS4: bool, + const PRID: u32, + const FIR: u32, + const TLB_ENTRIES: usize, + const R10K_OPS: bool, +> { + downstream: Arc, + llbit: UnsafeCell, + lladdr: UnsafeCell, + /// Somewhere to decode into. Not a cache line — there is no caching. + fetch_scratch: UnsafeCell, + /// CP0 ECC on the way in to a store, and on the way out of a load. + ecc_in: UnsafeCell, + ecc_out: UnsafeCell, + ic: UnsafeCell, + dc: UnsafeCell, + l2: UnsafeCell, + /// tcache: ppmem's window base, its mapped-region bitmap (published into + /// directly on every remap) and the jitv2 generation window. This cache + /// reads nothing through them — `read`/`write` go to `downstream`, whose + /// ppmem path uses the same window — they are here only so jitv2's inline + /// tcache path can serve this model too (`jit_dc_geometry`). + #[cfg(feature = "tcache")] + tc_base: UnsafeCell<*mut u8>, + #[cfg(feature = "tcache")] + tc_bitmap: UnsafeCell, + #[cfg(all(feature = "tcache", feature = "jitv2"))] + tc_gen: UnsafeCell<*mut std::sync::atomic::AtomicU64>, +} + +// Safety: the CPU thread is the only accessor, as for every other cache model +// in this crate. +unsafe impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > Send + for ShadowCache +{ +} +unsafe impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > Sync + for ShadowCache +{ +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > ShadowCache +{ + const IC_LINES: usize = if IC_LINE == 0 { 0 } else { IC_SIZE / IC_LINE }; + const DC_LINES: usize = if DC_LINE == 0 { 0 } else { DC_SIZE / DC_LINE }; + const L2_LINES: usize = if L2_LINE == 0 { 0 } else { L2_SIZE / L2_LINE }; + + pub fn new(downstream: Arc) -> Self { + Self { + downstream, + llbit: UnsafeCell::new(false), + lladdr: UnsafeCell::new(0), + fetch_scratch: UnsafeCell::new(DecodedInstr::default()), + ecc_in: UnsafeCell::new(0), + ecc_out: UnsafeCell::new(0), + ic: UnsafeCell::new(Shadow::new(Self::IC_LINES, 0)), + dc: UnsafeCell::new(Shadow::new(Self::DC_LINES, 0)), + // Only the secondary keeps a data shadow: it is the one a PROM + // walks with Index_Store_Data, and a 1 MB array is cheap once. + l2: UnsafeCell::new(Shadow::new(Self::L2_LINES, L2_SIZE / 8)), + #[cfg(feature = "tcache")] + tc_base: UnsafeCell::new(std::ptr::null_mut()), + #[cfg(feature = "tcache")] + tc_bitmap: UnsafeCell::new(0), + #[cfg(all(feature = "tcache", feature = "jitv2"))] + tc_gen: UnsafeCell::new(std::ptr::null_mut()), + } + } + + #[allow(clippy::mut_from_ref)] + fn shadow(&self, sel: u32) -> &mut Shadow { + unsafe { + match sel { + CACH_PI => &mut *self.ic.get(), + CACH_PD => &mut *self.dc.get(), + _ => &mut *self.l2.get(), + } + } + } + + /// Tag slot for a CACHE index operation. + /// + /// Bit 0 of the index selects the way; the set number starts above the + /// line offset. Observed directly: the PROM initialises the secondary + /// cache at `…1000`, `…1001`, `…1080`, `…1081`, stepping by the 128-byte + /// line with the low bit alternating. + /// + /// Folding the way bit away instead — on the reasoning that it sits below + /// line granularity — made the two ways of a set alias onto one slot, so + /// a tag written to way 1 overwrote way 0 and the PROM read back the + /// wrong one. That is the whole of the "TAG walking 1s" failure. + fn tag_slot(&self, sel: u32, virt_addr: u64) -> usize { + let (line, lines) = match sel { + CACH_PI => (IC_LINE, Self::IC_LINES), + CACH_PD => (DC_LINE, Self::DC_LINES), + _ => (L2_LINE, Self::L2_LINES), + }; + if line == 0 || lines == 0 { + return 0; + } + let way = (virt_addr as usize) & (WAYS - 1); + let set = ((virt_addr as usize) / line) % (lines / WAYS).max(1); + (set * WAYS + way) % lines + } + + /// Data slot for an R10000 `Index_Load_Data` / `Index_Store_Data`. + /// + /// Same shape: way in bit 0, the rest addressing the array. The PROM walks + /// it at `…00`, `…01`, `…10`, `…11`, `…20` — one doubleword per operation + /// with the way bit shifted in beneath it. + fn data_slot(&self, len: usize, virt_addr: u64) -> usize { + if len == 0 { + return 0; + } + let way = (virt_addr as usize) & (WAYS - 1); + let word = (virt_addr as usize) >> 4; + (word * WAYS + way) % len + } +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > From> + for ShadowCache +{ + fn from(downstream: Arc) -> Self { + Self::new(downstream) + } +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > CpuModel + for ShadowCache +{ + const MIPS4: bool = MIPS4; + const PRID: u32 = PRID; + const FIR: u32 = FIR; + const TLB_ENTRIES: usize = TLB_ENTRIES; + // The R10000 implements 44 virtual address bits where the R4x00 implements + // 40; the shadow cache is only used for it, but key this off the same flag + // that selects its cache encodings rather than asserting it unconditionally. + const VA_BITS: u32 = if R10K_OPS { 44 } else { 40 }; + const NAME: &'static str = "shadow"; + const R10K_CACHE_OPS: bool = R10K_OPS; +} + +impl< + const IC_SIZE: usize, const IC_LINE: usize, const DC_SIZE: usize, const DC_LINE: usize, + const L2_SIZE: usize, const L2_LINE: usize, const MIPS4: bool, const PRID: u32, + const FIR: u32, const TLB_ENTRIES: usize, const R10K_OPS: bool, + > MipsCache + for ShadowCache +{ + const IC_SIZE: usize = IC_SIZE; + const IC_LINE: usize = IC_LINE; + const IC_WAYS: usize = 1; + const DC_SIZE: usize = DC_SIZE; + const DC_LINE: usize = DC_LINE; + const DC_WAYS: usize = 1; + const L2_SIZE: usize = L2_SIZE; + const L2_LINE: usize = L2_LINE; + + #[cfg(feature = "tcache")] + unsafe fn set_tcache_window(&self, base: *mut u8) { + unsafe { *self.tc_base.get() = base }; + } + + #[cfg(feature = "tcache")] + fn tcache_bitmap_ptr(&self) -> *mut u64 { self.tc_bitmap.get() } + + #[cfg(feature = "tcache")] + fn tcache_base_ptr(&self) -> *mut u8 { unsafe { *self.tc_base.get() } } + + #[cfg(all(feature = "tcache", feature = "jitv2"))] + fn tcache_gen_ptr(&self) -> *mut u8 { unsafe { *self.tc_gen.get() as *mut u8 } } + + #[cfg(all(feature = "tcache", feature = "jitv2"))] + unsafe fn set_tcache_gen_window(&self, gen_base: *mut std::sync::atomic::AtomicU64) { + unsafe { *self.tc_gen.get() = gen_base }; + } + + /// Under tcache, a tagless geometry: the inline path is the window alone. + /// Declined until both windows are published, as `CpuCache` does, so the + /// emitted code can assume them. Without tcache there is no inline path: + /// every access calls out, exactly as for `PassthroughCache`. + fn jit_dc_geometry(&self) -> crate::mips_cache_v2::JitDcGeometry { + #[cfg(all(feature = "tcache", feature = "jitv2"))] + if !unsafe { *self.tc_base.get() }.is_null() && !unsafe { *self.tc_gen.get() }.is_null() { + return crate::mips_cache_v2::JitDcGeometry { + supported: true, + tagless: true, + ..crate::mips_cache_v2::JitDcGeometry::unsupported() + }; + } + crate::mips_cache_v2::JitDcGeometry::unsupported() + } + + fn fetch(&self, _virt_addr: u64, phys_addr: u64) -> FetchInstrResult { + let r = self.downstream.read32(phys_addr as u32); + if r.is_ok() { + let slot = unsafe { &mut *self.fetch_scratch.get() }; + slot.flags = FLAG_NOT_DECODED; + slot.raw = r.data; + FetchInstrResult::hit(slot as *const DecodedInstr) + } else { + FetchInstrResult::exception(r.status) + } + } + + fn read(&self, _virt_addr: u64, phys_addr: u64) -> BusRead64 { + const { + assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") + }; + let a = phys_addr as u32; + if SIZE == 1 { + let r = self.downstream.read8(a); + BusRead64 { status: r.status, data: r.data as u64 } + } else if SIZE == 2 { + let r = self.downstream.read16(a); + BusRead64 { status: r.status, data: r.data as u64 } + } else if SIZE == 4 { + let r = self.downstream.read32(a); + BusRead64 { status: r.status, data: r.data as u64 } + } else { + self.downstream.read64(a) + } + } + + fn write(&self, _virt_addr: u64, phys_addr: u64, val: u64) -> u32 { + const { + assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") + }; + let a = phys_addr as u32; + if SIZE == 1 { + self.downstream.write8(a, val as u8) + } else if SIZE == 2 { + self.downstream.write16(a, val as u16) + } else if SIZE == 4 { + self.downstream.write32(a, val as u32) + } else { + self.downstream.write64(a, val) + } + } + + fn write64_masked(&self, _virt_addr: u64, phys_addr: u64, val: u64, mask: u64) -> u32 { + let aligned = (phys_addr & !7) as u32; + let r = self.downstream.read64(aligned); + if !r.is_ok() { + return r.status; + } + self.downstream.write64(aligned, (r.data & !mask) | (val & mask)) + } + + /// The whole point of the type. + /// + /// Every operation that only *moves data between cache and memory* — + /// invalidate, writeback, fill — is a genuine no-op here, because the + /// cache and memory can never disagree. The operations that move data + /// between the cache and a register are the ones with observable effects, + /// and those are served from the shadow. + fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u64 { + let sel = cache_op & 3; + let op = cache_op & 0x1C; + + // Tag operations under `op`, the whole data-array walk only under + // `hit` as well — see `l2_op_log`. + if l2_op_log(if matches!(op, C_IST | C_ILT) { CACHE_LOG_OP } else { CACHE_LOG_HIT }) { + crate::dlog!(LogModule::L2c, + "shadow: {:<22} raw={cache_op:#04x} va={virt_addr:#018x} arg={phys_addr:#018x}", + cache_op_name(cache_op)); + } + + match op { + // Index_Store_Tag / Index_Load_Tag. Stored and returned verbatim: + // a tag test is a round trip, and round trips do not require + // knowing what the bits mean. + C_IST => { + let idx = self.tag_slot(sel, virt_addr); + let mask = if matches!(sel, CACH_SI | CACH_SD) { L2_TAG_MASK } else { u64::MAX }; + if l2_op_log(CACHE_LOG_OP) { + crate::dlog!(LogModule::L2c, "shadow: -> IST va={virt_addr:#012x} idx={idx} stores {:#018x}", + phys_addr & mask & !MRU_BIT); + } + let s = self.shadow(sel); + if idx < s.tags.len() { + // The MRU bit belongs to the set, not to this way, so it + // is lifted out and kept separately. Everything else is + // stored verbatim: bit 32 in particular is *not* spare — + // TagHi[3:0] are tag address bits 35:32, and clearing it + // here destroyed real tag bits. + let set = (idx / WAYS).min(s.mru.len() - 1); + s.mru[set] = phys_addr & MRU_BIT != 0; + s.tags[idx] = phys_addr & mask & !MRU_BIT; + } + 0 + } + C_ILT => { + let idx = self.tag_slot(sel, virt_addr); + let s = self.shadow(sel); + let set = (idx / WAYS).min(s.mru.len() - 1); + let mut v = if idx < s.tags.len() { s.tags[idx] } else { 0 }; + // The bit belongs to the set: whichever way is read reports + // it. + if s.mru[set] { + v |= MRU_BIT; + } + if l2_op_log(CACHE_LOG_OP) { + crate::dlog!(LogModule::L2c, "shadow: -> ILT va={virt_addr:#012x} idx={idx} set={set} mru={} tag={:#018x}", + s.mru[set], if idx < s.tags.len() { s.tags[idx] } else { 0 }); + crate::dlog!(LogModule::L2c, "shadow: -> ILT slot={idx} returns {v:#018x}"); + } + v + } + + // R10000 reassigns 5/6/7, scoped to particular cache selects. + C_R10K_CBARRIER if R10K_OPS && sel == CACH_PI => 0, + C_R10K_ILD if R10K_OPS && matches!(sel, CACH_PI | CACH_PD | CACH_SD) => { + let s = self.shadow(sel); + let slot = self.data_slot(s.data.len(), virt_addr); + if s.data.is_empty() { + 0 + } else { + unsafe { *self.ecc_out.get() = s.ecc[slot] }; + if l2_op_log(CACHE_LOG_HIT) { + crate::dlog!(LogModule::L2c, "shadow: -> ILD slot={slot} data={:#018x} ecc={:#x}", + s.data[slot], s.ecc[slot]); + } + s.data[slot] + } + } + C_R10K_ISD if R10K_OPS && matches!(sel, CACH_SI | CACH_SD) => { + let s = self.shadow(sel); + let slot = self.data_slot(s.data.len(), virt_addr); + if !s.data.is_empty() { + s.data[slot] = phys_addr; + s.ecc[slot] = unsafe { *self.ecc_in.get() }; + if l2_op_log(CACHE_LOG_HIT) { + crate::dlog!(LogModule::L2c, "shadow: -> ISD slot={slot} data={phys_addr:#018x} ecc={:#x}", + s.ecc[slot]); + } + } + 0 + } + + // Invalidate, writeback, fill, and the R4000 hit operations. All + // no-ops: there is nothing held that could be stale or dirty. + _ => 0, + } + } + + fn get_config(&self, cache_target: u32) -> (usize, usize) { + match cache_target { + CACH_PI => (IC_SIZE, IC_LINE), + CACH_PD => (DC_SIZE, DC_LINE), + _ => (L2_SIZE, L2_LINE), + } + } + + /// The check bits ride with cache data on an R10000: a store takes them + /// from CP0 ECC and a load returns them there. + fn set_cache_ecc(&self, v: u32) { + unsafe { *self.ecc_in.get() = v }; + } + + fn cache_op_ecc(&self) -> u32 { + unsafe { *self.ecc_out.get() } + } + + fn downstream(&self) -> Arc { + self.downstream.clone() + } + + fn check_and_clear_llbit(&self, _phys_addr: u64) { + unsafe { *self.llbit.get() = false }; + } + fn get_llbit(&self) -> bool { + unsafe { *self.llbit.get() } + } + fn set_llbit(&self, val: bool) { + unsafe { *self.llbit.get() = val }; + } + fn get_lladdr(&self) -> u32 { + unsafe { *self.lladdr.get() } + } + fn set_lladdr(&self, addr: u32) { + unsafe { *self.lladdr.get() = addr }; + } +} + +/// SGI Indigo2 IMPACT R10000 (IP28). +/// +/// Real geometry as software reads it — 32 KB primaries with 64-byte +/// instruction lines and 32-byte data lines, a 1 MB secondary with 128-byte +/// lines — 64 TLB entries, MIPS IV, and the R10000 cache operation encodings. +/// Nothing of the microarchitecture: no ways, no LRU, no out-of-order. +pub type R10000ShadowCache = + ShadowCache<32768, 64, 32768, 32, 1048576, 128, true, 0x0000_0900, 0x0000_0900, 64, true>; + +#[cfg(test)] +mod tests { + use super::*; + use crate::mem::Memory; + use crate::mips_cache_v2::{CACH_PD, CACH_SD}; + + fn cache() -> R10000ShadowCache { + let mem: Arc = Arc::new(Memory::new(1024 * 1024)); + R10000ShadowCache::from(mem) + } + + /// A tag test is a round trip, so the shadow must return exactly the bits + /// it was given — including any the real layout would not use. The cache + /// model that preceded this one decoded TagLo into ptag/state/pidx fields + /// and re-encoded on the way out, which silently dropped the low seven + /// bits and every state code it did not recognise. A walking-1s test then + /// failed on its very first bit. + #[test] + fn tags_round_trip_every_bit() { + let c = cache(); + for bit in 0..64 { + // 32 and 63 are MRU control/state, not tag storage. + if bit == 32 || bit == 63 { continue; } + let v = 1u64 << bit; + c.cache_op(C_IST | CACH_SD, 0, v); + assert_eq!( + c.cache_op(C_ILT | CACH_SD, 0, 0), + v, + "bit {bit} did not survive a store/load tag round trip" + ); + } + } + + /// Index_Store_Data / Index_Load_Data likewise. + #[test] + fn secondary_data_round_trips() { + let c = cache(); + c.cache_op(C_R10K_ISD | CACH_SD, 0x40, 0xdeadbeef); + assert_eq!(c.cache_op(C_R10K_ILD | CACH_SD, 0x40, 0), 0xdeadbeef); + } + + /// The MRU bit is **shared between the ways of a set**: written through + /// one way, it must read back through the other. The IP28 PROM's + /// diagnostic does exactly this and nothing else satisfies it — a model + /// that records which way was marked reports nothing on the way that is + /// actually read. + #[test] + fn the_mru_bit_is_shared_between_the_ways_of_a_set() { + let c = cache(); + // Way 0 of set 0, then way 1 of set 0. + c.cache_op(C_IST | CACH_SD, 0, MRU_BIT); + assert_ne!( + c.cache_op(C_ILT | CACH_SD, 1, 0) & MRU_BIT, + 0, + "written through way 0, must be visible through way 1" + ); + // And the other direction. + c.cache_op(C_IST | CACH_SD, 1, MRU_BIT); + assert_ne!(c.cache_op(C_ILT | CACH_SD, 0, 0) & MRU_BIT, 0); + } + + /// Writing a tag without the bit clears it, again for the whole set. + #[test] + fn writing_a_tag_without_the_mru_bit_clears_it() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0, MRU_BIT); + c.cache_op(C_IST | CACH_SD, 0, 0); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 1, 0) & MRU_BIT, 0); + } + + /// Marking one set must not mark its neighbour. The PROM writes the + /// following set specifically to clobber the signal line between the two + /// checks, so a model where the bit leaks across sets passes the first + /// check and fails the next. + #[test] + fn the_mru_bit_does_not_leak_between_sets() { + let c = cache(); + let next_set = 128; // one line + c.cache_op(C_IST | CACH_SD, 0, MRU_BIT); + c.cache_op(C_IST | CACH_SD, next_set, 0); + assert_ne!(c.cache_op(C_ILT | CACH_SD, 1, 0) & MRU_BIT, 0, "set 0 keeps its bit"); + assert_eq!(c.cache_op(C_ILT | CACH_SD, next_set, 0) & MRU_BIT, 0, "set 1 has none"); + } + + /// Check bits are stored with the data and returned on load. The PROM's + /// ECC walk writes a different value through each way and reads them + /// back, so both the round trip and the separation matter. + /// + /// This exists because the methods carrying ECC were once declared on the + /// trait with defaults and simply never implemented here: the data + /// round-tripped perfectly and every check bit read back as zero. + #[test] + fn check_bits_travel_with_the_data() { + let c = cache(); + c.set_cache_ecc(0x001); + c.cache_op(C_R10K_ISD | CACH_SD, 0, 0x1111_2222_3333_4444); + c.set_cache_ecc(0x3fe); + c.cache_op(C_R10K_ISD | CACH_SD, 1, 0xaaaa_bbbb_cccc_dddd); + + c.set_cache_ecc(0); + assert_eq!(c.cache_op(C_R10K_ILD | CACH_SD, 0, 0), 0x1111_2222_3333_4444); + assert_eq!(c.cache_op_ecc(), 0x001, "way 0's check bits"); + assert_eq!(c.cache_op(C_R10K_ILD | CACH_SD, 1, 0), 0xaaaa_bbbb_cccc_dddd); + assert_eq!(c.cache_op_ecc(), 0x3fe, "way 1's check bits"); + } + + /// An untouched set reports no MRU, so an ordinary tag read is not + /// contaminated by it. + #[test] + fn an_untouched_set_has_no_mru_bit() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0, 0xdead_beef); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0, 0), 0xdead_beef); + } + + /// Separate lines must not alias onto one another. + #[test] + fn distinct_lines_hold_distinct_tags() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0, 0x1111_1111); + c.cache_op(C_IST | CACH_SD, 128, 0x2222_2222); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0, 0), 0x1111_1111); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 128, 0), 0x2222_2222); + } + + /// Bit 0 of a CACHE index selects the **way**, and the two ways of a set + /// must not alias. + /// + /// This is exactly the IP28 PROM's secondary-cache tag test: it stores one + /// tag to way 0 and a different one to way 1 of the same set, then reads + /// way 0 back. Folding the way bit away let the second store clobber the + /// first, and the PROM reported + /// `Expected: 0x0000000000000001 ... TAG walking 1s`. + #[test] + fn the_two_ways_of_a_set_are_independent() { + let c = cache(); + c.cache_op(C_IST | CACH_SD, 0x2000_0000, 0x0000_0001); + c.cache_op(C_IST | CACH_SD, 0x2000_0001, 0xffff_cdfe); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0x2000_0000, 0), 0x0000_0001, + "way 1's tag overwrote way 0's"); + assert_eq!(c.cache_op(C_ILT | CACH_SD, 0x2000_0001, 0), 0xffff_cdfe); + } + + /// Adjacent sets stay distinct once the way bit is accounted for. + #[test] + fn adjacent_sets_do_not_alias_through_the_way_bit() { + let c = cache(); + for set in 0..4u64 { + for way in 0..2u64 { + let va = set * 128 + way; + c.cache_op(C_IST | CACH_SD, va, 0x1000 + set * 16 + way); + } + } + for set in 0..4u64 { + for way in 0..2u64 { + let va = set * 128 + way; + assert_eq!(c.cache_op(C_ILT | CACH_SD, va, 0), + 0x1000 + set * 16 + way, + "set {set} way {way} aliased"); + } + } + } + + /// Nothing is ever held, so a load must see the store that preceded it + /// with no flush in between. This is the property that makes the design + /// safe, not merely fast. + #[test] + fn memory_is_the_only_store() { + let c = cache(); + c.write::<4>(0, 0x2000, 0x1234_5678); + assert_eq!(c.read::<4>(0, 0x2000).data, 0x1234_5678); + // An invalidate cannot lose it, and a writeback cannot be needed. + c.cache_op(crate::mips_cache_v2::C_IINV | CACH_PD, 0x2000, 0); + assert_eq!(c.read::<4>(0, 0x2000).data, 0x1234_5678); + } + + /// Geometry is what CP0 Config is built from, so it must be the real + /// part's even though the model holds nothing. + #[test] + fn reported_geometry_is_the_real_parts() { + let c = cache(); + assert_eq!(c.get_config(CACH_PI), (32768, 64)); + assert_eq!(c.get_config(CACH_PD), (32768, 32)); + assert_eq!(c.get_config(CACH_SD), (1048576, 128)); + } + + /// Under tcache the JIT serves this model through the window alone, and + /// only once both windows are published: the emitted code dereferences + /// them without a null check. + #[test] + #[cfg(all(feature = "tcache", feature = "jitv2"))] + fn tcache_offers_a_tagless_path_once_both_windows_exist() { + let c = cache(); + assert!(!c.jit_dc_geometry().supported, "no window yet"); + let mut window = [0u8; 8]; + unsafe { c.set_tcache_window(window.as_mut_ptr()) }; + assert!(!c.jit_dc_geometry().supported, "no generation window yet"); + let mut gens = [std::sync::atomic::AtomicU64::new(0)]; + unsafe { c.set_tcache_gen_window(gens.as_mut_ptr()) }; + let g = c.jit_dc_geometry(); + assert!(g.supported && g.tagless && !g.has_l2, + "tagless, and no L2 decode slots for a store to invalidate: {g:?}"); + } +} diff --git a/src/mips_cache_v2.rs b/src/mips_cache_v2.rs index 89e98ebf..8552c7d6 100644 --- a/src/mips_cache_v2.rs +++ b/src/mips_cache_v2.rs @@ -63,6 +63,16 @@ pub use crate::mips_isa::{ C_HINV, C_HWBINV, C_FILL, C_HWB, C_HSV, }; +// R10000 reassigns cache operations 5, 6 and 7. Where an R4000 has +// Hit_Invalidate, Hit_Writeback_Invalidate and Hit_Writeback, an R10000 has a +// cache barrier and index-addressed load/store of the cache *data* array. +// Confirmed against NetBSD's mips/include/cache_r10k.h, and against the IP28 +// PROM, whose secondary-cache SRAM test issues C_R10K_ISD(SD) — which this +// emulator was executing as a hit-writeback, so nothing was ever stored. +pub const C_R10K_CBARRIER: u32 = 5 << 2; +pub const C_R10K_ILD: u32 = 6 << 2; +pub const C_R10K_ISD: u32 = 7 << 2; + /// Decode a raw cache_op field (5-bit: op[4:2] | target[1:0]) to a human-readable name. /// Matches the disassembler mnemonic convention used by gas/objdump. pub fn cache_op_name(op: u32) -> &'static str { @@ -202,6 +212,11 @@ pub struct JitDcGeometry { /// `set | (way << num_lines_shift)` = extended tag index. Only meaningful /// when `ways > 1`. pub num_lines_shift: u32, + /// No L1-D model at all: loads and stores go straight to memory, so under + /// tcache the inline path is the ppmem-window half alone, with no tag + /// probe, no LRU and no dirty bit (the R10000's shadow cache). The other + /// fields are then unused. + pub tagless: bool, } impl JitDcGeometry { @@ -209,7 +224,7 @@ impl JitDcGeometry { Self { supported: false, line_shift: 0, num_lines_mask: 0, data_mask: 0, has_l2: false, l2_line_shift: 0, l2_num_lines_mask: 0, - ways: 1, num_lines_shift: 0, + ways: 1, num_lines_shift: 0, tagless: false, } } } @@ -431,6 +446,11 @@ pub trait CpuModel: MipsCache { const VA_BITS: u32 = 40; /// Name as the guest and the benchmark report see it. const NAME: &'static str; + /// Cache ops 5/6/7 carry their R10000 meanings rather than their R4000 ones. + /// The executor needs this: `Index_Store_Data` takes its value from TagLo + /// and `Index_Load_Data` returns into it, neither of which is true of the + /// R4000 hit operations that share those encodings. + const R10K_CACHE_OPS: bool = false; } pub trait MipsCache: Send + Sync { @@ -633,7 +653,19 @@ pub trait MipsCache: Send + Sync { /// /// For Index_Load_Tag operations, returns the tag value in TagLo CP0 register format /// For other operations, returns 0 - fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u32; + /// Perform a CACHE operation. For `Index_Load_Tag` the return is the + /// tag as software sees it, full width — an R10000 secondary tag does not + /// fit in 32 bits. + fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u64; + + /// CP0 `ECC` ($26) on the way in to a cache-data store. + /// + /// On an R10000 the check bits ride with the data: `Index_Store_Data` + /// takes them from this register and `Index_Load_Data` returns them + /// there. Models that do not keep them ignore both of these. + fn set_cache_ecc(&self, _v: u32) {} + /// CP0 `ECC` after the last cache operation. + fn cache_op_ecc(&self) -> u32 { 0 } /// Write back dirty L1-D (and, if present, L2) lines covering /// `[phys_addr, phys_addr + size)` to memory, without invalidating them. @@ -800,7 +832,7 @@ impl MipsCache for PassthroughCacheOf { self.downstream.write64(aligned_addr, new_val) } - fn cache_op(&self, _cache_op: u32, _virt_addr: u64, _phys_addr: u64) -> u32 { + fn cache_op(&self, _cache_op: u32, _virt_addr: u64, _phys_addr: u64) -> u64 { // No-op for passthrough cache - just return 0 0 } @@ -1064,6 +1096,7 @@ pub struct CpuCache< const L2_CACHE_SIZE: usize, const L2_LINE: usize, const L2_TAGS: usize, const L2_DATA: usize, const L2_NINSTRS: usize, const HAS_L2: bool, const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, + const MODEL: u8, > { downstream: Arc, @@ -1141,12 +1174,12 @@ unsafe impl Send for CpuCache {} + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Send for CpuCache {} unsafe impl Sync for CpuCache {} + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Sync for CpuCache {} // Per-level cache types, parameterised so each CPU model monomorphises its own. type ICacheT = @@ -1156,24 +1189,91 @@ type DCacheT = Cache; +/// Which processor a monomorphisation models, as the `MODEL` const parameter. +/// +/// This exists because the thing that distinguishes these parts *in this file* +/// is not any one of the shape parameters. It was originally inferred from +/// `IC_WAYS == 2`, which worked only while "2-way" and "R5000" named the same +/// processor. They do not: the R10000 is also 2-way, and it shares neither the +/// R5000's TagLo layout nor its cache-op semantics. Inferring identity from +/// shape would have quietly given an R10000 every R5000 behaviour in the 40-odd +/// places that branch on it, each one individually plausible. +pub mod model { + pub const R4400: u8 = 0; + pub const R5000: u8 = 1; + pub const R10000: u8 = 2; +} + /// SGI Indy R4400: direct-mapped 16K L1s, 1 MB unified L2 owning the decode slots. pub type R4400Cache = CpuCache<16384, 16, 1, 1024, 16384, 16, 1, 1024, 2048, 1048576, 128, 8192, 131072, 262144, true, - false, 0x0000_0440, 0x0000_0500, 48>; + false, 0x0000_0440, 0x0000_0500, 48, { model::R4400 }>; /// SGI Indy R5000: 2-way 32K L1s, no secondary cache; L1I owns its decode slots. pub type R5000Cache = CpuCache<32768, 32, 2, 1024, 32768, 32, 2, 1024, 4096, 128, 128, 1, 16, 0, false, - true, 0x0000_2321, 0x0000_2300, 48>; + true, 0x0000_2321, 0x0000_2300, 48, { model::R5000 }>; +/// SGI Indigo2 IMPACT R10000 (IP28), modelled for speed rather than fidelity. +/// +/// The real part has two-way 32 KB L1s and a two-way secondary cache. This +/// models all three **direct-mapped**, keeping the real total sizes and line +/// sizes (64-byte L1I lines, 32-byte L1D, 128-byte L2). +/// +/// That is deliberate. Associativity reaches software only through the +/// way-select bits of `CACHE Index_*`, and the only thing an operating system +/// does by index is flush the whole cache — which comes out the same for any +/// geometry holding the same lines. Emulating ways costs a victim-selection +/// and LRU update on every access and buys nothing IRIX can observe. +/// +/// It also buys correctness here, not just speed. A two-way L1 *with* an L2 is +/// a combination this file has never had: `fetch()` selects the two-way tag +/// probe and the L1I-resident decode slots under one condition, and such a +/// part needs the first with the second's alternative. Direct-mapped plus L2 +/// is exactly the R4400's shape, so every associativity and decode-slot branch +/// already does the right thing for this model, unchanged. +/// +/// What is *not* faked is anything software reads back: the PRId, the TLB +/// size, MIPS IV decoding, and the cache tag layout the PROM's diagnostics +/// inspect directly. +#[cfg(feature = "ip28")] +pub type R10000Cache = CpuCache<32768, 64, 1, 512, + 32768, 32, 1, 1024, 4096, + 1048576, 128, 8192, 131072, 262144, true, + true, 0x0000_0900, 0x0000_0900, 64, { model::R10000 }>; impl CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuCache { // Model discriminator: folds to a literal, so it replaces #[cfg(feature = "r5k")]. - const IS_R5K: bool = IC_WAYS == 2; + // Keyed on MODEL, not on associativity — see `mod model`. + const IS_R5K: bool = MODEL == model::R5000; + + /// IP28 bring-up watch: physical window to trace, from IRIS_IP28_WATCH. + /// Folds to a constant `None` on every model but the R10000, so the + /// non-IP28 hot path keeps no trace of this. + #[inline(always)] + fn ip28_watch() -> Option { + if !Self::R10K_CACHE_OPS { return None; } + static W: std::sync::OnceLock> = std::sync::OnceLock::new(); + *W.get_or_init(|| { + std::env::var("IRIS_IP28_WATCH").ok().and_then(|v| { + u64::from_str_radix(v.trim_start_matches("0x"), 16).ok() + }) + }) + } + + #[inline(always)] + fn ip28_trace(&self, what: &str, addr: u64, val: u64) { + if let Some(w) = Self::ip28_watch() { + if (addr & !0x7f) == (w & !0x7f) { + eprintln!("ip28watch: {what:<18} addr={addr:#018x} val={val:#018x}"); + } + } + } + // Logical L2 size; 0 means the model has no secondary cache. pub const L2_SIZE: usize = if HAS_L2 { L2_CACHE_SIZE } else { 0 }; @@ -1298,7 +1398,7 @@ impl From> for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> From> for CpuCache { fn from(downstream: Arc) -> Self { Self::new(downstream) } @@ -1308,7 +1408,7 @@ impl CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuCache { /// Check if we're tracking this physical address (for debug purposes) #[cfg(feature = "debug_cache")] #[inline] @@ -2811,19 +2911,24 @@ impl CpuModel for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuModel for CpuCache { const MIPS4: bool = MIPS4; const PRID: u32 = PRID; const FIR: u32 = FIR; const TLB_ENTRIES: usize = TLB_ENTRIES; - const NAME: &'static str = if IC_WAYS == 2 { "R5000" } else { "R4400" }; + const R10K_CACHE_OPS: bool = MODEL == model::R10000; + const NAME: &'static str = match MODEL { + model::R5000 => "R5000", + model::R10000 => "R10000", + _ => "R4400", + }; } impl MipsCache for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> MipsCache for CpuCache { fn set_l1i_counters(&mut self, hit: Arc, fetch: Arc) { self.l1i_hit_count = hit; self.l1i_fetch_count = fetch; @@ -2909,6 +3014,7 @@ impl(&self, virt_addr: u64, phys_addr: u64) -> BusRead64 { + self.ip28_trace("read", phys_addr, 0); const { assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") }; #[cfg(feature = "debug_cache")] { @@ -3194,6 +3301,7 @@ impl(&self, virt_addr: u64, phys_addr: u64, val: u64) -> u32 { + self.ip28_trace("write", phys_addr, val); const { assert!(SIZE == 1 || SIZE == 2 || SIZE == 4 || SIZE == 8, "invalid memory access SIZE") }; #[cfg(feature = "debug_cache")] { @@ -3320,7 +3428,55 @@ impl u32 { + fn cache_op(&self, cache_op: u32, virt_addr: u64, phys_addr: u64) -> u64 { + if Self::R10K_CACHE_OPS && std::env::var_os("IRIS_IP28_CACHEOPS").is_some() { + use std::sync::atomic::{AtomicU32, Ordering}; + static SEEN: AtomicU32 = AtomicU32::new(0); + let bit = 1u32 << (cache_op & 0x1F); + let all = matches!(std::env::var("IRIS_IP28_CACHEOPS").as_deref(), Ok("all")); + if all || SEEN.fetch_or(bit, Ordering::Relaxed) & bit == 0 { + eprintln!("ip28: op {} raw={:#04x} va={:#018x} arg={:#018x}", + cache_op_name(cache_op), cache_op, virt_addr, phys_addr); + } + } + if Self::ip28_watch().is_some() { + self.ip28_trace(cache_op_name(cache_op), virt_addr, phys_addr); + } + // R10000 ops 5/6/7 are not the R4000 hit operations that share these + // encodings — see C_R10K_ISD. Handled before the shared decode below. + if Self::R10K_CACHE_OPS { + // The R10000 meanings are scoped to particular cache selects; + // outside those the R4000 operation on the same encoding still + // applies. cache_r10k.h annotates each one, and the IP28 PROM uses + // both readings of op 5: `Cache_Barrier` against the instruction + // cache, and R4000 `Hit_Invalidate` against the secondary. + let sel = cache_op & 3; + match cache_op & 0x1C { + // An ordering barrier. Nothing to do in a model with no + // speculative memory pipeline to hold back. + C_R10K_CBARRIER if sel == CACH_PI => return 0, + C_R10K_ILD if matches!(sel, CACH_PI | CACH_PD | CACH_SD) => { + let is_l2 = sel == CACH_SD; + // The index is a byte offset into the data array. Bit 0 + // selects the way on real silicon; this model is + // direct-mapped, and bit 0 falls below the u64 slot index, + // so it drops out without any special case. + let slot = (virt_addr as usize) >> 3; + if is_l2 && HAS_L2 { + return self.l2.data()[slot & (L2_DATA - 1)]; + } + return self.dc.data()[slot & (DC_DATA - 1)]; + } + C_R10K_ISD if matches!(sel, CACH_SI | CACH_SD) => { + let slot = (virt_addr as usize) >> 3; + if HAS_L2 { + self.l2.data_mut()[slot & (L2_DATA - 1)] = phys_addr; + } + return 0; + } + _ => {} + } + } // Decode cache operation let cache_target = cache_op & 0x3; // bits [17:16] let operation = cache_op & 0x1C; // bits [20:18] (shifted by 2) @@ -3410,6 +3566,10 @@ impl 0, L2_CS_CLEAN_EXCLUSIVE => 4, @@ -3418,7 +3578,7 @@ impl 7, _ => 0, }; - (tag.ptag() << 13) | (state << 10) | (tag.pidx() << 7) + ((tag.ptag() << 13) | (state << 10) | (tag.pidx() << 7)) as u64 } else if is_icache { let tag: L1ITag = self.ic.get_tag(idx); let raw_ptag = (tag.ptag >> L1_PTAG_SHIFT) as u32 & L1_PTAG_MASK; @@ -3426,11 +3586,11 @@ impl 3u32, _ => 0u32, }; - (raw_ptag << 8) | (pstate << 6) + ((raw_ptag << 8) | (pstate << 6)) as u64 } } } // Index Store Tag — write CP0 TagLo into internal tag C_IST => { - let tag_lo = phys_addr as u32; + let tag_lo = phys_addr as u32; // R4000-family tags are 32-bit if is_l2 { + if MODEL == model::R10000 { + eprintln!("ip28: C_IST(SD) idx={idx:#x} taglo={tag_lo:#010x} phys={phys_addr:#018x}"); + } // L2 TagLo format: [31:13] ptag [12:10] state [9:7] PIdx let ptag = (tag_lo >> 13) & L2_PTAG_MASK; let state = (tag_lo >> 10) & 0x7; @@ -4027,7 +4190,7 @@ impl Drop for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Drop for CpuCache { fn drop(&mut self) { self.ic.stop.store(true, Ordering::Relaxed); } @@ -4039,7 +4202,7 @@ impl Resettable for CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> Resettable for CpuCache { fn power_on(&self) { self.ic.tags_mut().fill(L1ITag::default()); self.dc.tags_mut().fill(L1DTag::default()); @@ -4070,7 +4233,7 @@ impl CpuCache { + const MIPS4: bool, const PRID: u32, const FIR: u32, const TLB_ENTRIES: usize, const MODEL: u8> CpuCache { fn save_tags_as_u32>(tags: &[TAG]) -> Vec { tags.iter().map(|&t| t.into()).collect() } @@ -4219,6 +4382,32 @@ mod tests { R4400Cache::new(mem as Arc) } + /// Deliberately the R5000's exact shape, differing *only* in `MODEL`. If + /// model identity is ever inferred from a shape parameter again, this + /// aliases onto the R5000 and the test below fails. + type NotAnR5000 = CpuCache<32768, 32, 2, 1024, + 32768, 32, 2, 1024, 4096, + 128, 128, 1, 16, 0, false, + true, 0x0000_0900, 0x0000_0900, 64, { model::R10000 }>; + + /// Two-way associativity is not an identity. + /// + /// `IS_R5K` used to be `IC_WAYS == 2`, which was true of every 2-way part + /// the file knew about. The R10000 is also 2-way and shares neither the + /// R5000's TagLo layout nor its cache-op semantics, so that inference would + /// have handed it R5000 behaviour at all 40-odd sites that branch on it — + /// silently, and each one plausible on its own. + #[test] + fn model_identity_is_explicit_and_not_inferred_from_associativity() { + assert_eq!(::NAME, "R4400"); + assert_eq!(::NAME, "R5000"); + assert_eq!(::NAME, "R10000"); + + assert!(R5000Cache::IS_R5K, "the R5000 is the R5000"); + assert!(!NotAnR5000::IS_R5K, "a 2-way cache is not what makes an R5000"); + assert!(!R4400Cache::IS_R5K); + } + // Same helper for whichever CPU model a test wants to exercise. fn make_cache_of>>(mem: Arc) -> C { C::from(mem as Arc) @@ -4301,7 +4490,7 @@ mod tests { cache.cache_op(C_IST | CACH_PD, va, written as u64); let read_back = cache.cache_op(C_ILT | CACH_PD, va, idx); assert_eq!((read_back >> 6) & 0x3, 3, "R4400 DirtyExclusive must round-trip"); - assert_eq!(read_back >> 8, raw_ptag, "physical tag must round-trip"); + assert_eq!(read_back >> 8, raw_ptag as u64, "physical tag must round-trip"); } // R5000: valid+dirty is D=1,V=1 (0xC0). Under the old code this was @@ -4320,9 +4509,9 @@ mod tests { let read_back = cache.cache_op(C_ILT | CACH_PD, va, idx); assert_ne!(read_back & (1 << 6), 0, "R5000 {label} line must read back V=1"); - assert_eq!(read_back & (1 << 7), d_bit, + assert_eq!(read_back & (1 << 7), d_bit as u64, "R5000 {label} line must round-trip its D bit"); - assert_eq!(read_back >> 8, raw_ptag, "physical tag must round-trip"); + assert_eq!(read_back >> 8, raw_ptag as u64, "physical tag must round-trip"); } } } diff --git a/src/mips_core.rs b/src/mips_core.rs index dd0312e8..571fb0bd 100644 --- a/src/mips_core.rs +++ b/src/mips_core.rs @@ -28,6 +28,38 @@ pub const STATUS_CU1: u32 = 1 << 29; // Coprocessor 1 (FPU) Usable pub const STATUS_CU2: u32 = 1 << 30; // Coprocessor 2 Usable pub const STATUS_CU3: u32 = 1 << 31; // Coprocessor 3 Usable +/// Is CP0 register tracing on? `log mips mask cp0` on the monitor. +/// +/// Traces everything the cache-error machinery touches — CP0 ECC (26) and +/// CacheErr (27) accesses, Status.DE transitions, and CACHE ops. +/// +/// Built to explain why SGI's own IDE field diagnostic reports "Failure +/// detected on the CPU module" on an otherwise healthy emulated R10000. What +/// it found: the IDE ends in a bounded 3712-iteration loop reading CacheErr +/// and ECC, both of which return zero every pass, because nothing in this +/// emulator ever writes CacheErr — we do not detect or log a cache error at +/// all. Note it does this with Status.DE *set*: DE suppresses only the trap, +/// while real hardware still records the error, which is what a poll-based +/// parity test relies on. (An earlier guess that the IDE wanted a Cache Error +/// *exception* was refuted by this tracer — DE is never cleared outside the +/// PROM's own memory sizing.) +/// +/// Two relaxed atomic loads when off, and unlike the `IRIS_IP28_CACHEDIAG` +/// environment variable this replaced it can be turned on, masked and +/// redirected to a file part-way through a boot. +#[inline(always)] +pub fn cachediag_on() -> bool { + cp0_log() +} + +/// `log mips mask cp0` — CP0 register traffic. +#[inline(always)] +pub fn cp0_log() -> bool { + crate::devlog::devlog_is_active(crate::devlog::LogModule::Mips) + && (crate::devlog::devlog_mask(crate::devlog::LogModule::Mips) + & crate::mips_exec::MIPS_LOG_CP0) != 0 +} + // CP0 Cause Register bit definitions pub const CAUSE_EXCCODE_MASK: u32 = 0x1F << 2; // Exception Code mask pub const CAUSE_EXCCODE_SHIFT: u32 = 2; // Exception Code shift @@ -399,10 +431,6 @@ pub struct MipsCore { /// ppmem window base — tcache's data source (`tc_base + phys`). #[cfg(all(feature = "jitv2", feature = "tcache"))] pub jit_tc_base: *mut u8, - /// 64MB-granularity mapped-region bitmap; bit `phys >> 26` set = the - /// region is fully mapped RAM and reachable through the window. - #[cfg(all(feature = "jitv2", feature = "tcache"))] - pub jit_tc_bitmap: *const u64, /// Base of the L2 tag array (`[L2Tag]`, `u32` each). The inline tcache /// store clears `has_code` on the written line so L1-I refills cannot /// reuse stale decoded instructions — mirroring `tc_invalidate_l2_code`. @@ -816,7 +844,14 @@ pub struct MipsCore { pub cp0_xcontext: u64, // 20: Extended Context (64-bit) pub cp0_ecc: u32, // 26: ECC Register pub cp0_cacheerr: u32, // 27: Cache Error - pub cp0_taglo: u32, // 28: Cache Tag Low + /// 28: Cache Tag Low. 64 bits, not 32. + /// + /// An R4000's TagLo fits in 32, but an R10000's secondary cache tag does + /// not — it carries a 40-bit physical address, and the IP28 PROM's tag + /// diagnostic writes and expects back values like 0x0000000f_ffffcdfe. + /// Truncating to 32 lost the top nibble and the PROM reported the + /// difference. + pub cp0_taglo: u64, // 28: Cache Tag Low pub cp0_taghi: u32, // 29: Cache Tag High pub cp0_errorepc: u64, // 30: Error Exception PC @@ -1253,8 +1288,6 @@ impl MipsCore { #[cfg(all(feature = "jitv2", feature = "tcache"))] jit_tc_base: std::ptr::null_mut(), #[cfg(all(feature = "jitv2", feature = "tcache"))] - jit_tc_bitmap: std::ptr::null(), - #[cfg(all(feature = "jitv2", feature = "tcache"))] jit_l2_tags: std::ptr::null_mut(), #[cfg(all(feature = "jitv2", feature = "tcache"))] jit_tc_gen: std::ptr::null_mut(), @@ -1466,6 +1499,24 @@ impl MipsCore { /// Write a GPR by index. Unconditionally re-zeros gpr[0] to avoid a branch. #[inline(always)] pub fn write_gpr(&mut self, reg: u32, value: u64) { + // IP28 bring-up: `IRIS_IP28_WATCHGPR=` reports every write to that + // GPR with the PC that did it. One cached bool and a compare against a + // register number already in hand; unarmed it is a predictable branch. + #[cfg(debug_assertions)] + let _ = (); + { + static WATCH: std::sync::OnceLock> = std::sync::OnceLock::new(); + let watch = *WATCH.get_or_init(|| { + std::env::var("IRIS_IP28_WATCHGPR").ok().and_then(|v| v.trim().parse().ok()) + }); + if watch == Some(reg) { + // The register number is a parameter, so it stays an + // environment variable; only the output moves, so it can be + // redirected with `log mips file `. + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28gpr: ${reg} = {value:#018x} at pc={:#018x}", self.pc); + } + } unsafe { *self.gpr.get_unchecked_mut(reg as usize) = value; } self.gpr[0] = 0; } @@ -1545,9 +1596,21 @@ impl MipsCore { eprintln!("[ip7] MFC0 PerfCnt (reg 25) read -> 0"); 0 } - 26 => self.cp0_ecc as u64, - 27 => self.cp0_cacheerr as u64, - 28 => self.cp0_taglo as u64, + 26 => { + if cachediag_on() { + eprintln!("ip28cd: MFC0 ECC -> {:#010x} pc={:#018x}", + self.cp0_ecc, self.pc); + } + self.cp0_ecc as u64 + } + 27 => { + if cachediag_on() { + eprintln!("ip28cd: MFC0 CacheErr -> {:#010x} pc={:#018x}", + self.cp0_cacheerr, self.pc); + } + self.cp0_cacheerr as u64 + } + 28 => self.cp0_taglo, 29 => self.cp0_taghi as u64, 30 => self.cp0_errorepc, _ => 0, // Unimplemented registers read as 0 @@ -1579,7 +1642,7 @@ impl MipsCore { 20 => self.cp0_xcontext, 26 => self.cp0_ecc as u64, 27 => self.cp0_cacheerr as u64, - 28 => self.cp0_taglo as u64, + 28 => self.cp0_taglo, 29 => self.cp0_taghi as u64, 30 => self.cp0_errorepc, _ => 0, @@ -1913,6 +1976,16 @@ impl MipsCore { self.reanchor_count_and_reschedule(); } 10 => { // always use 64bit mask because the entries need to be valid in 64 bit mode even when they were set from 32 bit mode + // `log mips mask cp0` shows what the guest wrote against + // what survives the mask. The mask is R4400's 40-bit virtual + // address; the R10000 implements 44. + if cp0_log() && (value & !0xC000_00FF_FFFF_E0FF) != 0 { + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28ehi: wrote {value:#018x} -> kept {:#018x} (lost {:#018x}) pc={:#018x}", + value & 0xC000_00FF_FFFF_E0FF, + value & !0xC000_00FF_FFFF_E0FF, + self.pc); + } self.cp0_entryhi = value & 0xC000_00FF_FFFF_E0FF; }, 11 => { @@ -2036,6 +2109,18 @@ impl MipsCore { 12 => { let old = self.cp0_status; self.cp0_status = value as u32; + // Status.DE gates cache error exceptions. A diagnostic that + // means to provoke one has to clear it first, so a DE + // transition is the guest announcing its intent. + if cachediag_on() && (old ^ self.cp0_status) & STATUS_DE != 0 { + eprintln!("ip28cd: Status.DE {} pc={:#018x}", + if self.cp0_status & STATUS_DE != 0 { + "SET (cache exceptions DISABLED)" + } else { + "CLEAR (cache exceptions ENABLED)" + }, + self.pc); + } // Trace every change to the IP7 mask bit (Status.IM7). Linux's // mips_cpu_irq_controller masks IM7 on interrupt entry // (irq_ack) and unmasks on EOI; if the unmask never comes, no @@ -2063,7 +2148,17 @@ impl MipsCore { let mask = CAUSE_IP0 | CAUSE_IP1; self.cp0_cause = (self.cp0_cause & !mask) | ((value as u32) & mask); } - 14 => self.cp0_epc = value, + 14 => { + // A guest write of a 32-bit value here is where a 64-bit + // return address would lose its top half, so the ERET that + // follows lands at a truncated PC. `log mips mask cp0`. + if cp0_log() { + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28epc: write EPC={value:#018x} (was {:#018x}) from pc={:#018x}", + self.cp0_epc, self.pc); + } + self.cp0_epc = value; + } 15 => { /* PRId is read-only */ } 16 => { // Bits 5:0 always writable (K0, CU, DB, IB). @@ -2087,14 +2182,37 @@ impl MipsCore { 17 => self.cp0_lladdr = value as u32, 18 => self.cp0_watchlo = value as u32, 19 => self.cp0_watchhi = value as u32, - 20 => self.cp0_xcontext = value, + 20 => { + // IP28 bring-up: IRIX's XTLB refill handler builds the page + // table base itself and expects XContext's PTEBase to be + // zero, so anything landing in bits [63:33] becomes a wild + // pointer inside the handler. `log mips mask cp0`. + if cp0_log() { + crate::dlog!(crate::devlog::LogModule::Mips, + "ip28xctx: write {value:#018x} (ptebase={:#018x}) pc={:#018x}", + value & 0xFFFF_FFFE_0000_0000, self.pc); + } + self.cp0_xcontext = value; + } 25 => { #[cfg(feature = "developer_ip7")] eprintln!("[ip7] MTC0 PerfCnt (reg 25) write {:#018x} (ignored)", value); } - 26 => self.cp0_ecc = value as u32, - 27 => self.cp0_cacheerr = value as u32, - 28 => self.cp0_taglo = value as u32, + 26 => { + if cachediag_on() { + eprintln!("ip28cd: MTC0 ECC = {:#010x} (was {:#010x}) pc={:#018x}", + value as u32, self.cp0_ecc, self.pc); + } + self.cp0_ecc = value as u32; + } + 27 => { + if cachediag_on() { + eprintln!("ip28cd: MTC0 CacheErr = {:#010x} (was {:#010x}) pc={:#018x}", + value as u32, self.cp0_cacheerr, self.pc); + } + self.cp0_cacheerr = value as u32; + } + 28 => self.cp0_taglo = value, 29 => self.cp0_taghi = value as u32, 30 => self.cp0_errorepc = value, _ => {} // Writes to unimplemented registers are ignored diff --git a/src/mips_exec.rs b/src/mips_exec.rs index c1308662..f70a8471 100644 --- a/src/mips_exec.rs +++ b/src/mips_exec.rs @@ -24,6 +24,8 @@ pub const MIPS_LOG_INSN: u32 = 0x0001; // per-instruction disassembly trace pub const MIPS_LOG_TLB: u32 = 0x0002; // TLB read/write/probe pub const MIPS_LOG_MEM: u32 = 0x0004; // uncached memory accesses pub const MIPS_LOG_FPU: u32 = 0x0008; // FP compare/condmove/convert operand+result trace +pub const MIPS_LOG_CP0: u32 = 0x0010; // CP0 register traffic: EPC/EntryHi/XContext writes, + // ECC and CacheErr accesses, Status.DE transitions /// Opt-in gate for the `developerx` Coprocessor-Unusable break (see the three /// `handle_exception*` wrappers). Off unless `IRIS_BREAK_CPU=1`. @@ -44,6 +46,17 @@ fn mips_log(bit: u32) -> bool { devlog_is_active(LogModule::Mips) && (devlog_mask(LogModule::Mips) & bit) != 0 } +/// Like `mips_log`, but live in every build. +/// +/// The IP28 bring-up traces this gates were plain environment variables that +/// worked in release, and the machines they diagnose are booted in release — +/// so they keep that reach. The cost is two relaxed atomic loads, against the +/// `getenv` per call some of them used to do. +#[inline(always)] +fn mips_log_always(bit: u32) -> bool { + devlog_is_active(LogModule::Mips) && (devlog_mask(LogModule::Mips) & bit) != 0 +} + // Without `developer`, dlog_dev! is a no-op, so callers gate on this constant `false` // instead of paying an atomic load per call site to check a flag that can never fire. #[cfg(not(feature = "developer"))] @@ -333,7 +346,7 @@ struct CpuSnapshot { cp0_xcontext: u64, cp0_ecc: u32, cp0_cacheerr: u32, - cp0_taglo: u32, + cp0_taglo: u64, cp0_taghi: u32, cp0_errorepc: u64, @@ -1159,6 +1172,18 @@ impl MipsCpuConfig { pub const fn indy() -> Self { Self { tlb_entries: 48 } } + + /// The JTLB size the CPU model declares. + /// + /// `core.tlb_entries` is already taken from the model, so sizing the TLB + /// itself from anything else leaves the two disagreeing: Random cycles + /// over a range the array does not have, and a TLBWI to an index past the + /// end is silently dropped. The R10000 has 64 entries where the R4400 has + /// 48, and SGI's IP28 diagnostic writes index 48 on its first cache-alias + /// test — every one of its reported failures was that write going nowhere. + pub fn for_model() -> Self { + Self { tlb_entries: C::TLB_ENTRIES } + } } /// MIPS Execution Engine - combines CPU core with memory interface and TLB @@ -2704,6 +2729,37 @@ impl MipsExecutor { config |= ss << CONFIG_TR_SS; } + // R10000 lays Config out completely differently from an R4000, and the + // PROM sizes its cache diagnostics from it. Fields, per NetBSD's + // mips/include/cpuregs.h (MIPS4_CONFIG_*): + // [31:29] IC primary I-cache size, as 4096 << field + // [28:26] DC primary D-cache size, likewise + // [18:16] SS secondary cache size + // [15] BE big endian + // [13] SB secondary block size, 0 = 64B, 1 = 128B + // [2:0] K0 kseg0 cacheability, which the PROM sets for itself + // Presenting an R4000 Config here told an R10000 PROM that its + // secondary cache size field was zero. + if C::R10K_CACHE_OPS { + let log2 = |n: usize| (n / 4096).trailing_zeros(); + let mut c = 0u32; + c |= log2(32768) << 29; // 32 KB L1I + c |= log2(32768) << 26; // 32 KB L1D + c |= 1 << 15; // big endian + if C::L2_LINE == 128 { c |= 1 << 13; } + // Secondary cache size. The encoding is not in anything to hand, + // so it was swept against the PROM rather than guessed. + // + // 1 is the value to keep: IRIX reports "Secondary unified + // instruction/data cache size: 1 Mbyte", which agrees with this + // model's own `L2_SIZE`, and the PROM's power-on diagnostics pass + // and IRIX boots with it. 4 also passes POST but has IRIX report + // 8 MB, contradicting the model. + c |= 1 << 16; + c |= 2; // K0 = uncached at reset + config = c; + } + core.cp0_config = config; core.tlb_entries = C::TLB_ENTRIES as u32; // MipsCore::new already ran reset_registers, so set both the reset value @@ -3038,7 +3094,6 @@ impl MipsExecutor { #[cfg(feature = "tcache")] { self.core.jit_tc_base = self.cache.tcache_base_ptr(); - self.core.jit_tc_bitmap = self.cache.tcache_bitmap_ptr() as *const u64; self.core.jit_tc_gen = self.cache.tcache_gen_ptr(); self.core.jit_l2_tags = self.cache.jit_l2_tags_ptr(); } @@ -4252,7 +4307,91 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", /// Handle an exception: update CP0 registers and jump to handler vector. /// Takes an ExecStatus with EXEC_IS_EXCEPTION set; extracts code and TLB-refill flag. + /// IP28 bring-up: dump everything about the one exception whose BadVAddr + /// is `IRIS_IP28_EXC_VADDR`. + /// + /// A first-N trace is useless for a fault that lands after thousands of + /// ordinary TLB misses, and reading the register file from the monitor + /// afterwards shows the guest's panic handler, not the fault. Costs one + /// relaxed load per exception on the R10000 model and folds away on every + /// other; exceptions are not a hot path. + #[inline] + fn ip28_trace_exception(&self, status: ExecStatus) { + if !C::R10K_CACHE_OPS { + return; + } + static WANT: std::sync::OnceLock> = std::sync::OnceLock::new(); + let want = *WANT.get_or_init(|| { + std::env::var("IRIS_IP28_EXC_VADDR").ok().and_then(|v| { + let v = v.trim(); + if v.is_empty() { return None; } + u64::from_str_radix(v.trim_start_matches("0x"), 16).ok() + }) + }); + let Some(want) = want else { return }; + + // Keep the run-up. The exception that panics the guest is the second + // one: the first is an ordinary miss, and its handler then faults. + // Only the run-up says what the handler was handed, and BadVAddr has + // already been overwritten by the time the match fires. + static RING: std::sync::Mutex> = std::sync::Mutex::new(Vec::new()); + const KEEP: usize = 12; + let line = format!( + "code={:<2} pc={:#018x} bd={} badvaddr={:#018x} xcontext={:#018x} \ + context={:#018x} entryhi={:#018x} ra={:#018x}", + (status & CAUSE_EXCCODE_MASK) >> 2, + self.core.pc, + self.core.in_delay_slot as u8, + self.core.cp0_badvaddr, + self.core.cp0_xcontext, + self.core.cp0_context, + self.core.cp0_entryhi, + self.core.read_gpr(31), + ); + if let Ok(mut ring) = RING.lock() { + if ring.len() == KEEP { + ring.remove(0); + } + ring.push(line); + if want == self.core.cp0_badvaddr { + eprintln!("ip28exc: --- last {} exceptions, oldest first ---", ring.len()); + for (i, l) in ring.iter().enumerate() { + eprintln!("ip28exc: [{i}] {l}"); + } + } + } + + if want != self.core.cp0_badvaddr { + return; + } + const NAMES: [&str; 32] = [ + "zero", "at", "v0", "v1", "a0", "a1", "a2", "a3", + "t0", "t1", "t2", "t3", "t4", "t5", "t6", "t7", + "s0", "s1", "s2", "s3", "s4", "s5", "s6", "s7", + "t8", "t9", "k0", "k1", "gp", "sp", "fp", "ra", + ]; + let code = (status & CAUSE_EXCCODE_MASK) >> 2; + eprintln!( + "ip28exc: code={code} pc={:#018x} delay_slot={} badvaddr={:#018x} status={:#010x}", + self.core.pc, self.core.in_delay_slot, self.core.cp0_badvaddr, + self.core.cp0_status, + ); + for i in 0..32u32 { + if self.core.read_gpr(i) == self.core.cp0_badvaddr { + eprintln!("ip28exc: {} (${i}) holds the bad address", NAMES[i as usize]); + } + } + for chunk in (0..32u32).collect::>().chunks(4) { + let mut line = String::from("ip28exc: "); + for &i in chunk { + line.push_str(&format!(" {:>4}={:#018x}", NAMES[i as usize], self.core.read_gpr(i))); + } + eprintln!("{line}"); + } + } + fn handle_exception(&mut self, status: ExecStatus) -> ExecStatus { + self.ip28_trace_exception(status); // In developer builds, bus/address error exceptions break into the // monitor at the fault site rather than dispatching to the MIPS // vector — must be decided before deliver_exception runs, since that @@ -6663,7 +6802,19 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", let op = cache_op & 0x1C; // Determine if this is a Hit operation that needs address translation - let needs_translation = matches!(op, C_CDX | C_HINV | C_HWBINV | C_HWB | C_HSV); + // On an R10000 the encodings for 5/6/7 are index operations, not the + // R4000 hit operations — translating them would fault on an index that + // is not a valid virtual address. See C_R10K_ISD in mips_cache_v2. + let sel = cache_op & 3; + let r10k_index_op = C::R10K_CACHE_OPS + && match op { + C_R10K_CBARRIER => sel == CACH_PI, + C_R10K_ILD => matches!(sel, CACH_PI | CACH_PD | CACH_SD), + C_R10K_ISD => matches!(sel, CACH_SI | CACH_SD), + _ => false, + }; + let needs_translation = + !r10k_index_op && matches!(op, C_CDX | C_HINV | C_HWBINV | C_HWB | C_HSV); let phys_addr = if needs_translation { // Hit operations need address translation @@ -6677,19 +6828,73 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", // For Index_Store_Tag, pass TagLo via phys_addr let op = cache_op & 0x1C; - let phys_addr_or_taglo = if op == C_IST { - self.core.cp0_taglo as u64 + // A cache tag is TagHi:TagLo, not TagLo alone. The IP28 PROM writes + // the low 32 bits to $28 and the bits above to $29 — a 36-bit + // secondary tag arrives as TagLo=0xffffcdfe, TagHi=0xf — so a model + // that reads only TagLo sees a different tag from the one written. + // Harmless for R4400/R5000, whose tags fit in 32 bits and who leave + // TagHi zero: the shift-in contributes nothing there. + let stores_taglo = op == C_IST || (r10k_index_op && op == C_R10K_ISD); + let phys_addr_or_taglo = if stores_taglo { + ((self.core.cp0_taghi as u64) << 32) | (self.core.cp0_taglo as u32 as u64) } else { phys_addr }; + // On an R10000 the check bits travel with cache data through CP0 ECC. + if C::R10K_CACHE_OPS { + self.cache.set_cache_ecc(self.core.cp0_ecc); + } + // IP28 cache-error investigation (`log mips mask cp0`). Two kinds + // of op are worth seeing: one carrying non-zero check bits, which is a + // guest staging a parity error on purpose, and any op from outside the + // PROM — i.e. from a loaded diagnostic, which runs out of XKPHYS. + // + // Both halves of that gate were learned the hard way. Gating on ECC + // alone hid every op the IDE issued, because the IDE's ECC reads back + // zero: the exact symptom under investigation was also blinding the + // instrument to its cause. The cap then has to be generous, because a + // cap of 600 silently truncated a run at precisely the boundary and + // made a partial picture look like the whole one. It exists only so a + // hot loop cannot rewrite the timing it is measuring. + if crate::mips_core::cachediag_on() { + let from_prom = (self.core.pc >> 32) == 0xFFFF_FFFF; + // CBARRIER is pure ordering: it names no line and carries no check + // bits, and it outnumbers everything else ~8:1 (83534 of 94322 in + // one IDE run). Tracing it swamped the log and slowed the guest + // enough that the run no longer reached the loop being studied — + // so drop it unless it is carrying staged check bits. + let noise = op == C_R10K_CBARRIER && self.core.cp0_ecc == 0; + if (self.core.cp0_ecc != 0 || !from_prom) && !noise { + const CACHE_TRACE_CAP: u32 = 200_000; + static SEEN: std::sync::atomic::AtomicU32 = + std::sync::atomic::AtomicU32::new(0); + let n = SEEN.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + if n < CACHE_TRACE_CAP { + crate::dlog!(LogModule::Mips, + "ip28cd: CACHE op={:#04x} sel={} vaddr={:#018x} ECC={:#010x} \ + TagHi:Lo={:#010x}:{:#018x} pc={:#018x}", + op, sel, virt_addr, self.core.cp0_ecc, + self.core.cp0_taghi, self.core.cp0_taglo, self.core.pc); + } else if n == CACHE_TRACE_CAP { + crate::dlog!(LogModule::Mips, + "ip28cd: CACHE trace capped at {CACHE_TRACE_CAP} ops \ + — output past this point is incomplete"); + } + } + } // Call unified cache interface let result = self.cache.cache_op(cache_op, virt_addr, phys_addr_or_taglo); + if C::R10K_CACHE_OPS && op == C_R10K_ILD { + self.core.cp0_ecc = self.cache.cache_op_ecc(); + } // For Index_Load_Tag, update CP0 TagLo from result - if op == C_ILT { - self.core.cp0_taglo = result; - self.core.cp0_taghi = 0; + if op == C_ILT || (r10k_index_op && op == C_R10K_ILD) { + // Split back the way it arrived. Zeroing TagHi unconditionally + // threw away the top of every tag wider than 32 bits. + self.core.cp0_taglo = result & 0xFFFF_FFFF; + self.core.cp0_taghi = (result >> 32) as u32; } self.handle_exec_complete() @@ -6917,6 +7122,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", let rt_reg = d.rt as u32; let rd_val = d.rd as u32; let value = self.core.read_cp0(rd_val); + self.ip28_cp0_trace("mfc0", rd_val, value); // Sign-extend 32-bit value to 64 bits self.core.write_gpr(rt_reg, value as u32 as i32 as i64 as u64); self.handle_exec_complete() @@ -6927,6 +7133,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", let rt_reg = d.rt as u32; let rd_val = d.rd as u32; let value = self.core.read_cp0(rd_val); + self.ip28_cp0_trace("dmfc0", rd_val, value); self.core.write_gpr(rt_reg, value); self.handle_exec_complete() } @@ -6935,13 +7142,16 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", /// The CP0 registers that are 64 bits wide. /// /// EntryLo0/1, Context, BadVAddr, EntryHi, EPC, XContext and ErrorEPC are - /// 64-bit in MIPS III and stay so in MIPS IV. + /// 64-bit in MIPS III and stay so in MIPS IV. TagLo is 32-bit on R4x00 but + /// 64-bit on the R10000, which is why it is asked of the model rather than + /// listed flat. fn cp0_is_64bit(reg: u32) -> bool { - matches!(reg, 2 | 3 | 4 | 8 | 10 | 14 | 20 | 30) + matches!(reg, 2 | 3 | 4 | 8 | 10 | 14 | 20 | 30) || (C::R10K_CACHE_OPS && reg == 28) } fn exec_mtc0(&mut self, d: &DecodedInstr) -> ExecStatus { let rt_val = self.core.read_gpr(d.rt as u32); + self.ip28_cp0_trace("mtc0", d.rd as u32, rt_val); let rd_val = d.rd as u32; // MTC0 moves the whole register. DMTC0 differs in what the @@ -6976,8 +7186,22 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", } // DMTC0 - Doubleword Move To CP0 (MIPS III) + /// IP28 bring-up: watch the cache-tag CP0 registers (26 ECC, 28 TagLo, + /// 29 TagHi) under `log mips mask cp0`. Only the R10000 model compiles + /// this in. + /// + /// The env-var form this replaced did an uncached `getenv` on every + /// DMTC0/MTC0, armed or not. + #[inline(always)] + fn ip28_cp0_trace(&self, what: &str, reg: u32, val: u64) { + if C::R10K_CACHE_OPS && matches!(reg, 26 | 28 | 29) && mips_log_always(MIPS_LOG_CP0) { + crate::dlog!(LogModule::Mips, "ip28cp0: {what} ${reg} = {val:#018x}"); + } + } + fn exec_dmtc0(&mut self, d: &DecodedInstr) -> ExecStatus { let rt_val = self.core.read_gpr(d.rt as u32); + self.ip28_cp0_trace("dmtc0", d.rd as u32, rt_val); let rd_val = d.rd as u32; self.core.write_cp0(rd_val, rt_val); self.handle_cp0_side_effects(rd_val); @@ -7143,6 +7367,31 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", // TLBWI - Write Indexed TLB Entry // Writes CP0.EntryHi, CP0.EntryLo0, CP0.EntryLo1, and CP0.PageMask to the TLB entry indexed by CP0.Index + /// Report every TLB write, under `log mips mask tlb`. + /// + /// `IRIS_IP28_TLBW=wired` additionally narrows it to writes below CP0 + /// Wired — the mappings a kernel means never to be evicted. That stays an + /// environment variable because it selects a *subset*, which the module + /// mask has no way to express; the on/off is devlog's. + #[inline] + fn ip28_trace_tlb_write(&self, op: &str, index: usize, entry: &crate::mips_tlb::TlbEntry) { + if !mips_log_always(MIPS_LOG_TLB) { + return; + } + static WIRED_ONLY: std::sync::OnceLock = std::sync::OnceLock::new(); + let wired_only = *WIRED_ONLY.get_or_init(|| { + matches!(std::env::var("IRIS_IP28_TLBW").as_deref(), Ok("wired")) + }); + if wired_only && index >= self.core.cp0_wired as usize { + return; + } + crate::dlog!(LogModule::Mips, + "ip28tlbw: {op} idx={index:<2} wired={} entryhi={:#018x} lo0={:#018x} lo1={:#018x} \ + mask={:#x} pc={:#018x}", + self.core.cp0_wired, entry.entry_hi, entry.entry_lo[0], entry.entry_lo[1], + entry.page_mask, self.core.pc); + } + fn exec_tlbwi(&mut self) -> ExecStatus { // The slot is Index[5:0]. Masking rather than `%` matters twice: // @@ -7168,7 +7417,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", return self.handle_exec_complete(); } let entry = self.create_tlb_entry_from_cp0(); - //eprintln!("TLBWI idx={} entryhi={:#018x} lo0={:#018x} lo1={:#018x} pc={:#018x}", index, entry.entry_hi, entry.entry_lo[0], entry.entry_lo[1], self.core.pc); + self.ip28_trace_tlb_write("tlbwi", index, &entry); self.tlb.write(index, entry); // Flushes the nutlb too — the TLB it caches just changed. self.nanotlb_invalidate(); @@ -7192,6 +7441,7 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", self.core.update_random(); let index = (self.core.cp0_random as usize) % self.tlb.num_entries(); let entry = self.create_tlb_entry_from_cp0(); + self.ip28_trace_tlb_write("tlbwr", index, &entry); self.tlb.write(index, entry); self.nanotlb_invalidate(); @@ -7361,6 +7611,15 @@ va={:#018x} phys={:#010x} (code pfn {:#x}, page {:#010x}, word {}/{})", // *guarantees* the next access misses and therefore calls `translate_fn`. self.resync_privilege_state(); + // `log mips mask cp0`. If the target's top half is already gone by + // the time we get here, the guest wrote a truncated EPC; if it is + // intact and the next fetch is truncated anyway, the loss is + // downstream of this. + if C::R10K_CACHE_OPS && mips_log_always(MIPS_LOG_CP0) { + crate::dlog!(LogModule::Mips, + "ip28epc: eret -> {target:#018x} (epc={:#018x} errorepc={:#018x} status={:#010x})", + self.core.cp0_epc, self.core.cp0_errorepc, self.core.cp0_status); + } // ERET jumps immediately without delay slot self.core.pc = target; @@ -14143,7 +14402,7 @@ impl Saveable for MipsCpu cp0u32!(cp0_status); cp0u32!(cp0_cause); cp0u32!(cp0_prid); cp0u32!(cp0_config); cp0u32!(cp0_lladdr); cp0u32!(cp0_watchlo); cp0u32!(cp0_watchhi); cp0u32!(cp0_ecc); cp0u32!(cp0_cacheerr); - cp0u32!(cp0_taglo); cp0u32!(cp0_taghi); + cp0u64!(cp0_taglo); cp0u32!(cp0_taghi); cp0u64!(cp0_badvaddr); cp0u64!(cp0_epc); cp0u64!(cp0_errorepc); cp0u64!(cp0_entrylo0); cp0u64!(cp0_entrylo1); cp0u64!(cp0_context); cp0u64!(cp0_pagemask); cp0u64!(cp0_entryhi); cp0u64!(cp0_xcontext); @@ -14208,7 +14467,7 @@ impl Saveable for MipsCpu c.count_hz_atomic.store(c.count_hz, std::sync::atomic::Ordering::Relaxed); ld32!(cp0_status); ld32!(cp0_cause); ld32!(cp0_prid); ld32!(cp0_config); ld32!(cp0_lladdr); ld32!(cp0_watchlo); ld32!(cp0_watchhi); - ld32!(cp0_ecc); ld32!(cp0_cacheerr); ld32!(cp0_taglo); ld32!(cp0_taghi); + ld32!(cp0_ecc); ld32!(cp0_cacheerr); ld64!(cp0_taglo); ld32!(cp0_taghi); ld64!(cp0_entrylo0); ld64!(cp0_entrylo1); ld64!(cp0_context); ld64!(cp0_pagemask); ld64!(cp0_badvaddr); ld64!(cp0_entryhi); ld64!(cp0_xcontext); ld64!(cp0_epc); ld64!(cp0_errorepc); @@ -14657,6 +14916,7 @@ mod xcontext_layout_tests { } } + #[cfg(test)] mod round_to_int_mode_tests { use super::*; diff --git a/src/mips_tlb.rs b/src/mips_tlb.rs index d52861ec..1c387bd7 100644 --- a/src/mips_tlb.rs +++ b/src/mips_tlb.rs @@ -4,9 +4,21 @@ use crate::mips_exec::CacheAttr; use std::fmt::Write; use crate::snapshot::{u64_slice_to_toml, load_u64_slice}; -/// Number of TLB entries in R4000 (48 dual-entries = 96 pages) +/// Default JTLB size: the R4000/R4400/R5000 count (48 dual-entries = 96 pages). +/// +/// The *live* size is per-CPU-model and lives in `MipsTlb::num_entries`; this +/// is only the default for `Default::default()` and for callers that do not +/// name a model. pub const TLB_NUM_ENTRIES: usize = 48; +/// Array capacity: the largest JTLB any modelled CPU has. +/// +/// The R10000 has 64 entries where the R4400 has 48. Sizing the arrays to the +/// maximum and carrying the live count separately keeps one concrete `MipsTlb` +/// type — the lookup walks an MRU list that only ever contains live slots, so +/// the unused tail costs nothing on the hot path. +pub const TLB_MAX_ENTRIES: usize = 64; + // ── TLB statistics (feature = "tlbstats") ──────────────────────────────────── #[cfg(feature = "tlbstats")] @@ -508,12 +520,13 @@ impl ShadowEntry { /// Real R4000 TLB implementation /// -/// Implements a fully associative JTLB (Joint TLB) with 48 dual-entries. +/// Implements a fully associative JTLB (Joint TLB); `num_entries` dual-entries, +/// 48 on R4x00/R5000 and 64 on the R10000. /// /// **32-bit mode (and 64-bit sign-extended ±2GB) fast path**: a 512KB `vmap` /// array indexed by VA[31:13] gives O(1) lookup. Each slot holds the TLB /// entry index (0-47) or VMAP_MISS. After the index is found we still verify -/// ASID/Global and the valid/dirty bits — but the linear scan over 48 entries +/// ASID/Global and the valid/dirty bits — but the linear scan over the entries /// is eliminated. /// /// For 64-bit VAs that are sign-extended 32-bit values (upper 32 bits all-zero @@ -528,14 +541,17 @@ impl ShadowEntry { #[derive(Clone)] pub struct MipsTlb { /// Architectural TLB entries (read/written by TLBR/TLBWI/TLBWR/TLBP). - entries: [TlbEntry; TLB_NUM_ENTRIES], + entries: [TlbEntry; TLB_MAX_ENTRIES], /// Cache-friendly shadow used by translate() and probe(). Kept in sync /// with `entries` — rebuilt whenever an entry is written. - shadow: [ShadowEntry; TLB_NUM_ENTRIES], + shadow: [ShadowEntry; TLB_MAX_ENTRIES], /// Head of each MRU list (slot index, or MRU_NONE). mru_head: [u8; MRU_LISTS], /// `mru_next[list][slot]` — next slot in that list, or MRU_NONE. - mru_next: [[u8; TLB_NUM_ENTRIES]; MRU_LISTS], + mru_next: [[u8; TLB_MAX_ENTRIES]; MRU_LISTS], + + /// Live JTLB entries for the CPU model in use (<= TLB_MAX_ENTRIES). + num_entries: usize, /// O(1) lookup for 32-bit (and sign-extended 64-bit) VAs. /// Indexed by VA[31:13] (19 bits). Value = entry index or VMAP_MISS. vmap: [u8; VMAP_SIZE], @@ -555,13 +571,14 @@ pub struct MipsTlb { impl MipsTlb { pub fn new(num_entries: usize) -> Self { - assert_eq!(num_entries, TLB_NUM_ENTRIES, - "MipsTlb currently requires exactly {} entries", TLB_NUM_ENTRIES); + assert!(num_entries > 1 && num_entries <= TLB_MAX_ENTRIES, + "MipsTlb supports 2..={} entries, got {}", TLB_MAX_ENTRIES, num_entries); let mut tlb = Self { - entries: [TlbEntry::new(); TLB_NUM_ENTRIES], - shadow: [ShadowEntry::invalid(); TLB_NUM_ENTRIES], + entries: [TlbEntry::new(); TLB_MAX_ENTRIES], + shadow: [ShadowEntry::invalid(); TLB_MAX_ENTRIES], mru_head: [0u8; MRU_LISTS], - mru_next: [[MRU_NONE; TLB_NUM_ENTRIES]; MRU_LISTS], + mru_next: [[MRU_NONE; TLB_MAX_ENTRIES]; MRU_LISTS], + num_entries, vmap: [VMAP_MISS; VMAP_SIZE], #[cfg(feature = "tlbcheck")] vmap_touched: std::collections::HashSet::new(), @@ -570,10 +587,12 @@ impl MipsTlb { }; for list in 0..MRU_LISTS { tlb.mru_head[list] = 0; - for i in 0..TLB_NUM_ENTRIES - 1 { + for i in 0..num_entries - 1 { tlb.mru_next[list][i] = (i + 1) as u8; } - // slot 47 already MRU_NONE from array initialisation + // The last live slot is already MRU_NONE from array + // initialisation, and so is every slot past it — the unused tail + // of a 48-entry model never joins a list, so lookups never see it. } tlb } @@ -667,7 +686,7 @@ impl MipsTlb { let (old_start, old_count) = old_range; self.vmap_clear(old_start, old_count); let old_end = old_start + old_count; - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { if i == index { continue; } @@ -707,12 +726,12 @@ impl MipsTlb { // Two entries conflict if their VPN2 ranges intersect (after masking // by the wider of the two page masks) and they'd both be visible to // the same lookup: either entry is global, or both share an ASID. - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let a = &self.entries[i]; if !a.is_valid_even() && !a.is_valid_odd() { continue; // fully-invalid entries can't conflict } - for j in (i + 1)..TLB_NUM_ENTRIES { + for j in (i + 1)..self.num_entries { let b = &self.entries[j]; if !b.is_valid_even() && !b.is_valid_odd() { continue; @@ -741,7 +760,7 @@ impl MipsTlb { // (an all-zero architectural entry) derives 0 for the same fields. // Both are correctly unmatchable — the bit patterns just differ by // construction — so comparing them here would be a false positive. - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let expected = ShadowEntry::from_entry(&self.entries[i]); let s = &self.shadow[i]; if s.valid_dirty != expected.valid_dirty || s.asid != expected.asid @@ -766,12 +785,12 @@ impl MipsTlb { // 3. MRU list integrity: each list must visit every slot exactly once. for list in 0..MRU_LISTS { - let mut seen = [false; TLB_NUM_ENTRIES]; + let mut seen = [false; TLB_MAX_ENTRIES]; let mut cur = self.mru_head[list]; let mut count = 0usize; while cur != MRU_NONE { let slot = cur as usize; - if slot >= TLB_NUM_ENTRIES { + if slot >= self.num_entries { report(format!("MRU list {} contains out-of-range slot {}", list, slot)); break; } @@ -783,9 +802,9 @@ impl MipsTlb { count += 1; cur = self.mru_next[list][slot]; } - let missing: Vec = (0..TLB_NUM_ENTRIES).filter(|&i| !seen[i]).collect(); + let missing: Vec = (0..self.num_entries).filter(|&i| !seen[i]).collect(); if !missing.is_empty() { - report(format!("MRU list {} is missing slot(s) {:?} (visited {} of {})", list, missing, count, TLB_NUM_ENTRIES)); + report(format!("MRU list {} is missing slot(s) {:?} (visited {} of {})", list, missing, count, self.num_entries)); } } @@ -817,7 +836,7 @@ impl MipsTlb { { use std::collections::HashMap; let mut owners: HashMap = HashMap::new(); // vpn2 -> (most_recent_entry_idx, multiple) - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let (start, count) = Self::vmap_range(&self.entries[i]); for k in 0..count { let vpn2 = start.wrapping_add(k); @@ -882,7 +901,7 @@ impl MipsTlb { eprintln!("TLBCHECK VIOLATION [{}]: {}", context, msg); } eprintln!("TLBCHECK: dumping full TLB state for [{}]:", context); - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { eprintln!("{}", self.format_entry(i)); } true @@ -1055,7 +1074,7 @@ impl Tlb for MipsTlb { } fn write(&mut self, index: usize, mut entry: TlbEntry) { - if index < self.entries.len() { + if index < self.num_entries { let mask = entry.page_mask | 0x1FFF; entry.selector_bit_shift = (mask.trailing_ones() - 1) as u8; entry.vcmp32 = !mask & 0x0000_0000_FFFF_E000; @@ -1076,7 +1095,7 @@ impl Tlb for MipsTlb { } fn read(&self, index: usize) -> TlbEntry { - if index < self.entries.len() { + if index < self.num_entries { self.entries[index] } else { TlbEntry::new() @@ -1117,7 +1136,7 @@ impl Tlb for MipsTlb { } fn format_entry(&self, index: usize) -> String { - if index >= self.entries.len() { + if index >= self.num_entries { return format!("Index {} out of bounds", index); } let e = &self.entries[index]; @@ -1188,14 +1207,14 @@ impl Tlb for MipsTlb { } fn power_on(&mut self) { - self.entries = [TlbEntry::new(); TLB_NUM_ENTRIES]; - self.shadow = [ShadowEntry::invalid(); TLB_NUM_ENTRIES]; + self.entries = [TlbEntry::new(); TLB_MAX_ENTRIES]; + self.shadow = [ShadowEntry::invalid(); TLB_MAX_ENTRIES]; for list in 0..MRU_LISTS { self.mru_head[list] = 0; - for i in 0..TLB_NUM_ENTRIES - 1 { + for i in 0..self.num_entries - 1 { self.mru_next[list][i] = (i + 1) as u8; } - self.mru_next[list][TLB_NUM_ENTRIES - 1] = MRU_NONE; + self.mru_next[list][self.num_entries - 1] = MRU_NONE; } self.vmap.fill(VMAP_MISS); #[cfg(feature = "tlbcheck")] @@ -1204,7 +1223,9 @@ impl Tlb for MipsTlb { fn save_state(&self) -> toml::Value { // Each entry stored as [page_mask, entry_hi, entry_lo0, entry_lo1] - let arr: Vec = self.entries.iter().map(|e| { + // Only the live entries: a 48-entry model must still produce a + // 48-entry snapshot even though the array has room for 64. + let arr: Vec = self.entries[..self.num_entries].iter().map(|e| { let words = [e.page_mask, e.entry_hi, e.entry_lo[0], e.entry_lo[1]]; u64_slice_to_toml(&words) }).collect(); @@ -1214,7 +1235,7 @@ impl Tlb for MipsTlb { fn load_state(&mut self, v: &toml::Value) -> Result<(), String> { if let toml::Value::Array(arr) = v { for (i, item) in arr.iter().enumerate() { - if i >= TLB_NUM_ENTRIES { break; } + if i >= self.num_entries { break; } let mut words = [0u64; 4]; load_u64_slice(item, &mut words); let page_mask = words[0]; @@ -1240,19 +1261,19 @@ impl Tlb for MipsTlb { }; } } - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { self.shadow[i] = ShadowEntry::from_entry(&self.entries[i]); } // Reset MRU lists to canonical order so snapshot restores are deterministic. for list in 0..MRU_LISTS { self.mru_head[list] = 0; - for i in 0..TLB_NUM_ENTRIES - 1 { + for i in 0..self.num_entries - 1 { self.mru_next[list][i] = (i + 1) as u8; } - self.mru_next[list][TLB_NUM_ENTRIES - 1] = MRU_NONE; + self.mru_next[list][self.num_entries - 1] = MRU_NONE; } self.vmap.fill(VMAP_MISS); - for i in 0..TLB_NUM_ENTRIES { + for i in 0..self.num_entries { let (start, count) = Self::vmap_range(&self.entries[i]); self.vmap_fill_range(i, start, count); } diff --git a/src/mips_tlb_test.rs b/src/mips_tlb_test.rs index 57077e6d..c5a82467 100644 --- a/src/mips_tlb_test.rs +++ b/src/mips_tlb_test.rs @@ -440,4 +440,57 @@ mod tests { .join() .expect("thread panicked"); } + +/// The JTLB is as big as the CPU model says, and no bigger. +/// +/// The R10000 has 64 entries where the R4400 has 48. The count used to be a +/// compile-time 48 for every model while `core.tlb_entries` was taken from the +/// model, so on an R10000 the two disagreed: Random cycled over a range the +/// array did not have, and a TLBWI to index 48..63 went nowhere. SGI's IP28 +/// diagnostic writes index 48 on its first cache-alias test, and every failure +/// it reported was that write being dropped. +#[test] +fn tlb_size_follows_the_cpu_model() { + #[cfg(feature = "ip28")] + use crate::mips_cache_shadow::R10000ShadowCache; + use crate::mips_cache_v2::{R4400Cache, R5000Cache}; + use crate::mips_exec::MipsCpuConfig; + + assert_eq!(MipsCpuConfig::for_model::().tlb_entries, 48); + assert_eq!(MipsCpuConfig::for_model::().tlb_entries, 48); + #[cfg(feature = "ip28")] + assert_eq!(MipsCpuConfig::for_model::().tlb_entries, 64); +} + +/// An index past the live count is dropped; one inside it is kept. +#[test] +fn tlb_write_past_the_live_entry_count_is_dropped() { + let mut entry = TlbEntry::new(); + entry.entry_hi = 0xC000_0000_0004_0000; + + let mut r4400 = MipsTlb::new(48); + r4400.write(48, entry); + assert_eq!(r4400.read(48).entry_hi, 0, "48-entry TLB must drop index 48"); +} + +/// The other half of the pair, deliberately in its own `#[test]`. +/// +/// `MipsTlb` carries its arrays inline and is large enough that **two** of +/// them alive at once overflows a test thread's stack in a debug build — the +/// single test these two replace aborted the whole binary with SIGABRT. It +/// went unnoticed because every test run in this project passes `--release`, +/// where the temporaries collapse; `cargo test` on its own is what an +/// upstream reviewer would run. One TLB per test, as every neighbouring test +/// already does. +#[test] +fn tlb_write_inside_a_larger_entry_count_is_kept() { + let mut entry = TlbEntry::new(); + entry.entry_hi = 0xC000_0000_0004_0000; + + let mut r10000 = MipsTlb::new(64); + r10000.write(48, entry); + assert_eq!(r10000.read(48).entry_hi, entry.entry_hi, "64-entry TLB must keep it"); + r10000.write(63, entry); + assert_eq!(r10000.read(63).entry_hi, entry.entry_hi, "...and its last slot"); +} } diff --git a/src/physical.rs b/src/physical.rs index b5f90d2b..b27bd5e8 100644 --- a/src/physical.rs +++ b/src/physical.rs @@ -211,7 +211,24 @@ const PROM_END: u32 = 0x1FD00000; // accesses go through the normal lomem device_map entries — no direct bank pointer needed. const ALIAS_BASE: u32 = 0x00000000; const ALIAS_END: u32 = 0x00080000; -const ALIAS_OFFSET: u32 = LOMEM_BASE; + +/// Where the 512 KB alias at physical 0 points. +/// +/// The MC mirrors the bottom 512 KB of *memory*, and which physical address +/// that is depends on the machine: LOMEM_BASE on IP22/IP24, 0x20000000 on +/// IP28, whose RAM starts there and has nothing at lomem at all. With the +/// offset fixed at LOMEM_BASE the whole window read back as zero on IP28. +/// +/// That window is not spare space. ARCS builds its system parameter block at +/// physical 0x1000 and its firmware vector table at 0x1800, and a 64-bit sash +/// loads its firmware pointer straight out of 0x1018 — so the PROM's writes +/// were being discarded and sash then dereferenced the null it read back. +/// +/// Taken from the machine profile, never from the environment. +fn alias_offset_for(ip28: bool) -> u32 { + if ip28 { HIMEM_BASE } else { LOMEM_BASE } +} + /// What one 64 KB `device_map` slot should point at after a MEMCFG write. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -308,6 +325,10 @@ pub struct Physical { /// `u64` before that — never null, so the test needs no guard. #[cfg(feature = "ppmem")] ppmem_bitmap: *const u64, + /// IP28: the bank placement ppmem's window was last built for, so a + /// MEMCFG write that moves nothing leaves the window alone. + #[cfg(all(feature = "ppmem", feature = "ip28"))] + ppmem_last_placement: Option<([Option<(u32, u32, u32)>; 4], [usize; 4])>, pub rex3: Option>, /// Second Newport head (dual-head Indigo2 / `graphics.heads = 2`). @@ -343,6 +364,10 @@ pub struct Physical { /// two, and a bank parked out there has to be unmapped again when it /// moves — see `plan_bank_slots`. banks_outside_windows: Vec, + /// Where the 512 KB alias at physical 0 points — see `alias_offset_for`. + alias_offset: u32, + /// This machine is an IP28. Only used to label the bank-map trace. + is_ip28: bool, trace: AtomicBool, start_tick: u64, @@ -399,6 +424,8 @@ impl Physical { mc: MemoryController, hpc3: Hpc3, prom: PromPort, + // IP28: RAM starts at 0x20000000, so the low-memory alias follows it. + ip28: bool, ) -> Self { let host_freq = crate::platform::get_host_tick_frequency(); let start_tick = crate::platform::get_host_ticks(); @@ -409,7 +436,7 @@ impl Physical { let gio_bus_error = GioBusErrorDevice { mc: mc.clone() }; // Alias targets will be set in build_device_map once Physical is in final location let unmapped_ram = UnmappedRam; - let alias_bus = AliasBus::new(std::ptr::null::(), ALIAS_OFFSET); + let alias_bus = AliasBus::new(std::ptr::null::(), alias_offset_for(ip28)); // VINO GIO alias: 0x1F08xxxx → 0x0008xxxx (subtract 0x1F000000 = add 0xFF000000) // GIO64 VINO aperture sits at 0x1F080000; the chip's primary registers // live at physical 0x00080000 (VINO_BASE). To map 0x1F080000 → 0x00080000 @@ -467,6 +494,8 @@ impl Physical { ppmem_gen_base, #[cfg(feature = "ppmem")] ppmem_bitmap, + #[cfg(all(feature = "ppmem", feature = "ip28"))] + ppmem_last_placement: None, rex3, rex3_head1, gr2, @@ -487,6 +516,8 @@ impl Physical { black_hole, device_map, banks_outside_windows: Vec::new(), + alias_offset: alias_offset_for(ip28), + is_ip28: ip28, trace: AtomicBool::new(false), start_tick, host_freq, @@ -571,16 +602,11 @@ impl Physical { self.device_map[i as usize] = gr2_ptr; } } else if let Some(mgras_ptr) = mgras_ptr { - // Indigo2 IMPACT preview: MGRAS stub spans gfx + populated expansion slots. + // IMPACT in the graphics slot. The expansion slots stay unmapped so + // their probes bus-error, as empty slots do. for i in (NEWPORT_BASE >> 16)..((NEWPORT_END - 1) >> 16) + 1 { self.device_map[i as usize] = mgras_ptr; } - for i in (GIO_SLOT0_BASE >> 16)..((GIO_SLOT0_END - 1) >> 16) + 1 { - self.device_map[i as usize] = mgras_ptr; - } - for i in (GIO_SLOT1_BASE >> 16)..((GIO_SLOT1_END - 1) >> 16) + 1 { - self.device_map[i as usize] = mgras_ptr; - } } // else: GIO timeout from layer 2 already covers the Newport slot @@ -631,7 +657,7 @@ impl Physical { self.device_map[i as usize] = prom_ptr; } - // Alias: points back into Physical itself with ALIAS_OFFSET added. + // Alias: points back into Physical itself with `alias_offset()` added. // So alias accesses go: AliasBus → Physical::read/write(addr + LOMEM_BASE) // → device_map lookup → whichever bank is mapped at LOMEM_BASE. // This way alias automatically tracks whatever MEMCFG maps at LOMEM_BASE. @@ -647,6 +673,7 @@ impl Physical { self.vino_gio_alias.target = self as *const Physical as *const dyn BusDevice; let vino_gio_alias_ptr: *const dyn BusDevice = &self.vino_gio_alias; self.device_map[(0x1F080000u32 >> 16) as usize] = vino_gio_alias_ptr; + } /// Remap memory banks in device_map. @@ -669,12 +696,38 @@ impl Physical { &self.banks[3], ]; - // ppmem: drop every mapping before re-placing the banks. Safe to leave - // the window briefly unmapped — remapping runs inside the CPU's MEMCFG - // store during PROM POST, before DMA is running (design doc §5.1). + // IP28: a MEMCFG write that leaves every bank where it was (IRIX's + // kernel rewrites MEMCFG1 during boot to fix the refresh bits) must + // not tear down and rebuild ppmem's window. JIT compile workers and + // the DMA thread are running by then, and the rebuild is exactly the + // window in which they used to fault. + #[cfg(all(feature = "ppmem", feature = "ip28"))] + let ppmem_placement_changed = { + let sizes = [0, 1, 2, 3].map(|i| self.banks[i].size()); + let key = (bank_addrs, sizes); + let changed = self.ppmem_last_placement != Some(key); + self.ppmem_last_placement = Some(key); + changed + }; + #[cfg(all(feature = "ppmem", not(feature = "ip28")))] + let ppmem_placement_changed = true; + + // ppmem: drop every mapping before re-placing the banks. The comment + // this replaced said the window is safe to leave unmapped because + // remapping only runs during PROM POST, before DMA; that is not true + // on IP28 (see above), which is why ppmem scrubs rather than unmaps (see `AddrSpace::scrub`). #[cfg(feature = "ppmem")] - if let Some(sp) = &self.ppmem_space { - sp.clear_mappings(); + if ppmem_placement_changed { + if let Some(sp) = &self.ppmem_space { + sp.clear_mappings(); + } + // A bank now answering at an address another bank answered at + // must look changed to the JIT even if the two counters happen to + // be equal, so move every counter. + #[cfg(all(feature = "ip28", feature = "jitv2"))] + for b in &self.banks { + b.bump_gen_all(); + } } for (bank_idx, maybe_bank) in bank_addrs.iter().enumerate() { @@ -683,6 +736,9 @@ impl Physical { continue; }; + if self.is_ip28 { + eprintln!("iris: IP28 experiment: bank {bank_idx} -> base {conf_base:#010x} mask {addr_mask:#010x} limit {limit:#010x}"); + } dlog_dev!(LogModule::Mc, "[MEMCFG] bank {} mapped at 0x{:08x}..0x{:08x} addr_mask={:08x} limit={:08x} ({}MB visible, {}MB per rank)", bank_idx, conf_base, conf_base + limit, addr_mask, limit, limit >> 20, (addr_mask + 1) >> 20); @@ -693,7 +749,7 @@ impl Physical { // undersized bank repeats to fill `limit`, which is exactly the // SIMM mirroring `addr_mask` encodes — see docs/ppmem-design.md §5. #[cfg(feature = "ppmem")] - if let Some(sp) = &self.ppmem_space { + if let (true, Some(sp)) = (ppmem_placement_changed, &self.ppmem_space) { // `addr_mask + 1` is the SIMM's mirror period, and `limit` the // configured slot. They are independent: a dual-rank SIMM has a // slot half the size of the bank (each rank placed separately), @@ -740,8 +796,8 @@ impl Physical { // re-dispatch. AliasBus stays installed in device_map as the bus-path // equivalent; both see identical memory. #[cfg(feature = "ppmem")] - if let Some(sp) = &self.ppmem_space { - let bank0_mapped = bank_addrs[0].is_some_and(|(base, _, _)| base == LOMEM_BASE); + if let (true, Some(sp)) = (ppmem_placement_changed, &self.ppmem_space) { + let bank0_mapped = bank_addrs[0].is_some_and(|(base, _, _)| base == self.alias_offset); if bank0_mapped { let alias_len = (ALIAS_END - ALIAS_BASE) as u64; if (self.banks[0].size() as u64) >= alias_len { diff --git a/src/platform_profile_tests.rs b/src/platform_profile_tests.rs index bb3014a4..03e4eb16 100644 --- a/src/platform_profile_tests.rs +++ b/src/platform_profile_tests.rs @@ -15,7 +15,7 @@ mod tests { }; use crate::eeprom_93c56::Eeprom93c56; use crate::ioc::{Ioc, IOC_BASE, IOC_SYS_ID, l1_regs, IOC_INT3_L1_STAT}; - use crate::mgras::{self, reg as mgras_reg, Mgras, MGRAS_REG_OFF, MGRAS_SLOT_GFX_BASE}; + use crate::mgras::{Mgras, GIO_ID, MGRAS_SLOT_GFX_BASE}; use crate::traits::{BusDevice, Saveable}; use crate::dev::gr2::{Gr2, Gr2Stats, Gr2Variant, GR2_BASE}; @@ -185,35 +185,12 @@ mod tests { } #[test] - fn mgras_solid_impact_board_id() { - let cfg = ImpactSection { - gfx: ImpactSlot::Solid, - exp0: ImpactSlot::None, - exp1: ImpactSlot::None, - }; - let m = Mgras::new(&cfg); - let addr = MGRAS_SLOT_GFX_BASE + MGRAS_REG_OFF + mgras_reg::BOARD_ID; - let id = m.read32(addr).data; - assert_eq!(id, mgras_reg::board_id_for(ImpactSlot::Solid)); - } - - #[test] - fn mgras_maximum_three_slot_map() { - let cfg = ImpactSection { - gfx: ImpactSlot::Solid, - exp0: ImpactSlot::High, - exp1: ImpactSlot::Max, - }; - let m = Mgras::new(&cfg); - assert!(m.any_slot()); - for (slot, kind) in [ - (mgras::MGRAS_SLOT_GFX_BASE, ImpactSlot::Solid), - (mgras::MGRAS_SLOT_EXP0_BASE, ImpactSlot::High), - (mgras::MGRAS_SLOT_EXP1_BASE, ImpactSlot::Max), - ] { - let addr = slot + MGRAS_REG_OFF + mgras_reg::BOARD_ID; - assert_eq!(m.read32(addr).data, mgras_reg::board_id_for(kind)); - } + fn mgras_answers_the_gio_id_probe() { + let cfg = ImpactSection { gfx: ImpactSlot::Solid, exp0: ImpactSlot::None, exp1: ImpactSlot::None }; + let hb = Arc::new(std::sync::atomic::AtomicU64::new(0)); + let ioc = crate::ioc::Ioc::new(false); + let m = Mgras::new(&cfg, ioc, hb.clone(), hb); + assert_eq!(m.read32(MGRAS_SLOT_GFX_BASE).data, GIO_ID); } #[test] diff --git a/src/ppmem/map_unix.rs b/src/ppmem/map_unix.rs index 98076dd7..2e3e380d 100644 --- a/src/ppmem/map_unix.rs +++ b/src/ppmem/map_unix.rs @@ -276,6 +276,45 @@ impl AddrSpace { } } +impl AddrSpace { + /// Replace `[at, at+len)` with private, zero-filled, read-write memory, + /// **retaining the claim** like `unmap` does. + /// + /// For ranges something may still point into. ppmem's generation window + /// is read by JIT compile workers through pointers taken while a bank was + /// mapped, and the data window by the DMA thread after a bitmap check; a + /// remap runs on the CPU thread meanwhile. With `PROT_NONE` there, a + /// reader in that window took the whole emulator down (SIGBUS in + /// `PhysicalCodePage::current_gen`, twice on IP28, where IRIX's kernel + /// rewrites MEMCFG1 during boot). Here it reads zeros, which the + /// generation check sees as a change and the bitmap never lets reach + /// guest-visible state. + /// + /// # Safety + /// + /// As `map`: whatever was mapped in the range is gone. + pub unsafe fn scrub(&self, at: usize, len: usize) -> io::Result<()> { + self.check_range(at, len); + let want = self.base.add(at); + let got = libc::mmap( + want as *mut libc::c_void, + len, + libc::PROT_READ | libc::PROT_WRITE, + libc::MAP_PRIVATE | libc::MAP_ANONYMOUS | libc::MAP_FIXED | libc::MAP_NORESERVE, + -1, + 0, + ); + if got == libc::MAP_FAILED { + return Err(oserr("mmap(scrub)")); + } + assert_eq!( + got as usize, want as usize, + "ppmem: scrub landed at {got:p}, wanted {want:p}" + ); + Ok(()) + } +} + impl Drop for AddrSpace { fn drop(&mut self) { // One munmap releases the whole reservation including every view diff --git a/src/ppmem/map_windows.rs b/src/ppmem/map_windows.rs index 8acdfffa..20aad16e 100644 --- a/src/ppmem/map_windows.rs +++ b/src/ppmem/map_windows.rs @@ -272,6 +272,18 @@ impl AddrSpace { } Ok(()) } + + /// The Unix backend's `scrub`. Not done here: a placeholder cannot be + /// made readable without committing a view, and IP28 is not built for + /// Windows. Falls back to `unmap`, which keeps the old fault-on-access + /// behaviour. + /// + /// # Safety + /// + /// Same as `unmap`. + pub unsafe fn scrub(&self, at: usize, len: usize) -> io::Result<()> { + unsafe { self.unmap(at, len) } + } } impl Drop for AddrSpace { diff --git a/src/ppmem/ppmem.rs b/src/ppmem/ppmem.rs index a5ccff12..085335e1 100644 --- a/src/ppmem/ppmem.rs +++ b/src/ppmem/ppmem.rs @@ -319,7 +319,7 @@ impl PpMemory { } #[cfg(feature = "jitv2")] - fn bump_gen_all(&self) { + pub fn bump_gen_all(&self) { for i in 0..self.gen_count() { unsafe { (*self.gen_base.add(i)).fetch_add(1, Ordering::Relaxed) }; } @@ -768,7 +768,8 @@ impl MappedMemory for PpMemSpace { for m in std::mem::take(&mut st.mappings) { // Best-effort: a failure here leaves the range mapped, which is // still safe — it just isn't reverted. - let _ = unsafe { self.space.unmap(m.at as usize, m.len as usize) }; + // `scrub`, not `unmap`: see `AddrSpace::scrub`. + let _ = unsafe { self.space.scrub(m.at as usize, m.len as usize) }; #[cfg(feature = "jitv2")] { let gen_off = m.at / GEN_RATIO; @@ -776,7 +777,7 @@ impl MappedMemory for PpMemSpace { let gran = super::map::granularity() as u64; if gen_len >= gran && gen_off % gran == 0 && gen_len % gran == 0 { let _ = unsafe { - self.gen_space.unmap(gen_off as usize, gen_len as usize) + self.gen_space.scrub(gen_off as usize, gen_len as usize) }; } } @@ -1105,6 +1106,39 @@ mod tests { } } + /// Clearing the mappings must leave the window readable. JIT compile + /// workers hold generation-counter pointers taken while a bank was mapped, + /// and read them from other threads while the CPU thread remaps; with + /// `PROT_NONE` there, that read was a SIGBUS that killed the emulator + /// (twice in IP28 boots). Without the fix this test dies the same way. + #[cfg(all(unix, feature = "jitv2"))] + #[test] + fn a_cleared_window_is_still_readable_through_old_pointers() { + let (space, _banks) = PpMemSpace::with_bank_sizes(&[8]).unwrap(); + let base = 0x0800_0000u64; + space.map_bank(0, base, 8 * MB as u64, 8 * MB as u64).unwrap(); + let data = unsafe { space.window_base().add(base as usize + 0x100) as *const u32 }; + let gen = unsafe { space.gen_window_base().add((base as usize + 0x100) >> 12) }; + unsafe { + *(data as *mut u32) = 0xFEED_FACE; + (*gen).fetch_add(7, std::sync::atomic::Ordering::Relaxed); + } + + space.clear_mappings(); + + unsafe { + assert_eq!(*data, 0, "a cleared data range reads as zeros"); + assert_eq!((*gen).load(std::sync::atomic::Ordering::Relaxed), 0, + "a cleared generation range reads as zero, which differs from the live counter"); + } + // Mapped back, the bank's own contents and counter are there again. + space.map_bank(0, base, 8 * MB as u64, 8 * MB as u64).unwrap(); + unsafe { + assert_eq!(*data, 0xFEED_FACE); + assert_eq!((*gen).load(std::sync::atomic::Ordering::Relaxed), 7); + } + } + /// A bank mapped into the window and the same bank accessed through /// `BusDevice` must be the same memory — that is what makes ppmem /// pluggable into the existing bus rather than a parallel universe. diff --git a/src/ps2.rs b/src/ps2.rs index c321403b..e707c66e 100644 --- a/src/ps2.rs +++ b/src/ps2.rs @@ -917,7 +917,7 @@ impl Device for Ps2Controller { fn get_clock(&self) -> u64 { 0 } fn register_commands(&self) -> Vec<(String, String)> { - vec![("ps2".to_string(), "PS/2 commands: ps2 debug | ps2 type | ps2 enter | ps2 status".to_string())] + vec![("ps2".to_string(), "PS/2 commands: ps2 debug | ps2 type | ps2 enter | ps2 mouse [buttons] | ps2 status".to_string())] } fn execute_command(&self, cmd: &str, args: &[&str], mut writer: Box) -> Result<(), String> { @@ -953,6 +953,16 @@ impl Device for Ps2Controller { writeln!(writer, "PS/2: pressed Enter").unwrap(); return Ok(()); } + if !args.is_empty() && args[0] == "mouse" { + let n = |i: usize| args.get(i).and_then(|v| v.parse::().ok()); + let (Some(dx), Some(dy)) = (n(1), n(2)) else { + return Err("Usage: ps2 mouse [buttons: 1 left, 2 right, 4 middle]".to_string()); + }; + let buttons = n(3).unwrap_or(0) as u8; + self.push_mouse_input(buttons, dx, dy, 0); + writeln!(writer, "PS/2: mouse {} {} buttons {}", dx, dy, buttons).unwrap(); + return Ok(()); + } if !args.is_empty() && args[0] == "status" { let s = self.state.lock(); writeln!(writer, "PS/2 state: running={} scanning_enabled={} mouse_enabled={} mouse_id={} rx_queue_len={} mouse_queue_bytes={} scancode_set={} config={:02x} last_read={:02x}", @@ -961,7 +971,7 @@ impl Device for Ps2Controller { s.mouse_queue_bytes, s.scancode_set, s.config, s.last_read).unwrap(); return Ok(()); } - return Err("Usage: ps2 debug | ps2 type | ps2 enter | ps2 status".to_string()); + return Err("Usage: ps2 debug | ps2 type | ps2 enter | ps2 mouse [buttons] | ps2 status".to_string()); } Err("Command not found".to_string()) } diff --git a/src/testdev.rs b/src/testdev.rs index 54503472..3aff30e2 100644 --- a/src/testdev.rs +++ b/src/testdev.rs @@ -348,7 +348,7 @@ pub fn dump_json(core: &MipsCore, tag: u32) -> String { ("XContext", hex64(core.cp0_xcontext)), ("ECC", hex32(core.cp0_ecc)), ("CacheErr", hex32(core.cp0_cacheerr)), - ("TagLo", hex32(core.cp0_taglo)), + ("TagLo", hex32(core.cp0_taglo as u32)), ("TagHi", hex32(core.cp0_taghi)), ]; let body: Vec = cp0.iter().map(|(n, v)| format!(" \"{}\": {}", n, v)).collect();