From cb96f420e356974b5b2fe7e53dc2a5f5c82d177a Mon Sep 17 00:00:00 2001 From: Dani Sarfati Date: Wed, 16 Sep 2026 22:49:10 -0400 Subject: [PATCH 1/3] cpu-tests: loads and stores right behind a load, and through the TLB mem/load_then_load, mem/load_then_load_evict, mem/load_then_store and tlb/load_then_mapped_load: the second access right behind a load that does not name the loaded register, in every mix of cold and warm D-cache lines, with an eviction and a write-back between the two, with LWL's merge, KSEG1 against KSEG0, and with a TLB walk on either side. Plain architecture; a core whose memory stage reads a synchronous cache RAM is where these go wrong. On the sgiindy_MiSTer core (build 37 RTL) they found a stalled load being released on the completion of the load ahead of it; with that fixed the suite is 2409/0 over 250 tests in its simulator, as is the build 36 control that stalls every load behind a load. Not yet run on an Indy. Co-Authored-By: Claude Opus 5 --- cpu-tests/README.md | 10 +- cpu-tests/tests/mem/mem.c | 221 ++++++++++++++++++++++++++++++++++++++ cpu-tests/tests/tlb/tlb.c | 63 +++++++++++ 3 files changed, 292 insertions(+), 2 deletions(-) diff --git a/cpu-tests/README.md b/cpu-tests/README.md index 79b200dd..224a9552 100644 --- a/cpu-tests/README.md +++ b/cpu-tests/README.md @@ -134,10 +134,16 @@ yet. Each is plain architecture rather than a part-specific quirk: | `mem/load_then_trap` | a syscall, break, trap, overflow or reserved instruction right behind a load: exactly one exception, EPC on it, the load complete | | `mem/load_then_more` | a CP0 read, a divide, a jump, and nearby loads and stores right behind a load | | `fpu/trap_behind_a_load` | an FP trap right behind an integer load, while the load may still be waiting on its fill | +| `mem/load_then_load` | a load right behind a load that does not name its register - same line, two lines, the same word at two widths, three in a row, the first load's value as the third's base, the second's used at once, after a store, before a store, an LWL merge, KSEG1 against KSEG0 - with both lines cold, both warm, and each one warm alone | +| `mem/load_then_load_evict` | the second of two loads evicts the first's D-cache line, or finds its own line just written back to memory | +| `mem/load_then_store` | a store right behind a load, in the load's line and another, read back, in the same four cache states | +| `tlb/load_then_mapped_load` | two loads where one or both walk the TLB, and the second's value is used at once | The first results for them are from the `sgiindy_MiSTer` FPGA core presenting as -an R4600: all pass (2259 checks over 246 tests in its simulator, both with every -load stalling execute and with loads that stall only when they must). +an R4600: all pass (2409 checks over 250 tests in its simulator, both with every +load stalling execute and with loads that stall only when they must, including +behind another load or store). `mem/load_then_load` found that core releasing a +stalled load's execute hold on the completion of the load ahead of it. ## Writing a test diff --git a/cpu-tests/tests/mem/mem.c b/cpu-tests/tests/mem/mem.c index 04e151c0..268ba9c8 100644 --- a/cpu-tests/tests/mem/mem.c +++ b/cpu-tests/tests/mem/mem.c @@ -687,6 +687,224 @@ static void t_load_then_more(void) } } +/* + * A load right behind a load, and a store right behind a load, where neither + * names the first load's register. A core whose memory stage reads a + * synchronous cache RAM has the second access's address on that RAM while the + * first one is still being answered, so these are the sequences where it hands + * one access the other's word. Every sequence runs in four cache states: both + * lines cold, both warm, only the first warm, only the second warm - a fill, + * a hit, and each of the two mixed. + * + * t_load_then_load_evict adds the case where the second load evicts the + * first one's line, or finds its own line just written back. + */ +#define LL_WARM(first, second) \ + do { \ + q[0] = v0; q[1] = (u64)(s64)(s32)(unsigned long)&q[8]; q[2] = 0; \ + q[3] = 0; q[4] = v4; q[5] = v5; q[6] = 0; q[7] = 0; q[8] = v8; \ + SYNC(); \ + dcache_wb_invalidate_range(q, 9 * 8); \ + SYNC(); \ + if (first) { u64 w = q[0] ^ q[1]; (void)w; } \ + if (second) { u64 w = q[4] ^ q[5]; (void)w; } \ + } while (0) + +static void t_load_then_load(void) +{ + volatile u64 *q = (volatile u64 *)(_scratch_start + 1536); + volatile u64 *q1; /* the same lines, uncached */ + const u64 v0 = 0x0123456789ABCDEFull; + const u64 v4 = 0x5555AAAA3333CCCCull; + const u64 v5 = 0x0F1E2D3C4B5A6978ull; + const u64 v8 = 0x7766554433221100ull; + u64 r, s, t; + int pass; + + q1 = (volatile u64 *)K1_PTR(q); + + for (pass = 0; pass < 4; pass++) { + /* Same line, consecutive words. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%2)\n\tld $9, 8(%2)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q) : "$8", "$9"); + CHECK_EQ_AT("same line first", pass, r, v0); + CHECK_EQ_AT("same line second", pass, s, (u64)(s64)(s32)(unsigned long)&q[8]); + + /* Two lines. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%2)\n\tld $9, 32(%2)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q) : "$8", "$9"); + CHECK_EQ_AT("two lines first", pass, r, v0); + CHECK_EQ_AT("two lines second", pass, s, v4); + + /* The same word twice, at two widths and offsets. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "lw $8, 4(%2)\n\tlbu $9, 1(%2)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q) : "$8", "$9"); + CHECK_EQ_AT("same word lw", pass, r, 0xFFFFFFFF89ABCDEFull); + CHECK_EQ_AT("same word lbu", pass, s, 0x23u); + + /* Three in a row, each from its own line. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%3)\n\tld $9, 32(%3)\n\tld $10, 64(%3)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero\n\tdaddu %2, $10, $zero" Z + : "=r"(r), "=r"(s), "=r"(t) : "r"(q) : "$8", "$9", "$10"); + CHECK_EQ_AT("three first", pass, r, v0); + CHECK_EQ_AT("three second", pass, s, v4); + CHECK_EQ_AT("three third", pass, t, v8); + + /* The first load's value is the base of the third instruction, which + * reads it while the second load is in the memory stage. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 8(%2)\n\tld $9, 32(%2)\n\tld $10, 0($8)\n\t" + "daddu %0, $9, $zero\n\tdaddu %1, $10, $zero" Z + : "=r"(r), "=r"(s) : "r"(q) : "$8", "$9", "$10"); + CHECK_EQ_AT("chase behind second", pass, r, v4); + CHECK_EQ_AT("chase value", pass, s, v8); + + /* The second load's value used at once: that one has to hold. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%2)\n\tld $9, 32(%2)\n\tdaddu $10, $9, $9\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $10, $zero" Z + : "=r"(r), "=r"(s) : "r"(q) : "$8", "$9", "$10"); + CHECK_EQ_AT("second used first", pass, r, v0); + CHECK_EQ_AT("second used", pass, s, v4 + v4); + + /* A store, then two loads: the first of them waits out READWAIT. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "sd %2, 16(%3)\n\tld $8, 16(%3)\n\tld $9, 40(%3)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(OPAQUE((u64)v8)), "r"(q) + : "$8", "$9", "memory"); + CHECK_EQ_AT("store load load first", pass, r, v8); + CHECK_EQ_AT("store load load second", pass, s, v5); + + /* Two loads, a store, and the store read back. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%3)\n\tld $9, 32(%3)\n\tsd $8, 48(%3)\n\tld $10, 48(%3)\n\t" + "daddu %0, $9, $zero\n\tdaddu %1, $10, $zero\n\tdaddu %2, $8, $zero" Z + : "=r"(r), "=r"(s), "=r"(t) : "r"(q) : "$8", "$9", "$10", "memory"); + CHECK_EQ_AT("load load store second", pass, r, v4); + CHECK_EQ_AT("load load store reload", pass, s, v0); + CHECK_EQ_AT("load load store first", pass, t, v0); + + /* LWL merges with the register's old value, behind a load. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "lui $9, 0x1234\n\tori $9, $9, 0x5678\n\t" + "ld $8, 0(%2)\n\tlwl $9, 33(%2)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q) : "$8", "$9"); + CHECK_EQ_AT("lwl first", pass, r, v0); + CHECK_EQ_AT("lwl merged", pass, s, 0x0000000055AAAA78ull); + + /* Uncached (KSEG1) and cached (KSEG0) loads of the same memory. */ + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%3)\n\tld $9, 32(%2)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q), "r"(q1) : "$8", "$9"); + CHECK_EQ_AT("kseg1 then kseg0 first", pass, r, v0); + CHECK_EQ_AT("kseg1 then kseg0 second", pass, s, v4); + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%2)\n\tld $9, 32(%3)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q), "r"(q1) : "$8", "$9"); + CHECK_EQ_AT("kseg0 then kseg1 first", pass, r, v0); + CHECK_EQ_AT("kseg0 then kseg1 second", pass, s, v4); + } +} + +/* + * The second load evicts, or is evicted by, the first. p16 is 16 KB above q: + * another line of memory at the same index in a direct-mapped D-cache of up to + * 16 KB (on a two-way cache the two simply share a set). Writing it in C + * leaves its line dirty in the cache, so the load of q that follows writes it + * back before filling, and the load of p16 right behind has to find the + * written-back word in memory. + */ +static void t_load_then_load_evict(void) +{ + volatile u64 *q = (volatile u64 *)(_scratch_start + 1536); + volatile u64 *p16 = (volatile u64 *)(_scratch_start + 1536 + 16384); + const u64 v0 = 0x0123456789ABCDEFull; + const u64 w0 = 0x6B6B6B6B00000000ull; + u64 r, s; + int pass; + + for (pass = 0; pass < 2; pass++) { + q[0] = v0; p16[0] = 0; + SYNC(); + dcache_wb_invalidate_range(q, 8); + dcache_wb_invalidate_range(p16, 8); + SYNC(); + + /* p16's line cached and dirty; load q (write-back, fill), then p16. */ + p16[0] = w0 + (u64)pass; + __asm__ __volatile__(A "ld $8, 0(%2)\n\tld $9, 0(%3)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q), "r"(p16) : "$8", "$9"); + CHECK_EQ_AT("evict q", pass, r, v0); + CHECK_EQ_AT("evicted p16", pass, s, w0 + (u64)pass); + + /* And the other way round, with q's line dirty. */ + q[0] = v0 + (u64)pass; + __asm__ __volatile__(A "ld $8, 0(%3)\n\tld $9, 0(%2)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(q), "r"(p16) : "$8", "$9"); + CHECK_EQ_AT("evict p16", pass, r, w0 + (u64)pass); + CHECK_EQ_AT("evicted q", pass, s, v0 + (u64)pass); + } +} + +/* + * A store right behind a load, not storing or addressing through the loaded + * register, read back afterwards - in the load's own line and in another, in + * the same four cache states as above. + */ +static void t_load_then_store(void) +{ + volatile u64 *q = (volatile u64 *)(_scratch_start + 1536); + const u64 v0 = 0x0123456789ABCDEFull; + const u64 v4 = 0x5555AAAA3333CCCCull; + const u64 v5 = 0x0F1E2D3C4B5A6978ull; + const u64 v8 = 0x7766554433221100ull; + const u64 x = 0x3C3C3C3CA5A5A5A5ull; + u64 r, s; + int pass; + + for (pass = 0; pass < 4; pass++) { + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%3)\n\tsd %2, 16(%3)\n\tld $9, 16(%3)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(OPAQUE((u64)x)), "r"(q) + : "$8", "$9", "memory"); + CHECK_EQ_AT("store own line value", pass, r, v0); + CHECK_EQ_AT("store own line reload", pass, s, x); + CHECK_EQ_AT("store own line memory", pass, q[2], x); + + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 0(%3)\n\tsw %2, 44(%3)\n\tld $9, 40(%3)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(OPAQUE((u64)x)), "r"(q) + : "$8", "$9", "memory"); + CHECK_EQ_AT("store other line value", pass, r, v0); + CHECK_EQ_AT("store other line reload", pass, s, (v5 & 0xFFFFFFFF00000000ull) | (x & 0xFFFFFFFFull)); + + LL_WARM(pass == 1 || pass == 2, pass == 1 || pass == 3); + __asm__ __volatile__(A "ld $8, 32(%3)\n\tsb %2, 3(%3)\n\tld $9, 0(%3)\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $9, $zero" Z + : "=r"(r), "=r"(s) : "r"(OPAQUE((u64)x)), "r"(q) + : "$8", "$9", "memory"); + CHECK_EQ_AT("sb value", pass, r, v4); + CHECK_EQ_AT("sb reload", pass, s, (v0 & ~0x000000FF00000000ull) | 0x000000A500000000ull); + (void)v8; + } +} +#undef LL_WARM + static const struct test tests[] = { TEST("mem/load_widths_sign", t_load_widths_and_sign, CPU_ALL), TEST("mem/store_widths", t_store_widths, CPU_ALL), @@ -709,6 +927,9 @@ static const struct test tests[] = { TEST("mem/load_then_use", t_load_then_use, CPU_ALL), TEST("mem/load_then_trap", t_load_then_trap, CPU_ALL), TEST("mem/load_then_more", t_load_then_more, CPU_ALL), + TEST("mem/load_then_load", t_load_then_load, CPU_ALL), + TEST("mem/load_then_load_evict", t_load_then_load_evict, CPU_ALL), + TEST("mem/load_then_store", t_load_then_store, CPU_ALL), }; const struct test_group group_mem = { diff --git a/cpu-tests/tests/tlb/tlb.c b/cpu-tests/tests/tlb/tlb.c index 9a418cad..22ceff81 100644 --- a/cpu-tests/tests/tlb/tlb.c +++ b/cpu-tests/tests/tlb/tlb.c @@ -352,6 +352,68 @@ static void t_refill_vector_and_context(void) cp0_entryhi_set(saved_hi); } +/* ── a load behind a load, through the TLB ───────────────────────────────── */ + +/* + * The second of two loads needs a TLB walk, or the first does, and the + * instruction after the second uses its value. A core that lets the first load + * leave execute without holding it runs the second load's walk and its own + * load stall while the first load is still being answered - and must not take + * the first load's completion for the second's. Rewriting the entry before + * each sequence makes the mapped access walk the TLB rather than hit a cached + * translation. Both passes: lines not in the D-cache, then cached. + */ +static void t_load_then_mapped_load(void) +{ + u64 saved_hi = cp0_entryhi(); + u64 phys = scratch_phys(); + volatile u64 *qk = (volatile u64 *)(_scratch_start + 1536); + volatile u64 *qt = (volatile u64 *)(unsigned long)(TEST_VA + 1536); + const u64 v0 = 0x1357924680ACEBDFull; + const u64 v1 = 0x0246813579BDFACEull; + const u64 v4 = 0x7FEDCBA987654321ull; + u64 r, s; + int pass; + + for (pass = 0; pass < 2; pass++) { + qk[0] = v0; qk[1] = v1; qk[4] = v4; + SYNC(); + dcache_wb_invalidate_range(qk, 5 * 8); + SYNC(); + if (pass == 1) { u64 w = qk[0] ^ qk[4]; (void)w; } + + write_entry(3, TEST_VA, entrylo(phys, ELO_V | ELO_D | ELO_G), 0, PM_4K); + exc_clear(); + __asm__ __volatile__(A "ld $8, 32(%2)\n\tld $9, 0(%3)\n\tdaddu $10, $9, $9\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $10, $zero" Z + : "=r"(r), "=r"(s) : "r"(qk), "r"(qt) : "$8", "$9", "$10"); + CHECK_EQ_AT("kseg0 then mapped: first", pass, r, v4); + CHECK_EQ_AT("kseg0 then mapped: second used", pass, s, v0 + v0); + CHECK_EQ_AT("kseg0 then mapped: exceptions", pass, exc.count, 0u); + + write_entry(3, TEST_VA, entrylo(phys, ELO_V | ELO_D | ELO_G), 0, PM_4K); + exc_clear(); + __asm__ __volatile__(A "ld $8, 32(%3)\n\tld $9, 8(%2)\n\tdaddu $10, $9, $9\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $10, $zero" Z + : "=r"(r), "=r"(s) : "r"(qk), "r"(qt) : "$8", "$9", "$10"); + CHECK_EQ_AT("mapped then kseg0: first", pass, r, v4); + CHECK_EQ_AT("mapped then kseg0: second used", pass, s, v1 + v1); + CHECK_EQ_AT("mapped then kseg0: exceptions", pass, exc.count, 0u); + + write_entry(3, TEST_VA, entrylo(phys, ELO_V | ELO_D | ELO_G), 0, PM_4K); + exc_clear(); + __asm__ __volatile__(A "ld $8, 32(%2)\n\tld $9, 8(%2)\n\tdaddu $10, $9, $9\n\t" + "daddu %0, $8, $zero\n\tdaddu %1, $10, $zero" Z + : "=r"(r), "=r"(s) : "r"(qt) : "$8", "$9", "$10"); + CHECK_EQ_AT("mapped then mapped: first", pass, r, v4); + CHECK_EQ_AT("mapped then mapped: second used", pass, s, v1 + v1); + CHECK_EQ_AT("mapped then mapped: exceptions", pass, exc.count, 0u); + } + + write_entry(3, 0x1FFFE000ull, 0, 0, PM_4K); + cp0_entryhi_set(saved_hi); +} + static const struct test tests[] = { TEST("tlb/tlbwi_tlbr", t_tlbwi_tlbr_round_trip, CPU_ALL), TEST("tlb/all_entries", t_all_entries_round_trip, CPU_ALL), @@ -363,6 +425,7 @@ static const struct test tests[] = { TEST("tlb/asid_match", t_asid_match_and_mismatch, CPU_ALL), TEST("tlb/global_ignores_asid", t_global_entry_ignores_asid, CPU_ALL), TEST("tlb/refill_context", t_refill_vector_and_context, CPU_ALL), + TEST("tlb/load_then_mapped_load", t_load_then_mapped_load, CPU_ALL), }; const struct test_group group_tlb = { From 16ee36d44c66bab07253e3aa9abf6966bfc85163 Mon Sep 17 00:00:00 2001 From: Dani Sarfati Date: Mon, 28 Sep 2026 18:08:05 -0400 Subject: [PATCH 2/3] cpu-tests: umode - User mode the way IRIX 6 runs it (UX=1 under KX=0) IRIX 6 runs n32 processes with Status.UX = 1 under a 32-bit kernel with KX = 0, so every exception and ERET switches addressing modes; the rest of the suite runs KX = SX = UX = 1 and IRIX 5 runs all three clear, so nothing covered the switch. Each test runs a few words at kuseg 0x00400000 in User mode under all64 / irix5 / irix6 Status settings and records every exception (vector, Cause, EPC, BadVAddr, Context, XContext, EntryHi): syscalls first/second after ERET (libc _getuid), break, a pending interrupt, data and fetch TLB refills (fixed up and retried), KSEG0 and misaligned loads, and 64-bit ops with UX. `make ONLY=group_x` builds identity + one group for slow targets. IRIS main passes all 12. An IRIS build from 2026-08-31 failed the refill-vector and KSEG0-from-User checks; current main passes them. On the sgiindy_MiSTer FPGA core, umode/syscall_second reproduced IRIX 6.5's init death exactly (TLBL at the general vector's own address, EPC on _getuid's syscall) before the core's fix. Derived from the R4000 manual; not yet measured on silicon. Co-Authored-By: Claude Opus 5.5 --- cpu-tests/Makefile | 2 +- cpu-tests/harness/tests.c | 9 + cpu-tests/tests/umode/umode.c | 702 ++++++++++++++++++++++++++++++++++ 3 files changed, 712 insertions(+), 1 deletion(-) create mode 100644 cpu-tests/tests/umode/umode.c diff --git a/cpu-tests/Makefile b/cpu-tests/Makefile index d3a9def8..ba405276 100644 --- a/cpu-tests/Makefile +++ b/cpu-tests/Makefile @@ -38,7 +38,7 @@ CFLAGS := $(ARCHFLAGS) -ffreestanding -nostdlib -nostdinc \ -fno-delete-null-pointer-checks -fomit-frame-pointer \ -O1 -g -std=gnu11 \ -Wall -Wextra -Werror -Wno-unused-parameter \ - -Iharness + -Iharness $(if $(ONLY),-DCPUTEST_ONLY=$(ONLY)) ASFLAGS := $(ARCHFLAGS) -ffreestanding -nostdlib -Iharness -g -Wa,--fatal-warnings LDFLAGS := -EB -T harness/link.ld -nostdlib --no-warn-mismatch diff --git a/cpu-tests/harness/tests.c b/cpu-tests/harness/tests.c index 7803ce19..c6d7cf63 100644 --- a/cpu-tests/harness/tests.c +++ b/cpu-tests/harness/tests.c @@ -16,6 +16,7 @@ DECLARE_GROUP(group_branch); DECLARE_GROUP(group_excep); DECLARE_GROUP(group_cp0); DECLARE_GROUP(group_tlb); +DECLARE_GROUP(group_umode); DECLARE_GROUP(group_fpu); DECLARE_GROUP(group_fpu_trap); DECLARE_GROUP(group_fpu_denorm); @@ -28,6 +29,12 @@ DECLARE_GROUP(group_cache); DECLARE_GROUP(group_mips4); DECLARE_GROUP(group_mips4_fp); +/* `make ONLY=group_umode` builds a suite of identity + that one group, for + * iterating on a slow target (a simulator). */ +#ifdef CPUTEST_ONLY +DECLARE_GROUP(CPUTEST_ONLY); +const struct test_group *const all_groups[] = { &group_identity, &CPUTEST_ONLY }; +#else const struct test_group *const all_groups[] = { &group_identity, &group_alu, @@ -37,6 +44,7 @@ const struct test_group *const all_groups[] = { &group_excep, &group_cp0, &group_tlb, + &group_umode, &group_fpu, &group_fpu_trap, &group_fpu_denorm, @@ -49,5 +57,6 @@ const struct test_group *const all_groups[] = { &group_mips4, &group_mips4_fp, }; +#endif const unsigned n_groups = sizeof(all_groups) / sizeof(all_groups[0]); diff --git a/cpu-tests/tests/umode/umode.c b/cpu-tests/tests/umode/umode.c new file mode 100644 index 00000000..54658f81 --- /dev/null +++ b/cpu-tests/tests/umode/umode.c @@ -0,0 +1,702 @@ +/* umode — User mode the way IRIX 6 runs it. + * + * IRIX 6 runs n32 processes with Status.UX = 1 (64-bit operations and the + * 64-bit user address space) under a kernel that runs with KX = 0 on the + * 32-bit kernels (IP22 among them). So every exception taken from such a + * process switches the CPU between 64-bit and 32-bit addressing, and every + * ERET back switches it again. The rest of the suite never does that: it runs + * with KX = SX = UX = 1 throughout, and IRIX 5's o32 processes run with all + * three clear, so neither exercises the switch. + * + * Each test runs a few words of code at a kuseg address in User mode, in + * three Status settings: + * + * all64 KX = SX = UX = 1 what the rest of the suite runs + * irix5 KX = SX = UX = 0 o32 under a 32-bit kernel + * irix6 KX = 0, UX = 1 n32 under a 32-bit kernel + * + * and records every exception the code takes: which vector, Cause, EPC, + * BadVAddr, Context, XContext and EntryHi. The code ends in a syscall, which + * returns to Kernel mode. Any other exception is resumed in User mode by the + * test's policy: stepped over, or (TLB misses) fixed up by writing a TLB + * entry and retried, or (interrupts) acknowledged and retried. + * + * The case that matters most is an exception taken by the FIRST instruction + * executed after an ERET into User mode - a syscall right after an interrupt + * returns, or a load that misses the TLB right after a refill returns. IRIX 6 + * does that all the time. + * + * Derived from the R4000 manual (exception vectors, Status.UX, the 64-bit + * address spaces), not yet measured on silicon. + */ + +#include "testlib.h" +#include "cp0.h" +#include "excoff.h" + +#define A ".set push; .set mips3; .set noreorder; .set nomacro; .set noat\n\t" +#define Z "\n\t.set pop" + +extern char _scratch_start[]; + +#define ST_KSU_USER 0x00000010u + +/* Virtual layout. The code page and data page share one TLB entry (even and + * odd page of one VPN2); the other two are mapped on demand by the fix-up + * policy, i.e. the first touch of them is a TLB refill. */ +#define UMX_CODE_VA 0x00400000ull +#define UMX_DATA_VA 0x00401000ull +#define UMX_DATA2_VA 0x00800000ull +#define UMX_CODE2_VA 0x00810000ull + +/* Physical pages, offsets into the suite's scratch area (16 KB aligned). Each + * one has the same address bits 13:12 as the virtual page it is mapped at, so + * no virtually indexed cache sees two colours of it. */ +#define UMX_P_CODE 0x0000u +#define UMX_P_DATA 0x1000u +#define UMX_P_DATA2 0x4000u +#define UMX_P_CODE2 0x8000u + +#define UMX_TLB_INDEX 20u /* code + data */ +#define UMX_FIX_INDEX 21u /* written by the fix-up policy */ + +#define UMX_NREC 8 +#define UMX_RUNAWAY 12 /* this many exceptions and it goes home */ + +#define UMX_POLICY_SKIP 0 /* step over anything but Sys and Int */ +#define UMX_POLICY_MAP 1 /* TLB miss: map it (map_lo0/1) and retry */ + +struct umx_rec { + u32 vector; /* VECID_* */ + u32 cause; + u32 status; + u32 pad; + u64 epc; + u64 badvaddr; + u64 context; + u64 xcontext; + u64 entryhi; +}; + +struct umx_state { + u32 n; /* 0: exceptions taken (may exceed UMX_NREC) */ + u32 policy; /* 4 */ + u64 return_pc; /* 8: where a syscall goes home to */ + u64 map_lo0; /* 16: EntryLo0/1 the MAP policy writes */ + u64 map_lo1; /* 24 */ + struct umx_rec rec[UMX_NREC]; /* 32, 56 bytes each */ +}; + +_Static_assert(sizeof(struct umx_rec) == 56, "umx_rec layout is fixed by umx_handler"); +_Static_assert(__builtin_offsetof(struct umx_state, rec) == 32, "umx.rec"); + +extern volatile struct umx_state umx; +extern void umx_handler(void); + +/* + * The exception hook. exc_dispatch has already filled `exc` with this + * exception's registers and saved $at/$v0/$v1; $k0/$k1 are free. + */ +__asm__( + " .text\n" + " .set push; .set mips3; .set noreorder; .set nomacro; .set noat\n" + " .globl umx_handler\n" + " .ent umx_handler\n" + "umx_handler:\n" + " lui $k1, %hi(umx)\n" + " addiu $k1, $k1, %lo(umx)\n" + " lui $at, %hi(exc)\n" + " addiu $at, $at, %lo(exc)\n" + " lw $v0, 0($k1)\n" + " sltiu $v1, $v0, 8\n" /* UMX_NREC */ + " beqz $v1, 2f\n" + " nop\n" + " sll $v1, $v0, 6\n" + " sll $k0, $v0, 3\n" + " subu $v1, $v1, $k0\n" /* n * 56 */ + " addu $v1, $v1, $k1\n" + " lw $k0, 12($at)\n" /* EXC_O_VECTOR */ + " sw $k0, 32($v1)\n" + " lw $k0, 8($at)\n" /* EXC_O_CAUSE */ + " sw $k0, 36($v1)\n" + " lw $k0, 4($at)\n" /* EXC_O_STATUS */ + " sw $k0, 40($v1)\n" + " ld $k0, 24($at)\n" /* EXC_O_EPC */ + " sd $k0, 48($v1)\n" + " ld $k0, 32($at)\n" /* EXC_O_BADVADDR */ + " sd $k0, 56($v1)\n" + " ld $k0, 56($at)\n" /* EXC_O_CONTEXT */ + " sd $k0, 64($v1)\n" + " ld $k0, 64($at)\n" /* EXC_O_XCONTEXT */ + " sd $k0, 72($v1)\n" + " ld $k0, 48($at)\n" /* EXC_O_ENTRYHI */ + " sd $k0, 80($v1)\n" + "2: addiu $v0, $v0, 1\n" + " sw $v0, 0($k1)\n" + " sltiu $v1, $v0, 12\n" /* UMX_RUNAWAY */ + " beqz $v1, umx_home\n" + " nop\n" + " lw $v0, 8($at)\n" /* Cause */ + " andi $v0, $v0, 0x7c\n" + " xori $v1, $v0, 0x20\n" /* ExcCode 8, Sys */ + " beqz $v1, umx_home\n" + " nop\n" + " beqz $v0, umx_int\n" /* ExcCode 0, Int */ + " nop\n" + " lw $v1, 4($k1)\n" /* policy */ + " beqz $v1, umx_skip\n" + " nop\n" + " xori $v1, $v0, 0x08\n" /* ExcCode 2, TLBL */ + " beqz $v1, umx_map\n" + " nop\n" + " xori $v1, $v0, 0x0c\n" /* ExcCode 3, TLBS */ + " bnez $v1, umx_skip\n" + " nop\n" + "umx_map:\n" /* EntryHi: loaded by the miss */ + " li $v1, 21\n" /* UMX_FIX_INDEX */ + " mtc0 $v1, $0\n" + " mtc0 $zero, $5\n" + " ld $v1, 16($k1)\n" + " dmtc0 $v1, $2\n" + " ld $v1, 24($k1)\n" + " dmtc0 $v1, $3\n" + " nop\n" + " nop\n" + " tlbwi\n" + " nop\n" + " nop\n" + " nop\n" + " nop\n" + " b umx_eret\n" /* retry: EPC unchanged */ + " nop\n" + "umx_int:\n" + " mtc0 $zero, $13\n" /* drop the software interrupt */ + " nop\n" + " nop\n" + " b umx_eret\n" /* retry */ + " nop\n" + "umx_skip:\n" + " dmfc0 $k0, $14\n" + " nop\n" + " daddiu $k0, $k0, 4\n" + " dmtc0 $k0, $14\n" + " nop\n" + " b umx_eret\n" + " nop\n" + "umx_home:\n" + " mfc0 $k0, $12\n" + " nop\n" + " ori $k0, $k0, 0x18\n" + " xori $k0, $k0, 0x18\n" /* KSU = Kernel; EXL still set */ + " mtc0 $k0, $12\n" + " nop\n" + " ld $k0, 8($k1)\n" + " dmtc0 $k0, $14\n" + " nop\n" + "umx_eret:\n" + " lui $k0, %hi(exc_save)\n" + " addiu $k0, $k0, %lo(exc_save)\n" + " ld $at, 0($k0)\n" + " ld $v0, 8($k0)\n" + " ld $v1, 16($k0)\n" + " ssnop\n" + " ssnop\n" + " ssnop\n" + " ssnop\n" + " eret\n" + " nop\n" + " .end umx_handler\n" + " .set pop\n" + " .section .bss\n" + " .align 3\n" + " .globl umx\n" + "umx:\n" + " .space 480\n" /* sizeof(struct umx_state) */ + " .text\n"); + +_Static_assert(sizeof(struct umx_state) == 480, "umx storage in umx_handler's .space"); + +/* ── the User-mode code ───────────────────────────────────────────────────── */ +/* + * Assembled here, copied to the code pages at run time. On entry $4, $5 and + * $6 hold the runner's a0/a1/a2; $5 is always UMX_DATA_VA. + */ +#define UCODE(name, body) \ + __asm__(" .section .rodata\n" \ + " .set push; .set mips3; .set noreorder; .set nomacro; .set noat\n" \ + " .align 2\n" \ + " .globl " #name "\n" #name ":\n" body \ + " .globl " #name "_end\n" #name "_end:\n" \ + " .set pop\n" \ + " .text\n"); \ + extern const u32 name[], name##_end[] + +/* A syscall as the first instruction. */ +UCODE(uc_sys, + " syscall\n" + " nop\n"); + +/* libc's _getuid: li v0, 1024; syscall. */ +UCODE(uc_getuid, + " li $2, 1024\n" + " syscall\n" + " nop\n"); + +/* Six nops for a pending interrupt to land in, then _getuid. */ +UCODE(uc_int_sled, + " nop\n" + " nop\n" + " nop\n" + " nop\n" + " nop\n" + " nop\n" + " li $2, 1024\n" + " syscall\n" + " nop\n"); + +UCODE(uc_break, + " break\n" + " syscall\n" + " nop\n"); + +/* A load through $4 as the first instruction; the value goes to data+16. */ +UCODE(uc_load_first, + " lw $8, 0($4)\n" + " sw $8, 16($5)\n" + " syscall\n" + " nop\n"); + +UCODE(uc_load_third, + " sd $4, 24($5)\n" /* the address, for the log */ + " nop\n" + " lw $8, 0($4)\n" + " sw $8, 16($5)\n" + " syscall\n" + " nop\n"); + +/* Jump through $6. */ +UCODE(uc_jump, + " jr $6\n" + " nop\n"); + +/* 64-bit operations, stored to data+0 and data+8. */ +UCODE(uc_dword, + " daddiu $8, $0, 1\n" + " dsll32 $8, $8, 4\n" + " sd $8, 0($5)\n" + " ld $9, 0($5)\n" + " daddu $10, $8, $9\n" + " sd $10, 8($5)\n" + " syscall\n" + " nop\n"); + +/* ── the runner ───────────────────────────────────────────────────────────── */ + +#define N_MODES 3 +static const u32 modes[N_MODES] = { ST_KX | ST_SX | ST_UX, 0, ST_UX }; +static const char *const mode_names[N_MODES] = { "all64", "irix5", "irix6" }; + +static u64 scratch_phys(void) +{ + return (u64)((u32)(unsigned long)_scratch_start & 0x1FFFFFFFu); +} + +static u64 umx_lo(u32 page_off) +{ + return (((scratch_phys() + page_off) >> 12) << ELO_PFN_SHIFT) | + ((u64)CA_CACHEABLE_NC << ELO_C_SHIFT) | ELO_V | ELO_D | ELO_G; +} + +static void tlb_write(u32 index, u64 hi, u64 lo0, u64 lo1) +{ + cp0_index_set(index); + cp0_pagemask_set(PM_4K); + cp0_entryhi_set(hi); + cp0_entrylo0_set(lo0); + cp0_entrylo1_set(lo1); + tlb_write_indexed(); +} + +/* Park an entry on a VPN2 of its own that nothing uses, invalid. */ +static void tlb_retire(u32 index) +{ + tlb_write(index, 0x1FF00000ull + ((u64)index << 13), 0, 0); +} + +/* Make sure no entry maps `va`. */ +static void tlb_unmap(u64 va) +{ + cp0_entryhi_set(va); + tlb_probe(); + if ((cp0_index() & 0x80000000u) == 0) tlb_retire(cp0_index() & 0x3F); +} + +static void put_code(u32 page_off, const u32 *code, const u32 *end) +{ + volatile u32 *page = (volatile u32 *)(_scratch_start + page_off); + unsigned n = (unsigned)(end - code), i; + for (i = 0; i < n; i++) page[i] = code[i]; + SYNC(); + dcache_wb_invalidate_range(page, n * 4); + icache_invalidate_range(page, n * 4); +} + +static volatile u32 *data_page(void) +{ + return (volatile u32 *)(_scratch_start + UMX_P_DATA); +} + +/* + * Run `code` in User mode at `entry` with Status = the suite's own plus + * `mode` (KX/SX/UX, IE/IM) and Cause = `cause` (software interrupt bits). + */ +static void umx_run(const u32 *code, const u32 *code_end, u64 entry, u32 mode, + u32 cause, u32 policy, u64 a0, u64 a2) +{ + u64 saved_hi = cp0_entryhi(); + u32 saved_pm = cp0_pagemask(); + u32 saved_status = cp0_status(); + u64 a1 = UMX_DATA_VA; + u32 status; + unsigned i; + + put_code(UMX_P_CODE, code, code_end); + + tlb_unmap(UMX_CODE_VA); + tlb_unmap(UMX_DATA2_VA); + tlb_unmap(UMX_CODE2_VA); + tlb_retire(UMX_FIX_INDEX); + tlb_write(UMX_TLB_INDEX, UMX_CODE_VA, umx_lo(UMX_P_CODE), umx_lo(UMX_P_DATA)); + /* The same physical lines through the mapping, so nothing cached from an + * earlier run of the code page survives into this one. */ + icache_invalidate_range(SEXT_PTR((u32)UMX_CODE_VA), (u32)(code_end - code) * 4); + + umx.n = 0; + umx.policy = policy; + for (i = 0; i < UMX_NREC; i++) { + umx.rec[i].vector = 0; + umx.rec[i].cause = 0; + umx.rec[i].status = 0; + umx.rec[i].epc = 0; + umx.rec[i].badvaddr = 0; + umx.rec[i].context = 0; + umx.rec[i].xcontext = 0; + umx.rec[i].entryhi = 0; + } + exc_clear(); + exc_user_handler = (u32)(unsigned long)&umx_handler; + + status = (saved_status & ~(ST_CU0 | ST_KX | ST_SX | ST_UX | 0x18u | ST_IE | ST_IM_MASK)) + | ST_EXL | ST_KSU_USER | mode; + __asm__ __volatile__(A + "dla $8, 1f\n\t" + "sd $8, 8(%0)\n\t" /* umx.return_pc */ + "mfc0 $9, $12\n\t" + "nop\n\t" + "ori $9, $9, 0x2\n\t" /* EXL first: still Kernel */ + "mtc0 $9, $12\n\t" + "nop; nop; nop\n\t" + "mtc0 %2, $13\n\t" /* software interrupts, if any */ + "mtc0 %1, $12\n\t" /* KSU, KX/SX/UX, IE/IM; EXL kept */ + "nop; nop; nop\n\t" + "move $4, %4\n\t" + "move $5, %5\n\t" + "move $6, %6\n\t" + "dmtc0 %3, $14\n\t" + "nop; nop; nop\n\t" + "eret\n\t" + "nop\n\t" + "1:" Z + :: "r"(&umx), "r"(status), "r"(cause), "r"(entry), "r"(a0), "r"(a1), "r"(a2) + : "$2", "$3", "$4", "$5", "$6", "$7", "$8", "$9", "$10", "$11", + "$12", "$13", "$14", "$15", "$24", "$25", "memory"); + + exc_user_handler = 0; + cp0_status_set(saved_status); + cp0_cause_set(0); + + tlb_retire(UMX_TLB_INDEX); + tlb_retire(UMX_FIX_INDEX); + cp0_entryhi_set(saved_hi); + cp0_pagemask_set(saved_pm); +} + +/* What the run recorded, for a failing test's log. */ +static void umx_dump(int m) +{ + u32 i, n = umx.n < UMX_NREC ? umx.n : UMX_NREC; + con_printf(" [%s: %u exception(s)]\n", mode_names[m], umx.n); + for (i = 0; i < n; i++) { + con_printf(" #%u vec=%u exc=%u epc=", i, umx.rec[i].vector, + CAUSE_EXC(umx.rec[i].cause)); + con_hex64(umx.rec[i].epc); + con_puts(" badva="); + con_hex64(umx.rec[i].badvaddr); + con_printf(" sr=%x\n", umx.rec[i].status); + } +} + +/* Check record `i`: its ExcCode, vector and EPC. */ +static void expect(int m, u32 i, u32 code, u32 vecid, u64 epc) +{ + CHECK_EQ_AT(mode_names[m], (int)i, CAUSE_EXC(umx.rec[i].cause), code); + CHECK_EQ_AT(mode_names[m], (int)i, umx.rec[i].vector, vecid); + CHECK_EQ_AT(mode_names[m], (int)i, umx.rec[i].epc, epc); + /* Taken from User mode: KSU still says User inside the handler. */ + CHECK_EQ_AT(mode_names[m], (int)i, umx.rec[i].status & 0x18u, ST_KSU_USER); +} + +#define DUMP_IF_FAILED(m, before) do { if (cur_test_fails != (before)) umx_dump(m); } while (0) + +/* An EPC somewhere in uc_int_sled's nops or its li. */ +#define CHECK_AT_SLED(m, epc) \ + CHECK_EQ_AT(mode_names[m], 0, \ + (epc) >= UMX_CODE_VA && (epc) <= UMX_CODE_VA + 24 && ((epc) & 3) == 0, 1) + +static u32 refill_vecid(int m) +{ + return (modes[m] & ST_UX) ? (u32)VECID_XTLB : (u32)VECID_TLB; +} + +/* ── tests ────────────────────────────────────────────────────────────────── */ + +/* A syscall as the first instruction after ERET into User mode. */ +static void t_syscall_first(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + umx_run(uc_sys, uc_sys_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_SKIP, 0, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 1u); + expect(m, 0, EXC_SYS, VECID_GENERAL, UMX_CODE_VA); + DUMP_IF_FAILED(m, before); + } +} + +/* libc's _getuid: the syscall is the second instruction. */ +static void t_syscall_second(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + umx_run(uc_getuid, uc_getuid_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_SKIP, 0, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 1u); + expect(m, 0, EXC_SYS, VECID_GENERAL, UMX_CODE_VA + 4); + DUMP_IF_FAILED(m, before); + } +} + +/* A break first, stepped over; then the syscall is the first instruction + * after the handler's ERET. */ +static void t_break_then_syscall(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + umx_run(uc_break, uc_break_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_SKIP, 0, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 2u); + expect(m, 0, EXC_BP, VECID_GENERAL, UMX_CODE_VA); + expect(m, 1, EXC_SYS, VECID_GENERAL, UMX_CODE_VA + 4); + DUMP_IF_FAILED(m, before); + } +} + +/* + * A software interrupt is pending when ERET enters User mode. It is taken in + * User mode, from one of the first instructions - which one is a matter of + * pipeline latency the architecture does not pin down, hence the nop sled - + * and the handler acknowledges it and returns there. Then _getuid's syscall. + */ +static void t_interrupt_then_syscall(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + umx_run(uc_int_sled, uc_int_sled_end, UMX_CODE_VA, modes[m] | ST_IE | (1u << ST_IM_SHIFT), + 1u << CAUSE_IP_SHIFT, UMX_POLICY_SKIP, 0, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 2u); + CHECK_EQ_AT(mode_names[m], 0, CAUSE_EXC(umx.rec[0].cause), (u32)EXC_INT); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].vector, (u32)VECID_GENERAL); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].status & 0x18u, ST_KSU_USER); + CHECK_AT_SLED(m, umx.rec[0].epc); + expect(m, 1, EXC_SYS, VECID_GENERAL, UMX_CODE_VA + 28); + DUMP_IF_FAILED(m, before); + } +} + +/* + * A load that misses the TLB. The refill goes to the XTLB vector when UX is + * set (the address is in the 64-bit user space) and to the 32-bit one when it + * is not; the fix-up maps the page and retries, so the load then runs as the + * first instruction after the refill's ERET, and the syscall after it. + */ +static void load_miss(const u32 *code, const u32 *end, u64 load_pc) +{ + int m; + volatile u32 *data2 = (volatile u32 *)(_scratch_start + UMX_P_DATA2); + + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + + data2[0] = 0x5EC0DE00u + (u32)m; + data_page()[4] = 0; + SYNC(); + dcache_wb_invalidate_range(data2, 16); + dcache_wb_invalidate_range(data_page(), 32); + + umx.map_lo0 = umx_lo(UMX_P_DATA2); + umx.map_lo1 = 0; + umx_run(code, end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_MAP, UMX_DATA2_VA, 0); + + CHECK_EQ_AT(mode_names[m], 0, umx.n, 2u); + expect(m, 0, EXC_TLBL, refill_vecid(m), load_pc); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].badvaddr, UMX_DATA2_VA); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].entryhi & ~0x1FFFull, UMX_DATA2_VA); + CHECK_EQ_AT(mode_names[m], 0, (umx.rec[0].context >> 4) & 0x7FFFFull, UMX_DATA2_VA >> 13); + if (modes[m] & ST_UX) { + /* XContext: R (32:31) = 0 for xuseg, BadVPN2 (30:4) = VA[39:13]. */ + CHECK_EQ_AT(mode_names[m], 0, (umx.rec[0].xcontext >> 4) & 0x7FFFFFFull, UMX_DATA2_VA >> 13); + CHECK_EQ_AT(mode_names[m], 0, (umx.rec[0].xcontext >> 31) & 3ull, 0u); + } + expect(m, 1, EXC_SYS, VECID_GENERAL, load_pc + 8); + CHECK_EQ_AT(mode_names[m], 0, data_page()[4], 0x5EC0DE00u + (u32)m); + DUMP_IF_FAILED(m, before); + } +} + +static void t_load_miss_first(void) +{ + load_miss(uc_load_first, uc_load_first_end, UMX_CODE_VA); +} + +static void t_load_miss_third(void) +{ + load_miss(uc_load_third, uc_load_third_end, UMX_CODE_VA + 8); +} + +/* + * An instruction fetch that misses the TLB: a jump to an unmapped page + * (`jump`), or ERET straight to one (`entry`, so the miss comes before any + * User instruction has executed). The fix-up maps the second code page, + * whose first instruction is a syscall. + */ +static void fetch_miss(int via_jump) +{ + int m; + + put_code(UMX_P_CODE2, uc_sys, uc_sys_end); + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + + umx.map_lo0 = umx_lo(UMX_P_CODE2); + umx.map_lo1 = 0; + if (via_jump) + umx_run(uc_jump, uc_jump_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_MAP, 0, UMX_CODE2_VA); + else + umx_run(uc_sys, uc_sys_end, UMX_CODE2_VA, modes[m], 0, UMX_POLICY_MAP, 0, 0); + + CHECK_EQ_AT(mode_names[m], 0, umx.n, 2u); + expect(m, 0, EXC_TLBL, refill_vecid(m), UMX_CODE2_VA); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].badvaddr, UMX_CODE2_VA); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].entryhi & ~0x1FFFull, UMX_CODE2_VA); + expect(m, 1, EXC_SYS, VECID_GENERAL, UMX_CODE2_VA); + DUMP_IF_FAILED(m, before); + } +} + +static void t_fetch_miss_jump(void) { fetch_miss(1); } +static void t_fetch_miss_entry(void) { fetch_miss(0); } + +/* A load from KSEG0 in User mode, as the first instruction: Address Error. */ +static void t_kseg0_load_first(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + umx_run(uc_load_first, uc_load_first_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_SKIP, + 0xFFFFFFFF80000000ull, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 2u); + expect(m, 0, EXC_ADEL, VECID_GENERAL, UMX_CODE_VA); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].badvaddr, 0xFFFFFFFF80000000ull); + expect(m, 1, EXC_SYS, VECID_GENERAL, UMX_CODE_VA + 8); + DUMP_IF_FAILED(m, before); + } +} + +/* The same, third instruction: the first two are not the ones after ERET. */ +static void t_kseg0_load_third(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + umx_run(uc_load_third, uc_load_third_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_SKIP, + 0xFFFFFFFF80000000ull, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 2u); + expect(m, 0, EXC_ADEL, VECID_GENERAL, UMX_CODE_VA + 8); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].badvaddr, 0xFFFFFFFF80000000ull); + expect(m, 1, EXC_SYS, VECID_GENERAL, UMX_CODE_VA + 16); + if (cur_test_fails != before) { + con_printf(" [%s: address ", mode_names[m]); + con_hex64(*(volatile u64 *)((volatile char *)data_page() + 24)); + con_printf(" loaded %x]\n", data_page()[4]); + } + DUMP_IF_FAILED(m, before); + } +} + +/* A misaligned word load as the first instruction: Address Error, in every + * mode (nothing about it depends on the address space). */ +static void t_unaligned_load_first(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + umx_run(uc_load_first, uc_load_first_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_SKIP, + UMX_DATA_VA + 2, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 2u); + expect(m, 0, EXC_ADEL, VECID_GENERAL, UMX_CODE_VA); + CHECK_EQ_AT(mode_names[m], 0, umx.rec[0].badvaddr, UMX_DATA_VA + 2); + expect(m, 1, EXC_SYS, VECID_GENERAL, UMX_CODE_VA + 8); + DUMP_IF_FAILED(m, before); + } +} + +/* With UX set, 64-bit operations are legal in User mode. */ +static void t_dword_ops_with_ux(void) +{ + int m; + for (m = 0; m < N_MODES; m++) { + u32 before = cur_test_fails; + volatile u64 *d = (volatile u64 *)data_page(); + if (!(modes[m] & ST_UX)) continue; + d[0] = 0; + d[1] = 0; + SYNC(); + dcache_wb_invalidate_range(d, 16); + umx_run(uc_dword, uc_dword_end, UMX_CODE_VA, modes[m], 0, UMX_POLICY_SKIP, 0, 0); + CHECK_EQ_AT(mode_names[m], 0, umx.n, 1u); + expect(m, 0, EXC_SYS, VECID_GENERAL, UMX_CODE_VA + 24); + CHECK_EQ_AT(mode_names[m], 0, d[0], 1ull << 36); + CHECK_EQ_AT(mode_names[m], 0, d[1], 2ull << 36); + DUMP_IF_FAILED(m, before); + } +} + +static const struct test tests[] = { + TEST("umode/syscall_first", t_syscall_first, CPU_ALL), + TEST("umode/syscall_second", t_syscall_second, CPU_ALL), + TEST("umode/break_then_syscall", t_break_then_syscall, CPU_ALL), + TEST("umode/interrupt_then_sys", t_interrupt_then_syscall, CPU_ALL), + TEST("umode/load_miss_first", t_load_miss_first, CPU_ALL), + TEST("umode/load_miss_third", t_load_miss_third, CPU_ALL), + TEST("umode/fetch_miss_jump", t_fetch_miss_jump, CPU_ALL), + TEST("umode/fetch_miss_entry", t_fetch_miss_entry, CPU_ALL), + TEST("umode/kseg0_load_first", t_kseg0_load_first, CPU_ALL), + TEST("umode/kseg0_load_third", t_kseg0_load_third, CPU_ALL), + TEST("umode/unaligned_load_first", t_unaligned_load_first, CPU_ALL), + TEST("umode/dword_ops_with_ux", t_dword_ops_with_ux, CPU_ALL), +}; + +const struct test_group group_umode = { + "umode", tests, sizeof(tests) / sizeof(tests[0]) +}; From 0f17838da5ee6019d21a60c4a5b835c434ab69b1 Mon Sep 17 00:00:00 2001 From: Dani Sarfati Date: Wed, 30 Sep 2026 06:29:47 -0400 Subject: [PATCH 3/3] cpu-tests docs: the umode group and the later load tests README: `make ONLY=`, the umode row in "not yet run on silicon", what the group found on the FPGA core, and the tests/ layout line. docs/status.md: mem 24, tlb 11, umode 12, total 262, and which tests postdate the 2026-09-11 silicon run. Co-Authored-By: Claude Opus 5.5 --- cpu-tests/README.md | 16 +++++++++++++++- cpu-tests/docs/status.md | 11 +++++++---- 2 files changed, 22 insertions(+), 5 deletions(-) diff --git a/cpu-tests/README.md b/cpu-tests/README.md index 224a9552..546e913f 100644 --- a/cpu-tests/README.md +++ b/cpu-tests/README.md @@ -24,6 +24,10 @@ make # -> build/cputest.elf No root? `make toolchain-local` unpacks the same packages into `~/.local/opt`; the Makefile finds them automatically. +`make ONLY=group_umode` (any `group_*` from `harness/tests.c`) builds a suite of +`identity` plus that one group - for iterating on a slow target such as an HDL +simulator. Do a clean build when switching between it and the full suite. + ## Run ```sh @@ -138,6 +142,7 @@ yet. Each is plain architecture rather than a part-specific quirk: | `mem/load_then_load_evict` | the second of two loads evicts the first's D-cache line, or finds its own line just written back to memory | | `mem/load_then_store` | a store right behind a load, in the load's line and another, read back, in the same four cache states | | `tlb/load_then_mapped_load` | two loads where one or both walk the TLB, and the second's value is used at once | +| `umode/*` (12 tests) | User mode the way IRIX 6 runs it: n32 processes with `Status.UX` = 1 under a 32-bit kernel (KX = 0), so every exception and ERET switches addressing mode. Each test runs a few words at kuseg `0x00400000` under three Status settings - KX = SX = UX = 1, all clear (IRIX 5), and UX alone (IRIX 6) - and records every exception: vector, Cause, EPC, BadVAddr, Context, XContext, EntryHi. Syscalls as the first and second instruction after ERET (libc's `_getuid`), a break, a pending interrupt, load and fetch TLB refills (XTLB vector when UX = 1; fixed up and retried), KSEG0 and misaligned loads (AdEL), and 64-bit operations with UX set | The first results for them are from the `sgiindy_MiSTer` FPGA core presenting as an R4600: all pass (2409 checks over 250 tests in its simulator, both with every @@ -145,6 +150,15 @@ load stalling execute and with loads that stall only when they must, including behind another load or store). `mem/load_then_load` found that core releasing a stalled load's execute hold on the completion of the load ahead of it. +The `umode` group came out of IRIX 6.5's installer dying on that core with `init +died (why = 3, what = 0xb)`: init's saved frame showed a TLB miss at the general +exception vector's own address, reported on the `syscall` in `_getuid`. The core +built the vector address with the kernel's 32-bit mode and fetched it under the +process's UX = 1. Before its fix `umode/syscall_second` reproduced init's frame +exactly; after it, 2752 checks pass over 267 tests on the FPGA, all but +`umode/fetch_miss_entry` (the core still takes an instruction-TLB miss on an ERET +target as a nested exception). IRIS passes all 12. + ## Writing a test Add a function to the right file in `tests/`, register it in that file's table, @@ -185,7 +199,7 @@ and the CPU each did about it. ``` harness/ startup, exception vectors, CHECK macros, console -tests/ identity alu muldiv mem branch excep cp0 tlb fpu cache mips4 +tests/ identity alu muldiv mem branch excep cp0 tlb umode fpu cache mips4 gen/ fpvectors.py — computes the FP expectation tables run/ run-local.sh matrix.sh run-prom.sh bare.toml boot.toml docs/ findings gotchas status oracle memory-map toolchain diff --git a/cpu-tests/docs/status.md b/cpu-tests/docs/status.md index fe3a1087..e28bd657 100644 --- a/cpu-tests/docs/status.md +++ b/cpu-tests/docs/status.md @@ -16,20 +16,23 @@ | `identity` | 5 | PRId, FIR, cache geometry, Config.K0, TLB size | | `alu` | 29 | sign extension across the 32/64-bit boundary, overflow traps, shifts, logic, SLT | | `muldiv` | 18 | mult/div in both widths, HI/LO, the unspecified cases | -| `mem` | 21 | load/store widths, the whole unaligned family at every offset, alignment faults, KSEG0/KSEG1 | +| `mem` | 24 | load/store widths, the whole unaligned family at every offset, alignment faults, KSEG0/KSEG1, loads and stores right behind a load | | `branch` | 15 | every conditional, likely-nullification, link registers, delay slots, faults in delay slots | | `excep` | 17 | traps, reserved instructions, coprocessor usability, EXL/ERET, vector selection | | `cp0` | 21 | read-only registers, reserved-bit masks, 64-bit access, Count/Compare, LL/SC | -| `tlb` | 10 | entry round-trip over all 48, TLBP, every page size, real translation, V/D bits, ASIDs, refill | +| `tlb` | 11 | entry round-trip over all 48, TLBP, every page size, real translation, V/D bits, ASIDs, refill | +| `umode` | 12 | User mode under KX/SX/UX all set, all clear, and UX alone (IRIX 6 n32): syscalls, interrupts, TLB refill vectors and Context/XContext, address errors, 64-bit ops | | `fpu` | 89 | see below | | `cache` | 8 | geometry, tag round-trip, cached/uncached views, I-cache coherency | | `mips4` | 13 | every MIPS IV addition — computes on R5000, must raise RI on R4400 | -| **total** | **246** | | +| **total** | **262** | | Six of these (`excep/cp0_unusable_user`, `excep/cp0_usable_cu0`, `mem/load_then_use`, `mem/load_then_trap`, `mem/load_then_more`, `fpu/trap_behind_a_load`) were added after the 2026-09-11 run and are not in -its numbers; `identity/config_k0` was also re-enabled. +its numbers; `identity/config_k0` was also re-enabled. `mem/load_then_load`, +`mem/load_then_load_evict`, `mem/load_then_store`, `tlb/load_then_mapped_load` +and the whole `umode` group came later still and are not in them either. ### Inside `fpu`