diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml new file mode 100644 index 0000000..48dc1fe --- /dev/null +++ b/.github/workflows/check.yml @@ -0,0 +1,149 @@ +# Host-side checks for conquertron: everything that can be proved without a +# game dump or a Switch. +# +# recompiler builds recomp and its two verification tools against the +# pinned Capstone, with warnings shown +# shims runs hosttest/sync_harness.c -- the real event, mutex, +# semaphore and FS-completion shims, not a model of them -- +# normally and under ThreadSanitizer +# verify blaster's verify.sh: C programs compiled for PowerPC, +# recompiled, then built for the host and for ARM64 (run under +# QEMU) and compared with a native build of the same source +# +# None of this touches the retail RPX, and nothing here is uploaded as an +# artifact. The Switch build stays on the self-hosted runner in jouster. +# +# verify checks out Arkchemy/blaster. That works while blaster is public; if +# it is made private, add a read-only token as the BLASTER_TOKEN secret. +name: check + +on: + push: + paths-ignore: ['**.md', 'findings/**', 'docs/**'] + pull_request: + paths-ignore: ['**.md', 'findings/**', 'docs/**'] + workflow_dispatch: + +permissions: + contents: read + +env: + CAPSTONE_VERSION: 5.0.3 + ZIG_VERSION: 0.13.0 + +jobs: + recompiler: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + + - name: Cache Capstone + id: capstone + uses: actions/cache@v4 + with: + path: ~/devtools/capstone-install + key: capstone-${{ env.CAPSTONE_VERSION }}-ppc-${{ runner.os }} + + - name: Build Capstone ${{ env.CAPSTONE_VERSION }} (PowerPC only) + if: steps.capstone.outputs.cache-hit != 'true' + run: | + git clone -q --depth 1 --branch "$CAPSTONE_VERSION" https://github.com/capstone-engine/capstone.git /tmp/capstone + cmake -S /tmp/capstone -B /tmp/capstone/build -DCMAKE_BUILD_TYPE=Release \ + -DCAPSTONE_ARCHITECTURE_DEFAULT=OFF -DCAPSTONE_PPC_SUPPORT=ON \ + -DCAPSTONE_BUILD_TESTS=OFF -DCAPSTONE_BUILD_CSTOOL=OFF \ + -DCMAKE_INSTALL_PREFIX="$HOME/devtools/capstone-install" + cmake --build /tmp/capstone/build -j"$(nproc)" + cmake --install /tmp/capstone/build + + - name: Build recomp + run: | + cmake -S . -B build -DCMAKE_BUILD_TYPE=Release -DCAPSTONE_PREFIX="$HOME/devtools/capstone-install" + cmake --build build -j"$(nproc)" + + shims: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + + - name: Headers compile clean + run: | + printf '#include "ppc_runtime.h"\n#include "cafeos_coreinit_fs.h"\n#include "cafeos_coreinit_sync.h"\nvoid ppc_dispatch(PpcContext *c, uint32_t a) { (void)c; (void)a; }\n' > /tmp/tu.c + for cc in gcc clang; do + $cc -std=gnu11 -fsyntax-only -Wall -Wextra -Werror \ + -Wno-unused-function -Wno-unused-parameter -Wno-unused-variable \ + -I include /tmp/tu.c + done + + - name: sync_harness + run: | + gcc -O1 -g -w -I include hosttest/sync_harness.c include/cafeos_state.c \ + -o /tmp/sync_harness -lpthread -lm + # several runs: the cases are about races, and one pass proves little + for i in 1 2 3 4 5; do timeout 120 /tmp/sync_harness; done + + - name: sync_harness under ThreadSanitizer + run: | + gcc -O1 -g -w -fsanitize=thread -I include hosttest/sync_harness.c include/cafeos_state.c \ + -o /tmp/sync_harness_tsan -lpthread -lm + # The diagnostic counters (g_ark_sev_*, g_ark_wev_*, g_arkchemy_event_*) + # are unlocked on purpose. Anything else racing fails the job. + TSAN_OPTIONS="halt_on_error=0 exitcode=0" setarch "$(uname -m)" -R timeout 600 \ + /tmp/sync_harness_tsan 2> /tmp/tsan.txt + # Every report must be on one of those globals: count all reports, + # count the allowed ones, and fail on any difference -- a race on + # heap or struct memory has no "Location is global" line at all. + total=$(grep -c 'SUMMARY: ThreadSanitizer' /tmp/tsan.txt || true) + allowed=$(grep 'Location is global' /tmp/tsan.txt \ + | grep -cE "'(g_ark_sev_|g_ark_wev_|g_arkchemy_event_|g_case_done|limit)" || true) + echo "ThreadSanitizer: $total report(s), $allowed on diagnostic counters" + if [ "$total" != "$allowed" ]; then + echo "::error::ThreadSanitizer found a race outside the diagnostic counters" + cat /tmp/tsan.txt + exit 1 + fi + + verify: + runs-on: ubuntu-24.04 + needs: recompiler + steps: + - uses: actions/checkout@v4 + with: + path: conquertron + + - uses: actions/checkout@v4 + with: + repository: Arkchemy/blaster + path: blaster + token: ${{ secrets.BLASTER_TOKEN || github.token }} + + - name: Tools + run: | + sudo apt-get update -q + sudo apt-get install -y -q qemu-user-static + mkdir -p "$HOME/devtools" + curl -sSL "https://ziglang.org/download/${ZIG_VERSION}/zig-linux-x86_64-${ZIG_VERSION}.tar.xz" | tar -xJ -C "$HOME/devtools" + mv "$HOME/devtools/zig-linux-x86_64-${ZIG_VERSION}" "$HOME/devtools/zig" + + - name: Cache Capstone + id: capstone + uses: actions/cache@v4 + with: + path: ~/devtools/capstone-install + key: capstone-${{ env.CAPSTONE_VERSION }}-ppc-${{ runner.os }} + + - name: Build Capstone ${{ env.CAPSTONE_VERSION }} (PowerPC only) + if: steps.capstone.outputs.cache-hit != 'true' + run: | + git clone -q --depth 1 --branch "$CAPSTONE_VERSION" https://github.com/capstone-engine/capstone.git /tmp/capstone + cmake -S /tmp/capstone -B /tmp/capstone/build -DCMAKE_BUILD_TYPE=Release \ + -DCAPSTONE_ARCHITECTURE_DEFAULT=OFF -DCAPSTONE_PPC_SUPPORT=ON \ + -DCAPSTONE_BUILD_TESTS=OFF -DCAPSTONE_BUILD_CSTOOL=OFF \ + -DCMAKE_INSTALL_PREFIX="$HOME/devtools/capstone-install" + cmake --build /tmp/capstone/build -j"$(nproc)" + cmake --install /tmp/capstone/build + + - name: verify.sh + env: + CONQUERTRON: ${{ github.workspace }}/conquertron + QEMU_AARCH64: /usr/bin/qemu-aarch64-static + run: sh blaster/verify.sh diff --git a/ROADMAP.md b/ROADMAP.md index b76e971..9072747 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -30,16 +30,28 @@ scrutiny than the translator because they are hand-written and look simple. - [x] guest thread stacks reserve an EABI linkage area — a missing 16 bytes corrupted a heap and stalled boot for four sessions +- [x] OSEvent AUTO mode wakes exactly one waiter per signal, and a signal + landing mid-pump is kept rather than lost (2026-09-24, EVCREDIT; + pinned by seven sync_harness cases) +- [x] the FS completion queue is locked and the pumps claim atomically; the + unlocked queue lost and duplicated completions under contention + (2026-09-24). Next hardware run should say whether this moves the + loading stall -- judge it on `served/back`, not bytes - [ ] **audit every shim against real Cafe OS semantics**, especially anything that sets up guest register state: thread creation, TLS, callbacks, anything that fabricates a stack frame - [ ] `memset`/`memcpy` bypass `ppc_store_u32`, so they are invisible to the store watch. Either route them through it under a debug flag, or make that limitation impossible to forget. -- [ ] `setjmp`/`longjmp` are recompiled PowerPC. A `longjmp` restoring guest - registers cannot unwind the **host** call stack the recompiled code runs - on. Not yet implicated in a real bug, but it cannot work as written and - will matter to any code using it for error handling. +- [x] `setjmp`/`longjmp` are done on the host (2026-09-24). A recompiled + `longjmp` restored guest registers but could not unwind the host call + stack. Calls to them are now recognised by name and replaced -- the + host `setjmp` runs inline in the caller's own C function, and `longjmp` + jumps back to it with the guest register file restored. XenonRecomp's + approach, adapted to C. Pinned by blaster's verify.sh at guest -O0 and + -O1. **Unconfirmed against the retail binary:** the default names are + `setjmp`/`_setjmp`/`__setjmp` and their `longjmp` twins; if GHS's + libc spells them differently, pass `--setjmp-name`/`--longjmp-name` - [x] GX2 is no longer "shims that mostly record and discard". As of 2026-09-16 the path is real end to end: surfaces are sized and allocated, each one keeps its own deko3d image across re-binds, draws diff --git a/docs/prior-art-2026-09.md b/docs/prior-art-2026-09.md new file mode 100644 index 0000000..80ae338 --- /dev/null +++ b/docs/prior-art-2026-09.md @@ -0,0 +1,147 @@ +# Prior art, September 2026 update + +A follow-up to the "External prior art — useful repos" note (last edited +2026-09-12), which already covers igRewrite8, re_nsyshid, Texthead1, +spyrosadventure, hYdos, NefariousTechSupport, LG-RZ, decaf-emu, +kinnay/Nintendo-File-Formats, CafeGLSL and the GX2 shader samples. Nothing +from that page is repeated here. + +Same rule as that page: **the licence decides whether a project can be built +on or only read.** Arkchemy's own licence is permission-required, so anything +GPL can be studied but not copied in. + +Researched 2026-09-24. Star counts and commit counts are as of that day. + +## The headline: other people are statically recompiling Wii U code now + +When the prior-art page was written, the only other Skylanders +recompilation effort was an empty placeholder (`sky2015-recomp`). There are +now at least three Wii U static recompilers in public, one of them +running a real game. + +| Project | Licence | State | Worth | +| --- | --- | --- | --- | +| [BlackLineInteractive/nWiiURecomp](https://github.com/BlackLineInteractive/nWiiURecomp) | **GPL-3.0** | *Wind Waker HD* EU v0 "authenticates, maps its sections and relocations, initializes Cafe ABI state, and runs deterministically through cooperative startup". One validated title. 6★, 11 commits. | **Read, compare notes.** The closest project to conquertron that exists: RPX in, C++ out, an HLE Cafe OS runtime, and an Espresso interpreter as a fallback for code it did not compile. It says it runs "deterministically through cooperative startup" -- the same cooperative-scheduling territory the loading stall lives in. Spun out of [NWiiRecomp](https://github.com/BlackLineInteractive/NWiiRecomp) (GameCube/Wii, 25★, 226 commits). | +| [ApfelTeeSaft/RebrewU](https://github.com/ApfelTeeSaft/RebrewU) | none stated | RPX/RPL → C++. Function discovery, CFGs, jump-table detection; tools to inspect sections, relocations, imports, exports. 5★, 5 commits. | **Read only** (no licence = all rights reserved). Its jump-table detection is the one piece worth comparing against. | +| [chrissotraidis/DolRecomp](https://github.com/chrissotraidis/DolRecomp) | **GPL-3.0** | GameCube/Wii recompiler with an "experimental" RPX frontend its authors say is not maintained. 236 opcodes; *Luigi's Mansion* reaches its title screen. | **Read.** Its decoder, CPU-behaviour and RPX tests are a second opinion on instruction semantics -- useful for the instruction audit in ROADMAP.md, where every bug so far has been a silent one. | +| [hardkiller2565123123/Wii-U-Recomp](https://github.com/hardkiller2565123123/Wii-U-Recomp) | not yet | Announced, no source. | Watch. | + +None of these ports a game to the Switch, and none targets Skylanders or +Alchemy. They are peers, not overlap -- but the people behind nWiiURecomp in +particular are the ones most likely to have already met whatever +conquertron meets next. + +## The best permissively-licensed reference for conquertron itself + +### [hedge-dev/XenonRecomp](https://github.com/hedge-dev/XenonRecomp) — MIT + +The Xbox 360 recompiler behind the *Unleashed Recompiled* port. The Xbox +360's CPU is PowerPC, so most of the hard problems are the same ones, and +**it is MIT: it can be built on, with attribution.** Three things in it map +directly onto open items: + +- **`setjmp`/`longjmp`.** ROADMAP.md: "a `longjmp` restoring guest registers + cannot unwind the host call stack the recompiled code runs on." XenonRecomp's + answer is to not recompile them at all: calls to the guest's `setjmp`/ + `longjmp` are redirected to the host's own, with guest CPU state saved + alongside. That is the known-good shape for the fix here. +- **Jump tables.** It detects `mtctr` … `bctr` patterns and turns them into C + `switch` statements, with per-game TOML for the cases the pattern misses. + Its README is candid that there is "no fully generic solution". +- **Mid-assembly hooks.** Calls to host functions injected at chosen guest + addresses, with register arguments, and options to return or branch + afterwards. conquertron's probes (ark_blockprobe.h) are hand-rolled + versions of the same idea; a general mechanism would stop each probe + needing its own codegen touch. + +It also documents a subtlety conquertron should check: the FPU leaves +denormals alone while the vector unit flushes them. The Espresso has no VMX, +but it does have paired singles, which have their own rounding behaviour. + +## Cafe OS semantics: where to check a shim against + +### [cemu-project/Cemu](https://github.com/cemu-project/Cemu) — MPL-2.0 + +Already cited in the shim comments. Two findings from reading it on +2026-09-24: + +1. **The event fix is right.** `coreinit_Synchronization.cpp`: in AUTO mode, + `OSSignalEvent` latches if the wait queue is empty and otherwise calls + `wakeupSingleThreadWaitQueue` -- one thread. `OSSignalEventAll` latches if + empty and otherwise wakes the whole queue. `OSWaitEventWithTimeout` + returns false on timeout and true on a signal. That is exactly what the + EVCREDIT rewrite implements and what sync_harness now pins. +2. **Real FS callbacks run on dedicated threads.** `coreinit_FS.cpp` + comments that on hardware async FS completion processing "delegates the + processing to the AppIO threads" (Cemu itself runs it on its IPC thread). + conquertron runs callbacks cooperatively, on whichever thread next calls + an import that pumps -- which is why OSWaitEvent, OSWaitEventWithTimeout, + OSLockMutex and now OSWaitSemaphore all have to pump, and why the archive + pump exists at all. **A dedicated host thread that delivers completions, + the way AppIO does, is the architecture that would retire the pumps.** + Not attempted here: it changes which thread runs guest callbacks, which + is a hardware question. + + Cemu also queues FS commands by priority (`__FSQueueCmdByPriority`), so + completion order on hardware is not strictly submission order. + conquertron's FIFO queue is stricter than hardware, which is the safe + direction. + +MPL-2.0 is file-level copyleft: a file adapted from Cemu must stay MPL-2.0, +but it can sit in a project under another licence. Reading it to confirm +semantics, as the shims do, carries no obligation at all. + +### [devkitPro/wut](https://github.com/devkitPro/wut) — zlib + +The homebrew Wii U SDK. Its `include/coreinit/*.h` headers are the +signatures and structure layouts the shims already cite, and its +[generated docs](https://wut.devkitpro.org/group__coreinit__event.html) are +the quickest place to check a function's documented contract. zlib licence: +usable freely. + +### [WiiUBrew: Coreinit.rpl](https://wiiubrew.org/wiki/Coreinit.rpl) and [/dev/fsa](https://wiiubrew.org/wiki//dev/fsa) + +Community documentation of coreinit exports and the filesystem IPC device. + +## Tools for reading the binary + +| Project | Licence | Why | +| --- | --- | --- | +| [Maschell/GhidraRPXLoader](https://github.com/Maschell/GhidraRPXLoader) | GPL-3.0 (tool) | Opens `.rpx`/`.rpl` directly in Ghidra, with Espresso (paired-singles) processor definitions and a script that names imports. Using a GPL tool puts no obligation on what it is used to read. Every "read the instruction, not the name" lesson in the Loading Deadlock notes gets cheaper with a decompiler view next to the probes. | +| [wiiu-env/RPXParserLib](https://github.com/wiiu-env/RPXParserLib) | Apache-2.0 | A Java library that parses RPX/RPL: symbols, imports, exports. Usable, and a cross-check for `elf_loader.cpp`, particularly compressed sections and import resolution. | +| [BullyWiiPlaza/RPL-Studio](https://github.com/BullyWiiPlaza/RPL-Studio) | see repo | GUI pack/unpack for RPX/RPL. | + +## Paired singles: toolchain support is arriving + +- [llvm/llvm-project#211463](https://github.com/llvm/llvm-project/pull/211463) + adds a `ppc750cl` CPU and about 40 paired-single instructions to LLVM's + assembler and disassembler (not codegen yet). Open, approved in review, + last active 2026-09-24. **Once merged, clang -- and so the zig that + verify.sh uses -- can assemble paired-single test programs**, which would + let verify.sh check conquertron's `PPC_INS_ARKCHEMY_PS_*` handling the way + it already checks integer and float code. Today nothing does. +- [capstone#476](https://github.com/capstone-engine/capstone/issues/476): + Capstone still lacks paired singles, which is why conquertron decodes them + itself. + +## The Switch side + +| Project | Licence | Why | +| --- | --- | --- | +| [devkitPro/uam](https://github.com/devkitPro/uam) | see repo (mesa/nouveau-derived) | The offline GLSL → DKSH compiler blaster's shader path targets. | +| [averne/libuam](https://github.com/averne/libuam) | see repo | A library form of uam. If shaders ever have to be compiled when the game first binds them, rather than ahead of time by `build-shaders.sh`, this is the route. 1★. | + +## Suggested next steps, in order of payoff + +1. **Build and run the conquertron branch on hardware.** Judge it on + `served/back`, per the Loading Deadlock notes -- the FS queue race is the + first candidate cause for the stall that is a certain bug rather than a + theory. +2. **Port XenonRecomp's `setjmp`/`longjmp` approach** (MIT, with + attribution). It is a ROADMAP item with a known-good answer. +3. **Prototype an AppIO-style completion thread** behind a flag, and compare + it with the pumps on hardware. +4. **Contact the nWiiURecomp author.** Two projects doing the same thing to + the same OS will keep finding the same bugs. +5. **Watch LLVM #211463.** When it lands, add paired-single programs to + verify.sh. diff --git a/hosttest/README.md b/hosttest/README.md index e76c6a3..ad38826 100644 --- a/hosttest/README.md +++ b/hosttest/README.md @@ -22,10 +22,15 @@ silently returning a plausible value. ## sync_harness.c - gcc -O1 -g -w -I include -I ../jouster/game/include \ + gcc -O1 -g -w -I include \ hosttest/sync_harness.c include/cafeos_state.c \ -o /tmp/sync_harness -lpthread -lm && /tmp/sync_harness +Exits non-zero if any semantic check fails. Worth also running once under +ThreadSanitizer (`-fsanitize=thread`) after touching the shims; the only races +it should report are the unlocked diagnostic counters (`g_ark_sev_*`, +`g_ark_wev_*`, `g_arkchemy_event_*`), which are deliberately unsynchronised. + Drives the real event, mutex, semaphore and FS-completion shims -- not a model of them -- with a watchdog per case, so a deadlocked case is reported with the phase counters rather than hanging the run. @@ -52,6 +57,33 @@ On hardware this showed as the game thread stopping at an identical three console cycles narrowed it to "somewhere in OSSignalEvent"; the harness answered it in minutes and gave the blast radius for free. +## 2026-09-24: two more shim bugs, both on the loading path + +**AUTO events released every waiter.** Each parked waiter compared one +per-event epoch that every signal bumped, so a single `OSSignalEvent` on an +AUTO event released all of them, and a second signal landing before the first +woken waiter ran was lost outright. EVSTATE had measured two waiters parked on +the loading event, so this was live. Replaced by a count of wake credits; +see "EVCREDIT" in `cafeos_coreinit_sync.h`. Three new cases fail on the old +code: one signal releasing both of two waiters, a second signal vanishing, +and two timed waiters both being told TRUE. + +**The FS completion queue had no lock.** It is filled by whichever thread +issues a read and drained by whichever thread next pumps, and the pump's +re-entrancy flag was a plain int two threads could both see clear. Four +producers and four pumpers, 16,000 completions: the unlocked queue delivered +one completion twice, then wrapped its count to 4,294,967,295 -- after which +it reads as permanently full and every later completion is dropped. A +completion that never runs is a work item that never reaches status 2 and a +block that is never released, which is the shape of the loading stall. +Whether it is *the* cause is for a hardware run to say; the race itself is +not in doubt. The queue now has a mutex, and both pump guards are atomic +claims. + +The Sep 16 hardware log reports `skip=0` and `dropped=0` on every line, so +neither the message-queue completion gap nor a full queue was losing reads in +that run. + ## Result so far `tlsf_create` builds the full 5 MB arena and `tlsf_memalign` is correct in diff --git a/hosttest/sync_harness.c b/hosttest/sync_harness.c index 7c7c186..456baad 100644 --- a/hosttest/sync_harness.c +++ b/hosttest/sync_harness.c @@ -24,6 +24,7 @@ */ #include +#include #include #include #include @@ -97,11 +98,12 @@ static void begin(const char *name, double limit_s) pthread_t th; pthread_create(&th, NULL, watchdog, &limit); pthread_detach(th); - printf(" %-52s", name); + printf(" %-60s", name); fflush(stdout); } -static void pass(void) { g_case_done = 1; printf("ok\n"); } +static int g_case_failed; +static void pass(void) { g_case_done = 1; printf(g_case_failed ? "FAILED\n" : "ok\n"); g_case_failed = 0; } /* ---- the event under test --------------------------------------------- */ @@ -129,11 +131,7 @@ static void queue_one_completion(PpcContext *ctx, uint32_t event_addr) ppc_store_u32(ctx, ad + 0, 0xdead0000u); ppc_store_u32(ctx, ad + 4, 0u); ppc_store_u32(ctx, ad + 8, 0u); - if (g_arkchemy_fs_pending_n < ARKCHEMY_FS_ASYNC_QUEUE) { - ArkchemyFsPending *q = &g_arkchemy_fs_pending[g_arkchemy_fs_pending_n++]; - q->async_data = ad; q->client = 0; q->block = 0; q->status = 1; - g_arkchemy_fs_queued++; - } + arkchemy_fs_enqueue(ad, 0, 0, 1); } /* ---- cases ------------------------------------------------------------- */ @@ -291,6 +289,375 @@ static void case_wait_with_timeout_keeps_its_timeout(void) free(ctx); } + +/* ---- AUTO / MANUAL semantics --------------------------------------------- + * + * What Cafe OS guarantees, and what the loading stall depends on: + * + * AUTO OSSignalEvent wakes exactly ONE queued waiter. With nobody + * queued it latches, and the next waiter consumes the latch. + * OSSignalEventAll wakes every queued waiter and latches nothing. + * MANUAL a signal releases everyone and stays set until OSResetEvent. + * + * Before 2026-09-24 the shim broke the AUTO rule twice over. A signal bumped + * one per-event epoch that every parked waiter was comparing against, so one + * signal released ALL of them. And a second signal arriving before the + * first woken waiter had run found waiting_count still non-zero, bumped the + * epoch again and latched nothing -- so it was lost. + * + * These cases fail on that code and pass on the credit scheme. Each uses its + * own event address so a failure cannot leak into the next case. */ + +static int g_semantic_failures; + +static void expect(int ok, const char *what) +{ + if (!ok) { g_semantic_failures++; g_case_failed = 1; fprintf(stderr, "\n * %s ", what); } +} + +static void sleep_ms(long ms) +{ + struct timespec t = { ms / 1000, (ms % 1000) * 1000000L }; + nanosleep(&t, NULL); +} + +static int waiting_on(uint32_t addr) +{ + ArkchemyEventEntry *e = arkchemy_event_get(addr, 0, 1); + pthread_mutex_lock(&e->lock); + int n = e->waiting_count; + pthread_mutex_unlock(&e->lock); + return n; +} + +static void wait_until_waiting(uint32_t addr, int n) +{ + for (int i = 0; i < 400 && waiting_on(addr) < n; i++) sleep_ms(5); +} + +static void init_event(uint32_t addr, int value, int mode) +{ + PpcContext *c = make_ctx(); + c->r[3] = addr; c->r[4] = (uint32_t)value; c->r[5] = (uint32_t)mode; + ppc_import_coreinit_OSInitEvent(c); + free(c); +} + +static void signal_event(uint32_t addr) +{ + PpcContext *c = make_ctx(); + c->r[3] = addr; + ppc_import_coreinit_OSSignalEvent(c); + free(c); +} + +static void signal_event_all(uint32_t addr) +{ + PpcContext *c = make_ctx(); + c->r[3] = addr; + ppc_import_coreinit_OSSignalEventAll(c); + free(c); +} + +static void reset_event(uint32_t addr) +{ + PpcContext *c = make_ctx(); + c->r[3] = addr; + ppc_import_coreinit_OSResetEvent(c); + free(c); +} + +/* A timed wait from the calling thread; returns the shim's BOOL. Timeout in + * milliseconds, converted to the 62.15625 MHz timer the shim expects. */ +static int timed_wait(uint32_t addr, long ms) +{ + PpcContext *c = make_ctx(); + int64_t ticks = (int64_t)ms * (ARKCHEMY_ESPRESSO_TIMER_CLOCK / 1000); + c->r[3] = addr; + c->r[5] = (uint32_t)((uint64_t)ticks >> 32); + c->r[6] = (uint32_t)ticks; + ppc_import_coreinit_OSWaitEventWithTimeout(c); + int r = (int)c->r[3]; + free(c); + return r; +} + +typedef struct { uint32_t addr; long timeout_ms; volatile int done; volatile int result; } Waiter; + +static void *untimed_waiter(void *arg) +{ + Waiter *w = arg; + PpcContext *c = make_ctx(); + c->r[3] = w->addr; + ppc_import_coreinit_OSWaitEvent(c); + free(c); + w->result = 1; + __atomic_store_n(&w->done, 1, __ATOMIC_SEQ_CST); + return NULL; +} + +static void *timed_waiter(void *arg) +{ + Waiter *w = arg; + w->result = timed_wait(w->addr, w->timeout_ms); + __atomic_store_n(&w->done, 1, __ATOMIC_SEQ_CST); + return NULL; +} + +static int count_done(Waiter *w, int n) +{ + int k = 0; + for (int i = 0; i < n; i++) k += __atomic_load_n(&w[i].done, __ATOMIC_SEQ_CST); + return k; +} + +#define EV_AUTO_ONE 0x3000u +#define EV_AUTO_TWICE 0x3100u +#define EV_AUTO_TIMED 0x3200u +#define EV_MANUAL 0x3300u +#define EV_AUTO_ALL 0x3400u +#define EV_RESET 0x3500u +#define EV_REINIT 0x3600u + +/* The stall's own shape: two threads parked on one AUTO event, one signal. */ +static void case_auto_signal_wakes_exactly_one(void) +{ + begin("AUTO: one signal wakes exactly one of two waiters", 10.0); + init_event(EV_AUTO_ONE, 0, 1); + Waiter w[2] = {{EV_AUTO_ONE}, {EV_AUTO_ONE}}; + pthread_t th[2]; + for (int i = 0; i < 2; i++) pthread_create(&th[i], NULL, untimed_waiter, &w[i]); + wait_until_waiting(EV_AUTO_ONE, 2); + signal_event(EV_AUTO_ONE); + sleep_ms(150); + expect(count_done(w, 2) == 1, "one signal released both waiters (or neither)"); + signal_event(EV_AUTO_ONE); /* releases the other */ + for (int i = 0; i < 2; i++) pthread_join(th[i], NULL); + pass(); +} + +/* Two signals before the woken waiter has run: the first wakes it, the + * second must latch rather than vanish. */ +static void case_auto_second_signal_latches(void) +{ + begin("AUTO: a second signal while the first is in flight latches", 10.0); + init_event(EV_AUTO_TWICE, 0, 1); + Waiter w = {EV_AUTO_TWICE}; + pthread_t th; + pthread_create(&th, NULL, untimed_waiter, &w); + wait_until_waiting(EV_AUTO_TWICE, 1); + signal_event(EV_AUTO_TWICE); + signal_event(EV_AUTO_TWICE); + pthread_join(th, NULL); + expect(timed_wait(EV_AUTO_TWICE, 50) == 1, "the second signal was lost"); + expect(timed_wait(EV_AUTO_TWICE, 50) == 0, "the latch was consumed twice"); + pass(); +} + +/* The timeout variant, which the job-queue workers park in: one signal must + * return TRUE to exactly one of them, and the other must time out FALSE. */ +static void case_auto_timed_waiters_one_true(void) +{ + begin("AUTO timed: one signal, one TRUE and one FALSE", 10.0); + init_event(EV_AUTO_TIMED, 0, 1); + Waiter w[2] = {{EV_AUTO_TIMED, 400}, {EV_AUTO_TIMED, 400}}; + pthread_t th[2]; + for (int i = 0; i < 2; i++) pthread_create(&th[i], NULL, timed_waiter, &w[i]); + wait_until_waiting(EV_AUTO_TIMED, 2); + signal_event(EV_AUTO_TIMED); + for (int i = 0; i < 2; i++) pthread_join(th[i], NULL); + expect(w[0].result + w[1].result == 1, "both or neither reported being signalled"); + expect(waiting_on(EV_AUTO_TIMED) == 0, "a timed-out waiter was left counted"); + pass(); +} + +static void case_manual_releases_all_and_stays_set(void) +{ + begin("MANUAL: releases every waiter and stays set", 10.0); + init_event(EV_MANUAL, 0, 0); + Waiter w[3] = {{EV_MANUAL}, {EV_MANUAL}, {EV_MANUAL}}; + pthread_t th[3]; + for (int i = 0; i < 3; i++) pthread_create(&th[i], NULL, untimed_waiter, &w[i]); + wait_until_waiting(EV_MANUAL, 3); + signal_event(EV_MANUAL); + for (int i = 0; i < 3; i++) pthread_join(th[i], NULL); + expect(timed_wait(EV_MANUAL, 50) == 1, "MANUAL did not stay set"); + expect(timed_wait(EV_MANUAL, 50) == 1, "MANUAL was consumed like AUTO"); + reset_event(EV_MANUAL); + expect(timed_wait(EV_MANUAL, 50) == 0, "OSResetEvent did not clear MANUAL"); + pass(); +} + +static void case_auto_signal_all_wakes_all_latches_nothing(void) +{ + begin("AUTO: SignalEventAll wakes all, latches nothing", 10.0); + init_event(EV_AUTO_ALL, 0, 1); + Waiter w[3] = {{EV_AUTO_ALL}, {EV_AUTO_ALL}, {EV_AUTO_ALL}}; + pthread_t th[3]; + for (int i = 0; i < 3; i++) pthread_create(&th[i], NULL, untimed_waiter, &w[i]); + wait_until_waiting(EV_AUTO_ALL, 3); + signal_event_all(EV_AUTO_ALL); + for (int i = 0; i < 3; i++) pthread_join(th[i], NULL); + expect(timed_wait(EV_AUTO_ALL, 50) == 0, "SignalEventAll with waiters left a latch"); + pass(); +} + +static void case_reset_forgets_a_latched_signal(void) +{ + begin("OSResetEvent forgets a latched AUTO signal", 10.0); + init_event(EV_RESET, 0, 1); + signal_event(EV_RESET); /* nobody waiting: latches */ + reset_event(EV_RESET); + expect(timed_wait(EV_RESET, 50) == 0, "a reset latch still woke a waiter"); + pass(); +} + +/* Re-initialising an event with a waiter parked on it is undefined on real + * hardware, but must not wedge the shim: the waiter is released, and the + * event behaves as freshly initialised afterwards. */ +static void case_reinit_releases_waiters(void) +{ + begin("OSInitEvent on a live event releases its waiters", 10.0); + init_event(EV_REINIT, 0, 1); + Waiter w = {EV_REINIT}; + pthread_t th; + pthread_create(&th, NULL, untimed_waiter, &w); + wait_until_waiting(EV_REINIT, 1); + init_event(EV_REINIT, 0, 1); + pthread_join(th, NULL); + expect(waiting_on(EV_REINIT) == 0, "re-init left a waiter counted"); + expect(timed_wait(EV_REINIT, 50) == 0, "re-init left the event signalled"); + pass(); +} + +/* ---- the completion queue under contention --------------------------------- + * + * The FS completion queue is written by the thread that issues a read and + * drained by whichever thread next pumps -- the game thread, a job-queue + * worker in OSWaitEventWithTimeout, an FMOD thread in OSLockMutex. Until + * 2026-09-24 it had no lock, and the pump's re-entrancy flag was a plain int + * two threads could both see clear. A completion taken twice, or shifted out + * from under a concurrent enqueue, is a read whose callback never runs: its + * work item never reaches status 2 and its block is never released. + * + * This hammers the queue from several producers and several pumpers and + * counts every delivery by id. Each completion must arrive exactly once. */ + +#define STRESS_PRODUCERS 4 +#define STRESS_PER_PRODUCER 4000 +#define STRESS_PUMPERS 4 +#define STRESS_TOTAL (STRESS_PRODUCERS * STRESS_PER_PRODUCER) + +static volatile uint32_t g_stress_seen[STRESS_TOTAL]; +static volatile int g_stress_stop; + +/* The id rides in `client`; the callback records it. */ +static void count_delivery(PpcContext *ctx, uint32_t addr) +{ + (void)addr; + uint32_t id = ctx->r[3]; + if (id < STRESS_TOTAL) __atomic_add_fetch(&g_stress_seen[id], 1, __ATOMIC_SEQ_CST); +} + +static void *stress_producer(void *arg) +{ + uint32_t base = (uint32_t)(uintptr_t)arg * STRESS_PER_PRODUCER; + for (uint32_t i = 0; i < STRESS_PER_PRODUCER; i++) { + /* a full queue is back-pressure, not loss: retry until it takes */ + while (!arkchemy_fs_enqueue(0x8000u, base + i, 0, 1)) sched_yield(); + } + return NULL; +} + +static void *stress_pumper(void *arg) +{ + PpcContext *c = make_ctx(); + (void)arg; + while (!__atomic_load_n(&g_stress_stop, __ATOMIC_SEQ_CST)) arkchemy_fs_pump_completions(c); + arkchemy_fs_pump_completions(c); + free(c); + return NULL; +} + +static void case_queue_delivers_each_completion_once(void) +{ + begin("FS queue: every completion delivered exactly once", 30.0); + PpcContext *c = make_ctx(); + ppc_store_u32(c, 0x8000u + 0, 0xdead0000u); /* non-zero callback */ + ppc_store_u32(c, 0x8000u + 4, 0u); + free(c); + g_on_dispatch = count_delivery; + g_stress_stop = 0; + pthread_t prod[STRESS_PRODUCERS], pump[STRESS_PUMPERS]; + for (int i = 0; i < STRESS_PUMPERS; i++) pthread_create(&pump[i], NULL, stress_pumper, NULL); + for (int i = 0; i < STRESS_PRODUCERS; i++) pthread_create(&prod[i], NULL, stress_producer, (void *)(uintptr_t)i); + for (int i = 0; i < STRESS_PRODUCERS; i++) pthread_join(prod[i], NULL); + __atomic_store_n(&g_stress_stop, 1, __ATOMIC_SEQ_CST); + for (int i = 0; i < STRESS_PUMPERS; i++) pthread_join(pump[i], NULL); + PpcContext *d = make_ctx(); + arkchemy_fs_pump_completions(d); /* anything left behind */ + free(d); + int lost = 0, doubled = 0; + for (int i = 0; i < STRESS_TOTAL; i++) { + if (g_stress_seen[i] == 0) lost++; + if (g_stress_seen[i] > 1) doubled++; + } + if (lost || doubled) { + char msg[128]; + snprintf(msg, sizeof msg, "%d of %d lost, %d delivered more than once", lost, STRESS_TOTAL, doubled); + expect(0, msg); + } + g_on_dispatch = NULL; + pass(); +} + +/* ---- a semaphore waiter must pump too -------------------------------------- + * + * OSWaitSemaphore used to pump once on entry and then park on the condvar + * for good. That is the same starvation OSWaitEvent was cured of on + * 2026-09-12: if the release it waits for comes from a completion queued + * AFTER it parked, and no other thread happens to pump, nothing delivers it. + * The loading path takes a count-1 semaphore (SEMINIT, 0x4503598), so this + * is not a hypothetical shape. */ + +#define SEM_A 0x5000u + +static void completion_signals_semaphore(PpcContext *ctx, uint32_t addr) +{ + (void)addr; + uint32_t saved = ctx->r[3]; + ctx->r[3] = SEM_A; + ppc_import_coreinit_OSSignalSemaphore(ctx); + ctx->r[3] = saved; +} + +static void *late_semaphore_completion(void *arg) +{ + (void)arg; + sleep_ms(100); + g_on_dispatch = completion_signals_semaphore; + arkchemy_fs_enqueue(0x8000u, 0, 0, 1); + return NULL; +} + +static void case_semaphore_waiter_pumps_its_own_rescue(void) +{ + begin("OSWaitSemaphore pumps the completion that releases it", 5.0); + PpcContext *c = make_ctx(); + ppc_store_u32(c, 0x8000u + 0, 0xdead0000u); + c->r[3] = SEM_A; c->r[4] = 0; + ppc_import_coreinit_OSInitSemaphore(c); + pthread_t th; + pthread_create(&th, NULL, late_semaphore_completion, NULL); + c->r[3] = SEM_A; + ppc_import_coreinit_OSWaitSemaphore(c); + pthread_join(th, NULL); + expect(c->r[3] == 1, "OSWaitSemaphore did not return the previous count"); + g_on_dispatch = NULL; + free(c); + pass(); +} + int main(void) { printf("sync harness -- the real shims, no Switch\n\n"); @@ -301,6 +668,16 @@ int main(void) case_wait_for_already_queued_completion(); case_signal_from_another_thread_wakes_waiter(); case_waiter_pumps_its_own_rescue(); + printf("\n"); + case_auto_signal_wakes_exactly_one(); + case_auto_second_signal_latches(); + case_auto_timed_waiters_one_true(); + case_manual_releases_all_and_stays_set(); + case_auto_signal_all_wakes_all_latches_nothing(); + case_reset_forgets_a_latched_signal(); + case_reinit_releases_waiters(); + case_queue_delivers_each_completion_once(); + case_semaphore_waiter_pumps_its_own_rescue(); printf("\nOSSignalEvent enter=%u entry=%u lock=%u exit=%u\n", g_ark_sev_enter, g_ark_sev_got_entry, g_ark_sev_got_lock, g_ark_sev_exit); printf("OSWaitEvent enter=%u entry=%u lock=%u parked=%u slices=%u exit=%u\n", @@ -308,6 +685,10 @@ int main(void) g_ark_wev_parked, g_ark_wev_slices, g_ark_wev_exit); printf("fs pump queued=%u delivered=%u pending=%u\n", g_arkchemy_fs_queued, g_arkchemy_fs_delivered, g_arkchemy_fs_pending_n); + if (g_semantic_failures) { + printf("\n%d semantic check(s) FAILED\n", g_semantic_failures); + return 1; + } printf("\nall cases completed\n"); return 0; } diff --git a/include/cafeos_coreinit_fs.h b/include/cafeos_coreinit_fs.h index 2456f70..b2c4f84 100644 --- a/include/cafeos_coreinit_fs.h +++ b/include/cafeos_coreinit_fs.h @@ -10,6 +10,7 @@ #include #include "ppc_runtime.h" +#include /* * Phase 1d CafeOS runtime shim -- coreinit's high-level filesystem API @@ -469,6 +470,53 @@ volatile uint32_t g_arkchemy_fs_dropped = 0; __attribute__((weak)) #endif volatile int g_arkchemy_fs_pumping = 0; +/* Guards g_arkchemy_fs_pending and g_arkchemy_fs_pending_n. Added + * 2026-09-24: the queue is written by whichever thread issues a read and + * drained by whichever thread next pumps, and had no lock at all. Under + * contention in sync_harness.c the unlocked version delivered a completion + * twice and then wrapped the count to 4294967295, after which the queue + * read as permanently full and every later completion would have been + * dropped. Held only to push or pop one entry -- never across a callback, + * which runs guest code that can issue another read. */ +#ifdef __GNUC__ +__attribute__((weak)) +#endif +pthread_mutex_t g_arkchemy_fs_queue_lock = PTHREAD_MUTEX_INITIALIZER; + +/* Queue one completion. Returns 0 if the queue is full; the caller counts + * that as a drop, loudly. */ +static inline int arkchemy_fs_enqueue(uint32_t async_data, uint32_t client, uint32_t block, int32_t status) { + int ok = 0; + pthread_mutex_lock(&g_arkchemy_fs_queue_lock); + if (g_arkchemy_fs_pending_n < ARKCHEMY_FS_ASYNC_QUEUE) { + ArkchemyFsPending *q = &g_arkchemy_fs_pending[g_arkchemy_fs_pending_n]; + q->async_data = async_data; q->client = client; q->block = block; q->status = status; + /* atomic only so the pump's lock-free "anything queued?" peek is + * well defined; the lock is what orders the queue itself */ + __atomic_store_n(&g_arkchemy_fs_pending_n, g_arkchemy_fs_pending_n + 1, __ATOMIC_RELEASE); + g_arkchemy_fs_queued++; + ok = 1; + } + pthread_mutex_unlock(&g_arkchemy_fs_queue_lock); + return ok; +} + +/* Take the oldest completion, if any. FIFO, as before: the loader's state + * machine expects its reads to complete in the order it issued them. */ +static inline int arkchemy_fs_dequeue(ArkchemyFsPending *out) { + int ok = 0; + pthread_mutex_lock(&g_arkchemy_fs_queue_lock); + if (g_arkchemy_fs_pending_n > 0) { + *out = g_arkchemy_fs_pending[0]; + for (uint32_t k = 1; k < g_arkchemy_fs_pending_n; k++) + g_arkchemy_fs_pending[k-1] = g_arkchemy_fs_pending[k]; + __atomic_store_n(&g_arkchemy_fs_pending_n, g_arkchemy_fs_pending_n - 1, __ATOMIC_RELEASE); + g_arkchemy_fs_delivered++; + ok = 1; + } + pthread_mutex_unlock(&g_arkchemy_fs_queue_lock); + return ok; +} static inline void ppc_import_coreinit_FSOpenFile(PpcContext *ctx) { char guest_path[512], real_path[512], mode[8]; @@ -830,10 +878,18 @@ static inline void ppc_fs_invoke_async_callback(PpcContext *ctx, uint32_t async_ /* Deliver any queued FS completions. Safe to call from any import site: it is * between guest instructions, and the function that issued the read has * already returned. Re-entrancy guarded, because a completion callback may - * itself call into a shim that pumps. */ + * itself call into a shim that pumps. + * + * One pumper at a time, across all threads, so completions are delivered in + * the order they were queued. The guard used to be a plain test-then-set + * that two threads could both pass; it is now an atomic claim. A thread + * that loses the claim returns at once -- whoever holds it will deliver + * what is queued, and every waiter here loops on a 1ms slice anyway. */ static inline void arkchemy_fs_pump_completions(PpcContext *ctx) { - if (g_arkchemy_fs_pending_n == 0 || g_arkchemy_fs_pumping) return; - g_arkchemy_fs_pumping = 1; + int unclaimed = 0; + if (__atomic_load_n(&g_arkchemy_fs_pending_n, __ATOMIC_ACQUIRE) == 0) return; + if (!__atomic_compare_exchange_n(&g_arkchemy_fs_pumping, &unclaimed, 1, 0, + __ATOMIC_ACQ_REL, __ATOMIC_ACQUIRE)) return; /* The pump borrows the caller's context to run a guest callback, and * ppc_fs_invoke_async_callback sets r3-r6 to that callback's arguments. * Unless they are put back, the pump silently eats the argument of @@ -856,17 +912,12 @@ static inline void arkchemy_fs_pump_completions(PpcContext *ctx) { uint32_t saved_r[10]; for (int i = 0; i < 10; i++) saved_r[i] = ctx->r[3 + i]; /* r3..r12 */ uint32_t saved_lr = ctx->lr; - while (g_arkchemy_fs_pending_n > 0) { - ArkchemyFsPending p = g_arkchemy_fs_pending[0]; - for (uint32_t k = 1; k < g_arkchemy_fs_pending_n; k++) - g_arkchemy_fs_pending[k-1] = g_arkchemy_fs_pending[k]; - g_arkchemy_fs_pending_n--; - g_arkchemy_fs_delivered++; + ArkchemyFsPending p; + while (arkchemy_fs_dequeue(&p)) ppc_fs_invoke_async_callback(ctx, p.async_data, p.client, p.block, p.status); - } for (int i = 0; i < 10; i++) ctx->r[3 + i] = saved_r[i]; ctx->lr = saved_lr; - g_arkchemy_fs_pumping = 0; + __atomic_store_n(&g_arkchemy_fs_pumping, 0, __ATOMIC_RELEASE); } /* Drive the guest's archive system one step, from inside a cooperative wait. @@ -918,15 +969,21 @@ static inline void arkchemy_archive_pump(PpcContext *ctx) { uint32_t saved_lr; int i; if (!g_arkchemy_archive_pump_enabled) return; - if (g_arkchemy_archive_pumping) return; - g_arkchemy_archive_pumping = 1; + { + /* An atomic claim, for the same reason as the completion pump: this + * runs guest code (updateArchiveSystem), and two threads that both + * saw the old plain flag clear would have run it at once. */ + uint32_t unclaimed = 0; + if (!__atomic_compare_exchange_n(&g_arkchemy_archive_pumping, &unclaimed, 1u, 0, + __ATOMIC_ACQ_REL, __ATOMIC_ACQUIRE)) return; + } for (i = 0; i < 10; i++) saved_r[i] = ctx->r[3 + i]; /* r3..r12 */ saved_lr = ctx->lr; g_arkchemy_archive_pumps++; ppc_dispatch(ctx, ARKCHEMY_ARCHIVE_PUMP_ADDR); for (i = 0; i < 10; i++) ctx->r[3 + i] = saved_r[i]; ctx->lr = saved_lr; - g_arkchemy_archive_pumping = 0; + __atomic_store_n(&g_arkchemy_archive_pumping, 0u, __ATOMIC_RELEASE); } static inline void ppc_import_coreinit_FSReadFileWithPosAsync(PpcContext *ctx) { @@ -968,13 +1025,8 @@ static inline void ppc_import_coreinit_FSReadFileWithPosAsync(PpcContext *ctx) { ark_fst_end(ark_now_ns()); /* see FSTIME: the read itself is done here */ g_arkchemy_fs_last_result = result; for (int w = 0; w < 4; w++) g_arkchemy_fs_head[w] = ppc_load_u32(ctx, buffer_addr + w * 4); - if (g_arkchemy_fs_pending_n < ARKCHEMY_FS_ASYNC_QUEUE) { - ArkchemyFsPending *q = &g_arkchemy_fs_pending[g_arkchemy_fs_pending_n++]; - q->async_data = async_data_addr; q->client = client; q->block = block; q->status = result; - g_arkchemy_fs_queued++; - } else { + if (!arkchemy_fs_enqueue(async_data_addr, client, block, result)) g_arkchemy_fs_dropped++; /* never silently: reported every frame */ - } ctx->r[3] = (uint32_t)ARKCHEMY_FS_STATUS_OK; } diff --git a/include/cafeos_coreinit_sync.h b/include/cafeos_coreinit_sync.h index 6aa1412..2588f7a 100644 --- a/include/cafeos_coreinit_sync.h +++ b/include/cafeos_coreinit_sync.h @@ -63,18 +63,11 @@ enum { * matches this exactly with no extra bookkeeping needed. * * OSEvent's manual/auto-reset semantics (confirmed against wut's docs - * and Cemu's real branching logic) are implemented with the standard, - * race-free mutex+condvar+flag+epoch pattern: `signaled` is a sticky - * flag for "signaled with nobody currently waiting" (persists for the - * next waiter), `epoch` is bumped on every broadcast-style wake so - * every *currently* blocked waiter reliably wakes exactly once without - * a shared single-use flag racing between them (needed because real - * hardware's explicit per-thread wake queue doesn't have a direct - * analog in POSIX condvars, which always require a recheck loop). - * `waiting_count` tracks whether anyone is currently blocked, to - * replicate the real "empty queue -> stays signaled" vs. "non-empty - * queue -> wakes waiters, doesn't persist" branch Cemu's source shows - * for both OSSignalEvent and OSSignalEventAll in AUTO mode. + * and Cemu's real branching logic) are implemented with a mutex, a + * condvar, a sticky `signaled` latch and a count of wake credits granted + * to parked waiters -- see "EVCREDIT" above ppc_import_coreinit_OSInitEvent + * for why a credit count and not an epoch, which is what this used until + * 2026-09-24 and which released every AUTO waiter on a single signal. */ /* Was 64. An engine of this size uses far more than 64 OSMutexes -- @@ -182,8 +175,12 @@ typedef struct { pthread_cond_t cond; int signaled; int mode; /* 0 = OS_EVENT_MODE_MANUAL, 1 = OS_EVENT_MODE_AUTO */ - uint64_t epoch; + uint64_t epoch; /* diagnostic only: signals that reached a parked waiter */ int waiting_count; + /* Added at the end so the positional initializer of the fallback entry + * in cafeos_state.c stays valid. See "EVCREDIT" below. */ + int wake_credits; /* wakes granted to parked waiters, not yet taken */ + uint64_t init_epoch; /* moves only in OSInitEvent */ } ArkchemyEventEntry; /* Real definitions in cafeos_state.c -- see its own file comment. */ extern ArkchemyEventEntry g_arkchemy_events[ARKCHEMY_SYNC_TABLE_SIZE]; @@ -212,6 +209,8 @@ static inline ArkchemyEventEntry *arkchemy_event_get(uint32_t addr, int create_w e->mode = create_with_mode; e->epoch = 0; e->waiting_count = 0; + e->wake_credits = 0; + e->init_epoch = 0; result = e; } if (result == NULL) { /* see arkchemy_mutex_get */ @@ -222,6 +221,100 @@ static inline ArkchemyEventEntry *arkchemy_event_get(uint32_t addr, int create_w return result; } +/* EVCREDIT -- how a signal reaches a waiter. Rewritten 2026-09-24. + * + * What Cafe OS does, from wut's event.h and Cemu's coreinit_Synchronization: + * + * AUTO OSSignalEvent wakes exactly one queued waiter. With nobody + * queued the event latches, and the next waiter consumes the latch. + * OSSignalEventAll wakes every queued waiter and latches nothing. + * MANUAL a signal releases every waiter and stays set until OSResetEvent. + * + * What this shim did before: every parked waiter remembered one per-event + * epoch, every signal bumped it, and the wait loop ran while + * `epoch == my_epoch`. So on an AUTO event + * + * - one signal released EVERY parked waiter, not one of them. EVSTATE + * measured two waiters parked on the loading event, so this was live; + * - a second signal arriving before the first woken waiter had run still + * saw waiting_count > 0, bumped the epoch again and latched nothing, so + * it was lost. The window is wide here, because a waiter drops the lock + * to run the pumps while staying counted -- deliberately, since the pump + * runs guest code that signals this very event. + * + * Both are ours, not the engine's, which is why this is a fix rather than a + * workaround behind a flag. sync_harness.c pins all of it: three of its + * AUTO cases fail on the old code. + * + * The scheme. `wake_credits` counts wakes granted to parked waiters and not + * yet taken, and never exceeds `waiting_count`. An AUTO signal grants one + * more credit if some waiter has not been granted one, and otherwise latches + * `signaled` -- which is exactly "wake one queued waiter, or latch if the + * queue is empty", with "queue" meaning waiters not already on their way + * out. The credit lives in the struct, so a signal that lands while its + * waiter is off running the pump is still there when it gets back. A waiter + * leaves by taking a credit first, then the latch. Broadcast, not + * pthread_cond_signal: the waiter a condvar signal would pick may be the one + * outside the wait, running the pump, and the signal would be spent on + * nobody. The others re-check, find no credit, and sleep again. + * + * One divergence, stated: real Cafe OS wakes the highest-priority queued + * thread, and a thread that starts waiting after a signal can never take + * that signal. Here any parked waiter can take any credit, including one + * that arrived after the grant. The count is always right -- one signal, one + * wake -- but not which thread gets it. + * + * `epoch` used to be both the release mechanism and what EVSTATE printed. It + * is now only the second: signals that reached a parked waiter. OSInitEvent + * has its own `init_epoch`, the one thing the wait predicate still compares, + * so re-initialising a live event releases its waiters without any signal + * doing so. */ + +/* An AUTO signal: grant a wake if a parked waiter has none, else latch. + * Caller holds e->lock. */ +static inline void arkchemy_event_signal_auto_locked(ArkchemyEventEntry *e) { + if (e->wake_credits < e->waiting_count) { + e->wake_credits++; + e->epoch++; + pthread_cond_broadcast(&e->cond); + } else { + e->signaled = 1; + } +} + +/* Is there anything for a waiter that started under `my_init` to leave on? */ +static inline int arkchemy_event_ready_locked(const ArkchemyEventEntry *e, uint64_t my_init) { + return e->wake_credits > 0 || e->signaled || e->init_epoch != my_init; +} + +/* A waiter leaving. Takes what it is leaving on, in order: a re-init takes + * nothing, then a credit, then the latch (which AUTO consumes and MANUAL + * keeps). Returns 1 if it was signalled, 0 if it left for any other reason + * -- a timeout, EVWAKE, or a re-init. Caller holds e->lock. + * + * The clamp at the end keeps wake_credits <= waiting_count. It can only + * trip when a waiter leaves without a credit while credits are outstanding, + * which is the re-init path; the excess are signals meant for a queue that + * no longer has anyone in it, and on AUTO that is a latch. */ +static inline int arkchemy_event_leave_locked(ArkchemyEventEntry *e, uint64_t my_init) { + int signalled = 0; + if (e->init_epoch == my_init) { + if (e->wake_credits > 0) { + e->wake_credits--; + signalled = 1; + } else if (e->signaled) { + if (e->mode == 1 /* AUTO */) e->signaled = 0; + signalled = 1; + } + } + e->waiting_count--; + if (e->wake_credits > e->waiting_count) { + e->wake_credits = e->waiting_count; + if (e->mode == 1) e->signaled = 1; + } + return signalled; +} + static inline void ppc_import_coreinit_OSInitEvent(PpcContext *ctx) { /* void OSInitEvent(OSEvent *event, BOOL value, OSEventMode mode) */ uint32_t addr = ctx->r[3]; @@ -230,11 +323,14 @@ static inline void ppc_import_coreinit_OSInitEvent(PpcContext *ctx) { ArkchemyEventEntry *e = arkchemy_event_get(addr, value, mode); /* Re-initializing an address already in this table (a real Init * called twice) resets its state -- matches real hardware, which - * always fully re-initializes the struct. */ + * always fully re-initializes the struct. Anyone parked on the old + * incarnation is released: init_epoch moving is what they wait for. */ pthread_mutex_lock(&e->lock); e->signaled = value; e->mode = mode; - e->epoch++; + e->wake_credits = 0; + e->init_epoch++; + pthread_cond_broadcast(&e->cond); pthread_mutex_unlock(&e->lock); } @@ -245,19 +341,12 @@ static inline void ppc_import_coreinit_OSSignalEvent(PpcContext *ctx) { g_arkchemy_event_signals++; g_arkchemy_event_last_signal = ctx->r[3]; pthread_mutex_lock(&e->lock); g_ark_sev_got_lock++; - if (!e->signaled) { - if (e->mode == 1 /* AUTO */) { - if (e->waiting_count == 0) { - e->signaled = 1; - } else { - e->epoch++; - pthread_cond_signal(&e->cond); /* wake exactly one */ - } - } else { /* MANUAL */ - e->signaled = 1; - e->epoch++; - pthread_cond_broadcast(&e->cond); - } + if (e->mode == 1 /* AUTO */) { + arkchemy_event_signal_auto_locked(e); + } else if (!e->signaled) { /* MANUAL */ + e->signaled = 1; + if (e->waiting_count) e->epoch++; + pthread_cond_broadcast(&e->cond); } pthread_mutex_unlock(&e->lock); g_ark_sev_exit++; @@ -267,24 +356,27 @@ static inline void ppc_import_coreinit_OSSignalEventAll(PpcContext *ctx) { g_arkchemy_event_signals++; g_arkchemy_event_last_signal = ctx->r[3]; ArkchemyEventEntry *e = arkchemy_event_get(ctx->r[3], 0, 1); pthread_mutex_lock(&e->lock); - if (!e->signaled) { - if (e->mode == 1 /* AUTO */) { - if (e->waiting_count == 0) { - e->signaled = 1; - } else { - e->epoch++; - pthread_cond_broadcast(&e->cond); /* wake everyone currently waiting; doesn't persist */ - } - } else { /* MANUAL */ + if (e->mode == 1 /* AUTO */) { + if (e->waiting_count == 0) { e->signaled = 1; + } else { + /* everyone currently parked gets a wake; nothing persists */ + e->wake_credits = e->waiting_count; e->epoch++; pthread_cond_broadcast(&e->cond); } + } else if (!e->signaled) { /* MANUAL */ + e->signaled = 1; + if (e->waiting_count) e->epoch++; + pthread_cond_broadcast(&e->cond); } pthread_mutex_unlock(&e->lock); } static inline void ppc_import_coreinit_OSResetEvent(PpcContext *ctx) { + /* Clears the latch only. Credits already granted are wakes that have + * happened -- on Cafe OS those threads are off the queue and running -- + * so a reset does not take them back. */ ArkchemyEventEntry *e = arkchemy_event_get(ctx->r[3], 0, 1); pthread_mutex_lock(&e->lock); e->signaled = 0; @@ -407,11 +499,11 @@ static inline void ppc_import_coreinit_OSWaitEvent(PpcContext *ctx) { * The slice is deliberately short and the loop re-checks the same * predicate, so a spurious wake or a missed signal cannot make this * return early -- it only costs a wakeup. */ - uint64_t my_epoch = e->epoch; + uint64_t my_init = e->init_epoch; uint32_t my_slices = 0; e->waiting_count++; g_ark_wev_parked++; - while (!e->signaled && e->epoch == my_epoch) { + while (!arkchemy_event_ready_locked(e, my_init)) { struct timespec deadline; g_ark_wev_slices++; clock_gettime(CLOCK_REALTIME, &deadline); @@ -424,8 +516,8 @@ static inline void ppc_import_coreinit_OSWaitEvent(PpcContext *ctx) { /* Drop the lock before pumping: a completion callback runs * guest code that can signal this very event, and e->lock is * not recursive. Leave the waiter counted across the gap so - * an AUTO signal arriving meanwhile wakes rather than being - * latched -- either outcome is handled by the re-check. */ + * an AUTO signal arriving meanwhile is granted to it as a + * credit, which waits in the struct until it is back. */ pthread_mutex_unlock(&e->lock); arkchemy_fs_pump_completions(ctx); /* Nothing queued means nothing will arrive on its own: the @@ -443,8 +535,7 @@ static inline void ppc_import_coreinit_OSWaitEvent(PpcContext *ctx) { } } } - e->waiting_count--; - if (e->signaled && e->mode == 1 /* AUTO */) e->signaled = 0; + arkchemy_event_leave_locked(e, my_init); } pthread_mutex_unlock(&e->lock); g_ark_wev_exit++; @@ -502,17 +593,18 @@ static inline void ppc_import_coreinit_OSWaitEventWithTimeout(PpcContext *ctx) { deadline.tv_sec += (time_t)secs; deadline.tv_nsec += ns; if (deadline.tv_nsec >= 1000000000L) { deadline.tv_nsec -= 1000000000L; deadline.tv_sec += 1; } - uint64_t my_epoch = e->epoch; + uint64_t my_init = e->init_epoch; e->waiting_count++; - while (!e->signaled && e->epoch == my_epoch) { + while (!arkchemy_event_ready_locked(e, my_init)) { if (pthread_cond_timedwait(&e->cond, &e->lock, &deadline) != 0) break; /* real timeout */ } - e->waiting_count--; - if (e->signaled) { - if (e->mode == 1) e->signaled = 0; - } else if (e->epoch == my_epoch) { - woke_signaled = 0; /* timed out, never woken */ - } + /* TRUE only for a credit or the latch. The old test here was + * `else if (epoch == my_epoch) FALSE`, which reported TRUE to a + * waiter whose epoch had moved because some OTHER waiter was + * signalled -- a timed-out worker told it had work. A credit is + * checked before the timeout is believed, so a grant that lands at + * the same instant as the deadline is still delivered. */ + woke_signaled = arkchemy_event_leave_locked(e, my_init); } pthread_mutex_unlock(&e->lock); if (woke_signaled) g_arkchemy_event_wakes++; else g_arkchemy_event_timeouts++; @@ -624,8 +716,26 @@ static inline void ppc_import_coreinit_OSWaitSemaphore(PpcContext *ctx) { ArkchemySemEntry *s = arkchemy_sem_get(ctx->r[3], 0); int32_t prev; pthread_mutex_lock(&s->lock); + /* Waits in 1ms slices and pumps FS completions between them, for the + * reason OSWaitEvent does: pumping is cooperative here, so a thread that + * parks for good cannot deliver the completion that would release it, + * and nothing guarantees another thread will. Added 2026-09-24 with the + * sync_harness case that deadlocked without it. The contract is + * unchanged -- this returns only once it has taken a count. Only the + * completion pump, not the archive pump: that one is a workaround for a + * specific wait, and does not belong on every semaphore. */ while (s->count <= 0) { - pthread_cond_wait(&s->cond, &s->lock); + struct timespec deadline; + clock_gettime(CLOCK_REALTIME, &deadline); + deadline.tv_nsec += ARKCHEMY_EVENT_PUMP_SLICE_NS; + if (deadline.tv_nsec >= 1000000000L) { deadline.tv_sec += 1; deadline.tv_nsec -= 1000000000L; } + if (pthread_cond_timedwait(&s->cond, &s->lock, &deadline) == ETIMEDOUT) { + /* the callback may signal this very semaphore; s->lock is not + * recursive, so it must not be held across the pump */ + pthread_mutex_unlock(&s->lock); + arkchemy_fs_pump_completions(ctx); + pthread_mutex_lock(&s->lock); + } } prev = s->count; s->count = prev - 1; diff --git a/include/cafeos_state.c b/include/cafeos_state.c index 2e4c25c..42c7e48 100644 --- a/include/cafeos_state.c +++ b/include/cafeos_state.c @@ -106,7 +106,7 @@ pthread_mutex_t g_arkchemy_sem_table_lock = PTHREAD_MUTEX_INITIALIZER; * on purpose, and counted so the wrongness is visible -- see * arkchemy_mutex_get in cafeos_coreinit_sync.h. */ pthread_mutex_t g_arkchemy_mutex_fallback = PTHREAD_MUTEX_INITIALIZER; -ArkchemyEventEntry g_arkchemy_event_fallback = { 0, 1, PTHREAD_MUTEX_INITIALIZER, PTHREAD_COND_INITIALIZER, 0, 1, 0, 0 }; +ArkchemyEventEntry g_arkchemy_event_fallback = { 0, 1, PTHREAD_MUTEX_INITIALIZER, PTHREAD_COND_INITIALIZER, 0, 1, 0, 0, 0, 0 }; ArkchemySemEntry g_arkchemy_sem_fallback = { 0, 1, PTHREAD_MUTEX_INITIALIZER, PTHREAD_COND_INITIALIZER, 0 }; unsigned g_arkchemy_sync_used[3]; diff --git a/include/ppc_runtime.h b/include/ppc_runtime.h index 8e2edcd..2699a54 100644 --- a/include/ppc_runtime.h +++ b/include/ppc_runtime.h @@ -2119,6 +2119,121 @@ typedef struct PpcContext { uint8_t reserve_valid; } PpcContext; +/* --- setjmp/longjmp, done on the host ------------------------------------ + * + * The game's setjmp and longjmp are ordinary PowerPC in the binary, and + * recompiling them does not work. A recompiled longjmp restores guest + * registers and then "returns" through whatever host C frames happen to be + * live -- it cannot unwind the host call stack the recompiled code runs on, + * so execution carries on in the function that called longjmp, with a stack + * pointer from somewhere else. (ROADMAP.md; XMLREAD's note in this file.) + * + * So codegen recognises calls to them (see setjmp_names/longjmp_names in + * codegen.cpp) and does both halves here, the approach XenonRecomp uses for + * the Xbox 360: + * + * setjmp the recompiled guest setjmp still runs, so the guest jmp_buf is + * written exactly as before. Then the guest register file is + * saved to a host-side slot keyed by the jmp_buf's guest address, + * and the HOST setjmp is called -- inline, in the caller's own C + * function, which is why this is a macro: a host setjmp inside a + * helper would be dead the moment the helper returned. + * longjmp never runs the recompiled body. Looks up the slot, and host + * longjmps to it; the setjmp site restores the guest registers + * and hands back the value (0 becomes 1, as C requires). + * + * The register file is the leading part of PpcContext up to `tb`: GPRs, + * FPRs and their paired-single lanes, LR, CTR, CR, GQRs and XER[CA]. The + * time base is left alone, and any lwarx reservation is dropped, as a + * context switch would. + * + * Limits, stated: a longjmp to a buffer this never saw a setjmp for aborts + * loudly rather than guessing; setjmp reached by an indirect call (bctrl) + * is not intercepted; and as in C, longjmp to a frame that has already + * returned is undefined. Pinned by blaster's verify.sh setjmp pipeline. */ +#include +#include +#include + +#define PPC_JMP_SLOTS 64 +typedef struct { + uint32_t guest_buf; /* 0 = free */ + uint32_t val; /* what setjmp returns after the longjmp */ + jmp_buf host; + unsigned char regs[offsetof(PpcContext, tb)]; +} PpcJmpSlot; + +#ifdef __GNUC__ +__attribute__((weak)) +#endif +PpcJmpSlot g_ppc_jmp_slots[PPC_JMP_SLOTS]; +#ifdef __GNUC__ +__attribute__((weak)) +#endif +volatile int g_ppc_jmp_lock = 0; +#ifdef __GNUC__ +__attribute__((weak)) +#endif +volatile uint32_t g_ppc_setjmps = 0, g_ppc_longjmps = 0, g_ppc_jmp_exhausted = 0; + +/* The slot for a guest jmp_buf; with `create`, claim one if there is none. + * A spinlock, not a pthread mutex: this header deliberately needs nothing + * beyond the C library, and the critical section is a 64-entry scan. */ +static inline PpcJmpSlot *ppc_jmp_slot(uint32_t guest_buf, int create) { + PpcJmpSlot *hit = NULL, *free_slot = NULL; + int i; + while (__atomic_exchange_n(&g_ppc_jmp_lock, 1, __ATOMIC_ACQUIRE)) { } + for (i = 0; i < PPC_JMP_SLOTS; i++) { + if (g_ppc_jmp_slots[i].guest_buf == guest_buf) { hit = &g_ppc_jmp_slots[i]; break; } + if (!free_slot && g_ppc_jmp_slots[i].guest_buf == 0) free_slot = &g_ppc_jmp_slots[i]; + } + if (!hit && create && guest_buf) { + if (free_slot) { free_slot->guest_buf = guest_buf; hit = free_slot; } + else g_ppc_jmp_exhausted++; + } + __atomic_store_n(&g_ppc_jmp_lock, 0, __ATOMIC_RELEASE); + return hit; +} + +static inline void ppc_jmp_save(const PpcContext *ctx, PpcJmpSlot *s) { + memcpy(s->regs, ctx, sizeof s->regs); +} + +static inline void ppc_jmp_restore(PpcContext *ctx, PpcJmpSlot *s) { + memcpy(ctx, s->regs, sizeof s->regs); + ctx->reserve_valid = 0; + ctx->r[3] = s->val; +} + +/* At a call to the guest's setjmp. `guest_call` is the ordinary call + * statement for the recompiled setjmp. The slot pointer is volatile so it + * is still valid when the host setjmp returns a second time. */ +#define PPC_HOST_SETJMP(ctx, guest_call) do { \ + PpcJmpSlot *volatile ppc_sj_slot_ = ppc_jmp_slot((ctx)->r[3], 1); \ + guest_call; \ + if (ppc_sj_slot_) { \ + g_ppc_setjmps++; \ + ppc_jmp_save((ctx), ppc_sj_slot_); \ + if (setjmp(ppc_sj_slot_->host) != 0) \ + ppc_jmp_restore((ctx), ppc_sj_slot_); \ + else \ + (ctx)->r[3] = 0; \ + } \ +} while (0) + +/* At a call to the guest's longjmp(jmp_buf env, int val). Does not return. */ +static inline void ppc_host_longjmp(PpcContext *ctx) { + PpcJmpSlot *s = ppc_jmp_slot(ctx->r[3], 0); + if (!s) { + fprintf(stderr, "ppc_host_longjmp: no setjmp was recorded for jmp_buf 0x%08x " + "(lr=0x%08x) -- refusing to guess where to go\n", ctx->r[3], ctx->lr); + abort(); + } + g_ppc_longjmps++; + s->val = ctx->r[4] ? ctx->r[4] : 1u; + longjmp(s->host, 1); +} + /* Real, found-the-hard-way sizing: the previous 4MB was "an arbitrary, * generous, documented placeholder, not a claim this matches real Wii U * game scale" -- turned out to be a real, severe bug once actually diff --git a/src/codegen.cpp b/src/codegen.cpp index 61c19d5..1e03ae0 100644 --- a/src/codegen.cpp +++ b/src/codegen.cpp @@ -143,6 +143,25 @@ bool is_synthetic_addr_lo_reloc(const ElfImage &img, uint32_t addr) { // target lands outside the current function. Returns "" if the target // can't be resolved through any of call_relocs/addr_to_name/ // import_trampolines. +// See add_setjmp_name in codegen.h. +std::set &setjmp_names() { + static std::set names = {"setjmp", "_setjmp", "__setjmp"}; + return names; +} +std::set &longjmp_names() { + static std::set names = {"longjmp", "_longjmp", "__longjmp"}; + return names; +} + +// A direct call to the named function, unless it is the guest's setjmp or +// longjmp, which are done on the host instead. +std::string direct_call_stmt(const std::string &name) { + std::string call = "ppc_" + name + "(ctx)"; + if (setjmp_names().count(name)) return "PPC_HOST_SETJMP(ctx, " + call + ");"; + if (longjmp_names().count(name)) return "ppc_host_longjmp(ctx);"; + return call + ";"; +} + std::string resolve_call_stmt(const ElfImage &img, const std::map &addr_to_name, uint32_t insn_addr, uint32_t target) { // A real linked .rpx/.rpl calls into other system libraries either via @@ -162,13 +181,13 @@ std::string resolve_call_stmt(const ElfImage &img, const std::mapsecond + "(ctx);"; + return direct_call_stmt(it->second); } // No relocation (already-linked binary): the raw immediate is a real // target address. auto it2 = addr_to_name.find(target); if (it2 != addr_to_name.end()) { - return "ppc_" + it2->second + "(ctx);"; + return direct_call_stmt(it2->second); } // Trampoline-stub case: target is the trampoline's own resolved // address (see ImportTrampoline). @@ -208,6 +227,9 @@ void emit_conditional_branch(std::ostream &out, const std::string &cond, uint32_ } // namespace +void add_setjmp_name(const std::string &name) { setjmp_names().insert(name); } +void add_longjmp_name(const std::string &name) { longjmp_names().insert(name); } + std::vector generate_function_c(const ElfImage &img, const ElfFunction &func, const DisasmResult &insns, const std::map &addr_to_name, std::ostream &out) { std::vector unhandled; diff --git a/src/codegen.h b/src/codegen.h index a2c8aec..0d315dd 100644 --- a/src/codegen.h +++ b/src/codegen.h @@ -10,6 +10,12 @@ #include "elf_loader.h" namespace recomp { +// Names of the guest's setjmp and longjmp. Calls to these are not emitted as +// ordinary calls: see PPC_HOST_SETJMP in include/ppc_runtime.h for why a +// recompiled longjmp cannot work. Defaults cover the usual spellings; main +// adds any given with --setjmp-name / --longjmp-name. +void add_setjmp_name(const std::string &name); +void add_longjmp_name(const std::string &name); // Emits one C function (named "ppc_") that reproduces `func`'s // behaviour against a PpcContext*. Any instruction outside the milestone's diff --git a/src/elf_loader.cpp b/src/elf_loader.cpp index 8675367..4a68236 100644 --- a/src/elf_loader.cpp +++ b/src/elf_loader.cpp @@ -526,6 +526,8 @@ bool load_elf(const std::string &path, ElfImage &out, std::string &error) { if (si != out.section_sizes.end() && si->second > text_size_for_bounds) text_size_for_bounds = si->second; } + std::map skipped_nonalloc; + auto rela_sh_size_entries = [](uint32_t size, uint32_t entsize) { return entsize ? size / entsize : 0u; }; for (int rela_idx : rela_other_indices) { const uint8_t *rela_sh = shdr(rela_idx); uint32_t rela_entsize = rd_be32(rela_sh + 36); @@ -534,6 +536,23 @@ bool load_elf(const std::string &path, ElfImage &out, std::string &error) { if (target_idx >= (uint32_t)e_shnum) continue; std::string target_name = secname(rd_be32(shdr(target_idx) + 0)); if (target_name.empty()) continue; + // Only sections the loader actually puts in memory. Relocations + // against .debug_info, .debug_frame and friends were being read too, + // which gave those sections synthetic bases in guest memory and + // wrote relocated pointers into them -- harmless to the program, + // but it is guest memory spent on nothing, and it made every object + // look as if it carried data of its own (see --extern-globals in + // main.cpp). Found 2026-09-24 when verify.sh's multi-object case + // failed on an object with no globals at all. + // + // Counted and printed by section, because this cannot be checked + // against the retail RPX from where it was written: if a regenerate + // ever lists anything here other than debug sections, a loaded + // section is flagged in a way this did not expect. + if (!(rd_be32(shdr(target_idx) + 8) & 0x2u /* SHF_ALLOC */)) { + skipped_nonalloc[target_name] += rela_sh_size_entries(rd_be32(rela_sh + 20), rela_entsize); + continue; + } uint32_t target_real_base = rd_be32(shdr(target_idx) + 12); // sh_addr const std::vector &rela_bytes = section_data[rela_idx]; @@ -607,6 +626,9 @@ bool load_elf(const std::string &path, ElfImage &out, std::string &error) { out.section_relocs.push_back(sr); } } + for (const auto &kv : skipped_nonalloc) + std::fprintf(stderr, "note: skipped %u relocation(s) into %s (not loaded: no SHF_ALLOC)\n", + kv.second, kv.first.c_str()); if (bad_text_relocs) { std::fprintf(stderr, "warning: %zu .text data relocations resolved outside .text and were skipped\n", bad_text_relocs); diff --git a/src/main.cpp b/src/main.cpp index 84b614e..a97d37a 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -25,12 +25,17 @@ int main(int argc, char **argv) { output_path = args[++i]; } else if (args[i] == "--entry-alias" && i + 1 < args.size()) { entry_alias = args[++i]; + } else if (args[i] == "--setjmp-name" && i + 1 < args.size()) { + recomp::add_setjmp_name(args[++i]); + } else if (args[i] == "--longjmp-name" && i + 1 < args.size()) { + recomp::add_longjmp_name(args[++i]); } else if (input_path.empty()) { input_path = args[i]; } } if (input_path.empty() || output_path.empty()) { - std::cerr << "usage: recomp [--stripped] [--entry-alias NAME] [--extern-globals] -o \n"; + std::cerr << "usage: recomp [--stripped] [--entry-alias NAME] [--extern-globals]\n" + " [--setjmp-name NAME] [--longjmp-name NAME] -o \n"; std::cerr << " --stripped: ignore any symbol table and recover function\n"; std::cerr << " boundaries via control-flow analysis instead\n"; std::cerr << " (see func_recovery.h)\n"; @@ -55,6 +60,10 @@ int main(int argc, char **argv) { std::cerr << " way, since dispatch tables aren't merged across\n"; std::cerr << " objects. Direct calls (bl) across objects work fine\n"; std::cerr << " regardless.\n"; + std::cerr << " --setjmp-name NAME, treat calls to NAME as the guest's setjmp or\n"; + std::cerr << " --longjmp-name NAME longjmp, done on the host (see PPC_HOST_SETJMP in\n"; + std::cerr << " ppc_runtime.h). setjmp/_setjmp/__setjmp and the\n"; + std::cerr << " matching longjmp names are always recognised.\n"; return 1; } @@ -68,7 +77,14 @@ int main(int argc, char **argv) { recomp::find_import_trampolines(img); recomp::resolve_data_imports(img); - if (extern_globals && !img.global_section_base.empty()) { + // .text is in global_section_base too -- it keeps its real address + // there (see assign_global_addrs) -- but it is code, not data anyone has + // to initialize. Counting it made this check refuse every object since + // 2026-08-29, which is how verify.sh's multi-object case went red. + bool has_own_data = false; + for (const auto &kv : img.global_section_base) + if (kv.first != ".text") { has_own_data = true; break; } + if (extern_globals && has_own_data) { std::cerr << "error: --extern-globals was passed, but this object has its own " "global/static data (would never get initialized -- see --help)\n"; return 1;