diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a5222462a..9d05739c8 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -543,7 +543,7 @@ jobs: extension-artifacts-native-android: name: Native Extensions (Android Packaging) needs: [affected, extension-artifacts-native-android-static, extension-artifacts-native-linux] - if: ${{ fromJson(needs.affected.outputs.extension_artifacts_native_matrix_android).include[0] != null }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.extension-artifacts-native-android-static.result == 'success' && needs.extension-artifacts-native-linux.result == 'success' && fromJson(needs.affected.outputs.extension_artifacts_native_matrix_android).include[0] != null }} uses: ./.github/workflows/extension-artifacts-native.yml with: phase: android-package @@ -591,7 +591,7 @@ jobs: needs: - affected - liboliphaunt-wasix-runtime - if: ${{ contains(fromJson(needs.affected.outputs.jobs), 'extension-artifacts-wasix') }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.liboliphaunt-wasix-runtime.result == 'success' && contains(fromJson(needs.affected.outputs.jobs), 'extension-artifacts-wasix') }} strategy: fail-fast: false matrix: ${{ fromJson(needs.affected.outputs.extension_artifacts_wasix_matrix) }} @@ -704,7 +704,7 @@ jobs: mobile-extension-packages-android: name: Extension Packages (Android) needs: [affected, extension-artifacts-native-android] - if: ${{ needs.affected.outputs.mobile_extension_package_native_targets_android_csv != '' }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.extension-artifacts-native-android.result == 'success' && needs.affected.outputs.mobile_extension_package_native_targets_android_csv != '' }} uses: ./.github/workflows/mobile-extension-packages.yml with: cache-save-if: ${{ needs.affected.outputs.cache_save_if == 'true' }} @@ -716,7 +716,7 @@ jobs: mobile-extension-packages-ios: name: Extension Packages (iOS) needs: [affected, extension-artifacts-native-ios] - if: ${{ needs.affected.outputs.mobile_extension_package_native_targets_ios_csv != '' }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.extension-artifacts-native-ios.result == 'success' && needs.affected.outputs.mobile_extension_package_native_targets_ios_csv != '' }} uses: ./.github/workflows/mobile-extension-packages.yml with: cache-save-if: ${{ needs.affected.outputs.cache_save_if == 'true' }} @@ -1102,7 +1102,7 @@ jobs: - liboliphaunt-native-desktop - liboliphaunt-native-ios - liboliphaunt-native-ios-abi - if: ${{ contains(fromJson(needs.affected.outputs.jobs), 'liboliphaunt-native-release-assets') }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.liboliphaunt-native-android.result == 'success' && needs.liboliphaunt-native-android-abi.result == 'success' && needs.liboliphaunt-native-desktop.result == 'success' && needs.liboliphaunt-native-ios.result == 'success' && needs.liboliphaunt-native-ios-abi.result == 'success' && contains(fromJson(needs.affected.outputs.jobs), 'liboliphaunt-native-release-assets') }} runs-on: ubuntu-24.04 timeout-minutes: 30 steps: @@ -1515,7 +1515,7 @@ jobs: - affected - liboliphaunt-native-ios-abi - swift-bindings - if: ${{ contains(fromJson(needs.affected.outputs.jobs), 'swift-sdk-package') }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.liboliphaunt-native-ios-abi.result == 'success' && needs.swift-bindings.result == 'success' && contains(fromJson(needs.affected.outputs.jobs), 'swift-sdk-package') }} runs-on: ubuntu-24.04 timeout-minutes: 90 steps: @@ -2274,7 +2274,7 @@ jobs: - affected - liboliphaunt-wasix-runtime - liboliphaunt-wasix-aot - if: ${{ contains(fromJson(needs.affected.outputs.jobs), 'liboliphaunt-wasix-release-assets') && (github.event_name != 'workflow_dispatch' || inputs.wasm_target == 'all') }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.liboliphaunt-wasix-runtime.result == 'success' && needs.liboliphaunt-wasix-aot.result == 'success' && contains(fromJson(needs.affected.outputs.jobs), 'liboliphaunt-wasix-release-assets') && (github.event_name != 'workflow_dispatch' || inputs.wasm_target == 'all') }} runs-on: ubuntu-24.04 timeout-minutes: 30 steps: @@ -3018,7 +3018,7 @@ jobs: - liboliphaunt-native-ios-abi - react-native-sdk-package - swift-sdk-package - if: ${{ contains(fromJson(needs.affected.outputs.jobs), 'mobile-build-ios') }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.js-sdk-package.result == 'success' && needs.mobile-extension-packages-ios.result == 'success' && needs.liboliphaunt-native-ios.result == 'success' && needs.liboliphaunt-native-ios-abi.result == 'success' && needs.react-native-sdk-package.result == 'success' && needs.swift-sdk-package.result == 'success' && contains(fromJson(needs.affected.outputs.jobs), 'mobile-build-ios') }} runs-on: macos-26 timeout-minutes: 180 env: @@ -3344,7 +3344,7 @@ jobs: needs: - affected - mobile-build-ios - if: ${{ contains(fromJson(needs.affected.outputs.jobs), 'mobile-build-ios') }} + if: ${{ !cancelled() && needs.affected.result == 'success' && needs.mobile-build-ios.result == 'success' && contains(fromJson(needs.affected.outputs.jobs), 'mobile-build-ios') }} runs-on: macos-26 timeout-minutes: 45 steps: diff --git a/src/docs/internal/OLIPHAUNT_PATCH_STACK.md b/src/docs/internal/OLIPHAUNT_PATCH_STACK.md index 14607e5e5..3d4536c4f 100644 --- a/src/docs/internal/OLIPHAUNT_PATCH_STACK.md +++ b/src/docs/internal/OLIPHAUNT_PATCH_STACK.md @@ -1,12 +1,23 @@ # Native PostgreSQL patch stack The ordered native recipe lives in -[`postgres/series`](../../runtimes/liboliphaunt/native/postgres/series), +[`postgres/series`](../../native/runtime/postgres/series), and the shared source pin lives in -[`source.toml`](../../postgres/versions/18/source.toml). +[`source.toml`](../../third-party/postgres/source.toml). Each patch header explains its change. Platform builders apply that series with `git apply --whitespace=error-all` before compiling PostgreSQL. Run the native runtime build and its C ABI, SQL, lifecycle, and extension tests for behavioral evidence. Source fragments, author headers, and a generated review table do not prove those behaviors and no longer gate qualification. + +## Correctness consolidation + +Patch 0021 models trusted embedded sessions: catalog identity, normal admission +and wraparound protections without pretending an absent postmaster exists. +Patch 0022 keeps cancellation, timers, wakeups, signal masks and COPY deadlines +inside the embedding boundary. Its narrower changes supersede the former +0021 wake-epoll-self-pipe patch; they are not an optional performance switch. + +Startup/cleanup, configured identity and host working-directory restoration +must be checked through the real C ABI, not inferred from patch formatting. diff --git a/src/docs/internal/WASIX_PATCH_STACK.md b/src/docs/internal/WASIX_PATCH_STACK.md index 22912d34e..4f84283e9 100644 --- a/src/docs/internal/WASIX_PATCH_STACK.md +++ b/src/docs/internal/WASIX_PATCH_STACK.md @@ -1,12 +1,37 @@ # WASIX PostgreSQL patch stack The ordered patches live in -[`patches/series`](../../runtimes/liboliphaunt/wasix/assets/build/postgres/patches/series); +[`postgres/series`](../../wasix/runtime/postgres/series); the PostgreSQL source pin lives in -[`versions/18/source.toml`](../../postgres/versions/18/source.toml). +[`source.toml`](../../third-party/postgres/source.toml). Each patch header explains its change. The WASIX builder applies that series before compiling the runtime. Use the built runtime's protocol and extension tests for behavioral evidence. Source fragments and generated review tables no longer gate qualification. The concurrent Postmaster runtime has its own source, patches, and recovery tests. + +## Retired patches + +These embedded WASIX numbers remain reserved for traceability; they do not +identify Native or Wasmer patches. The selected series is authoritative. + +| Numbers | Why they were removed | +| --- | --- | +| 0013, 0020 | Host-recovery symptom patches retained an exception boundary after its frame returned. Live guest-local recovery replaces them. | +| 0014 | The selected compiler already emitted the intended unaligned hash load; no isolated residual gain justified the fork. | +| 0015 | The current-XID shortcut lacked demonstrated isolated benefit. | +| 0016, 0026 | int4 comparison shortcuts did not prove active comparator identity and could bypass custom opclass semantics. | +| 0017 | Large stack scratch changed allocation ownership and recoverable failure behavior without an earned performance benefit. | +| 0024 | Byte searching could cross non-UTF8 multibyte character boundaries. The LIKE optimization is removed, not retained with an encoding fix. | +| 0028 | The semaphore-reset shortcut did not preserve PostgreSQL semantics. | +| 0030 | Cached WAL-segment arithmetic overflowed at an endpoint; a correction did not establish worthwhile speedup. | +| 0031 | Removing activity reporting discarded behavior without adequate justification. | +| 0035, 0036 | Scalar synchronization replacements lacked complete observer-exclusivity proof and an isolated performance win. | + +Correctness removals and unearned optimizations are different decisions; neither +should be reversed merely to improve a compound benchmark score. + +Guest recovery, startup identity and output contracts must be updated together +with the Rust and browser hosts. Patch 0044 establishes the host-selected catalog +principal before startup policy runs; a later `SET ROLE` is not equivalent. diff --git a/src/docs/maintainers/README.md b/src/docs/maintainers/README.md index 31c817d1d..47580bc4d 100644 --- a/src/docs/maintainers/README.md +++ b/src/docs/maintainers/README.md @@ -11,6 +11,7 @@ Executable configuration is authoritative. Documentation explains intent and ope | CI gates and test selection | `testing.md`, `tooling.md` | `.github/workflows/ci.yml`, `tools/ci/ci_plan.mts`, Moon project files | | Binary artifacts and WASIX provenance | `assets.md`, `compiler-caching.md` | runtime target metadata, exact producer SHA, runtime/AOT manifests and checksums, `src/wasix/runtime/tools/xtask` | | WASIX host APIs and storage | `wasix-usage.md` | WASIX binding source, host pins/patches, and product Moon tasks | +| Runtime stacks, buffers, and streaming budgets | `runtime-resource-budgets.md` | Owning runtime constants and `src/wasix/runtime/protocol-contract/contract.json` | | WASIX postmaster runtime and carrier | `wasix-postmaster.md` | `src/wasix/postmaster`, its Moon project, sealed-carrier policy, and release metadata | | Extension support and packaging | `extension-packaging-policy.md` | extension catalog, global target profiles, native-component contract, release catalog | | SDK contracts | `sdk-products-policy.md`, `sdk-parity-policy.md`, `sdk-api-surface.md` | SDK manifests, package manifests, generated extension metadata, clean-consumer tests | diff --git a/src/docs/maintainers/runtime-resource-budgets.md b/src/docs/maintainers/runtime-resource-budgets.md new file mode 100644 index 000000000..505731cb3 --- /dev/null +++ b/src/docs/maintainers/runtime-resource-budgets.md @@ -0,0 +1,374 @@ +# Runtime resource budgets + +Equal byte counts do not imply equal purposes. A stack allocation, an I/O batch, +a queue watermark, and a protocol rejection limit must have separate owners. +This guide describes the owning constants and their trade-offs, not optimal +values for every workload. MiB and KiB mean powers of 1024. + +## Read this first: what a size means + +- **A batch is a delivery box.** A 64 KiB batch moves up to that much at a time. + A 1 GiB result can use many boxes. A small result does not wait for a full box. + Bigger boxes may mean fewer trips, but require more space for each trip. +- **A queue is a waiting room.** It holds work or bytes until the next part can + accept them. More waiting space can absorb a burst; it does not make the + database execute a query faster. It can increase memory use and waiting time. +- **A cache keeps things for reuse.** More space can avoid repeated reads or + setup. It helps only when useful things would otherwise be thrown away. +- **A limit is a stop sign.** It rejects an oversized item. Raising a limit + permits larger items; it does not make already-accepted items faster. +- **A stack is the program's working trail.** Nested function calls need more + trail space. More space can allow deeper calls, not faster ordinary queries. + A safe depth check must stop the program before that space runs out. +- **An initial allocation is a starting container.** It may grow later. + Starting larger can avoid growth/copying, but wastes more space on small work. + +These are explanations of the mechanisms, not measured speedup claims. + +## Scope and ownership + +This guide covers first-party runtime, SDK, transport and build resource limits. +It does not duplicate dependency internals or describe test workloads, timeouts, +ports and binary field offsets as tuning choices. + +Central documentation does not mean one global constant: equal numbers with +different purposes stay separately owned. The shared protocol contract generates +the embedded bridge C and SDK Rust/TS values. Browser-host patch literals are +still separately pinned and checked by `src/wasix/browser-host/build-sdk.sh`. +SQL startup +defaults and broker-frame limits also still have multiple language owners. +Change and check every consumer together until those contracts are unified. + +## Is this PostgreSQL, our choice, or a platform rule? + +There are two questions: **who provides the mechanism**, and **who chooses its +size**. A standard PostgreSQL setting can still have an Oliphaunt-selected value. +"Our choice" also does not mean "proven optimization": some choices are safety +limits, some are compatibility requirements, and some are tuning candidates. + +| Inventory family | Origin and purpose | Ordinary PostgreSQL equivalent? | +| --- | --- | --- | +| `shared_buffers`, `wal_buffers`, `min_wal_size` | **PostgreSQL mechanisms; Oliphaunt startup policy.** We explicitly choose 128/4/80 MiB. | Yes, the same SQL settings. Matching a normal value is not a custom optimization. | +| `work_mem`, hash multiplier, maintenance/temp memory, `max_wal_size`, `max_stack_depth` | **PostgreSQL mechanisms/defaults**, unless a caller overrides them. | Yes. Memory for sorting, caching, logging and recursion checking exists in a normal server too. | +| PostgreSQL's own 8 KiB receive/send buffers | **PostgreSQL implementation choices.** Separate from our bridge buffers even when equal. | Yes, `src/backend/libpq/pqcomm.c`; send storage can grow. | +| Our 64 KiB batches, 4 MiB stream queue cap, 256 KiB channels, initial growing containers and compaction thresholds | **Oliphaunt transport/allocation tuning.** Trade calls/copies against memory and waiting. | A server also buffers I/O, but does not have these SDK bridge constants. Their exact values need workload evidence. | +| 64/256 work admission counts | **Oliphaunt overload control.** Bounds waiting work, not backend parallelism. | Client pools/server connection admission are analogous, but not the same queues or units. | +| OPFS bridge sizes, 32 spares, 16 parallel file operations, one cached runtime | **Oliphaunt browser/startup tuning.** Reuse and batching can reduce setup work. | No OPFS or compiled-Wasm module cache in a normal native server. | +| 128 MiB frontend/broker limits | **Oliphaunt safety/API policy.** Rejects oversized retained data on these paths. SQL results and collected tool output have no separate 64 MiB quota. | PostgreSQL has its own message/allocation rules; this cap is ours, not a SQL limit to attribute to PostgreSQL. | +| Diagnostic tails, startup/error text and archive/metadata ceilings | **Oliphaunt diagnostics/input protection.** | Similar needs exist elsewhere; our exact limits are not PostgreSQL query-performance settings. | +| Native backend 8 MiB thread stack, guest 8 MiB C stack, initial 128 MiB linear memory, Postmaster profile sizes | **Embedding/platform capacity choices.** Required resources with chosen budgets, not automatic speedups. | Native PostgreSQL also needs stack/heap space, but not these Wasm allocations or SDK thread defaults. | +| Wasmer execution stack and reserved-address layout | **Engine-owned mechanism**, with engine/product policy deciding capacity. | An ordinary native server uses its process/OS stack; it has no Wasmer coroutine stack. | +| Wasm page, TAR/wire headers, identifier/digest lengths, platform alignment | **Format/platform rules**, not free tuning knobs. | PostgreSQL-specific formats are shared; Wasm/TAR/Android rules belong to those respective formats/platforms. | +| Hash/copy batches and release/source/package-tool envelopes | **Build/install choices**, outside normal SQL execution. | Not PostgreSQL query settings. | + +The transport/storage tuning rows are the optimization-*motivated* choices. +This inventory does not establish that 64 KiB, 32 spares, or any other exact +selection is the optimum. Safety rows must earn their place through correct +limits and failure handling, not by producing a faster benchmark. + +## Protocol and streaming + +Paths in this table are relative to the repository root. Rust paths abbreviated +as `wasix-rust/...` are under `src/wasix/sdks/rust/src/oliphaunt/`. +TypeScript paths abbreviated as `wasix-ts/...` are under `src/wasix/sdks/ts/src/`. + +| Owner | Size | Meaning and allocation behavior | +| --- | --- | --- | +| `src/wasix/runtime/protocol-contract/contract.json`, `bufferedOutput.limitBytes` | 2 GiB minus 1 byte | Maximum length representable by the signed-i32 host bridge, not an application quota or reservation. Grows on demand; available memory may run out earlier. Allocation/overflow failures retire the session without publishing an incomplete buffered response. | +| Same contract, `streamedOutput.callbackChunkMaxBytes`; generated Rust `protocol_limits_generated.rs::PROTOCOL_CALLBACK_CHUNK_BYTES`; TS `core/database.ts::WASIX_PROTOCOL_CALLBACK_CHUNK_BYTES` | 64 KiB | Maximum bytes per host callback, not per result or COPY operation. Rust lends a slice for the synchronous call; JavaScript delivers an owned copy. | +| `src/wasix/pgwire-server/src/proxy.rs::PROXY_READ_BUFFER_BYTES` | 64 KiB | Local stack scratch space per socket-serving call. A PostgreSQL message can span many reads. | +| `wasix-rust/client.rs::DIRECT_TOOL_READ_BUFFER_BYTES` | 64 KiB | Separate local stack scratch space for the direct-tool socket. It is not owned by the callback ABI merely because its size matches. | +| `src/query/rust/src/wire.rs::MAX_FRONTEND_MESSAGE`; TS `protocol/pgwire-connection.ts::MAX_FRONTEND_MESSAGE_BYTES` | 128 MiB | Maximum total length of one frontend frame, including header. Readers validate the declared length; they do not allocate the maximum for every query. Not a total multi-frame COPY limit. | +| `wasix-ts/protocol/byte-channel.ts::WASIX_CHANNEL_BYTES` | 256 KiB + 1 byte | Fixed shared-memory ring allocation per channel, plus a separate five-word control block. One sentinel byte leaves 256 KiB usable. Full channels apply backpressure. | +| Same file, `WASIX_BYTE_CHANNEL_CHUNK_BYTES` | 64 KiB | Default read batch from the ring, independent of its capacity and of callback ownership. | +| `src/native/runtime/src/liboliphaunt_protocol.c::DEFAULT_STREAM_QUEUE_MAX_BYTES` | 4 MiB | Native stream queue byte cap, not eager allocation. Larger writes are split into queue-sized chunks and wait for the consumer to free space. Allocations follow queued bytes. | +| Same file, `INITIAL_BUFFERED_OUTPUT_BYTES` | 8 KiB | First native buffered-output allocation, grown geometrically as needed. Native lengths are host-sized, unlike the WASIX signed-i32 bridge. | +| `src/native/runtime/src/liboliphaunt_archive_tar.c::ARCHIVE_FILE_READ_CHUNK_BYTES` | 64 KiB | Stack scratch space for reading backup files. The buffered backup API accumulates the archive; the streaming API sends it to a callback without retaining the complete archive. | + +The protocol contract supplies shared numeric constants. Signatures live in +C declarations and typed host bindings. Use the existing +generator and compiled transport tests for a contract change; do not create a +second configuration file or import repository JSON at package runtime. Local +read buffers stay owned by their readers. The matching Rust/TS frontend limit +is a host admission policy, not PostgreSQL's universal maximum field size. + +COPY and streamed responses can exceed an individual chunk or queue size. +Streaming avoids retaining the complete transport output, but an SDK method +that collects decoded rows can still retain the full application result. An +error after streaming a prefix cannot retract bytes already delivered. + +Other similarly sized values have different owners: TS +`storage/opfs-provider.ts::DIRECT_BRIDGE_CAPACITY` supplies 1 MiB to the direct +filesystem bridge, not the PostgreSQL transport. TS +`direct-client-common.ts::CHROMIUM_SYNC_WASM_LIMIT_BYTES` is an 8 MiB module-size +admission threshold for synchronous extension loading in a Chromium Window; +larger modules require the worker placement, not a larger query buffer. Native +startup's `wal_buffers=4MB` configures PostgreSQL WAL buffering and is unrelated +to the 4 MiB stream queue. SQL working memory and shared buffers are separate +PostgreSQL configuration, not transport tunables. + +### Additional transport and allocation sizes + +| Owner (repository-relative) | Selection | In simple words; effect | +| --- | --- | --- | +| `src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge.c`, input reserve and output append | 8 KiB initially, doubles | Starting containers for guest input/output. Ordinary SQL results grow automatically. Output reset retains capacity for reuse, trading memory after a large response for fewer later allocations; streaming avoids retaining the complete response. | +| `src/native/sdks/rust/src/pgwire.rs` | 64 KiB read buffer; 8 KiB initial response vector | Socket delivery box and growing result container, respectively. Neither limits the total result to that size. | +| `src/native/sdks/rust/src/ipc.rs` and `src/native/sdks/ts/src/runtime/broker-frames.ts` | 128 MiB frame limit | Rejects a single oversized broker message; independent of the similarly sized PostgreSQL frontend-frame limit. | +| `wasix-rust/tools.rs::DIRECT_TOOL_SOCKET_BUFFER` | 256 KiB | Waiting space in the in-process tool socket. Larger capacity can absorb bursts, not speed up SQL itself. | +| Rust native/WASIX tools and JS tool collectors | Complete output in memory; no application byte quota | Available memory and host Vec/Buffer/string-size limits apply. Fallible capture failures discard partial stdout and stderr. Native pipes continue draining. Rust collectors grow amortized and release buffers on failure. These APIs do not yet offer a streaming sink; use an external tool for file/stream output. | +| `src/native/postgres-tools/crates/tools/src/lib.rs`, captured pipe reader | 32 KiB | Tool-output read batch; total capture is governed by the separate contract above. | +| Native `liboliphaunt_archive_tar.c::buffer_reserve` | 4 KiB initially, doubles | Backup archive starting container. The full archive still grows in memory; larger initial capacity does not solve large-backup memory use. | +| `wasix-ts/protocol/pgwire-connection.ts`, chunk-list compaction | 1,024 consumed chunks and at least half the list consumed | Removes old list entries in batches. More frequent removal frees references sooner but spends more time moving list entries. Not a byte limit. | +| `wasix-rust/postgres_mod/stdio.rs`, `sync_host_fs.rs`, tool fallback and browser stderr patch | 8 KiB write-readiness report | Says a file/stream is writable with this advertised amount; does not allocate an 8 KiB buffer or guarantee a complete write. | +| Guest `oliphaunt_wasix_bridge.c`, emulated `SO_SNDBUF`/`SO_RCVBUF` | 32 KiB | Socket-option compatibility answers, not real kernel socket allocations. Do not tune these as if they were queue capacities. | + +### Browser storage and counted resources + +| Owner | Selection | In simple words; effect | +| --- | --- | --- | +| `wasix-ts/storage/opfs-provider.ts::DIRECT_BRIDGE_CAPACITY` | 1 MiB | Maximum file-transfer batch offered to the browser filesystem bridge. Larger batches can reduce calls for big files; they do not make tiny reads faster. | +| Browser patch `0015-wasmer-js-add-sync-filesystem-bridge.patch`, `Backend::new` | Accepts 8 KiB–4 MiB bridge capacity | Allowed range for that batch setting, not a database-size limit. | +| Same patch, `READ_DIR_PAGE_CAPACITY` | 64 KiB | One page of directory names. Larger pages reduce trips for directories with many files, at a memory/copy cost. | +| `wasix-ts/storage/opfs-pool.ts::#ensureStagedCapacity` | At least 8 KiB on growth; doubles or meets requested size | Growing in-memory file storage before publication. Spare capacity saves repeated allocations but increases retained memory. No total database cap follows from this number. | +| Same file, `PREOPENED_FILE_RESERVE` | 32 spare files | Keeps files ready to use so creation can be quicker. Costs file handles/resources even before those spares hold useful data. | +| Same file, `MAX_PARALLEL_IO` | 16 operations | Limits simultaneous work in the helper's file batches. More may speed setup/publication, or compete for the same storage. Not 16 concurrent SQL queries. | +| `wasix-ts/hosts/browser/direct-client-common.ts::MAX_PREPARED_RUNTIMES` | 1 cached runtime identity | Avoids repeating preparation for the same assets. More entries help switching between runtime versions but retain more assets; eviction does not close live databases. | +| `src/wasix/sdks/rust/src/async_api.rs::OWNER_QUEUE_CAPACITY` | 64 ordinary work permits | Bounds ordinary admitted work in the async wrapper. More permits let more callers wait; they do not add backend execution parallelism. | +| `src/native/sdks/rust/src/executor.rs::ORDINARY_QUEUE_CAPACITY` | 256 ordinary queued commands | A separate SDK waiting room. Cleanup/recovery commands do not consume these slots. A command-count limit is not a byte-memory limit. | + +### PostgreSQL's own memory and disk choices + +Embedded defaults are set in native `liboliphaunt_runtime.c`, Rust +`postgres_mod.rs::DEFAULT_STARTUP_GUCS`, and TS `wasix-runtime.ts`. These are +startup settings, unlike compile-time transport constants. Caller settings can +override the applicable defaults; inspect the running database rather than +assuming a seed or caller has not changed them. + +| Setting | Oliphaunt embedded selection | PGlite 0.5.8 observed | Ordinary PostgreSQL / simple effect | +| --- | --- | --- | --- | +| `shared_buffers` | Explicit 128 MiB default | 128 MiB, configuration file | Standard setting, typically 128 MiB. Reuses data pages; larger caches cost memory per instance. Not the separate 128 MiB initial Wasm memory. | +| `wal_buffers` | Explicit 4 MiB default | 4 MiB, automatically selected (`boot_val=-1`) | Standard setting; normally auto-sized. Same value here, different selection policy: our fixed 4 MiB does not automatically follow a caller's larger cache. Holds recovery-log writes temporarily. | +| `min_wal_size` | Explicit 80 MiB default | 80 MiB, configuration file | Standard default 80 MiB. Recycled log-file space, not RAM. | +| `work_mem` | Inherited 4 MiB unless overridden | 4 MiB | Standard default. Per-sort/hash working space; several operations/sessions multiply use. | +| `hash_mem_multiplier` | Inherited 2 unless overridden | 2 | Standard default; hash operations can use twice `work_mem`. Not a separate fixed allocation. | +| `maintenance_work_mem` | Inherited 64 MiB unless overridden | 64 MiB | Standard default. Working space for index creation/vacuum, not a simple SELECT speed setting. | +| `temp_buffers` | Inherited 8 MiB with standard blocks | 8 MiB | Standard default. Temporary-table cache used as needed, not sorting memory. | +| `max_wal_size` | Inherited 1 GiB unless overridden | 1 GiB, configuration file | Standard default. Soft checkpoint-related target, not a hard disk quota. | +| `max_stack_depth` | PostgreSQL guard; ordinary startup commonly 2 MiB, inspect actual instance | 2 MiB | Standard SQL guard, not an allocated stack. Platform limits can affect the selected default. | + +PostgreSQL owns further settings not overridden by our runtime; this guide does +not duplicate its complete configuration manual. See its [memory settings](https://www.postgresql.org/docs/18/runtime-config-resource.html) +and [WAL settings](https://www.postgresql.org/docs/18/runtime-config-wal.html). +For actual values use `SHOW shared_buffers`, `SHOW wal_buffers`, `SHOW work_mem`, +and the other setting names. These descriptions explain potential effects; +this audit did not benchmark alternative settings. + +## PGlite comparison: embedding sizes + +Checked 2026-09-08 against the installed **`@electric-sql/pglite@0.5.8`** +distribution and its source maps. A fresh in-memory Node 24.18.0 instance +reported PostgreSQL **18.3**; Oliphaunt targets **18.4**. The SQL column above +comes from that running instance's `pg_settings` (including `unit`, `source` +and `boot_val`), not guesses from a build script. No directory/browser replay +or performance comparison was run for this documentation update. + +Build-source evidence is separately pinned: PGlite repository snapshot +`ae182ff8bd5ba4acb887d6c925d607a1498aa0b5`, PostgreSQL submodule +`b133782cd759f08b3aeb263b80a963b39c7b7af1`. These source snapshots are not claimed +to prove the exact compiler provenance of the published npm binary. + +| Topic | Oliphaunt | PGlite counterpart | What the comparison means | +| --- | --- | --- | --- | +| Guest C/shadow stack | 8 MiB link setting | 8 MiB in pinned backend link recipe [P1] | Same kind of stack and same selected size. Neither measures the host engine's execution stack. | +| Initial guest memory | 128 MiB link setting | 128 MiB default, caller `initialMemory` option [P2] | Comparable starting capacity, not a database-size limit. Memory can grow after startup. | +| Maximum guest memory | Product/engine-specific; Postmaster's 256 MiB profile is not the embedded default | 32,768 Wasm pages = 2 GiB in JS constructor [P2] | Do not compare a different Oliphaunt product's cap as if both ran under it. Native PostgreSQL has no equivalent one-piece Wasm ceiling. | +| Native execution stack | Wasmer's separate 1 MiB default in the retained Rust runtime; Postmaster selects its own size | JS engine controls native Wasm execution; no corresponding numeric capacity selected in inspected PGlite SDK | No valid "1 MiB versus 8 MiB" comparison: the latter is PGlite's *other* stack. | +| Growing result storage | Guest/native bridge output starts at 8 KiB | JS receive container starts at 1 MiB; grows, and resets to default on a later raw call [P3] | Similar job at different layers. Smaller starts save space for small work; larger starts avoid some growth. Both may also collect decoded rows. | +| Collected-output ceiling | WASIX signed-i32 bridge length (2 GiB minus 1); no separate application quota. Native uses host-sized lengths. | A constant named `MAX_BUFFER_SIZE` is 1 GiB, but see caveat below [P3] | Both can exhaust memory first. Neither collecting API promises bounded memory; use streaming when the whole result need not remain in memory. | +| Callback/read batches | Our 64 KiB callback maximum and separate reader batches | PGlite receives bytes through the guest callback; no matching fixed 64 KiB SDK callback cap found [P3] | PGlite's 1 MiB receive container is **not** its callback batch size. | +| PostgreSQL's internal protocol buffers | Separate from our bridge allocations | Pinned PGlite PostgreSQL still defines 8 KiB receive and initial send buffers [P4] | These come from PostgreSQL; matching 8 KiB bridge starts do not make them the same allocation. | +| 4 MiB queue / 256 KiB channels | Our native streaming / browser-tool transports | No direct matching fixed-byte waiting room found in inspected PGlite core | Compare streaming/copy counts and retained memory, not invented size parity. | +| 64/256 ordinary-work limits | Our async/SDK admission policies | Query/transaction mutexes serialize PGlite work; no corresponding numeric admission bound found [P3] | A mutex orders work; it is not itself a bounded request queue. | +| Browser filesystem transfer | 1 MiB bridge batch, 64 KiB directory page; 8 KiB starting staged files | OPFS AHP reads/writes requested file slices using synchronous access handles [P5] | Same filesystem problem, different route. No matching 1 MiB bridge/staging budget found there. | +| OPFS spare files | 32 maintained spares | Configurable 1,000 initially, 100 maintained [P5] | Directly comparable concept, not identical lifecycle. PGlite spends more setup/resources preparing spare files; fewer spares can mean more later replenishment. | +| OPFS batch concurrency | 16 helper operations | Pool work gathered with `Promise.all`; no matching 16-operation cap in inspected implementation [P5] | More parallel setup may help or contend for the same storage. Does not add SQL execution parallelism. | +| Cached prepared runtime | One retained identity | URL-keyed compiled-module cache; no numeric eviction bound found [P6] | Related reuse policy, not identical cached objects. More retained identities can avoid setup while retaining more memory. | +| Whole backup/archive handling | Initial containers, extraction ceilings and per-role checks documented below | TAR/compression helpers also gather whole data/chunks; no matching set of our archive ceilings found [P7] | Neither whole-result path becomes constant-memory just because reads are chunked. | +| Diagnostics, broker/tool caps, carrier proofs, build/download envelopes | Our product-specific policies | No directly comparable shared numeric contract established in this review | Mark as our policy, not a PGlite or PostgreSQL performance disadvantage. | +| Fixed-format sizes | Wasm page, TAR block, PostgreSQL page and headers | Same relevant formats; probe observed 8 KiB PG pages and 16 MiB WAL segments | Format compatibility is not an optimization. Check the cluster's WAL segment size rather than copying a test fixture. | + +**PGlite output-limit caveat:** in the inspected 0.5.8 `#defaultOnData`, the +allocation length is computed before `requiredSize` is adjusted to the named +maximum; the allocation uses that earlier length. Source inspection therefore +does not establish a hard 1 GiB ceiling. No giant allocation test was attempted +on this disk/memory-constrained host. This is a source-level finding, not a +measured failure threshold. + +**Durability caveat:** the PGlite probe reports SQL `fsync=off` (its startup uses +`-F`). PGlite also has filesystem-level synchronization and `relaxedDurability` +handling. Do not infer identical persistence guarantees—or no persistence—from +SQL settings alone. The same WAL buffer size cannot explain or normalize the +cost of commits across different storage implementations. No durability setting +was changed here. + +### PGlite evidence and refreshing the comparison + +- [P1 — backend linker settings](https://github.com/electric-sql/postgres-pglite/blob/b133782cd759f08b3aeb263b80a963b39c7b7af1/build-pglite.sh#L153). +- [P2 — pinned JS memory construction](https://github.com/electric-sql/pglite/blob/ae182ff8bd5ba4acb887d6c925d607a1498aa0b5/packages/pglite/src/pglite.ts#L317). +- [P3 — published 0.5.8 core source map](https://unpkg.com/@electric-sql/pglite@0.5.8/dist/index.js.map), original `../src/pglite.ts`: receive/growth code, raw execution, mutexes and startup flags. +- [P4 — pinned PostgreSQL protocol implementation](https://github.com/electric-sql/postgres-pglite/blob/b133782cd759f08b3aeb263b80a963b39c7b7af1/src/backend/libpq/pqcomm.c#L119). +- [P5 — published OPFS AHP source map](https://unpkg.com/@electric-sql/pglite@0.5.8/dist/fs/opfs-ahp.js.map), original `../../src/fs/opfs-ahp.ts`. +- [P6 — published module-cache source map](https://unpkg.com/@electric-sql/pglite@0.5.8/dist/chunk-NNS5RQRF.js.map), original `../../pglite-utils/src/utils.ts`. +- [P7 — published TAR helpers source map](https://unpkg.com/@electric-sql/pglite@0.5.8/dist/chunk-DDJLRBDX.js.map), original `../src/fs/tarUtils.ts`. + +To refresh, pin the package version, open a fresh instance with no startup +overrides, and query `pg_settings` for the names above. Convert `8kB` units to +bytes before comparing: `shared_buffers=16384` is 128 MiB, not 16 KiB. +Record storage mode, PostgreSQL version and caller overrides. Keep package +observations separate from newer source-main settings. "No counterpart found" +means no equivalent in the inspected path, not a claim about every PGlite +extension, third-party proxy or future release. + +## Stacks are a different resource + +| Owner | Existing setting | Meaning | +| --- | --- | --- | +| Native `liboliphaunt_runtime.c::DEFAULT_BACKEND_STACK_BYTES` | 8 MiB | Native PostgreSQL backend thread stack; `OLIPHAUNT_STACK_BYTES` is its existing override. It does not configure WASIX. | +| `src/wasix/runtime/assets/build/profile_flags.sh::OLIPHAUNT_WASM_GUEST_STACK_SIZE` | `8MB` | Guest C/shadow stack within Wasm linear memory. Shared by backend and initdb links, not an environment override. This is not the native machine-code call stack used by an AOT engine. | +| Same file, `OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE` | `128MB` | Initial guest linear-memory size, containing the C stack and heap; not all query memory is eagerly touched. Larger values increase the instance memory floor, not necessarily throughput. | +| Wasmer execution stack | Separate engine-owned allocation | AOT/native call frames use this stack. PostgreSQL's shadow-stack depth check alone cannot establish that enough native stack remains. | +| PostgreSQL `max_stack_depth` | SQL-configurable guard | Recursion safety threshold, not an allocation and not a transport budget. Raising it does not allocate either stack. | + +Wasmer's native execution stack can exhaust before PostgreSQL's guest C-stack +guard detects excessive recursion. Increasing `max_stack_depth` does not fix +this: it relaxes PostgreSQL's guard without allocating more engine stack. +There is no production remaining-native-stack guard in the embedded runtime; +an engine overflow is terminal, not a recoverable SQL error. + +The backend/initdb link sizes are included in the build-profile signature so +an incremental producer cannot silently reuse the old memory layout after a +constant changes. Changing them requires regenerated guest/AOT artifacts, not +a consumer runtime flag. The protocol generator likewise emits C, JavaScript, +and packaged Rust bounds from one JSON owner; generated-view checks detect drift. + +Postmaster has a separate carrier stack setting (currently a 32 MiB default in +its runner scripts). That is neither an embedded SDK default nor evidence that +the embedded recovery issue is fixed. Keep carrier and embedded validation +separate. + +## Diagnostic retention + +Rust `postgres_mod.rs::DIAGNOSTIC_TAIL_BYTES` retains the latest 8 KiB from each +split-initdb stdout and stderr stream. The backend separately retains 16 KiB +of stderr in `instantiate_wasix_module`; browser patch +`0018-wasmer-js-bound-direct-stderr.patch::STDERR_LIMIT_BYTES` also uses 16 KiB. +These matching backend values are not generated from the initdb constant. +`STARTUP_OUTCOME_MAX_PROTOCOL_BYTES` +allows up to 1 MiB of startup-error protocol payload and must match the guest +startup descriptor ABI in `oliphaunt_wasix_bridge.c`. These are diagnostic bounds, +not SQL result caps. Raising them retains/copies more diagnostic data; it cannot +speed up successful queries or repair error recovery. + +Native `liboliphaunt_internal.h::OLIPHAUNT_ERROR_CAPACITY` and Rust SDK +`liboliphaunt/ffi.rs::ERROR_CAPTURE_CAPACITY` use 1 KiB for error text. Native +temporary error strings also use 128/256/512/1,024 bytes; symbol-scope diagnostics +use 512 bytes. Deno's error-copy buffer starts at 1 KiB and can grow. +The JS broker startup-ready line has an 8 KiB limit in `runtime/node-adapter.ts`. +These determine how much diagnostic text is retained/accepted, not query speed. + +## Postmaster-specific budgets + +Do not apply these numbers to the embedded Rust/TS defaults above. + +| Owner under `src/wasix/postmaster/` | Selection | In simple words; effect | +| --- | --- | --- | +| `bin/run-release-carrier.sh`; builder/validation scripts and `lib/sealed-carrier.sh` | 32 MiB execution stack default | More space for native Wasm calls. Release runner uses `OLIPHAUNT_WASIX_POSTMASTER_STACK_SIZE`; build/validation uses `WASMER_STACK_SIZE`. Not a throughput buffer. | +| `lib/common.sh`, fresh linear-memory profile | 4,096 Wasm pages = 256 MiB maximum | A ceiling on guest linear-memory growth for that profile, not a fixed allocation per query. | +| Same profile and Wasmer patch `0008-postmaster-executor-and-build-closure.patch` | 65,536-page = 4 GiB static address bound, plus 2 GiB guard reservation on the specified U64 engine layout | Reserved address space for safe memory access. This does **not** mean 6 GiB of physical RAM is filled for every instance. It is part of the engine safety contract, not free memory to give to SQL. | +| `profiles/runtime-footprints/embedded-concurrent-v1.gucs` | 32 MiB shared buffers; 8 connections | This specific profile trades cache and connection capacity for a smaller footprint. Not the embedded SDK's 128 MiB cache default. | +| Wasmer patch `0007-wasix-instance-linker-and-sealed-runtime.patch` | 64 KiB preinitialized-image alignment | Required layout boundary for the memory image. Not a file-read batch. | +| Wasmer patch `0002-virtual-fs-file-description-and-writeback.patch::ZERO_WRITE_CHUNK_LEN` | 8 KiB | Reusable batch for writing zeros. Larger batches may reduce write calls while using more scratch space. | + +## Archives, installation and startup verification + +These sizes mostly affect opening, backup/restore, downloading or building—not +steady-state SQL execution. +An archive ceiling allows or rejects an archive; it does not allocate that much +memory in advance. A ceiling checked **after** decompression does not bound +peak decompression memory: TS `archive.ts::decompressIfNeeded` currently calls +`decompressZstd` before `extractTar` applies its archive limit. Do not describe +that limit as a streaming decompression memory guarantee. + +| Owner (repository-relative) | Selection | Meaning / cost | +| --- | --- | --- | +| `src/wasix/sdks/ts/src/resources/archive.ts` | 2 GiB tar; 512 MiB entry; 65,536 entries; 1,024-byte paths; 64 path levels | Runtime archive acceptance limits. Larger allowances accept larger archives and increase worst-case processing/memory exposure. | +| `src/extensions/contracts/extension-artifact-archive-policy.properties` | 128 MiB compressed; 512 MiB expanded; 256 MiB member; 4,096 members | Separate extension archive policy shared across consumers. Not the general runtime archive policy. | +| `src/native/sdks/ts/src/native/assets.ts` | 4,096 extension-runtime files; 48 MiB per file; 256 MiB total; 4,096-byte path check | Limits installed extension runtime resources, not query results. | +| `tools/packaging/portable-archive.mts::DEFAULT_PORTABLE_ARCHIVE_LIMITS` | 512 MiB archive/member; 1 GiB expanded; 32,768 entries | Default package-verifier envelope; callers can use a role-specific envelope. | +| Rust WASIX `aot.rs`; runtime asset/tools/AOT/ICU crate `build.rs` files | 128 KiB hash-read batches | Scratch used when checking binaries. Can change verification time and scratch use, not execution speed of a verified query. | +| `src/native/rust-bindings/src/liboliphaunt/root/fingerprint.rs` | 32 KiB | Root fingerprint read batch, separate from the AOT hash batch. | +| `src/third-party/tools/source-fetch-core.mts` | 1 GiB download; 8 GiB checkout; 500,000 checkout entries; 64 KiB marker; 16 MiB patch file; 1 MiB hash reads | Source acquisition/build protection, checked against the refreshed owner. Increasing these permits larger inputs; it does not make database queries faster. | +| `src/third-party/tools/source-archive.mts` | 1 GiB archive; 200,000 members; 2 GiB member; at most 4 GiB expanded, further bounded by 200× compressed size with a 64 MiB minimum allowance | Shared tar/ZIP source extraction envelope in current main. It rejects excessive archives; an accepted archive still consumes build-machine memory and disk. | +| `tools/dev/maintainer-tools.toml` | 8 MiB pinned helper archive allowances | Tool download envelopes, not runtime memory. | +| `src/native/sdks/swift/tools/swift-carrier-resolver.mts` | 2 GiB carrier; 512 MiB ZIP; 1 GiB member; 4 GiB expanded; 32,768 entries | Swift installation envelope. | +| `src/native/sdks/react-native/tools/stage-ios-app.mts` | Same byte envelopes; 4,096 ordinary entries, 16,384 bundled-resource entries; 1,024 legal files, 16 MiB each | Role-specific mobile installation limits. They are not all identical to the Swift limits. | +| `src/native/sdks/react-native/tools/mobile-extension-artifact-paths.mts` | 2 GiB artifact; 4,096 bundle members; 32,768 archive members; 4 GiB expanded | Mobile extension installation envelope. | +| `src/native/sdks/react-native/tools/ios-app-transport.mts` | 256 MiB ZIP directory; 1,000,000 entries; 4 KiB name; 64 KiB symlink target | App transport metadata limits; no direct SQL effect. | +| `src/native/sdks/swift/tools/render-extension-products.mts` | 32,768 files; 512 MiB file; 2 GiB tree | XCFramework inspection envelope. | +| Android Gradle plugin `ResolveOliphauntAndroidAssetsTask.java`, `OliphauntExtensionLegalCatalog.java`, `LinkOliphauntAndroidExtensionsTask.java` | 64 KiB artifact manifest; 1 MiB bundle/registry/legal-resource metadata; 4 MiB linker output | Metadata and diagnostic limits during Android builds. | +| Android archive/resolver/linker helpers | 128 KiB extraction/hash, 64 KiB reads, 16 KiB linker-output reads | Independent build-time I/O batches. | +| Postmaster `lib/durable_publication.py` | 16 MiB comparison; 256 MiB publication; 1 MiB copy batches | Safe publication of carrier metadata/files, not a cap on PGDATA. | +| Postmaster `lib/sealed_export_chain.py` and `linear_memory_transaction.py` | 512 MiB uninventoried input; 16 MiB small metadata; 1 MiB read/copy batches | Build/provenance validation envelopes. | +| Postmaster Wasmer patch `0008-postmaster-executor-and-build-closure.patch` | 4 MiB linear-memory receipt; 16 MiB export proof; 1 MiB manifest; 4 KiB proof line; 128 KiB hashes | Carrier admission/proof limits and verification batch, not runtime SQL buffers. | + +## Fixed formats, platform bounds, and unbounded growth + +- A normal WebAssembly page is 64 KiB; TAR records are 512 bytes. Standard + PostgreSQL data pages are 8 KiB, identifiers allow 63 bytes, and WAL segment + validation accepts power-of-two sizes from 1 MiB to 1 GiB. Those are format + rules, not interchangeable transport constants. Read the cluster's actual + segment size; a 1 MiB regression fixture is not the runtime default. +- The startup descriptor is 32 bytes; the broker header is 13 bytes; byte-channel + control is five 32-bit words. Digest lengths, TAR field widths, null + terminators and wire headers follow their formats. Changing them requires + changing the format/ABI, not performance tuning. +- Native path scratch includes a 4,096-byte cwd buffer and smaller leaf/name + arrays. Windows module-path discovery grows from `MAX_PATH` and stops + retrying after its capacity exceeds 32,768. These can affect accepted paths, + not SQL throughput. The Windows thread shim's 64 KiB `PTHREAD_STACK_MIN` + fallback is a platform lower bound, not the selected 8 MiB backend stack. +- Android `.so` packaging/linking uses 16 KiB alignment; other ZIP alignment + is 4 bytes, with 4 KiB signing padding. These are loading/layout constraints. +- Collected rows, native buffered output, whole backup archives and staged + in-memory files can grow beyond their starting buffers. A queue bounded in + *commands* does not bound the bytes held by those commands. This inventory + does not imply that total per-database memory is bounded. + +When changing a size, update this guide and its owning constant/contract, check +mirrored consumers, and test both small and large inputs. Include slow readers, +multiple databases and memory peaks. Do not add an environment flag merely to +make every number adjustable. + +## Small queries, large results, and tuning + +- Small queries do not wait to fill a read batch or queue. They do pay for any + eagerly allocated scratch/ring space and for each boundary crossing. Larger + buffers do not inherently improve their latency. +- Large operations may benefit from larger batches through fewer reads, + callbacks, or wakeups. Costs include larger scratch allocations, retained + capacity, longer producer bursts, and more queued data for slow consumers. +- Native `OLIPHAUNT_STREAM_QUEUE_MAX_BYTES` is an existing per-stream override + read when streaming starts. Lowering it increases backpressure; raising it + allows more data to accumulate. It is not a durability or query-size switch. +- Other transport sizes above are compile-time contract or implementation + choices, not consumer environment knobs. Change an ABI bound coherently across + guest and hosts; change an I/O batch only at its owner. Do not unify unrelated + constants just because both happen to be 64 KiB. +- Re-evaluate sizes using both query RTT and large/fragmented COPY/results, + including a slow consumer and cancellation/error recovery. Measure peak and + retained memory as well as throughput. A larger stack must also prove normal + SQL errors recover rather than merely moving the crash threshold. + +These budgets apply independently of memory-backed versus directory-backed +PGDATA. They do not relax WAL synchronization, filesystem durability, or Wasm +memory-access/trap semantics. diff --git a/src/docs/maintainers/wasix-postmaster.md b/src/docs/maintainers/wasix-postmaster.md index 09029b55d..483324981 100644 --- a/src/docs/maintainers/wasix-postmaster.md +++ b/src/docs/maintainers/wasix-postmaster.md @@ -81,6 +81,15 @@ expects it and must preserve the distinction between a still-running child, normal exit, signalled exit, and an unknown process. The supervisor owns cleanup; a failed backend cannot tear down another cluster's resources. +Inherited limitation: the pinned WASIX `pthread_sigmask` is a success-returning +stub. Signal delivery and child-wait support do **not** imply POSIX blocked masks, +handler `sa_mask`, or `sigsetjmp` mask restoration. Libc patch 0008 fixes argument +evaluation only. This is a correctness dependency for PostgreSQL startup, +error recovery and signal-sensitive critical sections; see the +[runtime capability inventory](../../wasix/postmaster/wasmer/README.md). +Do not treat host-adapter mask tests as proof of guest masking or replace the +missing delivery semantics with a libc-only remembered mask. + ## Sealed carrier The release carrier is compiler-free. Compilation happens in the trusted diff --git a/src/docs/maintainers/wasix-usage.md b/src/docs/maintainers/wasix-usage.md index 4af42933f..4b33512c2 100644 --- a/src/docs/maintainers/wasix-usage.md +++ b/src/docs/maintainers/wasix-usage.md @@ -154,6 +154,14 @@ TypeScript still has no cancellation or dedicated typed COPY API. Those absences are recorded in the SDK parity policy rather than represented as capability flags. +In the current Rust WASIX runtime, accepting SQL `statement_timeout` does not +guarantee interruption of CPU-bound guest queries. A caller-side future timeout +also does not stop guest execution. Neither is an execution-resource boundary; +the missing timer/interrupt delivery semantics are separate from SQL error +recovery and from Native PostgreSQL's timeout behavior. See +[resource budgets](runtime-resource-budgets.md) for the separate stack, buffer, +queue and output limits; increasing a buffer does not repair interruption. + Across both bindings, selecting extensions supplies exact runtime artifacts, dependencies, and required pre-start preload/GUC configuration only. Open and server start never execute database-local `CREATE EXTENSION`, `ALTER diff --git a/src/examples/wasix/browser/configured-identity-smoke.ts b/src/examples/wasix/browser/configured-identity-smoke.ts new file mode 100644 index 000000000..81794c751 --- /dev/null +++ b/src/examples/wasix/browser/configured-identity-smoke.ts @@ -0,0 +1,77 @@ +import Oliphaunt, { PostgresError } from '@oliphaunt/wasix-ts'; +import WorkerOliphaunt from '@oliphaunt/wasix-ts/worker'; +import { indexedDB } from '@oliphaunt/wasix-ts/storage/indexed-db'; + +/** Exercise real startup identity on the same durable root in both browser placements. */ +export async function expectConfiguredIdentity(): Promise { + const storage = indexedDB(`browser-identity-${crypto.randomUUID()}`); + const admin = await Oliphaunt.open({ storage }); + try { + for (const sql of [ + 'CREATE ROLE browser_reader LOGIN', + 'CREATE ROLE browser_no_login NOLOGIN', + 'ALTER ROLE browser_reader SET default_statistics_target = 137', + 'CREATE TABLE browser_login_events(who name)', + `CREATE FUNCTION browser_login_event() RETURNS event_trigger LANGUAGE plpgsql + SECURITY DEFINER SET search_path = pg_catalog, public AS $$ + BEGIN INSERT INTO public.browser_login_events VALUES (session_user); END $$`, + 'CREATE EVENT TRIGGER browser_login ON login EXECUTE FUNCTION browser_login_event()', + 'GRANT SELECT ON browser_login_events TO browser_reader', + ]) { + await admin.execute(sql); + } + } finally { + await admin.close(); + } + + let expectedLogins = 0; + for (const placement of [Oliphaunt, WorkerOliphaunt, Oliphaunt]) { + const database = await placement.open({ storage, username: 'browser_reader' }); + expectedLogins += 1; + try { + for (const reset of [undefined, 'RESET ROLE', 'DISCARD ALL']) { + if (reset !== undefined) await database.execute(reset); + const row = await database.queryRaw(` + SELECT current_user::text AS current_role, session_user::text AS session_role, + (system_user IS NULL)::text AS no_auth_identity, + current_setting('default_statistics_target') AS statistics_target, + (SELECT count(*)::text FROM browser_login_events + WHERE who = session_user) AS logins + `); + for (const [column, expected] of Object.entries({ + current_role: 'browser_reader', + session_role: 'browser_reader', + no_auth_identity: 'true', + statistics_target: '137', + logins: String(expectedLogins), + })) { + if (row.getText(0, column) !== expected) { + throw new Error( + `browser configured identity ${column}: expected ${expected}, got ${row.getText(0, column)}`, + ); + } + } + } + try { + await database.execute('SET ROLE postgres'); + throw new Error('browser configured role unexpectedly acquired superuser rights'); + } catch (error) { + if (!(error instanceof PostgresError) || error.sqlstate !== '42501') throw error; + } + } finally { + await database.close(); + } + } + + for (const placement of [Oliphaunt, WorkerOliphaunt]) { + let rejected = false; + try { + const unexpected = await placement.open({ storage, username: 'browser_no_login' }); + await unexpected.close(); + } catch (error) { + if (!(error instanceof PostgresError) || error.sqlstate !== '28000') throw error; + rejected = true; + } + if (!rejected) throw new Error('browser accepted a configured NOLOGIN role'); + } +} diff --git a/src/examples/wasix/browser/main.ts b/src/examples/wasix/browser/main.ts index e9da7f498..9b3b87518 100644 --- a/src/examples/wasix/browser/main.ts +++ b/src/examples/wasix/browser/main.ts @@ -13,6 +13,7 @@ import WorkerOliphaunt from '@oliphaunt/wasix-ts/worker'; import { standardSeed } from './resources.js'; import { expectStructuredApi } from './structured-api-smoke.js'; +import { expectConfiguredIdentity } from './configured-identity-smoke.js'; const status = requireElement('status'); const sql = requireElement('sql'); @@ -102,6 +103,7 @@ try { await readPgUuidv7(database); } await database.close(); + await expectConfiguredIdentity(); smokePhase('OPFS persistence'); const opfsAnswers = await expectOpfsPersistence(extensions); smokePhase('OPFS crash recovery'); @@ -115,6 +117,7 @@ try { pgtap: pgtapVersion, startupSqlstate: '3D000', directWorkers: 0, + configuredIdentity: true, opfsTransport: 'synchronous-access', opfsCrashAnswer: opfsCrash.answer, opfsCrashRelations: opfsCrash.relations, @@ -307,10 +310,11 @@ async function expectOwnedRawProtocolResponse(database: OliphauntDatabase): Prom simpleQuery("SELECT repeat('a', 10240) AS retained_payload"), ); const snapshot = retained.slice(); + // Many ordinary rows must not hit an application-specific 64 MiB quota. const large = await database.execProtocolRaw( - simpleQuery("SELECT repeat('z', 1048576) AS large_payload"), + simpleQuery("SELECT repeat('z', 8192) AS large_payload FROM generate_series(1, 10240)"), ); - if (large.byteLength < 1048576) { + if (large.byteLength < 80 * 1024 * 1024) { throw new Error( `browser worker returned a truncated large PGWire response: ${large.byteLength}`, ); @@ -321,6 +325,7 @@ async function expectOwnedRawProtocolResponse(database: OliphauntDatabase): Prom ) { throw new Error('browser worker response changed after the guest reused its output memory'); } + await expectAnswer(database); } async function expectClockConsistency(database: OliphauntDatabase): Promise { diff --git a/src/native/broker/src/server.rs b/src/native/broker/src/server.rs index 37a0d4a64..5c9bd6d7d 100644 --- a/src/native/broker/src/server.rs +++ b/src/native/broker/src/server.rs @@ -69,7 +69,7 @@ impl Write for Socket { type Reply = mpsc::Sender>; enum Command { - Startup(String, Reply>), + Startup(Option, Reply>), Execute(Vec, Socket, Reply<()>), Reset(Reply<()>), Backup(Reply>), @@ -222,7 +222,7 @@ fn run_backend( for command in receiver { match command { Command::Startup(application, reply) => { - let _ = reply.send(startup_parameters(&mut session, &application)); + let _ = reply.send(startup_parameters(&mut session, application.as_deref())); } Command::Execute(request, mut socket, reply) => { if shared.closing.load(Ordering::Acquire) { @@ -281,18 +281,25 @@ fn run_backend( session.close_terminal().map_err(other) } -fn startup_parameters(session: &mut NativeSession, application: &str) -> io::Result> { - let request = oliphaunt_query::extended_statement( - "SELECT set_config('application_name', $1, false)", - &[oliphaunt_query::Parameter::text(application)], - 0, - ) - .map_err(other)?; - parse_query_response( - &session.exec_protocol_raw(&request).map_err(other)?, - ExpectedProtocol::Extended, - ) - .map_err(other)?; +fn startup_parameters( + session: &mut NativeSession, + application: Option<&str>, +) -> io::Result> { + // Omission preserves PostgreSQL's configured default; an explicit empty + // string is still a client override, just like any other supplied value. + if let Some(application) = application { + let request = oliphaunt_query::extended_statement( + "SELECT set_config('application_name', $1, false)", + &[oliphaunt_query::Parameter::text(application)], + 0, + ) + .map_err(other)?; + parse_query_response( + &session.exec_protocol_raw(&request).map_err(other)?, + ExpectedProtocol::Extended, + ) + .map_err(other)?; + } let bytes = session.exec_simple_query("SELECT name, setting FROM pg_settings WHERE name IN ('server_version','server_encoding','client_encoding','application_name','DateStyle','IntervalStyle','TimeZone','integer_datetimes','standard_conforming_strings')").map_err(other)?; let result = parse_query_response(&bytes, ExpectedProtocol::Simple).map_err(other)?; let mut response = backend_frame(b'R', &0_i32.to_be_bytes()); @@ -459,7 +466,7 @@ fn sql_client(mut socket: Socket, shared: &Shared) -> io::Result<()> { let result = connected_sql( &mut socket, shared, - parameters.get("application_name").copied().unwrap_or(""), + parameters.get("application_name").copied(), ); shared.active.lock().unwrap().take(); let (reply, done) = mpsc::channel(); @@ -470,11 +477,15 @@ fn sql_client(mut socket: Socket, shared: &Shared) -> io::Result<()> { result } -fn connected_sql(socket: &mut Socket, shared: &Shared, application: &str) -> io::Result<()> { +fn connected_sql( + socket: &mut Socket, + shared: &Shared, + application: Option<&str>, +) -> io::Result<()> { let (reply, done) = mpsc::channel(); shared .commands - .send(Command::Startup(application.to_owned(), reply)) + .send(Command::Startup(application.map(str::to_owned), reply)) .map_err(other)?; let mut response = done.recv().map_err(other)??; let mut key = [0; 8]; diff --git a/src/native/broker/tests/postgres_client.rs b/src/native/broker/tests/postgres_client.rs index a8afc35fa..7497c9bcd 100644 --- a/src/native/broker/tests/postgres_client.rs +++ b/src/native/broker/tests/postgres_client.rs @@ -41,6 +41,8 @@ impl Broker { "127.0.0.1:0", "--control-listen", "127.0.0.1:0", + "--startup-guc", + "application_name=broker-default", ]) .env("OLIPHAUNT_BROKER_AUTH_TOKEN", "actual-consumer-test-secret") .stdout(Stdio::piped()) @@ -172,6 +174,46 @@ fn query(stream: &mut TcpStream, sql: &str) -> Vec { result } +#[test] +#[ignore = "requires a prepared native runtime; runs the actual broker executable"] +fn startup_application_name_distinguishes_omission_from_empty() { + for application in [None, Some(""), Some("client-application")] { + let broker = Broker::start(); + let _owner = broker.owner(); + let mut client = connect(&broker.sql); + let mut startup = oliphaunt_query::wire::PROTOCOL_3.to_be_bytes().to_vec(); + startup.extend_from_slice(b"user\0postgres\0database\0postgres\0"); + if let Some(value) = application { + startup.extend_from_slice(format!("application_name\0{value}\0").as_bytes()); + } + startup.push(0); + client + .write_all(&((startup.len() + 4) as u32).to_be_bytes()) + .unwrap(); + client.write_all(&startup).unwrap(); + assert_eq!(response(&mut client), frame(b'R', &3_i32.to_be_bytes())); + client + .write_all(&frame(b'p', b"actual-consumer-test-secret\0")) + .unwrap(); + loop { + let message = response(&mut client); + assert_ne!(message[0], b'E', "startup failed: {message:?}"); + if message[0] == b'Z' { + break; + } + } + let result = oliphaunt_query::parse_query_response( + &query(&mut client, "SHOW application_name"), + oliphaunt_query::ExpectedProtocol::Simple, + ) + .unwrap(); + assert_eq!( + result.get_text(0, "application_name").unwrap(), + Some(application.unwrap_or("broker-default")), + ); + } +} + #[test] #[ignore = "requires a prepared native runtime; runs the actual broker executable"] fn postgres_protocol_and_management_are_independent() { diff --git a/src/native/postgres-tools/crates/tools/README.md b/src/native/postgres-tools/crates/tools/README.md index 24bdc14b7..64b859b69 100644 --- a/src/native/postgres-tools/crates/tools/README.md +++ b/src/native/postgres-tools/crates/tools/README.md @@ -7,3 +7,9 @@ It selects the matching `oliphaunt-tools-*` artifact crate for the Cargo target and exposes thin `pg_dump` and non-interactive `psql` functions. The core `oliphaunt` SDK does not depend on this crate. Set `OLIPHAUNT_TOOLS_DIR` only when overriding packaged tool discovery during development. + +The API captures complete stdout and stderr in memory, subject to available +memory and host string-size limits. A capture failure discards both partial +outputs while the child pipes are drained and the child is reaped. For dumps +that should not stay in memory, use an external file/stream-output tool; this +API does not yet offer a streaming sink. diff --git a/src/native/postgres-tools/crates/tools/src/lib.rs b/src/native/postgres-tools/crates/tools/src/lib.rs index c8a650aa8..7dee97038 100644 --- a/src/native/postgres-tools/crates/tools/src/lib.rs +++ b/src/native/postgres-tools/crates/tools/src/lib.rs @@ -6,10 +6,11 @@ mod arguments; use std::error::Error as StdError; use std::ffi::OsString; use std::fmt; -use std::io::Write; +use std::io::{self, Read, Write}; use std::path::{Path, PathBuf}; use std::process::{Command, Stdio}; -use std::thread; +use std::sync::{Arc, Mutex}; +use std::thread::{self, JoinHandle}; use arguments::{validate_pg_dump_arguments, validate_psql_arguments}; @@ -19,7 +20,10 @@ pub const PRODUCT: &str = "oliphaunt-tools"; /// Artifact kind relayed by this facade crate. pub const KIND: &str = "native-tools"; -/// Options for a plain-text `pg_dump`. +/// Options for the in-memory, plain-text `pg_dump` convenience API. +/// +/// Complete stdout and stderr are retained in memory. Available memory and +/// host string sizes bound the output; a streaming API is not yet available. #[derive(Debug, Clone, Default, PartialEq, Eq)] pub struct PgDumpOptions { args: Vec, @@ -44,7 +48,10 @@ impl PgDumpOptions { } } -/// Options for a non-interactive `psql` invocation. +/// Options for an in-memory, non-interactive `psql` invocation. +/// +/// Complete stdout and stderr are retained in memory. Available memory and +/// host string sizes bound the output; a streaming API is not yet available. #[derive(Debug, Clone, Default, PartialEq, Eq)] pub struct PsqlOptions { args: Vec, @@ -96,14 +103,24 @@ pub struct PostgresToolError { /// Process exit status when the program started. pub exit_code: Option, /// UTF-8 standard output captured before failure. + /// + /// This is empty after output-capture failure so no partial prefix is + /// exposed. pub stdout: String, /// UTF-8 standard error captured before failure. + /// + /// This is empty after output-capture failure so no partial prefix is + /// exposed. pub stderr: String, source: Option, + output_capture_failure: Option, } impl fmt::Display for PostgresToolError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + if let Some(failure) = self.output_capture_failure { + return write!(formatter, "{} output capture failed: {failure}", self.tool); + } if let Some(source) = &self.source { return write!(formatter, "could not run {}: {source}", self.tool); } @@ -131,6 +148,9 @@ impl StdError for PostgresToolError { } /// Run packaged `pg_dump` against a PostgreSQL connection string. +/// +/// Returns complete PostgreSQL UTF-8 output in memory. Capture failures return +/// [`PostgresToolError`] without partial output. Streaming is not yet supported. pub fn pg_dump( connection_string: &str, options: PgDumpOptions, @@ -150,6 +170,9 @@ pub fn pg_dump( } /// Run packaged non-interactive `psql` against a PostgreSQL connection string. +/// +/// Returns complete PostgreSQL UTF-8 output in memory. Capture failures return +/// [`PostgresToolError`] without partial output. Streaming is not yet supported. pub fn psql(connection_string: &str, options: PsqlOptions) -> Result { validate_connection_string("psql", connection_string)?; validate_psql_arguments(&options.args) @@ -223,7 +246,25 @@ fn run_tool( stdout: String::new(), stderr: String::new(), source: Some(source), + output_capture_failure: None, })?; + let captured = Arc::new(Mutex::new(CapturedOutput::default())); + let stdout_reader = spawn_output_reader( + child + .stdout + .take() + .expect("stdout is piped for PostgreSQL tools"), + Arc::clone(&captured), + OutputChannel::Stdout, + ); + let stderr_reader = spawn_output_reader( + child + .stderr + .take() + .expect("stderr is piped for PostgreSQL tools"), + Arc::clone(&captured), + OutputChannel::Stderr, + ); // Drain stdout/stderr while a potentially large psql script is written. // Writing all stdin first can deadlock when the child fills an output pipe. let input_writer = stdin.and_then(|input| { @@ -232,51 +273,209 @@ fn run_tool( .take() .map(|mut writer| thread::spawn(move || writer.write_all(&input))) }); - let output = child - .wait_with_output() - .map_err(|source| PostgresToolError { - tool, - exit_code: None, - stdout: String::new(), - stderr: String::new(), - source: Some(source), - })?; + let status = child.wait(); + if status.is_err() { + // A failed wait can leave the process and its pipe readers alive. + // Terminate it before joining the readers so this error path cannot + // strand background threads. + let _ = child.kill(); + let _ = child.wait(); + } + let stdout_failure = join_output_reader(stdout_reader, "stdout"); + let stderr_failure = join_output_reader(stderr_reader, "stderr"); let input_failure = input_writer.and_then(|writer| match writer.join() { Ok(Ok(())) => None, Ok(Err(error)) => Some(error), Err(_) => Some(std::io::Error::other("psql stdin writer panicked")), }); - if !output.status.success() { + let captured = take_captured_output(&captured); + let exit_code = status + .as_ref() + .ok() + .and_then(std::process::ExitStatus::code); + if let Some(failure) = captured.failure { return Err(PostgresToolError { tool, - exit_code: output.status.code(), - stdout: String::from_utf8_lossy(&output.stdout).into_owned(), - stderr: String::from_utf8_lossy(&output.stderr).into_owned(), + exit_code, + stdout: String::new(), + stderr: String::new(), source: None, + output_capture_failure: Some(failure), + }); + } + let status = match status { + Ok(status) => status, + Err(source) => return Err(io_failure(tool, exit_code, captured, source)), + }; + if let Some(source) = stdout_failure.or(stderr_failure) { + return Err(io_failure(tool, status.code(), captured, source)); + } + let CapturedOutput { stdout, stderr, .. } = captured; + if !status.success() { + return Err(PostgresToolError { + tool, + exit_code: status.code(), + stdout: String::from_utf8_lossy(&stdout).into_owned(), + stderr: String::from_utf8_lossy(&stderr).into_owned(), + source: None, + output_capture_failure: None, }); } if let Some(source) = input_failure { return Err(PostgresToolError { tool, - exit_code: output.status.code(), - stdout: String::from_utf8_lossy(&output.stdout).into_owned(), - stderr: String::from_utf8_lossy(&output.stderr).into_owned(), + exit_code: status.code(), + stdout: String::from_utf8_lossy(&stdout).into_owned(), + stderr: String::from_utf8_lossy(&stderr).into_owned(), source: Some(source), + output_capture_failure: None, }); } - String::from_utf8(output.stdout).map_err(|error| PostgresToolError { + String::from_utf8(stdout).map_err(|error| PostgresToolError { tool, - exit_code: output.status.code(), + exit_code: status.code(), stdout: String::from_utf8_lossy(error.as_bytes()).into_owned(), stderr: format!( "{}{} produced non-UTF-8 output: {error}", - String::from_utf8_lossy(&output.stderr), + String::from_utf8_lossy(&stderr), tool ), source: None, + output_capture_failure: None, }) } +#[derive(Debug, Clone, Copy)] +enum OutputChannel { + Stdout, + Stderr, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum OutputCaptureFailure { + AllocationFailed, + LockPoisoned, +} + +impl fmt::Display for OutputCaptureFailure { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::AllocationFailed => { + formatter.write_str("could not reserve memory for tool output") + } + Self::LockPoisoned => { + formatter.write_str("stdout and stderr capture lock was poisoned") + } + } + } +} + +#[derive(Debug, Default)] +struct CapturedOutput { + stdout: Vec, + stderr: Vec, + failure: Option, +} + +impl CapturedOutput { + fn retain(&mut self, channel: OutputChannel, bytes: &[u8]) { + if self.failure.is_some() { + return; + } + let destination = match channel { + OutputChannel::Stdout => &mut self.stdout, + OutputChannel::Stderr => &mut self.stderr, + }; + // Vec checks length overflow and grows amortized across successive reads. + if destination.try_reserve(bytes.len()).is_err() { + self.fail(OutputCaptureFailure::AllocationFailed); + return; + } + destination.extend_from_slice(bytes); + } + + fn fail(&mut self, failure: OutputCaptureFailure) { + self.failure.get_or_insert(failure); + self.stdout = Vec::new(); + self.stderr = Vec::new(); + } +} + +fn spawn_output_reader( + reader: R, + captured: Arc>, + channel: OutputChannel, +) -> JoinHandle> +where + R: Read + Send + 'static, +{ + thread::spawn(move || drain_output(reader, &captured, channel)) +} + +fn drain_output( + mut reader: impl Read, + captured: &Mutex, + channel: OutputChannel, +) -> io::Result { + let mut buffer = [0_u8; 32 * 1024]; + let mut observed_bytes = 0_usize; + loop { + let count = match reader.read(&mut buffer) { + Err(error) if error.kind() == io::ErrorKind::Interrupted => continue, + result => result?, + }; + if count == 0 { + return Ok(observed_bytes); + } + observed_bytes = observed_bytes.saturating_add(count); + match captured.lock() { + Ok(mut captured) => captured.retain(channel, &buffer[..count]), + Err(poisoned) => { + let mut captured = poisoned.into_inner(); + captured.fail(OutputCaptureFailure::LockPoisoned); + } + } + } +} + +fn join_output_reader(reader: JoinHandle>, channel: &str) -> Option { + match reader.join() { + Ok(Ok(_)) => None, + Ok(Err(error)) => Some(error), + Err(_) => Some(io::Error::other(format!( + "PostgreSQL tool {channel} reader panicked" + ))), + } +} + +fn take_captured_output(captured: &Mutex) -> CapturedOutput { + let mut captured = match captured.lock() { + Ok(captured) => captured, + Err(poisoned) => { + let mut captured = poisoned.into_inner(); + captured.fail(OutputCaptureFailure::LockPoisoned); + captured + } + }; + std::mem::take(&mut *captured) +} + +fn io_failure( + tool: &'static str, + exit_code: Option, + captured: CapturedOutput, + source: io::Error, +) -> PostgresToolError { + PostgresToolError { + tool, + exit_code, + stdout: String::from_utf8_lossy(&captured.stdout).into_owned(), + stderr: String::from_utf8_lossy(&captured.stderr).into_owned(), + source: Some(source), + output_capture_failure: None, + } +} + fn resolve_tool(tool: &'static str) -> Result { let executable = if cfg!(windows) { format!("{tool}.exe") @@ -367,12 +566,14 @@ fn configuration_error(tool: &'static str, message: &str) -> PostgresToolError { stdout: String::new(), stderr: message.to_owned(), source: None, + output_capture_failure: None, } } #[cfg(test)] mod tests { use super::*; + use std::io::Cursor; #[test] fn psql_scripts_explicitly_read_standard_input() { @@ -395,4 +596,130 @@ mod tests { ); assert!(stdin.is_none()); } + + #[test] + fn captured_output_preserves_both_streams() { + let mut captured = CapturedOutput::default(); + captured.retain(OutputChannel::Stdout, b"out"); + captured.retain(OutputChannel::Stderr, b"err!"); + + assert_eq!(captured.failure, None); + assert_eq!(captured.stdout, b"out"); + assert_eq!(captured.stderr, b"err!"); + } + + #[test] + fn captured_output_accepts_more_than_64_mib() { + let mut captured = CapturedOutput::default(); + let chunk = vec![b'x'; 1024 * 1024]; + for _ in 0..65 { + captured.retain(OutputChannel::Stdout, &chunk); + } + captured.retain(OutputChannel::Stderr, b"diagnostic"); + assert_eq!(captured.failure, None); + assert_eq!(captured.stdout.len(), 65 * chunk.len()); + assert!(captured.stdout.iter().all(|&byte| byte == b'x')); + assert_eq!(captured.stderr, b"diagnostic"); + } + + #[test] + fn captured_output_discards_everything_after_failure() { + let mut captured = CapturedOutput::default(); + captured.retain(OutputChannel::Stdout, b"abc"); + captured.retain(OutputChannel::Stderr, b"def"); + captured.fail(OutputCaptureFailure::AllocationFailed); + captured.retain(OutputChannel::Stdout, b"later output"); + + assert_eq!( + captured.failure, + Some(OutputCaptureFailure::AllocationFailed) + ); + assert_eq!(captured.stdout.capacity(), 0); + assert_eq!(captured.stderr.capacity(), 0); + } + + #[test] + fn output_reader_continues_draining_after_capture_failure() { + let bytes = vec![b'x'; 32 * 1024 + 17]; + let mut failed = CapturedOutput::default(); + failed.fail(OutputCaptureFailure::AllocationFailed); + let captured = Mutex::new(failed); + + let observed = drain_output(Cursor::new(&bytes), &captured, OutputChannel::Stdout) + .expect("reader should drain all input"); + let captured = captured + .into_inner() + .expect("capture lock should remain usable"); + + assert_eq!(observed, bytes.len()); + assert_eq!( + captured.failure, + Some(OutputCaptureFailure::AllocationFailed) + ); + assert!(captured.stdout.is_empty()); + assert!(captured.stderr.is_empty()); + } + + #[test] + fn output_reader_retries_interrupted_reads() { + struct InterruptedOnce(bool); + impl Read for InterruptedOnce { + fn read(&mut self, _buffer: &mut [u8]) -> io::Result { + if !self.0 { + self.0 = true; + return Err(io::ErrorKind::Interrupted.into()); + } + Ok(0) + } + } + let captured = Mutex::new(CapturedOutput::default()); + assert_eq!( + drain_output(InterruptedOnce(false), &captured, OutputChannel::Stdout).unwrap(), + 0 + ); + assert_eq!(captured.lock().unwrap().failure, None); + } + + #[test] + fn output_capture_error_contains_no_partial_output() { + let error = PostgresToolError { + tool: "pg_dump", + exit_code: Some(0), + stdout: String::new(), + stderr: String::new(), + source: None, + output_capture_failure: Some(OutputCaptureFailure::AllocationFailed), + }; + + assert_eq!( + error.to_string(), + "pg_dump output capture failed: could not reserve memory for tool output" + ); + assert!(error.stdout.is_empty()); + assert!(error.stderr.is_empty()); + } + + #[test] + fn poisoned_capture_is_fail_closed_and_does_not_stop_draining() { + let captured = Arc::new(Mutex::new(CapturedOutput::default())); + let poison_target = Arc::clone(&captured); + let _ = std::thread::spawn(move || { + let _guard = poison_target.lock().unwrap(); + panic!("poison the capture lock"); + }) + .join(); + + let observed = drain_output( + Cursor::new(b"all bytes still drained"), + &captured, + OutputChannel::Stdout, + ) + .expect("poisoning must not stop pipe draining"); + let captured = take_captured_output(&captured); + + assert_eq!(observed, b"all bytes still drained".len()); + assert_eq!(captured.failure, Some(OutputCaptureFailure::LockPoisoned)); + assert!(captured.stdout.is_empty()); + assert!(captured.stderr.is_empty()); + } } diff --git a/src/native/postgres-tools/moon.yml b/src/native/postgres-tools/moon.yml index 3462e4809..dcb153b57 100644 --- a/src/native/postgres-tools/moon.yml +++ b/src/native/postgres-tools/moon.yml @@ -39,15 +39,21 @@ tasks: set -e cargo build -p oliphaunt-tools --locked bun run --cwd src/native/postgres-tools/npm build - inputs: ["crates/tools/**/*", "@group(cargo-workspace)", "npm/index.mts", "/tools/packaging/emit-javascript.mts"] + inputs: ["crates/tools/**/*", "@group(cargo-workspace)", "npm/index.mts", "npm/output-capture.mjs", "/tools/packaging/emit-javascript.mts"] outputs: ["npm/index.js", "/target/debug/liboliphaunt_tools.rlib"] options: runFromWorkspaceRoot: true test: tags: ["quality", "unit", "requires-rust"] - command: "cargo test -p oliphaunt-tools --locked" + script: | + set -e + cargo test -p oliphaunt-tools --locked + node --test src/native/postgres-tools/npm/output-capture.test.mjs inputs: - "crates/tools/**/*" + - "npm/index.mts" + - "npm/output-capture.mjs" + - "npm/output-capture.test.mjs" - "@group(cargo-workspace)" - project: "shared-test-fixtures" group: "fixtures" diff --git a/src/native/postgres-tools/npm/README.md b/src/native/postgres-tools/npm/README.md index 879a84036..fb68979af 100644 --- a/src/native/postgres-tools/npm/README.md +++ b/src/native/postgres-tools/npm/README.md @@ -17,3 +17,9 @@ non-interactive and accepts `command`, `script`, or ordinary passthrough arguments. Failures reject with `PostgresToolError`, including the exit code or signal and captured standard output and standard error. + +The API captures complete stdout and stderr in memory, subject to available +memory and host Buffer/string-size limits. A capture failure rejects the +invocation and discards both partial outputs; the child pipes are still drained +until it exits. For dumps that should not stay in memory, use an external tool +with file/stream output; this API does not yet offer a streaming sink. diff --git a/src/native/postgres-tools/npm/index.mts b/src/native/postgres-tools/npm/index.mts index 56277ba9a..1072ae507 100644 --- a/src/native/postgres-tools/npm/index.mts +++ b/src/native/postgres-tools/npm/index.mts @@ -1,6 +1,7 @@ import { spawn } from 'node:child_process'; import { createRequire } from 'node:module'; import path from 'node:path'; +import { createCapturedOutput, captureOutput, finishCapture } from './output-capture.mjs'; const require = createRequire(import.meta.url); // PostgreSQL 18 getopt_long optstrings. A value-taking option owns the rest @@ -267,53 +268,63 @@ async function runTool(tool, args, stdin) { rejectOnce(new PostgresToolError(tool, `could not start ${tool}`, { cause })); return; } - const stdout = []; - const stderr = []; - child.stdout.on('data', (chunk) => stdout.push(chunk)); - child.stderr.on('data', (chunk) => stderr.push(chunk)); + const captured = createCapturedOutput(); + child.stdout.on('data', (chunk) => captureOutput(captured, 'stdout', chunk)); + child.stderr.on('data', (chunk) => captureOutput(captured, 'stderr', chunk)); child.once('error', (cause) => { rejectOnce(new PostgresToolError(tool, `could not run ${tool}`, { cause })); }); child.once('close', (exitCode, signal) => { if (settled) return; settled = true; - const stdoutBytes = Buffer.concat(stdout); - const stderrBytes = Buffer.concat(stderr); - if (exitCode !== 0 || signal !== null) { - const stdoutText = decodeDiagnostics(stdoutBytes); - const stderrText = decodeDiagnostics(stderrBytes); - reject( - new PostgresToolError( - tool, - `${tool} ${signal === null ? `exited with status ${exitCode}` : `was terminated by ${signal}`}${stderrText.trim().length === 0 ? '' : `: ${stderrText.trim()}`}`, - { exitCode, signal, stdout: stdoutText, stderr: stderrText }, - ), - ); - return; - } - if (stdinFailure !== undefined) { - reject( - new PostgresToolError(tool, `could not write to ${tool}`, { - exitCode, - signal, - stdout: decodeDiagnostics(stdoutBytes), - stderr: decodeDiagnostics(stderrBytes), - cause: stdinFailure, - }), - ); - return; - } try { - resolve(new TextDecoder('utf-8', { fatal: true }).decode(stdoutBytes)); + const { stdout: stdoutBytes, stderr: stderrBytes } = finishCapture(captured); + if (exitCode !== 0 || signal !== null) { + const stdoutText = decodeDiagnostics(stdoutBytes); + const stderrText = decodeDiagnostics(stderrBytes); + reject( + new PostgresToolError( + tool, + `${tool} ${signal === null ? `exited with status ${exitCode}` : `was terminated by ${signal}`}${stderrText.trim().length === 0 ? '' : `: ${stderrText.trim()}`}`, + { exitCode, signal, stdout: stdoutText, stderr: stderrText }, + ), + ); + return; + } + if (stdinFailure !== undefined) { + reject( + new PostgresToolError(tool, `could not write to ${tool}`, { + exitCode, + signal, + stdout: decodeDiagnostics(stdoutBytes), + stderr: decodeDiagnostics(stderrBytes), + cause: stdinFailure, + }), + ); + return; + } + try { + resolve(new TextDecoder('utf-8', { fatal: true }).decode(stdoutBytes)); + } catch (cause) { + const stdoutText = decodeDiagnostics(stdoutBytes); + const stderrText = decodeDiagnostics(stderrBytes); + reject( + new PostgresToolError(tool, `${tool} produced non-UTF-8 output`, { + exitCode, + signal, + stdout: stdoutText, + stderr: stderrText, + cause, + }), + ); + } } catch (cause) { - const stdoutText = decodeDiagnostics(stdoutBytes); - const stderrText = decodeDiagnostics(stderrBytes); + // Buffer concatenation and JS string conversion have their own host + // size limits. Reject the invocation, not the event-loop callback. reject( - new PostgresToolError(tool, `${tool} produced non-UTF-8 output`, { + new PostgresToolError(tool, `${tool} output capture failed`, { exitCode, signal, - stdout: stdoutText, - stderr: stderrText, cause, }), ); diff --git a/src/native/postgres-tools/npm/output-capture.mjs b/src/native/postgres-tools/npm/output-capture.mjs new file mode 100644 index 000000000..b430fc324 --- /dev/null +++ b/src/native/postgres-tools/npm/output-capture.mjs @@ -0,0 +1,39 @@ +// Complete in-memory output: bounded by available memory and Buffer/string +// representability, not an application output quota. +export function createCapturedOutput() { + return { stdout: [], stderr: [], failure: undefined }; +} + +export function captureOutput(captured, channel, chunk) { + if (captured.failure !== undefined) return; + try { + captured[channel].push(chunk); + } catch (cause) { + discardCapture(captured, cause); + } +} + +export function finishCapture(captured) { + if (captured.failure !== undefined) throw captured.failure; + try { + const stdout = concatenate(captured.stdout); + const stderr = concatenate(captured.stderr); + captured.stdout = []; + captured.stderr = []; + return { stdout, stderr }; + } catch (cause) { + discardCapture(captured, cause); + throw cause; + } +} + +function discardCapture(captured, cause) { + // Keep draining both child pipes, but never return a partial dump. + captured.failure = cause; + captured.stdout = []; + captured.stderr = []; +} + +function concatenate(chunks) { + return chunks.length === 1 ? chunks[0] : Buffer.concat(chunks); +} diff --git a/src/native/postgres-tools/npm/output-capture.test.mjs b/src/native/postgres-tools/npm/output-capture.test.mjs new file mode 100644 index 000000000..787878b9e --- /dev/null +++ b/src/native/postgres-tools/npm/output-capture.test.mjs @@ -0,0 +1,57 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; +import { createCapturedOutput, captureOutput, finishCapture } from './output-capture.mjs'; + +test('capture preserves exact stdout and stderr bytes', () => { + const captured = createCapturedOutput(); + captureOutput(captured, 'stdout', Buffer.from([0, 255])); + captureOutput(captured, 'stdout', Buffer.from('a')); + captureOutput(captured, 'stderr', Buffer.from('err!')); + const { stdout, stderr } = finishCapture(captured); + assert.deepEqual(stdout, Buffer.from([0, 255, 97])); + assert.equal(stderr.toString(), 'err!'); +}); + +test('capture accepts more than 64 MiB', () => { + const captured = createCapturedOutput(); + const chunk = Buffer.alloc(1024 * 1024, 42); + for (let index = 0; index < 65; index++) captureOutput(captured, 'stdout', chunk); + captureOutput(captured, 'stderr', Buffer.from('diagnostic')); + const { stdout, stderr } = finishCapture(captured); + assert.equal(stdout.byteLength, 65 * chunk.byteLength); + for (let offset = 0; offset < stdout.length; offset += chunk.length) { + assert.deepEqual(stdout.subarray(offset, offset + chunk.length), chunk); + } + assert.equal(stderr.toString(), 'diagnostic'); +}); + +test('capture failure releases both streams and remains sticky', () => { + const captured = createCapturedOutput(); + captureOutput(captured, 'stdout', Buffer.from('out')); + captureOutput(captured, 'stderr', Buffer.from('err')); + const failure = new RangeError('allocation failed'); + captured.stdout.push = () => { + throw failure; + }; + captureOutput(captured, 'stdout', Buffer.from('more')); + captureOutput(captured, 'stderr', Buffer.from('later')); + assert.deepEqual(captured.stdout, []); + assert.deepEqual(captured.stderr, []); + assert.throws( + () => finishCapture(captured), + (error) => error === failure, + ); +}); + +test('concatenation failure releases chunks and never returns a prefix', () => { + const captured = createCapturedOutput(); + captureOutput(captured, 'stdout', Buffer.from('out')); + captureOutput(captured, 'stderr', Buffer.from('err')); + // Exercise Buffer.concat's error path without exhausting host memory. + captured.stderr.push(null); + assert.throws(() => finishCapture(captured), TypeError); + captureOutput(captured, 'stdout', Buffer.from('later')); + assert.deepEqual(captured.stdout, []); + assert.deepEqual(captured.stderr, []); + assert.throws(() => finishCapture(captured), TypeError); +}); diff --git a/src/native/postgres-tools/npm/package.json b/src/native/postgres-tools/npm/package.json index 2a82efc76..71a0ba1df 100644 --- a/src/native/postgres-tools/npm/package.json +++ b/src/native/postgres-tools/npm/package.json @@ -32,6 +32,7 @@ "files": [ "index.d.ts", "index.js", + "output-capture.mjs", "README.md", "LICENSE", "THIRD_PARTY_NOTICES.md" diff --git a/src/native/postgres-tools/tools/package-carriers.mts b/src/native/postgres-tools/tools/package-carriers.mts index f7a3b90f4..617271c65 100644 --- a/src/native/postgres-tools/tools/package-carriers.mts +++ b/src/native/postgres-tools/tools/package-carriers.mts @@ -108,7 +108,7 @@ function stageLiboliphauntToolsNpmFacade(version) { path.join(LIBOLIPHAUNT_NATIVE_TOOLS_FACADE_ROOT, 'index.mts'), path.join(stage, 'index.js'), ); - for (const descriptor of ['index.d.ts']) { + for (const descriptor of ['index.d.ts', 'output-capture.mjs']) { copyFileSync( path.join(LIBOLIPHAUNT_NATIVE_TOOLS_FACADE_ROOT, descriptor), path.join(stage, descriptor), @@ -164,6 +164,7 @@ export function liboliphauntToolsNpmTarballs( requiredMembers: [ 'package/index.js', 'package/index.d.ts', + 'package/output-capture.mjs', ...releaseNoticeRows({ profile: 'source-sdk' }).map((row) => `package/${row.member}`), ], }); diff --git a/src/native/runtime/README.md b/src/native/runtime/README.md index 9ac69cbc6..31a39f021 100644 --- a/src/native/runtime/README.md +++ b/src/native/runtime/README.md @@ -67,6 +67,32 @@ execution do not impose a synthetic query timeout; callers should use `oliphaunt_cancel` to interrupt long-running SQL. Ordinary SDK close is a lifecycle detach/wait boundary, not an implicit query cancellation primitive. +Direct mode is a trusted configured session, not a standalone maintenance +backend: the selected role and database must permit login and connection, role +and database settings apply, and login event triggers run. It does not perform +network authentication; hosts authorize the configured identity. It remains a +single-backend process-lifetime runtime, with synchronous I/O and no parallel +workers or WAL senders. + +`pg_import_system_collations` can import ICU collations in process. On POSIX, +trusted sessions emit a notice and skip libc locale enumeration, which requires +the external `locale -a` program; ordinary server and initdb discovery are +unchanged. Subprocess execution remains unavailable in the embedded backend. + +The embedded path leaves host signal handlers, signal masks, process timers, +and process-exit cleanup registration alone. Cancellation crosses an atomic +mailbox and a raw wake endpoint; PostgreSQL consumes it on its owning backend +thread. SQL deadlines are checked cooperatively, including protocol waits and +streaming backpressure. They cannot preempt host callbacks or extension code +that does not reach PostgreSQL interrupt checks. On POSIX, startup retains a +directory descriptor and restores the original working-directory identity even +if its path is renamed; startup still changes the process-wide working +directory temporarily. Use broker/server isolation when other host threads +must be insulated from that or from the documented PGDATA environment change. + +The C smoke tests exercise configured identities, process signals/timers, COPY +backpressure, and working-directory failure boundaries against the built library. + Hosts serialize ordinary non-cancel calls on one logical C handle; `oliphaunt_cancel` is the cross-thread exception. Streaming callbacks borrow each byte chunk only for the callback invocation. They may copy it, inspect an @@ -87,6 +113,8 @@ need different PostgreSQL settings do not need a new C ABI; they pass validated `-c name=value` startup arguments through `OliphauntConfig.startup_args`. Later arguments win, so SDKs and benchmark harnesses can apply concrete PostgreSQL GUC overrides above the stable C boundary without inventing tuning profiles. +Direct startup defaults to `-F` (`fsync=off`). Persistent storage +alone does not provide crash safety: pass `-c fsync=on` when it is required. SDKs must hydrate PGDATA from a packaged cluster seed before calling `oliphaunt_init`; the C boundary never runs `initdb` or initializes an empty diff --git a/src/native/runtime/include/oliphaunt.h b/src/native/runtime/include/oliphaunt.h index 680a8e231..407c0a9d9 100644 --- a/src/native/runtime/include/oliphaunt.h +++ b/src/native/runtime/include/oliphaunt.h @@ -51,6 +51,12 @@ typedef struct OliphauntStaticExtension { * releases a logical direct-mode lease but keeps the resident backend alive; * oliphaunt_close is terminal for the process lifetime and restores the caller's * previous PGDATA value, or unsets it if it was unset. + * Startup temporarily changes process cwd; POSIX restores its directory + * identity using a retained descriptor. Host signal handlers/masks and process + * timers are not backend-owned. Cancellation is consumed on the backend + * thread; deadlines cannot preempt a blocking host streaming callback. + * The configured identity is trusted by the host, but PostgreSQL login and + * connection eligibility, role/database settings, and login triggers apply. * * Every successful oliphaunt_init establishes a current * logical lease generation. Hosts with independent cleanup owners must capture @@ -77,7 +83,10 @@ typedef struct OliphauntConfig { const char *database; /* OLIPHAUNT_CONFIG_EXTERNAL_ROOT_LOCK or zero. */ uint64_t flags; - /* Zero or more `-c`, `name=value` pairs. Storage-routing GUCs are rejected. */ + /* + * Zero or more `-c`, `name=value` pairs. Storage-routing GUCs are rejected. + * Embedded startup defaults fsync to off; use fsync=on for crash safety. + */ const char *const *startup_args; size_t startup_arg_count; /* Optional existing ICU data directory. NULL preserves runtime/env discovery. */ diff --git a/src/native/runtime/moon.yml b/src/native/runtime/moon.yml index c4ad55fe7..b8122944f 100644 --- a/src/native/runtime/moon.yml +++ b/src/native/runtime/moon.yml @@ -112,6 +112,7 @@ tasks: - liboliphaunt-native:generation-lifecycle-test - liboliphaunt-native:stream-queue-test - liboliphaunt-native:archive-stream-test + - liboliphaunt-native:session-reset-test - liboliphaunt-native:ios-extension-packager-test - liboliphaunt-native:module-dir-resolver-test - liboliphaunt-native:postgis-reproducible-time-test @@ -119,6 +120,16 @@ tasks: - liboliphaunt-native:symbol-scope-test inputs: [] + session-reset-test: + command: node src/native/runtime/tools/test-session-reset.mjs + inputs: + - src/**/* + - include/oliphaunt.h + - tools/test-session-reset.mjs + options: + cache: true + internal: true + runFromWorkspaceRoot: true external-source-fetch-test: script: "set -e\nbash src/native/runtime/bin/fetch-pinned-git-checkout.test.sh\nbash src/native/runtime/bin/extension-source.test.sh\n" inputs: @@ -179,6 +190,8 @@ tasks: - /src/native/runtime/src/liboliphaunt_internal.h - /src/native/runtime/src/liboliphaunt_platform.h - /src/native/runtime/src/liboliphaunt_backup_state.c + - /src/native/runtime/src/liboliphaunt_config.c + - /src/native/runtime/src/liboliphaunt_runtime.c - /src/native/runtime/src/liboliphaunt_process.c - /src/native/runtime/smoke/liboliphaunt_generation_lifecycle.c - /src/native/runtime/tools/test-generation-lifecycle.sh diff --git a/src/native/runtime/patches/postgresql-18.4/0005-liboliphaunt-restore-host-cwd.patch b/src/native/runtime/patches/postgresql-18.4/0005-liboliphaunt-restore-host-cwd.patch index c8cf9e890..99685821a 100644 --- a/src/native/runtime/patches/postgresql-18.4/0005-liboliphaunt-restore-host-cwd.patch +++ b/src/native/runtime/patches/postgresql-18.4/0005-liboliphaunt-restore-host-cwd.patch @@ -4,41 +4,111 @@ Date: Wed, 13 May 2026 00:00:00 +0000 Subject: [PATCH] liboliphaunt: restore host cwd after embedded shutdown PostgreSQL's standalone startup changes the process working directory to -PGDATA. Embedded liboliphaunt must not leave the host process in that directory -after close, so preserve the caller cwd around the embedded backend lifetime -and restore it after PostgreSQL's normal exit cleanup callbacks run. +PGDATA. Embedded liboliphaunt must not leave the host process there after +terminal close, so preserve the caller cwd around the physical backend lifetime +and restore it after PostgreSQL's normal exit callbacks run. + +Capture is a startup precondition: if the caller cwd is already unresolvable, +fail before consuming PostgreSQL process state. On POSIX, retain a directory +descriptor and restore with fchdir() so a rename or replacement of the original +pathname cannot redirect cleanup. Use O_PATH on Linux/Android and O_SEARCH on +Apple: retaining a return location must not require directory-listing access. +Search permission is still required when restoring; the descriptor does not +override later permission changes. Preserve capture errno for the host's +startup error without entering PostgreSQL's error machinery. + +Windows has no portable directory-handle equivalent and retains path-based +restoration. Track the later ChangeToDataDir boundary separately so cleanup +never mistakes capture for a mutation owned by this invocation. + +This is terminal cleanup, not lifetime isolation. Direct execution still uses +the process-wide cwd until the physical backend exits; process-isolated callers +must use the broker. --- - src/backend/tcop/postgres.c | 7 +++++++ - 1 file changed, 7 insertions(+) + src/backend/tcop/postgres.c | 50 +++++++++++++++++++++++++++++++++++++++++-- + 1 file changed, 48 insertions(+), 2 deletions(-) diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index 31e14a1..56888c4 100644 +index 31e14a1..70f1baf 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -4199,6 +4199,8 @@ oliphaunt_embedded_main(int argc, char *argv[], +@@ -4195,15 +4195,42 @@ oliphaunt_embedded_main(int argc, char *argv[], + const char *dbname, const char *username, +- OliphauntEmbeddedIO *io) ++ OliphauntEmbeddedIO *io, int *cwd_capture_errno) + { + const char *switch_dbname = NULL; int socket_pair[2] = {-1, -1}; ClientSocket client_sock; struct sockaddr_in *addr; + char original_cwd[MAXPGPATH]; -+ bool have_original_cwd; - ++#ifndef WIN32 ++ int original_cwd_fd = -1; ++#endif ++ bool cwd_restore_required = false; + if (argv == NULL || argv[0] == NULL || dbname == NULL || - username == NULL || io == NULL) -@@ -4226,6 +4228,8 @@ oliphaunt_embedded_main(int argc, char *argv[], - if (!SelectConfigFiles(userDoption, progname)) +- username == NULL || io == NULL) ++ username == NULL || io == NULL || cwd_capture_errno == NULL) return -1; - -+ have_original_cwd = (getcwd(original_cwd, sizeof(original_cwd)) != NULL); + ++ *cwd_capture_errno = 0; ++ if (getcwd(original_cwd, sizeof(original_cwd)) == NULL) ++ { ++ *cwd_capture_errno = errno; ++ return -1; ++ } ++#ifndef WIN32 ++ /* Returning to a directory does not require listing its contents. */ ++#if defined(O_PATH) ++ original_cwd_fd = open(".", O_PATH | O_DIRECTORY | O_CLOEXEC); ++#elif defined(O_SEARCH) ++ original_cwd_fd = open(".", O_SEARCH | O_CLOEXEC); ++#else ++ original_cwd_fd = open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC); ++#endif ++ if (original_cwd_fd < 0) ++ { ++ *cwd_capture_errno = errno; ++ return -1; ++ } ++#endif + + Assert(!IsUnderPostmaster); + + if (progname == NULL) +@@ -4224,9 +4251,15 @@ oliphaunt_embedded_main(int argc, char *argv[], + errmsg("embedded liboliphaunt database name was provided twice"))); + + if (!SelectConfigFiles(userDoption, progname)) ++ { ++#ifndef WIN32 ++ close(original_cwd_fd); ++#endif + return -1; ++ } + checkDataDir(); ++ cwd_restore_required = true; ChangeToDataDir(); CreateDataDirLockFile(false); -@@ -4275,5 +4279,8 @@ oliphaunt_embedded_main(int argc, char *argv[], + LocalProcessControlFile(false); +@@ -4275,5 +4308,18 @@ oliphaunt_embedded_main(int argc, char *argv[], PostgresMain(dbname, username); oliphaunt_embedded_proc_exit(0); - -+ if (have_original_cwd && chdir(original_cwd) != 0) + ++#ifdef WIN32 ++ if (cwd_restore_required && chdir(original_cwd) != 0) ++ return -1; ++#else ++ if (cwd_restore_required && fchdir(original_cwd_fd) != 0) ++ { ++ close(original_cwd_fd); ++ return -1; ++ } ++ if (close(original_cwd_fd) != 0) + return -1; ++#endif + if (socket_pair[1] >= 0) close(socket_pair[1]); diff --git a/src/native/runtime/patches/postgresql-18.4/0008-liboliphaunt-clean-embedded-symbols.patch b/src/native/runtime/patches/postgresql-18.4/0008-liboliphaunt-clean-embedded-symbols.patch index 5c6333f33..5e8986a0e 100644 --- a/src/native/runtime/patches/postgresql-18.4/0008-liboliphaunt-clean-embedded-symbols.patch +++ b/src/native/runtime/patches/postgresql-18.4/0008-liboliphaunt-clean-embedded-symbols.patch @@ -136,7 +136,7 @@ index 74ae54f..8e3ab1e 100644 +typedef struct OliphauntEmbeddedIO OliphauntEmbeddedIO; +extern int oliphaunt_embedded_main(int argc, char *argv[], + const char *dbname, const char *username, -+ OliphauntEmbeddedIO *io); ++ OliphauntEmbeddedIO *io, int *cwd_capture_errno); extern void PostgresSingleUserMain(int argc, char *argv[], const char *username); extern void PostgresMain(const char *dbname, diff --git a/src/native/runtime/patches/postgresql-18.4/0009-liboliphaunt-guard-embedded-proc-exit.patch b/src/native/runtime/patches/postgresql-18.4/0009-liboliphaunt-guard-embedded-proc-exit.patch index 6ddf0d9bf..4cf699fdc 100644 --- a/src/native/runtime/patches/postgresql-18.4/0009-liboliphaunt-guard-embedded-proc-exit.patch +++ b/src/native/runtime/patches/postgresql-18.4/0009-liboliphaunt-guard-embedded-proc-exit.patch @@ -8,23 +8,33 @@ process. That is correct for backend processes, but embedded liboliphaunt runs a backend thread inside a host app. A FATAL during embedded startup must unwind to the liboliphaunt owner so the host can surface an error instead of exiting. -Install a thread-local embedded proc_exit handler around oliphaunt_embedded_main(). -The normal proc_exit_prepare() callback chain still runs, then control jumps -back to the embedded entrypoint with the PostgreSQL exit code. +Install a one-shot thread-local proc_exit handler only after its sigsetjmp target +exists. Keep every mutable cleanup value in a heap-owned lifecycle object: an +automatic object changed after sigsetjmp would be indeterminate after +siglongjmp. PostgreSQL's normal proc_exit_prepare() callback chain still runs; +the guarded path then disarms exit state and jumps to one ownership-aware cleanup +path. Ordinary startup failures use that same cleanup path after the handler is +installed. + +The lifecycle retains the POSIX caller-directory descriptor introduced by the +cwd patch, so rename-safe restoration and descriptor release also survive a +non-local exit. It closes both ends of its dummy socket pair after PostgreSQL's +callbacks: pqcomm deliberately invalidates its Port descriptor without closing +the underlying socket because an ordinary backend is about to exit. --- - src/backend/storage/ipc/ipc.c | 27 +++++++++++++++++++++++++ - src/backend/tcop/postgres.c | 38 +++++++++++++++++++++++++++++++++-- - src/include/storage/ipc.h | 3 +++ - 3 files changed, 66 insertions(+), 2 deletions(-) + src/backend/storage/ipc/ipc.c | 51 ++++++++++++++++++++ + src/backend/tcop/postgres.c | 127 ++++++++++++++++++++++++++++++----------- + src/include/storage/ipc.h | 5 ++ + 3 files changed, 149 insertions(+), 34 deletions(-) diff --git a/src/backend/storage/ipc/ipc.c b/src/backend/storage/ipc/ipc.c -index a2d0f8a..b0630ec 100644 +index a2d0f8a..97edd03 100644 --- a/src/backend/storage/ipc/ipc.c +++ b/src/backend/storage/ipc/ipc.c @@ -85,6 +85,17 @@ static int on_proc_exit_index, on_shmem_exit_index, before_shmem_exit_index; - + +#ifdef OLIPHAUNT_EMBEDDED +#ifdef _MSC_VER +#define OLIPHAUNT_THREAD_LOCAL __declspec(thread) @@ -36,17 +46,24 @@ index a2d0f8a..b0630ec 100644 +static OLIPHAUNT_THREAD_LOCAL void *oliphaunt_proc_exit_context = NULL; +#endif + - + /* ---------------------------------------------------------------- * proc_exit -@@ -111,6 +122,14 @@ proc_exit(int code) +@@ -111,6 +122,21 @@ proc_exit(int code) /* Clean up everything that must be cleaned up */ proc_exit_prepare(code); - + +#ifdef OLIPHAUNT_EMBEDDED + if (oliphaunt_proc_exit_handler != NULL) + { -+ oliphaunt_proc_exit_handler(code, oliphaunt_proc_exit_context); ++ oliphaunt_embedded_proc_exit_handler handler = oliphaunt_proc_exit_handler; ++ void *context = oliphaunt_proc_exit_context; ++ ++ /* The embedded escape is one-shot and must not leave exit state armed. */ ++ oliphaunt_proc_exit_handler = NULL; ++ oliphaunt_proc_exit_context = NULL; ++ proc_exit_inprogress = false; ++ handler(code, context); + abort(); + } +#endif @@ -54,23 +71,40 @@ index a2d0f8a..b0630ec 100644 #ifdef PROFILE_PID_DIR { /* -@@ -218,6 +237,14 @@ proc_exit_prepare(int code) +@@ -218,6 +244,31 @@ proc_exit_prepare(int code) } - + #ifdef OLIPHAUNT_EMBEDDED -+void -+oliphaunt_embedded_set_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, -+ void *context) ++bool ++oliphaunt_embedded_install_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, ++ void *context) +{ -+ oliphaunt_proc_exit_handler = handler; ++ if (handler == NULL || oliphaunt_proc_exit_handler != NULL) ++ return false; ++ + oliphaunt_proc_exit_context = context; ++ oliphaunt_proc_exit_handler = handler; ++ return true; ++} ++ ++bool ++oliphaunt_embedded_clear_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, ++ void *context) ++{ ++ if (oliphaunt_proc_exit_handler != handler || ++ oliphaunt_proc_exit_context != context) ++ return false; ++ ++ oliphaunt_proc_exit_handler = NULL; ++ oliphaunt_proc_exit_context = NULL; ++ return true; +} + /* * Run PostgreSQL's normal backend exit callbacks without terminating the host * process. This is only valid for the embedded thread owner that already diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index 56888c4..c96011f 100644 +index 70f1baf..226174c 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c @@ -27,6 +27,7 @@ @@ -80,90 +114,217 @@ index 56888c4..c96011f 100644 +#include #endif #include - -@@ -4180,6 +4181,20 @@ PostgresSingleUserMain(int argc, char *argv[], + +@@ -4180,6 +4181,34 @@ PostgresSingleUserMain(int argc, char *argv[], } - + #ifdef OLIPHAUNT_EMBEDDED -+typedef struct OliphauntEmbeddedProcExitGuard ++/* ++ * Keep every value read by cleanup behind an unchanged pointer. Automatic ++ * objects modified after sigsetjmp() would be indeterminate after siglongjmp(). ++ */ ++typedef struct OliphauntEmbeddedLifecycle +{ -+ sigjmp_buf env; -+ int code; -+} OliphauntEmbeddedProcExitGuard; ++ sigjmp_buf proc_exit_env; ++ int proc_exit_code; ++ int rc; ++ int socket_pair[2]; ++ char original_cwd[MAXPGPATH]; ++#ifndef WIN32 ++ int original_cwd_fd; ++#endif ++ bool cwd_restore_required; ++ bool proc_exit_handler_armed; ++} OliphauntEmbeddedLifecycle; + +static void +oliphaunt_embedded_throw_proc_exit(int code, void *context) +{ -+ OliphauntEmbeddedProcExitGuard *guard = (OliphauntEmbeddedProcExitGuard *) context; -+ guard->code = code; -+ siglongjmp(guard->env, 1); ++ OliphauntEmbeddedLifecycle *lifecycle = (OliphauntEmbeddedLifecycle *) context; ++ ++ lifecycle->proc_exit_code = code; ++ lifecycle->proc_exit_handler_armed = false; ++ siglongjmp(lifecycle->proc_exit_env, 1); +} + /* * oliphaunt_embedded_main * -@@ -4199,6 +4214,8 @@ oliphaunt_embedded_main(int argc, char *argv[], - int socket_pair[2] = {-1, -1}; +@@ -4196,43 +4225,63 @@ oliphaunt_embedded_main(int argc, char *argv[], + OliphauntEmbeddedIO *io, int *cwd_capture_errno) + { + const char *switch_dbname = NULL; +- int socket_pair[2] = {-1, -1}; ClientSocket client_sock; struct sockaddr_in *addr; -+ OliphauntEmbeddedProcExitGuard exit_guard; -+ int rc = 0; - char original_cwd[MAXPGPATH]; - bool have_original_cwd; - -@@ -4208,6 +4225,15 @@ oliphaunt_embedded_main(int argc, char *argv[], - +- char original_cwd[MAXPGPATH]; +-#ifndef WIN32 +- int original_cwd_fd = -1; +-#endif +- bool cwd_restore_required = false; ++ OliphauntEmbeddedLifecycle *lifecycle; ++ int result; + + if (argv == NULL || argv[0] == NULL || dbname == NULL || + username == NULL || io == NULL || cwd_capture_errno == NULL) + return -1; + + *cwd_capture_errno = 0; +- if (getcwd(original_cwd, sizeof(original_cwd)) == NULL) +- { +- *cwd_capture_errno = errno; +- return -1; +- } ++ lifecycle = calloc(1, sizeof(*lifecycle)); ++ if (lifecycle == NULL) ++ return -1; ++ lifecycle->socket_pair[0] = -1; ++ lifecycle->socket_pair[1] = -1; + #ifndef WIN32 ++ lifecycle->original_cwd_fd = -1; ++#endif ++ if (getcwd(lifecycle->original_cwd, ++ sizeof(lifecycle->original_cwd)) == NULL) ++ { ++ *cwd_capture_errno = errno; ++ free(lifecycle); ++ return -1; ++ } ++#ifndef WIN32 + /* Returning to a directory does not require listing its contents. */ + #if defined(O_PATH) +- original_cwd_fd = open(".", O_PATH | O_DIRECTORY | O_CLOEXEC); ++ lifecycle->original_cwd_fd = open(".", O_PATH | O_DIRECTORY | O_CLOEXEC); + #elif defined(O_SEARCH) +- original_cwd_fd = open(".", O_SEARCH | O_CLOEXEC); ++ lifecycle->original_cwd_fd = open(".", O_SEARCH | O_CLOEXEC); + #else +- original_cwd_fd = open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC); ++ lifecycle->original_cwd_fd = open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + #endif +- if (original_cwd_fd < 0) ++ if (lifecycle->original_cwd_fd < 0) + { + *cwd_capture_errno = errno; ++ free(lifecycle); + return -1; + } + #endif + Assert(!IsUnderPostmaster); - -+ memset(&exit_guard, 0, sizeof(exit_guard)); -+ oliphaunt_embedded_set_proc_exit_handler(oliphaunt_embedded_throw_proc_exit, -+ &exit_guard); -+ if (sigsetjmp(exit_guard.env, 1) != 0) + ++ if (sigsetjmp(lifecycle->proc_exit_env, 1) != 0) + { -+ rc = exit_guard.code; ++ lifecycle->rc = lifecycle->proc_exit_code; + goto embedded_cleanup; + } ++ if (!oliphaunt_embedded_install_proc_exit_handler( ++ oliphaunt_embedded_throw_proc_exit, lifecycle)) ++ { ++ lifecycle->rc = -1; ++ goto embedded_cleanup; ++ } ++ lifecycle->proc_exit_handler_armed = true; + if (progname == NULL) progname = get_progname(argv[0]); MyProcPid = getpid(); -@@ -4278,14 +4304,22 @@ oliphaunt_embedded_main(int argc, char *argv[], - +@@ -4238,14 +4287,12 @@ oliphaunt_embedded_main(int argc, char *argv[], + + if (!SelectConfigFiles(userDoption, progname)) + { +-#ifndef WIN32 +- close(original_cwd_fd); +-#endif +- return -1; ++ lifecycle->rc = -1; ++ goto embedded_cleanup; + } + + checkDataDir(); +- cwd_restore_required = true; ++ lifecycle->cwd_restore_required = true; + ChangeToDataDir(); + CreateDataDirLockFile(false); + LocalProcessControlFile(false); +@@ -4261,12 +4308,12 @@ oliphaunt_embedded_main(int argc, char *argv[], + PgStartTime = GetCurrentTimestamp(); + InitProcess(); + +- if (socketpair(AF_UNIX, SOCK_STREAM, 0, socket_pair) != 0) ++ if (socketpair(AF_UNIX, SOCK_STREAM, 0, lifecycle->socket_pair) != 0) + ereport(FATAL, + (errmsg("could not create embedded liboliphaunt socketpair: %m"))); + + memset(&client_sock, 0, sizeof(client_sock)); +- client_sock.sock = socket_pair[0]; ++ client_sock.sock = lifecycle->socket_pair[0]; + addr = (struct sockaddr_in *) &client_sock.raddr.addr; + addr->sin_family = AF_INET; + addr->sin_port = htons(5432); +@@ -4293,24 +4340,36 @@ oliphaunt_embedded_main(int argc, char *argv[], + PostgresMain(dbname, username); oliphaunt_embedded_proc_exit(0); -+ rc = 0; - ++ lifecycle->rc = 0; ++ +embedded_cleanup: -+ oliphaunt_embedded_set_proc_exit_handler(NULL, NULL); - if (have_original_cwd && chdir(original_cwd) != 0) ++ if (lifecycle->proc_exit_handler_armed && ++ !oliphaunt_embedded_clear_proc_exit_handler( ++ oliphaunt_embedded_throw_proc_exit, lifecycle)) ++ abort(); ++ lifecycle->proc_exit_handler_armed = false; + + #ifdef WIN32 +- if (cwd_restore_required && chdir(original_cwd) != 0) - return -1; -+ rc = rc == 0 ? -1 : rc; - - if (socket_pair[1] >= 0) - close(socket_pair[1]); -+ if (socket_pair[0] >= 0 && MyProcPort == NULL) -+ close(socket_pair[0]); - ++ if (lifecycle->cwd_restore_required && ++ chdir(lifecycle->original_cwd) != 0) ++ lifecycle->rc = lifecycle->rc == 0 ? -1 : lifecycle->rc; + #else +- if (cwd_restore_required && fchdir(original_cwd_fd) != 0) +- { +- close(original_cwd_fd); +- return -1; +- } +- if (close(original_cwd_fd) != 0) +- return -1; ++ if (lifecycle->cwd_restore_required && ++ fchdir(lifecycle->original_cwd_fd) != 0) ++ lifecycle->rc = lifecycle->rc == 0 ? -1 : lifecycle->rc; ++ if (lifecycle->original_cwd_fd >= 0 && ++ close(lifecycle->original_cwd_fd) != 0) ++ lifecycle->rc = lifecycle->rc == 0 ? -1 : lifecycle->rc; + #endif + +- if (socket_pair[1] >= 0) +- close(socket_pair[1]); ++ if (lifecycle->socket_pair[1] >= 0) ++ close(lifecycle->socket_pair[1]); ++ if (lifecycle->socket_pair[0] >= 0) ++ close(lifecycle->socket_pair[0]); + - return 0; -+ if (rc == 0 && exit_guard.code != 0) -+ rc = exit_guard.code; -+ -+ return rc; ++ result = lifecycle->rc; ++ free(lifecycle); ++ return result; } #endif - + diff --git a/src/include/storage/ipc.h b/src/include/storage/ipc.h -index 8bfa63a..69d2d80 100644 +index 8bfa63a..f6ac183 100644 --- a/src/include/storage/ipc.h +++ b/src/include/storage/ipc.h -@@ -68,7 +68,10 @@ extern PGDLLIMPORT bool shmem_exit_inprogress; +@@ -68,7 +68,12 @@ extern PGDLLIMPORT bool shmem_exit_inprogress; pg_noreturn extern void proc_exit(int code); extern void shmem_exit(int code); #ifdef OLIPHAUNT_EMBEDDED +typedef void (*oliphaunt_embedded_proc_exit_handler) (int code, void *context); extern void oliphaunt_embedded_proc_exit(int code); -+extern void oliphaunt_embedded_set_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, ++extern bool oliphaunt_embedded_install_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, + void *context); ++extern bool oliphaunt_embedded_clear_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, ++ void *context); #endif extern void on_proc_exit(pg_on_exit_callback function, Datum arg); extern void on_shmem_exit(pg_on_exit_callback function, Datum arg); diff --git a/src/native/runtime/patches/postgresql-18.4/0010-liboliphaunt-use-host-runtime-paths.patch b/src/native/runtime/patches/postgresql-18.4/0010-liboliphaunt-use-host-runtime-paths.patch index c4b4a1303..5bbb5c875 100644 --- a/src/native/runtime/patches/postgresql-18.4/0010-liboliphaunt-use-host-runtime-paths.patch +++ b/src/native/runtime/patches/postgresql-18.4/0010-liboliphaunt-use-host-runtime-paths.patch @@ -12,21 +12,22 @@ Use argv[0] as a host-provided install-root anchor for embedded mode when it is already absolute. This initializes my_exec_path and pgservice paths before InitStandaloneProcess(), avoiding find_my_exec()'s executable-bit requirement while preserving PostgreSQL's existing path derivation for share/etc resources. -When the host provides OLIPHAUNT_EMBEDDED_MODULE_DIR, use that as pkglib_path so -the embedded backend loads liboliphaunt-linked modules without mutating the -standalone runtime tree used by initdb. +When the host provides OLIPHAUNT_EMBEDDED_MODULE_DIR, require an absolute, +non-truncated value and use it as pkglib_path so the embedded backend loads +liboliphaunt-linked modules without mutating the standalone runtime tree used by +initdb. Reject overlong anchors instead of accepting a different truncated path. --- - src/backend/tcop/postgres.c | 60 +++++++++++++++++++++++++++++++++++-- - 1 file changed, 58 insertions(+), 2 deletions(-) + src/backend/tcop/postgres.c | 64 ++++++++++++++++++++++++++++++++++++++++++++- + 1 file changed, 63 insertions(+), 1 deletion(-) diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index c96011f..402b66b 100644 +index 226174c..2e6a788 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -4195,6 +4195,61 @@ oliphaunt_embedded_throw_proc_exit(int code, void *context) - siglongjmp(guard->env, 1); +@@ -4209,6 +4209,67 @@ oliphaunt_embedded_throw_proc_exit(int code, void *context) + siglongjmp(lifecycle->proc_exit_env, 1); } - + +static bool +oliphaunt_embedded_set_runtime_paths(const char *argv0) +{ @@ -43,17 +44,23 @@ index c96011f..402b66b 100644 + */ + if (my_exec_path[0] == '\0') + { -+ strlcpy(my_exec_path, argv0, MAXPGPATH); ++ if (strlcpy(my_exec_path, argv0, MAXPGPATH) >= MAXPGPATH) ++ ereport(FATAL, ++ (errmsg("embedded PostgreSQL runtime anchor is too long"))); + canonicalize_path(my_exec_path); + } + + if (pkglib_path[0] == '\0') + { + module_dir = getenv("OLIPHAUNT_EMBEDDED_MODULE_DIR"); -+ if (module_dir != NULL && module_dir[0] != '\0' && -+ is_absolute_path(module_dir)) ++ if (module_dir != NULL && module_dir[0] != '\0') + { -+ strlcpy(pkglib_path, module_dir, MAXPGPATH); ++ if (!is_absolute_path(module_dir)) ++ ereport(FATAL, ++ (errmsg("embedded PostgreSQL module directory must be absolute"))); ++ if (strlcpy(pkglib_path, module_dir, MAXPGPATH) >= MAXPGPATH) ++ ereport(FATAL, ++ (errmsg("embedded PostgreSQL module directory is too long"))); + canonicalize_path(pkglib_path); + } + else @@ -85,22 +92,13 @@ index c96011f..402b66b 100644 /* * oliphaunt_embedded_main * -@@ -4217,7 +4272,7 @@ oliphaunt_embedded_main(int argc, char *argv[], - OliphauntEmbeddedProcExitGuard exit_guard; - int rc = 0; - char original_cwd[MAXPGPATH]; -- bool have_original_cwd; -+ bool have_original_cwd = false; - - if (argv == NULL || argv[0] == NULL || dbname == NULL || - username == NULL || io == NULL) -@@ -4240,7 +4295,8 @@ oliphaunt_embedded_main(int argc, char *argv[], +@@ -4278,7 +4339,8 @@ oliphaunt_embedded_main(int argc, char *argv[], if (TopMemoryContext == NULL) MemoryContextInit(); (void) set_stack_base(); - set_pglocale_pgservice(argv[0], PG_TEXTDOMAIN("postgres")); + if (!oliphaunt_embedded_set_runtime_paths(argv[0])) + set_pglocale_pgservice(argv[0], PG_TEXTDOMAIN("postgres")); - + InitStandaloneProcess(argv[0]); InitializeGUCOptions(); diff --git a/src/native/runtime/patches/postgresql-18.4/0012-liboliphaunt-enable-event-triggers-in-embedded-backend.patch b/src/native/runtime/patches/postgresql-18.4/0012-liboliphaunt-enable-event-triggers-in-embedded-backend.patch index 24e1a496e..d2d5241ab 100644 --- a/src/native/runtime/patches/postgresql-18.4/0012-liboliphaunt-enable-event-triggers-in-embedded-backend.patch +++ b/src/native/runtime/patches/postgresql-18.4/0012-liboliphaunt-enable-event-triggers-in-embedded-backend.patch @@ -5,29 +5,45 @@ Subject: [PATCH] liboliphaunt: enable event triggers in embedded backend sessions PostgreSQL disables event triggers when !IsUnderPostmaster so standalone -single-user recovery has an escape hatch. liboliphaunt direct mode is not a -postmaster child, but it is also not a standalone recovery shell: it runs a -normal embedded backend session behind the frontend/backend protocol. +single-user recovery has an escape hatch. liboliphaunt's dedicated entrypoint +also starts from standalone infrastructure, but then serves a normal +frontend/backend protocol session and identifies it as B_BACKEND. -Keep upstream standalone behavior unchanged, but allow OLIPHAUNT_EMBEDDED builds -to fire event triggers when the event_triggers GUC is enabled. +Use private write-once admission state in the dedicated entrypoint before +PostgreSQL initialization. A compare-and-swap makes duplicate concurrent +entrypoint attempts race-free and admits exactly one backend. Once admitted, +even a startup failure consumes this PostgreSQL process lifetime; the state is +deliberately never reset. A rejected duplicate enters the existing embedded +cleanup path without claiming a cwd mutation or clearing PostgreSQL-global +resources it never owned. Successful admission is recorded in the heap +lifecycle before any FATAL-capable initialization so later cleanup can +distinguish the owner after a non-local exit. + +Admission is not a session capability and event-trigger hot paths do not read +it. In embedded builds, event triggers instead accept either an ordinary +postmaster child or PostgreSQL's existing regular-backend identity. The +dedicated entrypoint sets MyBackendType to B_BACKEND before PostgresMain, so +Native Direct and Broker sessions enable DDL and login event triggers while +bootstrap and genuine standalone single-user invocations keep the upstream +escape hatch. --- - src/backend/commands/event_trigger.c | 20 +++++++++++++++----- - 1 file changed, 15 insertions(+), 5 deletions(-) + src/backend/commands/event_trigger.c | 20 +++++++++++++----- + src/backend/tcop/postgres.c | 36 ++++++++++++++++++++++++++++++++++++ + 2 files changed, 51 insertions(+), 5 deletions(-) diff --git a/src/backend/commands/event_trigger.c b/src/backend/commands/event_trigger.c -index 074c476..af3c678 100644 +index 074c476..4defe1b 100644 --- a/src/backend/commands/event_trigger.c +++ b/src/backend/commands/event_trigger.c @@ -85,6 +85,16 @@ static EventTriggerQueryState *currentEventTriggerState = NULL; /* GUC parameter */ bool event_triggers = true; - + +static bool +EventTriggersHaveRunnableBackend(void) +{ +#ifdef OLIPHAUNT_EMBEDDED -+ return true; ++ return IsUnderPostmaster || AmRegularBackendProcess(); +#else + return IsUnderPostmaster; +#endif @@ -43,7 +59,7 @@ index 074c476..af3c678 100644 - if (!IsUnderPostmaster || !event_triggers) + if (!EventTriggersHaveRunnableBackend() || !event_triggers) return; - + runlist = EventTriggerCommonSetup(parsetree, @@ -782,7 +792,7 @@ EventTriggerDDLCommandEnd(Node *parsetree) * See EventTriggerDDLCommandStart for a discussion about why event @@ -52,7 +68,7 @@ index 074c476..af3c678 100644 - if (!IsUnderPostmaster || !event_triggers) + if (!EventTriggersHaveRunnableBackend() || !event_triggers) return; - + /* @@ -830,7 +840,7 @@ EventTriggerSQLDrop(Node *parsetree) * See EventTriggerDDLCommandStart for a discussion about why event @@ -61,7 +77,7 @@ index 074c476..af3c678 100644 - if (!IsUnderPostmaster || !event_triggers) + if (!EventTriggersHaveRunnableBackend() || !event_triggers) return; - + /* @@ -904,7 +914,7 @@ EventTriggerOnLogin(void) * triggers are disabled in single user mode or via a GUC. We also need a @@ -71,7 +87,7 @@ index 074c476..af3c678 100644 + if (!EventTriggersHaveRunnableBackend() || !event_triggers || !OidIsValid(MyDatabaseId) || !MyDatabaseHasLoginEventTriggers) return; - + @@ -1009,7 +1019,7 @@ EventTriggerTableRewrite(Node *parsetree, Oid tableOid, int reason) * See EventTriggerDDLCommandStart for a discussion about why event * triggers are disabled in single user mode or via a GUC. @@ -79,5 +95,73 @@ index 074c476..af3c678 100644 - if (!IsUnderPostmaster || !event_triggers) + if (!EventTriggersHaveRunnableBackend() || !event_triggers) return; - + /* +diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c +index 2e6a788..86c4a89 100644 +--- a/src/backend/tcop/postgres.c ++++ b/src/backend/tcop/postgres.c +@@ -61,6 +61,9 @@ + #include "pg_getopt.h" + #include "pg_trace.h" + #include "pgstat.h" ++#ifdef OLIPHAUNT_EMBEDDED ++#include "port/atomics.h" ++#endif + #include "postmaster/interrupt.h" + #include "postmaster/postmaster.h" + #include "replication/logicallauncher.h" +@@ -4181,6 +4184,32 @@ PostgresSingleUserMain(int argc, char *argv[], + } + + #ifdef OLIPHAUNT_EMBEDDED ++enum ++{ ++ OLIPHAUNT_EMBEDDED_ENTRYPOINT_UNCLAIMED = 0, ++ OLIPHAUNT_EMBEDDED_ENTRYPOINT_CLAIMED = 1, ++}; ++ ++/* ++ * PG18's pg_atomic_uint32 implementations are zero-initializable wrappers. ++ * Static initialization is intentional: lazy pg_atomic_init_u32() would itself ++ * race when duplicate entrypoint calls arrive concurrently. ++ */ ++static pg_atomic_uint32 embedded_entrypoint_admission = {0}; ++ ++static bool ++AdmitEmbeddedEntrypoint(void) ++{ ++ uint32 expected = OLIPHAUNT_EMBEDDED_ENTRYPOINT_UNCLAIMED; ++ ++ if (IsPostmasterEnvironment || IsUnderPostmaster) ++ return false; ++ ++ return pg_atomic_compare_exchange_u32(&embedded_entrypoint_admission, ++ &expected, ++ OLIPHAUNT_EMBEDDED_ENTRYPOINT_CLAIMED); ++} ++ + /* + * Keep every value read by cleanup behind an unchanged pointer. Automatic + * objects modified after sigsetjmp() would be indeterminate after siglongjmp(). +@@ -4197,6 +4226,7 @@ typedef struct OliphauntEmbeddedLifecycle + #endif + bool cwd_restore_required; + bool proc_exit_handler_armed; ++ bool entrypoint_admitted; + } OliphauntEmbeddedLifecycle; + + static void +@@ -4332,6 +4362,12 @@ oliphaunt_embedded_main(int argc, char *argv[], + goto embedded_cleanup; + } + lifecycle->proc_exit_handler_armed = true; ++ if (!AdmitEmbeddedEntrypoint()) ++ { ++ lifecycle->rc = -1; ++ goto embedded_cleanup; ++ } ++ lifecycle->entrypoint_admitted = true; + + if (progname == NULL) + progname = get_progname(argv[0]); diff --git a/src/native/runtime/patches/postgresql-18.4/0014-liboliphaunt-use-portable-embedded-socketpair.patch b/src/native/runtime/patches/postgresql-18.4/0014-liboliphaunt-use-portable-embedded-socketpair.patch index 1d57715c4..0a37b7ad2 100644 --- a/src/native/runtime/patches/postgresql-18.4/0014-liboliphaunt-use-portable-embedded-socketpair.patch +++ b/src/native/runtime/patches/postgresql-18.4/0014-liboliphaunt-use-portable-embedded-socketpair.patch @@ -6,18 +6,30 @@ Subject: [PATCH] liboliphaunt: use portable embedded socketpair The embedded backend routes protocol bytes through host-owned I/O callbacks, but PostgreSQL's backend libpq initialization still requires a real socket for address discovery, socket option setup, connection polling, and fixed wait-set -positions. Unix builds can satisfy that with socketpair(); Windows has no +positions. Unix builds can satisfy that with socketpair(); Windows has no socketpair(), so create an equivalent loopback pair with bind/listen/connect/ accept. This keeps the regular PostgreSQL socket path intact and keeps embedded -transport host-owned without adding Windows-specific logic outside PostgreSQL. +transport host-owned without spreading Windows-specific logic outside the +entrypoint. Initialize Winsock before PostgreSQL initialization and preload +libraries, report its native error codes, and balance the owned WSA reference +only after the admitted owner frees pq_init's session wait set and closes both +dummy sockets. The lifecycle remains the single cleanup authority after +non-local exits; a rejected concurrent contender cannot clear the admitted +backend's PostgreSQL-global wait set. pqcomm's exit callback intentionally +invalidates its Port descriptor without closing the underlying socket because +an ordinary backend is about to exit. + +The remaining standalone latch/wait-support and Windows signal resources are +process-lifetime PostgreSQL state, not socket-pair ownership; they require the +separate embedded lifecycle teardown design documented by the audit. --- - src/backend/tcop/postgres.c | 76 +++++++++++++++++++++++++++++++++---- - 1 file changed, 69 insertions(+), 7 deletions(-) + src/backend/tcop/postgres.c | 128 ++++++++++++++++++++++++++++++++++++++++----- + 1 file changed, 119 insertions(+), 9 deletions(-) diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index 402b66b..5511268 100644 +index 86c4a89..89b516f 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c @@ -26,7 +26,9 @@ @@ -30,27 +42,43 @@ index 402b66b..5511268 100644 #include #endif #include -@@ -4250,6 +4252,66 @@ oliphaunt_embedded_set_runtime_paths(const char *argv0) +@@ -78,6 +80,9 @@ + #include "storage/proc.h" + #include "storage/procsignal.h" + #include "storage/sinval.h" ++#ifdef OLIPHAUNT_EMBEDDED ++#include "storage/waiteventset.h" ++#endif + #include "tcop/backend_startup.h" + #include "tcop/fastpath.h" + #include "tcop/pquery.h" +@@ -4219,10 +4224,12 @@ typedef struct OliphauntEmbeddedLifecycle + sigjmp_buf proc_exit_env; + int proc_exit_code; + int rc; +- int socket_pair[2]; ++ pgsocket socket_pair[2]; + char original_cwd[MAXPGPATH]; + #ifndef WIN32 + int original_cwd_fd; ++#else ++ bool winsock_started; + #endif + bool cwd_restore_required; + bool proc_exit_handler_armed; +@@ -4300,6 +4307,64 @@ oliphaunt_embedded_set_runtime_paths(const char *argv0) return true; } - + +static int +oliphaunt_embedded_create_socketpair(pgsocket sockets[2]) +{ +#ifdef WIN32 -+ static bool wsa_started = false; -+ WSADATA wsaData; + pgsocket listener = PGINVALID_SOCKET; + struct sockaddr_in addr; + socklen_t addrlen = sizeof(addr); + int rc = -1; -+ -+ if (!wsa_started) -+ { -+ if (WSAStartup(MAKEWORD(2, 2), &wsaData) != 0) -+ return -1; -+ wsa_started = true; -+ } ++ int error = 0; + + listener = socket(AF_INET, SOCK_STREAM, 0); + if (listener == PGINVALID_SOCKET) @@ -78,6 +106,8 @@ index 402b66b..5511268 100644 + rc = 0; + +cleanup: ++ if (rc != 0) ++ error = WSAGetLastError(); + if (listener != PGINVALID_SOCKET) + closesocket(listener); + if (rc != 0) @@ -87,49 +117,114 @@ index 402b66b..5511268 100644 + if (sockets[1] != PGINVALID_SOCKET) + closesocket(sockets[1]); + sockets[0] = sockets[1] = PGINVALID_SOCKET; ++ if (error == 0) ++ error = WSASYSCALLFAILURE; + } -+ return rc; ++ return rc == 0 ? 0 : error; +#else -+ return socketpair(AF_UNIX, SOCK_STREAM, 0, sockets); ++ if (socketpair(AF_UNIX, SOCK_STREAM, 0, sockets) == 0) ++ return 0; ++ return errno != 0 ? errno : EIO; +#endif +} + /* * oliphaunt_embedded_main * -@@ -4266,7 +4328,7 @@ oliphaunt_embedded_main(int argc, char *argv[], - OliphauntEmbeddedIO *io) - { - const char *switch_dbname = NULL; -- int socket_pair[2] = {-1, -1}; -+ pgsocket socket_pair[2] = {PGINVALID_SOCKET, PGINVALID_SOCKET}; - ClientSocket client_sock; +@@ -4320,6 +4385,7 @@ oliphaunt_embedded_main(int argc, char *argv[], struct sockaddr_in *addr; - OliphauntEmbeddedProcExitGuard exit_guard; -@@ -4328,9 +4390,9 @@ oliphaunt_embedded_main(int argc, char *argv[], + OliphauntEmbeddedLifecycle *lifecycle; + int result; ++ int socket_error; + + if (argv == NULL || argv[0] == NULL || dbname == NULL || + username == NULL || io == NULL || cwd_capture_errno == NULL) +@@ -4328,8 +4394,8 @@ oliphaunt_embedded_main(int argc, char *argv[], + lifecycle = calloc(1, sizeof(*lifecycle)); + if (lifecycle == NULL) + return -1; +- lifecycle->socket_pair[0] = -1; +- lifecycle->socket_pair[1] = -1; ++ lifecycle->socket_pair[0] = PGINVALID_SOCKET; ++ lifecycle->socket_pair[1] = PGINVALID_SOCKET; + #ifndef WIN32 + lifecycle->original_cwd_fd = -1; + #endif +@@ -4375,6 +4441,19 @@ oliphaunt_embedded_main(int argc, char *argv[], + if (TopMemoryContext == NULL) + MemoryContextInit(); + (void) set_stack_base(); ++#ifdef WIN32 ++ { ++ WSADATA wsa_data; ++ int wsa_error; ++ ++ wsa_error = WSAStartup(MAKEWORD(2, 2), &wsa_data); ++ if (wsa_error != 0) ++ ereport(FATAL, ++ (errmsg("could not initialize Winsock: error code %d", ++ wsa_error))); ++ lifecycle->winsock_started = true; ++ } ++#endif + if (!oliphaunt_embedded_set_runtime_paths(argv[0])) + set_pglocale_pgservice(argv[0], PG_TEXTDOMAIN("postgres")); + +@@ -4410,9 +4489,19 @@ oliphaunt_embedded_main(int argc, char *argv[], PgStartTime = GetCurrentTimestamp(); InitProcess(); - -- if (socketpair(AF_UNIX, SOCK_STREAM, 0, socket_pair) != 0) -+ if (oliphaunt_embedded_create_socketpair(socket_pair) != 0) + +- if (socketpair(AF_UNIX, SOCK_STREAM, 0, lifecycle->socket_pair) != 0) ++ socket_error = oliphaunt_embedded_create_socketpair(lifecycle->socket_pair); ++ if (socket_error != 0) ++ { ++#ifdef WIN32 ++ ereport(FATAL, ++ (errmsg("could not create embedded liboliphaunt socket pair: error code %d", ++ socket_error))); ++#else ++ errno = socket_error; ereport(FATAL, - (errmsg("could not create embedded liboliphaunt socketpair: %m"))); + (errmsg("could not create embedded liboliphaunt socket pair: %m"))); - ++#endif ++ } + memset(&client_sock, 0, sizeof(client_sock)); - client_sock.sock = socket_pair[0]; -@@ -4367,10 +4429,10 @@ embedded_cleanup: - if (have_original_cwd && chdir(original_cwd) != 0) - rc = rc == 0 ? -1 : rc; - -- if (socket_pair[1] >= 0) -- close(socket_pair[1]); -- if (socket_pair[0] >= 0 && MyProcPort == NULL) -- close(socket_pair[0]); -+ if (socket_pair[1] != PGINVALID_SOCKET) -+ closesocket(socket_pair[1]); -+ if (socket_pair[0] != PGINVALID_SOCKET && MyProcPort == NULL) -+ closesocket(socket_pair[0]); - - if (rc == 0 && exit_guard.code != 0) - rc = exit_guard.code; + client_sock.sock = lifecycle->socket_pair[0]; +@@ -4464,10 +4553,31 @@ embedded_cleanup: + lifecycle->rc = lifecycle->rc == 0 ? -1 : lifecycle->rc; + #endif + +- if (lifecycle->socket_pair[1] >= 0) +- close(lifecycle->socket_pair[1]); +- if (lifecycle->socket_pair[0] >= 0) +- close(lifecycle->socket_pair[0]); ++ /* pq_init owns a session wait set that process exit normally reclaims. */ ++ if (lifecycle->entrypoint_admitted && FeBeWaitSet != NULL) ++ { ++ FreeWaitEventSet(FeBeWaitSet); ++ FeBeWaitSet = NULL; ++ } ++ ++ if (lifecycle->socket_pair[1] != PGINVALID_SOCKET) ++ { ++ closesocket(lifecycle->socket_pair[1]); ++ lifecycle->socket_pair[1] = PGINVALID_SOCKET; ++ } ++ if (lifecycle->socket_pair[0] != PGINVALID_SOCKET) ++ { ++ closesocket(lifecycle->socket_pair[0]); ++ lifecycle->socket_pair[0] = PGINVALID_SOCKET; ++ } ++#ifdef WIN32 ++ if (lifecycle->winsock_started) ++ { ++ if (WSACleanup() == SOCKET_ERROR && lifecycle->rc == 0) ++ lifecycle->rc = -1; ++ lifecycle->winsock_started = false; ++ } ++#endif + + result = lifecycle->rc; + free(lifecycle); diff --git a/src/native/runtime/patches/postgresql-18.4/0018-liboliphaunt-contain-embedded-proc-signals.patch b/src/native/runtime/patches/postgresql-18.4/0018-liboliphaunt-contain-embedded-proc-signals.patch index afa1a765a..9b3ba1fc8 100644 --- a/src/native/runtime/patches/postgresql-18.4/0018-liboliphaunt-contain-embedded-proc-signals.patch +++ b/src/native/runtime/patches/postgresql-18.4/0018-liboliphaunt-contain-embedded-proc-signals.patch @@ -22,7 +22,7 @@ diff --git a/src/backend/storage/ipc/procsignal.c b/src/backend/storage/ipc/proc index 4f9dcf1..b68d502 100644 --- a/src/backend/storage/ipc/procsignal.c +++ b/src/backend/storage/ipc/procsignal.c -@@ -268,6 +268,31 @@ ProcSignalClearSlot(ProcNumber procNumber) +@@ -269,6 +269,31 @@ ProcSignalClearSlot(ProcNumber procNumber) ConditionVariableBroadcast(&slot->pss_barrierCV); } @@ -54,7 +54,7 @@ index 4f9dcf1..b68d502 100644 /* * SendProcSignal -@@ -298,7 +324,7 @@ SendProcSignal(pid_t pid, ProcSignalReason reason, ProcNumber procNumber) +@@ -297,7 +323,7 @@ SendProcSignal(pid_t pid, ProcSignalReason reason, ProcNumber procNumber) slot->pss_signalFlags[reason] = true; SpinLockRelease(&slot->pss_mutex); /* Send signal */ @@ -63,7 +63,7 @@ index 4f9dcf1..b68d502 100644 } SpinLockRelease(&slot->pss_mutex); } -@@ -326,7 +352,7 @@ SendProcSignal(pid_t pid, ProcSignalReason reason, ProcNumber procNumber) +@@ -325,7 +351,7 @@ SendProcSignal(pid_t pid, ProcSignalReason reason, ProcNumber procNumber) slot->pss_signalFlags[reason] = true; SpinLockRelease(&slot->pss_mutex); /* Send signal */ @@ -72,7 +72,7 @@ index 4f9dcf1..b68d502 100644 } SpinLockRelease(&slot->pss_mutex); } -@@ -407,7 +433,8 @@ EmitProcSignalBarrier(ProcSignalBarrierType type) +@@ -406,7 +432,8 @@ EmitProcSignalBarrier(ProcSignalBarrierType type) /* see SendProcSignal for details */ slot->pss_signalFlags[PROCSIG_BARRIER] = true; SpinLockRelease(&slot->pss_mutex); @@ -86,7 +86,7 @@ diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c index d4cdf45..79222a9 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -4511,6 +4511,11 @@ PostgresMain(const char *dbname, const char *username) +@@ -4655,6 +4655,11 @@ PostgresMain(const char *dbname, const char *username) * midst of output during who-knows-what operation... */ pqsignal(SIGPIPE, SIG_IGN); diff --git a/src/native/runtime/patches/postgresql-18.4/0020-liboliphaunt-enforce-embedded-signal-boundary.patch b/src/native/runtime/patches/postgresql-18.4/0020-liboliphaunt-enforce-embedded-signal-boundary.patch index 860ef4dc9..cec2962ef 100644 --- a/src/native/runtime/patches/postgresql-18.4/0020-liboliphaunt-enforce-embedded-signal-boundary.patch +++ b/src/native/runtime/patches/postgresql-18.4/0020-liboliphaunt-enforce-embedded-signal-boundary.patch @@ -36,7 +36,7 @@ index b35b8cc..347803d 100644 /* * Windows has enough specialized port stuff that we push most of it off -@@ -531,7 +538,26 @@ extern void pqsignal(int signo, pqsigfunc func); +@@ -532,7 +539,26 @@ extern void pqsignal(int signo, pqsigfunc func); #endif typedef void (*pqsigfunc) (SIGNAL_ARGS); extern void pqsignal(int signo, pqsigfunc func); @@ -67,7 +67,7 @@ diff --git a/src/port/pqsignal.c b/src/port/pqsignal.c index 1e46997..082fed5 100644 --- a/src/port/pqsignal.c +++ b/src/port/pqsignal.c -@@ -69,7 +69,41 @@ StaticAssertDecl(SIGTERM < PG_NSIG, "SIGTERM >= PG_NSIG"); +@@ -70,7 +70,41 @@ StaticAssertDecl(SIGTERM < PG_NSIG, "SIGTERM >= PG_NSIG"); StaticAssertDecl(SIGALRM < PG_NSIG, "SIGALRM >= PG_NSIG"); static volatile pqsigfunc pqsignal_handlers[PG_NSIG]; @@ -109,7 +109,7 @@ index 1e46997..082fed5 100644 /* * Except when called with SIG_IGN or SIG_DFL, pqsignal() sets up this function * as the handler for all signals. This wrapper handler function checks that -@@ -127,7 +161,12 @@ pqsignal(int signo, pqsigfunc func) +@@ -128,7 +162,12 @@ pqsignal(int signo, pqsigfunc func) Assert(signo > 0); Assert(signo < PG_NSIG); diff --git a/src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-model-trusted-embedded-sessions.patch b/src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-model-trusted-embedded-sessions.patch new file mode 100644 index 000000000..7a2e263c6 --- /dev/null +++ b/src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-model-trusted-embedded-sessions.patch @@ -0,0 +1,698 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: liboliphaunt +Date: Thu, 27 Aug 2026 00:00:00 +0000 +Subject: [PATCH] liboliphaunt: model trusted embedded sessions + +Native serves FE/BE protocol from standalone process topology. Previously the +entrypoint selected B_BACKEND only after InitProcess, so PGPROC remained +non-regular; InitPostgres then installed bootstrap postgres identity, skipped +LOGIN and database policy, ignored role/database settings, and fired login +triggers as the wrong user. + +Add a private one-way UNSELECTED to PREPARED to ATTACHED lifecycle and carry +an explicit trusted-client flag into InitPostgres. Keep IsUnderPostmaster and +IsPostmasterEnvironment false. Pin synchronous I/O and zero worker topology +before shared-memory sizing, reject later live parallel settings, select +B_BACKEND before InitProcess, and populate the synthetic Port identity. + +The attached branch trusts the in-process host to select a role instead of +running HBA or ClientAuthentication_hook. It still resolves the catalog role, +enforces LOGIN and connection limits, checks database access, loads +pg_db_role_setting, and leaves system_user unset because no authentication +provider or principal ran. Genuine recovery standalone mode retains its bootstrap escape +hatch. + +Use the positive session capability only for enumerated normal-user semantics: +event triggers, text-search validation, OID allocation, XID and MultiXact stop +protection, and catalog-backed superuser checks. Keep XLOG ownership, worker +launch, postmaster signalling, latches, and supervisor topology on their +original predicates; reject reload, rotation, and promotion before +absent-postmaster signalling. +--- + src/backend/access/transam/multixact.c | 5 +- + src/backend/access/transam/varsup.c | 6 +- + src/backend/access/transam/xlogfuncs.c | 5 + + src/backend/commands/event_trigger.c | 2 +- + src/backend/commands/tsearchcmds.c | 2 +- + src/backend/storage/ipc/signalfuncs.c | 12 +++ + src/backend/tcop/backend_startup.c | 2 +- + src/backend/tcop/postgres.c | 32 +++++-- + src/backend/utils/init/Makefile | 1 + + src/backend/utils/init/embedded_session.c | 112 ++++++++++++++++++++++ + src/backend/utils/init/meson.build | 1 + + src/backend/utils/init/miscinit.c | 8 +- + src/backend/utils/init/postinit.c | 41 +++++++- + src/backend/utils/misc/guc_tables.c | 31 +++++- + src/backend/utils/misc/superuser.c | 2 +- + src/include/miscadmin.h | 27 ++++++ + src/include/tcop/tcopprot.h | 6 +- + 17 files changed, 266 insertions(+), 29 deletions(-) + create mode 100644 src/backend/utils/init/embedded_session.c + +diff --git a/src/backend/access/transam/multixact.c b/src/backend/access/transam/multixact.c +index da2a174..6d83697 100644 +--- a/src/backend/access/transam/multixact.c ++++ b/src/backend/access/transam/multixact.c +@@ -1248,7 +1248,7 @@ GetNewMultiXactId(int nmembers, MultiXactOffset *offset) + + LWLockRelease(MultiXactGenLock); + +- if (IsUnderPostmaster && ++ if (IsNormalUserSession() && + !MultiXactIdPrecedes(result, multiStopLimit)) + { + char *oldest_datname = get_database_name(oldest_datoid); +@@ -1257,7 +1257,8 @@ GetNewMultiXactId(int nmembers, MultiXactOffset *offset) + * Immediately kick autovacuum into action as we're already in + * ERROR territory. + */ +- SendPostmasterSignal(PMSIGNAL_START_AUTOVAC_LAUNCHER); ++ if (IsUnderPostmaster) ++ SendPostmasterSignal(PMSIGNAL_START_AUTOVAC_LAUNCHER); + + /* complain even if that DB has disappeared */ + if (oldest_datname) +diff --git a/src/backend/access/transam/varsup.c b/src/backend/access/transam/varsup.c +index fe89578..1cb34ad 100644 +--- a/src/backend/access/transam/varsup.c ++++ b/src/backend/access/transam/varsup.c +@@ -144,7 +144,7 @@ GetNewTransactionId(bool isSubXact) + if (IsUnderPostmaster && (xid % 65536) == 0) + SendPostmasterSignal(PMSIGNAL_START_AUTOVAC_LAUNCHER); + +- if (IsUnderPostmaster && ++ if (IsNormalUserSession() && + TransactionIdFollowsOrEquals(xid, xidStopLimit)) + { + char *oldest_datname = get_database_name(oldest_datoid); +@@ -578,7 +578,7 @@ GetNewObjectId(void) + */ + if (TransamVariables->nextOid < ((Oid) FirstNormalObjectId)) + { +- if (IsPostmasterEnvironment) ++ if (IsNormalUserSession()) + { + /* wraparound, or first post-initdb assignment, in normal mode */ + TransamVariables->nextOid = FirstNormalObjectId; +@@ -623,7 +623,7 @@ static void + SetNextObjectId(Oid nextOid) + { + /* Safety check, this is only allowable during initdb */ +- if (IsPostmasterEnvironment) ++ if (IsNormalUserSession()) + elog(ERROR, "cannot advance OID counter anymore"); + + /* Taking the lock is, therefore, just pro forma; but do it anyway */ +diff --git a/src/backend/access/transam/xlogfuncs.c b/src/backend/access/transam/xlogfuncs.c +index 8c30901..95bc950 100644 +--- a/src/backend/access/transam/xlogfuncs.c ++++ b/src/backend/access/transam/xlogfuncs.c +@@ -674,6 +674,11 @@ pg_promote(PG_FUNCTION_ARGS) + FILE *promote_file; + int i; + ++ if (IsTrustedEmbeddedSession()) ++ ereport(ERROR, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("standby promotion is not supported in a trusted embedded session"))); ++ + if (!RecoveryInProgress()) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), +diff --git a/src/backend/commands/event_trigger.c b/src/backend/commands/event_trigger.c +index 4defe1b..6f100b6 100644 +--- a/src/backend/commands/event_trigger.c ++++ b/src/backend/commands/event_trigger.c +@@ -89,7 +89,7 @@ static bool + EventTriggersHaveRunnableBackend(void) + { + #ifdef OLIPHAUNT_EMBEDDED +- return IsUnderPostmaster || AmRegularBackendProcess(); ++ return IsNormalUserSession(); + #else + return IsUnderPostmaster; + #endif +diff --git a/src/backend/commands/tsearchcmds.c b/src/backend/commands/tsearchcmds.c +index ab16d42..97a3ec6 100644 +--- a/src/backend/commands/tsearchcmds.c ++++ b/src/backend/commands/tsearchcmds.c +@@ -352,7 +352,7 @@ verify_dictoptions(Oid tmplId, List *dictoptions) + * that can't be translated into template1's encoding). We want to create + * them anyway, since they might be usable later in other databases. + */ +- if (!IsUnderPostmaster) ++ if (!IsNormalUserSession()) + return; + + tup = SearchSysCache1(TSTEMPLATEOID, ObjectIdGetDatum(tmplId)); +diff --git a/src/backend/storage/ipc/signalfuncs.c b/src/backend/storage/ipc/signalfuncs.c +index a3a670b..563bc04 100644 +--- a/src/backend/storage/ipc/signalfuncs.c ++++ b/src/backend/storage/ipc/signalfuncs.c +@@ -287,6 +287,11 @@ pg_terminate_backend(PG_FUNCTION_ARGS) + Datum + pg_reload_conf(PG_FUNCTION_ARGS) + { ++ if (IsTrustedEmbeddedSession()) ++ ereport(ERROR, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("configuration reload is not supported in a trusted embedded session"))); ++ + if (kill(PostmasterPid, SIGHUP)) + { + ereport(WARNING, +@@ -307,6 +312,13 @@ pg_reload_conf(PG_FUNCTION_ARGS) + Datum + pg_rotate_logfile(PG_FUNCTION_ARGS) + { ++ if (IsTrustedEmbeddedSession()) ++ { ++ ereport(WARNING, ++ (errmsg("log rotation is not supported in a trusted embedded session"))); ++ PG_RETURN_BOOL(false); ++ } ++ + if (!Logging_collector) + { + ereport(WARNING, +diff --git a/src/backend/tcop/backend_startup.c b/src/backend/tcop/backend_startup.c +index c89fe12..f9e8884 100644 +--- a/src/backend/tcop/backend_startup.c ++++ b/src/backend/tcop/backend_startup.c +@@ -121,7 +121,7 @@ BackendMain(const void *startup_data, size_t startup_data_len) + */ + MemoryContextSwitchTo(TopMemoryContext); + +- PostgresMain(MyProcPort->database_name, MyProcPort->user_name); ++ PostgresMain(MyProcPort->database_name, MyProcPort->user_name, 0); + } + + +diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c +index be79438..fd02019 100644 +--- a/src/backend/tcop/postgres.c ++++ b/src/backend/tcop/postgres.c +@@ -4185,7 +4185,7 @@ PostgresSingleUserMain(int argc, char *argv[], + * Now that sufficient infrastructure has been initialized, PostgresMain() + * can do the rest. + */ +- PostgresMain(dbname, username); ++ PostgresMain(dbname, username, 0); + } + + #ifdef OLIPHAUNT_EMBEDDED +@@ -4434,6 +4434,11 @@ oliphaunt_embedded_main(int argc, char *argv[], + goto embedded_cleanup; + } + lifecycle->entrypoint_admitted = true; ++ if (!PrepareTrustedEmbeddedSession()) ++ { ++ lifecycle->rc = -1; ++ goto embedded_cleanup; ++ } + + if (progname == NULL) + progname = get_progname(argv[0]); +@@ -4471,6 +4476,7 @@ oliphaunt_embedded_main(int argc, char *argv[], + lifecycle->rc = -1; + goto embedded_cleanup; + } ++ ConfigurePreparedTrustedEmbeddedSession(); + + checkDataDir(); + lifecycle->cwd_restore_required = true; +@@ -4487,7 +4493,11 @@ oliphaunt_embedded_main(int argc, char *argv[], + CreateSharedMemoryAndSemaphores(); + set_max_safe_fds(); + PgStartTime = GetCurrentTimestamp(); ++ MyBackendType = B_BACKEND; + InitProcess(); ++ if (!MyProc->isRegularBackend) ++ ereport(FATAL, ++ (errmsg("trusted embedded session did not initialize a regular backend process"))); + + socket_error = oliphaunt_embedded_create_socketpair(lifecycle->socket_pair); + if (socket_error != 0) +@@ -4512,8 +4522,10 @@ oliphaunt_embedded_main(int argc, char *argv[], + client_sock.raddr.salen = sizeof(struct sockaddr_in); + + whereToSendOutput = DestRemote; +- MyBackendType = B_BACKEND; + MyProcPort = pq_init(&client_sock); ++ MyProcPort->proto = PG_PROTOCOL_EARLIEST; ++ MyProcPort->database_name = pstrdup(dbname); ++ MyProcPort->user_name = pstrdup(username); + MyProcPort->oliphaunt_io = io; + + /* +@@ -4529,7 +4541,7 @@ oliphaunt_embedded_main(int argc, char *argv[], + pq_endmessage(&buf); + } + +- PostgresMain(dbname, username); ++ PostgresMain(dbname, username, INIT_PG_TRUSTED_CLIENT); + oliphaunt_embedded_proc_exit(0); + lifecycle->rc = 0; + +@@ -4590,14 +4602,16 @@ embedded_cleanup: + * postgres main loop -- all backends, interactive or otherwise loop here + * + * dbname is the name of the database to connect to, username is the +- * PostgreSQL user name to be used for the session. ++ * PostgreSQL user name to be used for the session, and init_postgres_flags ++ * carries explicit InitPostgres startup policy from the owning entrypoint. + * + * NB: Single user mode specific setup should go to PostgresSingleUserMain() + * if reasonably possible. + * ---------------------------------------------------------------- + */ + void +-PostgresMain(const char *dbname, const char *username) ++PostgresMain(const char *dbname, const char *username, ++ bits32 init_postgres_flags) + { + sigjmp_buf local_sigjmp_buf; + +@@ -4608,6 +4622,11 @@ PostgresMain(const char *dbname, const char *username) + + Assert(dbname != NULL); + Assert(username != NULL); ++#ifdef OLIPHAUNT_EMBEDDED ++ Assert((init_postgres_flags & ~INIT_PG_TRUSTED_CLIENT) == 0); ++#else ++ Assert(init_postgres_flags == 0); ++#endif + + Assert(GetProcessingMode() == InitProcessing); + +@@ -4709,7 +4728,8 @@ PostgresMain(const char *dbname, const char *username) + */ + InitPostgres(dbname, InvalidOid, /* database to connect to */ + username, InvalidOid, /* role to connect as */ +- (!am_walsender) ? INIT_PG_LOAD_SESSION_LIBS : 0, ++ init_postgres_flags | ++ ((!am_walsender) ? INIT_PG_LOAD_SESSION_LIBS : 0), + NULL); /* no out_dbname */ + + /* +diff --git a/src/backend/utils/init/Makefile b/src/backend/utils/init/Makefile +index 18c9474..5f9b641 100644 +--- a/src/backend/utils/init/Makefile ++++ b/src/backend/utils/init/Makefile +@@ -13,6 +13,7 @@ top_builddir = ../../../.. + include $(top_builddir)/src/Makefile.global + + OBJS = \ ++ embedded_session.o \ + globals.o \ + miscinit.o \ + postinit.o \ +diff --git a/src/backend/utils/init/embedded_session.c b/src/backend/utils/init/embedded_session.c +new file mode 100644 +index 0000000..d506094 +--- /dev/null ++++ b/src/backend/utils/init/embedded_session.c +@@ -0,0 +1,112 @@ ++/*------------------------------------------------------------------------- ++ * ++ * embedded_session.c ++ * trusted embedded session lifecycle and topology policy ++ * ++ * Portions Copyright (c) 1996-2025, PostgreSQL Global Development Group ++ * Portions Copyright (c) 1994, Regents of the University of California ++ * ++ * IDENTIFICATION ++ * src/backend/utils/init/embedded_session.c ++ * ++ *------------------------------------------------------------------------- ++ */ ++#include "postgres.h" ++ ++#ifdef OLIPHAUNT_EMBEDDED ++ ++#include "miscadmin.h" ++#include "optimizer/cost.h" ++#include "port/atomics.h" ++#include "replication/walsender.h" ++#include "storage/aio.h" ++#include "utils/guc.h" ++ ++typedef enum TrustedEmbeddedSessionLifecycle ++{ ++ TRUSTED_EMBEDDED_SESSION_UNSELECTED = 0, ++ TRUSTED_EMBEDDED_SESSION_PREPARED, ++ TRUSTED_EMBEDDED_SESSION_ATTACHED ++} TrustedEmbeddedSessionLifecycle; ++ ++/* ++ * Preparation and attachment are distinct one-way transitions. PREPARED is ++ * deliberately not an active user session: PostgreSQL startup still owns the ++ * decision to attach the trusted client at the catalog-safe point. ++ */ ++static pg_atomic_uint32 trusted_embedded_session_lifecycle = {0}; ++ ++bool ++PrepareTrustedEmbeddedSession(void) ++{ ++ uint32 expected = TRUSTED_EMBEDDED_SESSION_UNSELECTED; ++ ++ return pg_atomic_compare_exchange_u32( ++ &trusted_embedded_session_lifecycle, ++ &expected, ++ TRUSTED_EMBEDDED_SESSION_PREPARED); ++} ++ ++static bool ++trusted_embedded_settings_are_safe(void) ++{ ++ return io_method == IOMETHOD_SYNC && ++ max_worker_processes == 0 && ++ max_parallel_workers == 0 && ++ max_parallel_workers_per_gather == 0 && ++ max_parallel_maintenance_workers == 0 && ++ max_wal_senders == 0; ++} ++ ++void ++ConfigurePreparedTrustedEmbeddedSession(void) ++{ ++ if (pg_atomic_read_u32(&trusted_embedded_session_lifecycle) != ++ TRUSTED_EMBEDDED_SESSION_PREPARED) ++ ereport(FATAL, ++ (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), ++ errmsg("trusted embedded session was not prepared before configuration"))); ++ ++ /* ++ * Pin topology before shared-memory sizing. PGC_S_OVERRIDE wins over ++ * database and role defaults; GUC check hooks reject later live nonzero ++ * worker settings after attachment. ++ */ ++ SetConfigOption("io_method", "sync", PGC_POSTMASTER, PGC_S_OVERRIDE); ++ SetConfigOption("max_worker_processes", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_parallel_workers", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_parallel_workers_per_gather", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_parallel_maintenance_workers", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_wal_senders", "0", PGC_POSTMASTER, PGC_S_OVERRIDE); ++ ++ if (!trusted_embedded_settings_are_safe()) ++ ereport(FATAL, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("trusted embedded sessions require synchronous I/O and zero PostgreSQL workers"))); ++} ++ ++bool ++AttachPreparedTrustedEmbeddedSession(void) ++{ ++ uint32 expected = TRUSTED_EMBEDDED_SESSION_PREPARED; ++ ++ if (!trusted_embedded_settings_are_safe()) ++ return false; ++ return pg_atomic_compare_exchange_u32( ++ &trusted_embedded_session_lifecycle, ++ &expected, ++ TRUSTED_EMBEDDED_SESSION_ATTACHED); ++} ++ ++bool ++IsTrustedEmbeddedSession(void) ++{ ++ return pg_atomic_read_u32(&trusted_embedded_session_lifecycle) == ++ TRUSTED_EMBEDDED_SESSION_ATTACHED; ++} ++ ++#endif +diff --git a/src/backend/utils/init/meson.build b/src/backend/utils/init/meson.build +index c5d42e6..3d9af56 100644 +--- a/src/backend/utils/init/meson.build ++++ b/src/backend/utils/init/meson.build +@@ -1,6 +1,7 @@ + # Copyright (c) 2022-2025, PostgreSQL Global Development Group + + backend_sources += files( ++ 'embedded_session.c', + 'globals.c', + 'miscinit.c', + 'postinit.c', +diff --git a/src/backend/utils/init/miscinit.c b/src/backend/utils/init/miscinit.c +index df18257..2000cc1 100644 +--- a/src/backend/utils/init/miscinit.c ++++ b/src/backend/utils/init/miscinit.c +@@ -842,11 +842,11 @@ InitializeSessionUserId(const char *rolename, Oid roleid, + PGC_BACKEND, PGC_S_OVERRIDE); + + /* +- * These next checks are not enforced when in standalone mode, so that +- * there is a way to recover from sillinesses like "UPDATE pg_authid SET +- * rolcanlogin = false;". ++ * These checks apply to normal user sessions. Genuine recovery standalone ++ * mode uses InitializeSessionUserIdStandalone() and retains its escape ++ * hatch for damaged catalogs. + */ +- if (IsUnderPostmaster) ++ if (IsNormalUserSession()) + { + /* + * Is role allowed to login at all? (But background workers can +diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c +index c86ceef..497ce5e 100644 +--- a/src/backend/utils/init/postinit.c ++++ b/src/backend/utils/init/postinit.c +@@ -351,7 +351,7 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect + * a way to recover from disabling all access to all databases, for + * example "UPDATE pg_database SET datallowconn = false;". + */ +- if (IsUnderPostmaster) ++ if (IsNormalUserSession()) + { + /* + * Check that the database is currently allowing connections. +@@ -677,6 +677,9 @@ BaseInit(void) + * - INIT_PG_LOAD_SESSION_LIBS to honor [session|local]_preload_libraries. + * - INIT_PG_OVERRIDE_ALLOW_CONNS to connect despite !datallowconn. + * - INIT_PG_OVERRIDE_ROLE_LOGIN to connect despite !rolcanlogin. ++ * - INIT_PG_TRUSTED_CLIENT (embedded builds) to use a prepared ++ * host-authenticated identity ++ * while retaining standalone process topology. + * out_dbname: optional output parameter, see below; pass NULL if not used + * + * The database can be specified by name, using the in_dbname parameter, or by +@@ -716,6 +719,9 @@ InitPostgres(const char *in_dbname, Oid dboid, + { + bool bootstrap = IsBootstrapProcessingMode(); + bool am_superuser; ++#ifdef OLIPHAUNT_EMBEDDED ++ bool trusted_client = (flags & INIT_PG_TRUSTED_CLIENT) != 0; ++#endif + char *fullpath; + char dbname[NAMEDATALEN]; + int nfree = 0; +@@ -855,19 +861,44 @@ InitPostgres(const char *in_dbname, Oid dboid, + XactIsoLevel = XACT_READ_COMMITTED; + } + ++#ifdef OLIPHAUNT_EMBEDDED ++ if (trusted_client) ++ { ++ if (bootstrap || IsPostmasterEnvironment || IsUnderPostmaster || ++ !AmRegularBackendProcess() || MyProcPort == NULL) ++ ereport(FATAL, ++ (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), ++ errmsg("trusted client requires a standalone regular backend and protocol port"))); ++ if (!AttachPreparedTrustedEmbeddedSession()) ++ ereport(FATAL, ++ (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), ++ errmsg("trusted embedded session attach was not prepared or was already consumed"))); ++ } ++#endif ++ + /* + * Perform client authentication if necessary, then figure out our + * postgres user ID, and see if we are a superuser. + * +- * In standalone mode, autovacuum worker processes and slot sync worker +- * process, we use a fixed ID, otherwise we figure it out from the +- * authenticated user name. ++ * In recovery standalone mode, autovacuum worker processes and slot sync ++ * worker process, we use a fixed ID. A prepared trusted client has no HBA ++ * exchange, but still resolves the host-authenticated name through the ++ * catalogs and enforces normal login policy. + */ + if (bootstrap || AmAutoVacuumWorkerProcess() || AmLogicalSlotSyncWorkerProcess()) + { + InitializeSessionUserIdStandalone(); + am_superuser = true; + } ++#ifdef OLIPHAUNT_EMBEDDED ++ else if (trusted_client) ++ { ++ Assert(MyProcPort != NULL); ++ Assert(IsTrustedEmbeddedSession()); ++ InitializeSessionUserId(username, useroid, false); ++ am_superuser = superuser(); ++ } ++#endif + else if (!IsUnderPostmaster) + { + InitializeSessionUserIdStandalone(); +@@ -1311,7 +1342,7 @@ process_settings(Oid databaseid, Oid roleid) + Relation relsetting; + Snapshot snapshot; + +- if (!IsUnderPostmaster) ++ if (!IsNormalUserSession()) + return; + + relsetting = table_open(DbRoleSettingRelationId, AccessShareLock); +diff --git a/src/backend/utils/misc/guc_tables.c b/src/backend/utils/misc/guc_tables.c +index 6b82a23..49646da 100644 +--- a/src/backend/utils/misc/guc_tables.c ++++ b/src/backend/utils/misc/guc_tables.c +@@ -55,6 +55,7 @@ + #include "libpq/libpq.h" + #include "libpq/oauth.h" + #include "libpq/scram.h" ++#include "miscadmin.h" + #include "nodes/queryjumble.h" + #include "optimizer/cost.h" + #include "optimizer/geqo.h" +@@ -112,6 +113,30 @@ + #define PG_KRB_SRVTAB "" + #endif + ++#ifdef OLIPHAUNT_EMBEDDED ++static bool ++check_trusted_embedded_worker_limit(int *newval, void **extra, ++ GucSource source) ++{ ++ (void) extra; ++ ++ /* PGC_S_TEST validates stored values without changing this process. */ ++ if (IsTrustedEmbeddedSession() && ++ source != PGC_S_TEST && ++ source >= PGC_S_OVERRIDE && ++ *newval != 0) ++ { ++ GUC_check_errdetail("Trusted embedded sessions cannot launch PostgreSQL workers."); ++ return false; ++ } ++ ++ return true; ++} ++#define CHECK_TRUSTED_EMBEDDED_WORKER_LIMIT check_trusted_embedded_worker_limit ++#else ++#define CHECK_TRUSTED_EMBEDDED_WORKER_LIMIT NULL ++#endif ++ + /* + * Options for enum values defined in this module. + * +@@ -3620,7 +3645,7 @@ struct config_int ConfigureNamesInt[] = + }, + &max_parallel_maintenance_workers, + 2, 0, MAX_PARALLEL_WORKER_LIMIT, +- NULL, NULL, NULL ++ CHECK_TRUSTED_EMBEDDED_WORKER_LIMIT, NULL, NULL + }, + + { +@@ -3631,7 +3656,7 @@ struct config_int ConfigureNamesInt[] = + }, + &max_parallel_workers_per_gather, + 2, 0, MAX_PARALLEL_WORKER_LIMIT, +- NULL, NULL, NULL ++ CHECK_TRUSTED_EMBEDDED_WORKER_LIMIT, NULL, NULL + }, + + { +@@ -3642,7 +3667,7 @@ struct config_int ConfigureNamesInt[] = + }, + &max_parallel_workers, + 8, 0, MAX_PARALLEL_WORKER_LIMIT, +- NULL, NULL, NULL ++ CHECK_TRUSTED_EMBEDDED_WORKER_LIMIT, NULL, NULL + }, + + { +diff --git a/src/backend/utils/misc/superuser.c b/src/backend/utils/misc/superuser.c +index 5858b0a..eb896e2 100644 +--- a/src/backend/utils/misc/superuser.c ++++ b/src/backend/utils/misc/superuser.c +@@ -63,7 +63,7 @@ superuser_arg(Oid roleid) + return last_roleid_is_super; + + /* Special escape path in case you deleted all your users. */ +- if (!IsUnderPostmaster && roleid == BOOTSTRAP_SUPERUSERID) ++ if (!IsNormalUserSession() && roleid == BOOTSTRAP_SUPERUSERID) + return true; + + /* OK, look up the information in pg_authid */ +diff --git a/src/include/miscadmin.h b/src/include/miscadmin.h +index 9a7d733..addc346 100644 +--- a/src/include/miscadmin.h ++++ b/src/include/miscadmin.h +@@ -168,6 +168,30 @@ extern PGDLLIMPORT bool IsPostmasterEnvironment; + extern PGDLLIMPORT bool IsUnderPostmaster; + extern PGDLLIMPORT bool IsBinaryUpgrade; + ++#ifdef OLIPHAUNT_EMBEDDED ++extern bool PrepareTrustedEmbeddedSession(void); ++extern void ConfigurePreparedTrustedEmbeddedSession(void); ++extern bool AttachPreparedTrustedEmbeddedSession(void); ++extern bool IsTrustedEmbeddedSession(void); ++#else ++static inline bool ++IsTrustedEmbeddedSession(void) ++{ ++ return false; ++} ++#endif ++ ++/* ++ * A normal user session is either a real postmaster child or an explicitly ++ * attached trusted embedded client. This says nothing about supervisor, ++ * worker, socket, or XLOG topology. ++ */ ++static inline bool ++IsNormalUserSession(void) ++{ ++ return IsUnderPostmaster || IsTrustedEmbeddedSession(); ++} ++ + extern PGDLLIMPORT bool ExitOnAnyError; + + extern PGDLLIMPORT char *DataDir; +@@ -499,6 +523,9 @@ extern PGDLLIMPORT ProcessingMode Mode; + #define INIT_PG_LOAD_SESSION_LIBS 0x0001 + #define INIT_PG_OVERRIDE_ALLOW_CONNS 0x0002 + #define INIT_PG_OVERRIDE_ROLE_LOGIN 0x0004 ++#ifdef OLIPHAUNT_EMBEDDED ++#define INIT_PG_TRUSTED_CLIENT 0x0008 ++#endif + extern void pg_split_opts(char **argv, int *argcp, const char *optstr); + extern void InitializeMaxBackends(void); + extern void InitializeFastPathLocks(void); +diff --git a/src/include/tcop/tcopprot.h b/src/include/tcop/tcopprot.h +index 8e3ab1e..69164e1 100644 +--- a/src/include/tcop/tcopprot.h ++++ b/src/include/tcop/tcopprot.h +@@ -86,12 +86,14 @@ extern int oliphaunt_embedded_main(int argc, char *argv[], + extern void PostgresSingleUserMain(int argc, char *argv[], + const char *username); + extern void PostgresMain(const char *dbname, +- const char *username); ++ const char *username, ++ bits32 init_postgres_flags); + #else + pg_noreturn extern void PostgresSingleUserMain(int argc, char *argv[], + const char *username); + pg_noreturn extern void PostgresMain(const char *dbname, +- const char *username); ++ const char *username, ++ bits32 init_postgres_flags); + #endif + extern void ResetUsage(void); + extern void ShowUsage(const char *title); diff --git a/src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-wake-embedded-epoll-through-self-pipe.patch b/src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-wake-embedded-epoll-through-self-pipe.patch deleted file mode 100644 index e81da2d4b..000000000 --- a/src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-wake-embedded-epoll-through-self-pipe.patch +++ /dev/null @@ -1,49 +0,0 @@ ---- a/src/backend/storage/ipc/waiteventset.c -+++ b/src/backend/storage/ipc/waiteventset.c -@@ -103,6 +103,10 @@ - * By default, we use a self-pipe with poll() and a signalfd with epoll(), if - * available. For testing the choice can also be manually specified. - */ -+#if defined(OLIPHAUNT_EMBEDDED) && defined(WAIT_USE_EPOLL) -+/* The embedded host cancels from another thread, without PostgreSQL signals. */ -+#define WAIT_USE_SELF_PIPE -+#endif - #if defined(WAIT_USE_POLL) || defined(WAIT_USE_EPOLL) - #if defined(WAIT_USE_SELF_PIPE) || defined(WAIT_USE_SIGNALFD) - /* don't overwrite manual choice */ -@@ -185,7 +189,9 @@ - static int selfpipe_owner_pid = 0; - - /* Private function prototypes */ -+#if !defined(OLIPHAUNT_EMBEDDED) || !defined(WAIT_USE_EPOLL) - static void latch_sigurg_handler(SIGNAL_ARGS); -+#endif - static void sendSelfPipeByte(void); - #endif - -@@ -311,8 +317,10 @@ - ReserveExternalFD(); - ReserveExternalFD(); - -+#if !defined(OLIPHAUNT_EMBEDDED) || !defined(WAIT_USE_EPOLL) - pqsignal(SIGURG, latch_sigurg_handler); - #endif -+#endif - - #ifdef WAIT_USE_SIGNALFD - sigset_t signalfd_mask; -@@ -1892,12 +1900,14 @@ - * - * Wake up WaitLatch, if we're waiting. - */ -+#if !defined(OLIPHAUNT_EMBEDDED) || !defined(WAIT_USE_EPOLL) - static void - latch_sigurg_handler(SIGNAL_ARGS) - { - if (waiting) - sendSelfPipeByte(); - } -+#endif - - /* Send one byte to the self-pipe, to wake up WaitLatch */ - static void diff --git a/src/native/runtime/patches/postgresql-18.4/0022-liboliphaunt-preserve-host-process-boundaries.patch b/src/native/runtime/patches/postgresql-18.4/0022-liboliphaunt-preserve-host-process-boundaries.patch new file mode 100644 index 000000000..e0c4347ba --- /dev/null +++ b/src/native/runtime/patches/postgresql-18.4/0022-liboliphaunt-preserve-host-process-boundaries.patch @@ -0,0 +1,1861 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: liboliphaunt +Date: Thu, 27 Aug 2026 00:00:00 +0000 +Subject: [PATCH] liboliphaunt: preserve host process boundaries + +A Direct backend shares its process with the embedding application, but the +existing containment only reserved SIGUSR1. PostgresMain still replaced nine +host signal dispositions, the timeout subsystem repurposed ITIMER_REAL, and +latch wakeups emitted process-directed SIGURG. In a multithreaded host that +SIGURG can land outside the backend thread, both violating host ownership and +failing to wake cancellation promptly. + +Treat process signal state as host-owned. Direct pqsignal calls become inert; +nonzero kill and raise operations fail closed. Route public cancellation +through an atomic mailbox and a PostgreSQL-owned raw wake endpoint so only the +backend thread mutates PostgreSQL interrupt state. Preserve PostgreSQL's normal +wait provider in ordinary modes; Direct adds a hidden pipe/event to each wait +set and publishes only the stable raw wake token to the host. + +Keep PostgreSQL timeout ordering and handler semantics, but publish a lazy +one-shot deadline to the synchronous host waiter while bounding backend latch +and input waits directly. The private IO ABI is versioned, endpoint publication +is synchronized with host cancellation, and teardown unpublishes before close. +Direct teardown also releases retained latch wait sets and the Windows local +latch event that process exit would normally reclaim. +The hot interrupt path only reads an atomic cancel/timeout mailbox; exchanges +and clock reads stay on the uncommon slow path. Direct inherits the embedding +thread's signal mask instead of changing process-wide signal ownership. + +COPY FROM input waits consume Direct cancellation only at a validated frontend +frame boundary. This is safe because the Native ABI rejects truncated frontend +headers and bodies before publishing a request to PostgreSQL; ordinary protocol +and postmaster paths retain their upstream behavior. + +Reject PID-based SQL signalling, external pipe programs, and configured +archive/recovery shell commands because Direct cannot safely provide those +process capabilities. Broker isolation remains required for arbitrary native +extensions, external commands, blocking foreign code, and other process-global +behavior. This is a correctness boundary, not a performance optimization. + +The libc collation enumerator uses locale -a, so trusted sessions report and +skip that subprocess-only discovery while still importing ICU collations in +process. Ordinary server and initdb discovery remain unchanged. Put the signal +mask prototype beside PostgreSQL's sigset_t definition so Windows support +objects can include miscadmin.h without relying on backend include order. +--- +diff --git a/src/backend/access/transam/xact.c b/src/backend/access/transam/xact.c +index b885513f76541bc165a41ce78f83623703c2410a..ddd112f69b49e268f76e5e96d5314b94817b8ce4 100644 +--- a/src/backend/access/transam/xact.c ++++ b/src/backend/access/transam/xact.c +@@ -2866,7 +2866,12 @@ AbortTransaction(void) + * handler. We do this fairly early in the sequence so that the timeout + * infrastructure will be functional if needed while aborting. + */ +- sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ (void) SetTrustedEmbeddedThreadSignalMask(&UnBlockSig); ++ else ++#endif ++ sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); + + /* + * check the current transaction state +@@ -5271,7 +5276,12 @@ AbortSubTransaction(void) + * handler. We do this fairly early in the sequence so that the timeout + * infrastructure will be functional if needed while aborting. + */ +- sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ (void) SetTrustedEmbeddedThreadSignalMask(&UnBlockSig); ++ else ++#endif ++ sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); + + /* + * check the current transaction state +diff --git a/src/backend/commands/copyfromparse.c b/src/backend/commands/copyfromparse.c +index f5fc346e2013bb2332a7521d203eeb1a0ec2f920..5edf1b51109a00ac3c5c4d7bfeba0024b759effa 100644 +--- a/src/backend/commands/copyfromparse.c ++++ b/src/backend/commands/copyfromparse.c +@@ -271,7 +271,21 @@ CopyGetData(CopyFromState cstate, void *databuf, int minread, int maxread) + readmessage: + HOLD_CANCEL_INTERRUPTS(); + pq_startmsgread(); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ mtype = pq_getbyte_interruptible(); ++ else ++#endif + mtype = pq_getbyte(); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (mtype == PQ_READ_INTERRUPTED) ++ { ++ pq_endmsgread(); ++ RESUME_CANCEL_INTERRUPTS(); ++ CHECK_FOR_INTERRUPTS(); ++ goto readmessage; ++ } ++#endif + if (mtype == EOF) + ereport(ERROR, + (errcode(ERRCODE_CONNECTION_FAILURE), +diff --git a/src/backend/libpq/be-secure.c b/src/backend/libpq/be-secure.c +index b56bfb41d53166af4c92e0999bf0080e6ec4a6c2..7c2cae95139856dc9a0081c7d10cbd8101e56b8b 100644 +--- a/src/backend/libpq/be-secure.c ++++ b/src/backend/libpq/be-secure.c +@@ -31,6 +31,7 @@ + #include "miscadmin.h" + #include "tcop/tcopprot.h" + #include "utils/injection_point.h" ++#include "utils/timeout.h" + #include "utils/wait_event.h" + + char *ssl_library; +@@ -284,7 +285,8 @@ secure_raw_read(Port *port, void *ptr, size_t len) + + #ifdef OLIPHAUNT_EMBEDDED + if (port->oliphaunt_io != NULL && port->oliphaunt_io->read != NULL) +- return port->oliphaunt_io->read(port->oliphaunt_io->context, ptr, len); ++ return port->oliphaunt_io->read(port->oliphaunt_io->context, ptr, len, ++ GetEmbeddedTimeoutDelayMilliseconds()); + #endif + + /* +diff --git a/src/backend/libpq/pqcomm.c b/src/backend/libpq/pqcomm.c +index 4181d9c88411d02ccf47be17e9864a00b2dd02ec..bb230c1fe6643f2c46301f278fde004f20654d47 100644 +--- a/src/backend/libpq/pqcomm.c ++++ b/src/backend/libpq/pqcomm.c +@@ -903,7 +903,7 @@ socket_set_nonblocking(bool nonblocking) + * -------------------------------- + */ + static int +-pq_recvbuf(void) ++pq_recvbuf_internal(bool allow_trusted_interrupt) + { + if (PqRecvPointer > 0) + { +@@ -935,7 +935,16 @@ pq_recvbuf(void) + if (r < 0) + { + if (errno == EINTR) ++ { ++#ifdef OLIPHAUNT_EMBEDDED ++ if (allow_trusted_interrupt && ++ TrustedEmbeddedInterruptPending()) ++ return PQ_READ_INTERRUPTED; ++#else ++ (void) allow_trusted_interrupt; ++#endif + continue; /* Ok if interrupted */ ++ } + + /* + * Careful: an ereport() that tries to write to the client would +@@ -964,6 +973,12 @@ pq_recvbuf(void) + } + } + ++static int ++pq_recvbuf(void) ++{ ++ return pq_recvbuf_internal(false); ++} ++ + /* -------------------------------- + * pq_getbyte - get a single byte from connection, or return EOF + * -------------------------------- +@@ -981,6 +996,27 @@ pq_getbyte(void) + return (unsigned char) PqRecvBuffer[PqRecvPointer++]; + } + ++#ifdef OLIPHAUNT_EMBEDDED ++/* ++ * COPY-only variant that can expose a validated Native transport boundary. ++ * Generic socket callers retain pq_getbyte()'s ordinary EOF/retry contract. ++ */ ++int ++pq_getbyte_interruptible(void) ++{ ++ Assert(PqCommReadingMsg); ++ ++ while (PqRecvPointer >= PqRecvLength) ++ { ++ int rc = pq_recvbuf_internal(true); ++ ++ if (rc != 0) ++ return rc == PQ_READ_INTERRUPTED ? PQ_READ_INTERRUPTED : EOF; ++ } ++ return (unsigned char) PqRecvBuffer[PqRecvPointer++]; ++} ++#endif ++ + /* -------------------------------- + * pq_peekbyte - peek at next byte from connection + * +diff --git a/src/backend/port/win32/signal.c b/src/backend/port/win32/signal.c +index d051b15c0ddb38128d9f7b40c77630eabe32d8d2..01b748d56ad849c3178303c76ec2df42e266bea1 100644 +--- a/src/backend/port/win32/signal.c ++++ b/src/backend/port/win32/signal.c +@@ -36,6 +36,7 @@ static CRITICAL_SECTION pg_signal_crit_sec; + /* Note that array elements 0 are unused since they correspond to signal 0 */ + static struct sigaction pg_signal_array[PG_SIGNAL_COUNT]; + static pqsigfunc pg_signal_defaults[PG_SIGNAL_COUNT]; ++static bool pgwin32_embedded_signal_initialized = false; + + + /* Signal handling thread functions */ +@@ -74,12 +75,11 @@ pg_usleep(long microsec) + } + + +-/* Initialization */ +-void +-pgwin32_signal_initialize(void) ++/* Initialize state shared by normal and trusted embedded signal providers. */ ++static void ++pgwin32_signal_initialize_common(void) + { + int i; +- HANDLE signal_thread_handle; + + InitializeCriticalSection(&pg_signal_crit_sec); + +@@ -98,6 +98,41 @@ pgwin32_signal_initialize(void) + if (pgwin32_signal_event == NULL) + ereport(FATAL, + (errmsg_internal("could not create signal event: error code %lu", GetLastError()))); ++} ++ ++/* ++ * Initialize only PostgreSQL's thread-local emulation state and wake event. ++ * A trusted embedded backend must not create a process signal listener or ++ * claim the embedding application's console-control handler. ++ */ ++void ++pgwin32_signal_initialize_embedded(void) ++{ ++ pgwin32_signal_initialize_common(); ++ pgwin32_embedded_signal_initialized = true; ++} ++ ++void ++pgwin32_signal_shutdown_embedded(void) ++{ ++ if (!pgwin32_embedded_signal_initialized) ++ return; ++ if (pgwin32_signal_event != NULL) ++ { ++ (void) CloseHandle(pgwin32_signal_event); ++ pgwin32_signal_event = NULL; ++ } ++ DeleteCriticalSection(&pg_signal_crit_sec); ++ pgwin32_embedded_signal_initialized = false; ++} ++ ++/* Initialization */ ++void ++pgwin32_signal_initialize(void) ++{ ++ HANDLE signal_thread_handle; ++ ++ pgwin32_signal_initialize_common(); + + /* Create thread for handling signals */ + signal_thread_handle = CreateThread(NULL, 0, pg_signal_thread, NULL, 0, NULL); +diff --git a/src/backend/storage/file/fd.c b/src/backend/storage/file/fd.c +index 378c078c7c7f347e4d3c5bf00477c6804a600364..31d93e1f270fe329f5b8f311cab60efad7579d33 100644 +--- a/src/backend/storage/file/fd.c ++++ b/src/backend/storage/file/fd.c +@@ -2749,6 +2749,13 @@ OpenPipeStream(const char *command, const char *mode) + FILE *file; + int save_errno; + ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ ereport(ERROR, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("external programs are not supported in a trusted embedded backend"))); ++#endif ++ + DO_DB(elog(LOG, "OpenPipeStream: Allocated %d (%s)", + numAllocatedDescs, command)); + +diff --git a/src/backend/storage/ipc/ipc.c b/src/backend/storage/ipc/ipc.c +index 97edd034ca46a669feee954c25225fcd3f40218e..17ae5c22dcc7d87f15a0c97e0409a713f550f32e 100644 +--- a/src/backend/storage/ipc/ipc.c ++++ b/src/backend/storage/ipc/ipc.c +@@ -269,18 +269,6 @@ oliphaunt_embedded_clear_proc_exit_handler(oliphaunt_embedded_proc_exit_handler + return true; + } + +-/* +- * Run PostgreSQL's normal backend exit callbacks without terminating the host +- * process. This is only valid for the embedded thread owner that already +- * arranged for PostgresMain() to return on frontend Terminate. +- */ +-void +-oliphaunt_embedded_proc_exit(int code) +-{ +- proc_exit_prepare(code); +- +- proc_exit_inprogress = false; +-} + #endif + + /* ------------------ +diff --git a/src/backend/storage/ipc/latch.c b/src/backend/storage/ipc/latch.c +index beadeb5e46afa2c2d88c1a2a75bc951f5bd538aa..f29cccd1e3ec84e985e7d37a9c657777289b9aef 100644 +--- a/src/backend/storage/ipc/latch.c ++++ b/src/backend/storage/ipc/latch.c +@@ -56,6 +56,25 @@ InitializeLatchWaitSet(void) + } + } + ++#ifdef OLIPHAUNT_EMBEDDED ++/* ++ * A normal backend lets process exit reclaim this session-lifetime wait set. ++ * Direct returns to a retained host process, so release it explicitly before ++ * the provider-private wake endpoint is closed. ++ */ ++void ++ShutdownTrustedEmbeddedLatchWaitSet(void) ++{ ++ Assert(IsTrustedEmbeddedProcess()); ++ ++ if (LatchWaitSet != NULL) ++ { ++ FreeWaitEventSet(LatchWaitSet); ++ LatchWaitSet = NULL; ++ } ++} ++#endif ++ + /* + * Initialize a process-local latch. + */ +diff --git a/src/backend/storage/ipc/procsignal.c b/src/backend/storage/ipc/procsignal.c +index b4bb246dec4a1b6293eeda45304df6013514b56d..62c0581e095720d3942694d491a602657d23edb7 100644 +--- a/src/backend/storage/ipc/procsignal.c ++++ b/src/backend/storage/ipc/procsignal.c +@@ -278,21 +278,23 @@ static int + oliphaunt_send_proc_signal(pid_t pid) + { + #ifdef OLIPHAUNT_EMBEDDED +- int save_errno; +- +- if (pid != MyProcPid) ++ if (IsTrustedEmbeddedProcess()) + { +- errno = ESRCH; +- return -1; +- } ++ int save_errno; + +- save_errno = errno; +- procsignal_sigusr1_handler(SIGUSR1); +- errno = save_errno; +- return 0; +-#else +- return kill(pid, SIGUSR1); ++ if (pid != MyProcPid) ++ { ++ errno = ESRCH; ++ return -1; ++ } ++ ++ save_errno = errno; ++ procsignal_sigusr1_handler(SIGUSR1); ++ errno = save_errno; ++ return 0; ++ } + #endif ++ return kill(pid, SIGUSR1); + } + + /* +diff --git a/src/backend/storage/ipc/signalfuncs.c b/src/backend/storage/ipc/signalfuncs.c +index 563bc047bb22009099da9a2ac1ab273a014aa23e..05ee07dee9836ff9b42f6476170a9d278540d2c8 100644 +--- a/src/backend/storage/ipc/signalfuncs.c ++++ b/src/backend/storage/ipc/signalfuncs.c +@@ -135,7 +135,15 @@ pg_signal_backend(int pid, int sig) + Datum + pg_cancel_backend(PG_FUNCTION_ARGS) + { +- int r = pg_signal_backend(PG_GETARG_INT32(0), SIGINT); ++ int r; ++ ++ if (IsTrustedEmbeddedSession()) ++ ereport(ERROR, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("backend cancellation by process ID is not supported in a trusted embedded session"), ++ errhint("Use the embedding host's cancellation API for the active session."))); ++ ++ r = pg_signal_backend(PG_GETARG_INT32(0), SIGINT); + + if (r == SIGNAL_BACKEND_NOSUPERUSER) + ereport(ERROR, +@@ -243,6 +251,11 @@ pg_terminate_backend(PG_FUNCTION_ARGS) + pid = PG_GETARG_INT32(0); + timeout = PG_GETARG_INT64(1); + ++ if (IsTrustedEmbeddedSession()) ++ ereport(ERROR, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("backend termination by process ID is not supported in a trusted embedded session"))); ++ + if (timeout < 0) + ereport(ERROR, + (errcode(ERRCODE_NUMERIC_VALUE_OUT_OF_RANGE), +diff --git a/src/backend/storage/ipc/waiteventset.c b/src/backend/storage/ipc/waiteventset.c +index b0746521ae4251b459f85429661bb80e8227de4d..5088bb3cb7dd64272a5eb2af5474d0e3693cb2d3 100644 +--- a/src/backend/storage/ipc/waiteventset.c ++++ b/src/backend/storage/ipc/waiteventset.c +@@ -65,6 +65,9 @@ + #endif + + #include "libpq/pqsignal.h" ++#ifdef OLIPHAUNT_EMBEDDED ++#include "libpq/libpq-be.h" ++#endif + #include "miscadmin.h" + #include "pgstat.h" + #include "port/atomics.h" +@@ -77,6 +80,7 @@ + #include "storage/waiteventset.h" + #include "utils/memutils.h" + #include "utils/resowner.h" ++#include "utils/timeout.h" + + /* + * Select the fd readiness primitive to use. Normally the "most modern" +@@ -111,6 +115,17 @@ + #else + #define WAIT_USE_SELF_PIPE + #endif ++ ++#endif ++ ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++/* Stable raw wake endpoint owned by the selected Direct backend. */ ++static int trusted_embedded_wakeup_readfd = -1; ++static int trusted_embedded_wakeup_writefd = -1; ++static bool trusted_embedded_wakeup_reserved = false; ++ ++static bool InitializeTrustedEmbeddedWakeup(void); ++static void DrainTrustedEmbeddedWakeup(void); + #endif + + /* typedef in waiteventset.h */ +@@ -230,6 +245,67 @@ ResourceOwnerForgetWaitEventSet(ResourceOwner owner, WaitEventSet *set) + ResourceOwnerForget(owner, PointerGetDatum(set), &wait_event_set_resowner_desc); + } + ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++static bool ++InitializeTrustedEmbeddedWakeup(void) ++{ ++ int pipefd[2]; ++ int save_errno; ++ ++ if (trusted_embedded_wakeup_readfd != -1 || ++ trusted_embedded_wakeup_writefd != -1) ++ { ++ errno = EALREADY; ++ return false; ++ } ++ if (pipe(pipefd) < 0) ++ return false; ++ ++ trusted_embedded_wakeup_readfd = pipefd[0]; ++ trusted_embedded_wakeup_writefd = pipefd[1]; ++ if (fcntl(pipefd[0], F_SETFL, O_NONBLOCK) == -1 || ++ fcntl(pipefd[1], F_SETFL, O_NONBLOCK) == -1 || ++ fcntl(pipefd[0], F_SETFD, FD_CLOEXEC) == -1 || ++ fcntl(pipefd[1], F_SETFD, FD_CLOEXEC) == -1) ++ { ++ save_errno = errno; ++ (void) close(pipefd[0]); ++ (void) close(pipefd[1]); ++ trusted_embedded_wakeup_readfd = -1; ++ trusted_embedded_wakeup_writefd = -1; ++ errno = save_errno; ++ return false; ++ } ++ ++ ReserveExternalFD(); ++ ReserveExternalFD(); ++ trusted_embedded_wakeup_reserved = true; ++ return true; ++} ++ ++static void ++DrainTrustedEmbeddedWakeup(void) ++{ ++ char buf[128]; ++ ++ for (;;) ++ { ++ ssize_t rc = read(trusted_embedded_wakeup_readfd, ++ buf, sizeof(buf)); ++ ++ if (rc > 0) ++ continue; ++ if (rc < 0 && errno == EINTR) ++ continue; ++ if (rc < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) ++ return; ++ if (rc == 0) ++ elog(ERROR, "unexpected EOF on trusted embedded wakeup pipe"); ++ elog(ERROR, "read() on trusted embedded wakeup pipe failed: %m"); ++ } ++} ++#endif ++ + + /* + * Initialize the process-local wait event infrastructure. +@@ -240,8 +316,16 @@ ResourceOwnerForgetWaitEventSet(ResourceOwner owner, WaitEventSet *set) + void + InitializeWaitEventSupport(void) + { ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++ if (IsTrustedEmbeddedProcess() && !InitializeTrustedEmbeddedWakeup()) ++ elog(FATAL, "could not initialize trusted embedded wakeup pipe: %m"); ++#endif + #if defined(WAIT_USE_SELF_PIPE) +- int pipefd[2]; ++#ifdef OLIPHAUNT_EMBEDDED ++ if (!IsTrustedEmbeddedProcess()) ++#endif ++ { ++ int pipefd[2]; + + if (IsUnderPostmaster) + { +@@ -310,12 +394,16 @@ InitializeWaitEventSupport(void) + /* Tell fd.c about these two long-lived FDs */ + ReserveExternalFD(); + ReserveExternalFD(); +- + pqsignal(SIGURG, latch_sigurg_handler); ++ } + #endif + + #ifdef WAIT_USE_SIGNALFD +- sigset_t signalfd_mask; ++#ifdef OLIPHAUNT_EMBEDDED ++ if (!IsTrustedEmbeddedProcess()) ++#endif ++ { ++ sigset_t signalfd_mask; + + if (IsUnderPostmaster) + { +@@ -343,11 +431,15 @@ InitializeWaitEventSupport(void) + if (signal_fd < 0) + elog(FATAL, "signalfd() failed"); + ReserveExternalFD(); ++ } + #endif + + #ifdef WAIT_USE_KQUEUE + /* Ignore SIGURG, because we'll receive it via kqueue. */ +- pqsignal(SIGURG, SIG_IGN); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (!IsTrustedEmbeddedProcess()) ++#endif ++ pqsignal(SIGURG, SIG_IGN); + #endif + } + +@@ -381,7 +473,12 @@ CreateWaitEventSet(ResourceOwner resowner, int nevents) + #elif defined(WAIT_USE_KQUEUE) + sz += MAXALIGN(sizeof(struct kevent) * nevents); + #elif defined(WAIT_USE_POLL) ++ /* Direct reserves one extra provider-private wakeup slot. */ ++#ifdef OLIPHAUNT_EMBEDDED ++ sz += MAXALIGN(sizeof(struct pollfd) * (nevents + 1)); ++#else + sz += MAXALIGN(sizeof(struct pollfd) * nevents); ++#endif + #elif defined(WAIT_USE_WIN32) + /* need space for the pgwin32_signal_event */ + sz += MAXALIGN(sizeof(HANDLE) * (nevents + 1)); +@@ -406,7 +503,11 @@ CreateWaitEventSet(ResourceOwner resowner, int nevents) + data += MAXALIGN(sizeof(struct kevent) * nevents); + #elif defined(WAIT_USE_POLL) + set->pollfds = (struct pollfd *) data; ++#ifdef OLIPHAUNT_EMBEDDED ++ data += MAXALIGN(sizeof(struct pollfd) * (nevents + 1)); ++#else + data += MAXALIGN(sizeof(struct pollfd) * nevents); ++#endif + #elif defined(WAIT_USE_WIN32) + set->handles = (HANDLE) data; + data += MAXALIGN(sizeof(HANDLE) * nevents); +@@ -465,6 +566,29 @@ CreateWaitEventSet(ResourceOwner resowner, int nevents) + StaticAssertStmt(WSA_INVALID_EVENT == NULL, ""); + #endif + ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++ if (IsTrustedEmbeddedProcess()) ++ { ++#if defined(WAIT_USE_EPOLL) ++ struct epoll_event wakeup_event; ++ ++ MemSet(&wakeup_event, 0, sizeof(wakeup_event)); ++ wakeup_event.events = EPOLLIN | EPOLLERR | EPOLLHUP; ++ wakeup_event.data.ptr = NULL; ++ if (epoll_ctl(set->epoll_fd, EPOLL_CTL_ADD, ++ trusted_embedded_wakeup_readfd, &wakeup_event) < 0) ++ elog(ERROR, "could not register trusted embedded epoll wakeup: %m"); ++#elif defined(WAIT_USE_KQUEUE) ++ struct kevent wakeup_event; ++ ++ EV_SET(&wakeup_event, trusted_embedded_wakeup_readfd, ++ EVFILT_READ, EV_ADD, 0, 0, NULL); ++ if (kevent(set->kqueue_fd, &wakeup_event, 1, NULL, 0, NULL) < 0) ++ elog(ERROR, "could not register trusted embedded kqueue wakeup: %m"); ++#endif ++ } ++#endif ++ + return set; + } + +@@ -613,6 +737,17 @@ AddWaitEventToSet(WaitEventSet *set, uint32 events, pgsocket fd, Latch *latch, + { + set->latch = latch; + set->latch_pos = event->pos; ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++ if (IsTrustedEmbeddedProcess()) ++ { ++ event->fd = PGINVALID_SOCKET; ++#ifdef WAIT_USE_POLL ++ set->pollfds[event->pos].fd = -1; ++ set->pollfds[event->pos].events = 0; ++#endif ++ return event->pos; ++ } ++#endif + #if defined(WAIT_USE_SELF_PIPE) + event->fd = selfpipe_readfd; + #elif defined(WAIT_USE_SIGNALFD) +@@ -1070,6 +1205,19 @@ WaitEventSetWait(WaitEventSet *set, long timeout, + while (returned_events == 0) + { + int rc; ++#ifdef OLIPHAUNT_EMBEDDED ++ bool trusted_embedded_wait = IsTrustedEmbeddedProcess(); ++ bool embedded_timeout_selected = false; ++ long block_timeout; ++ ++ /* ++ * This is a blocking boundary, so consume unconditionally. The hot ++ * CHECK_FOR_INTERRUPTS predicate remains a plain atomic hint, while a ++ * provider wake can never be lost to a stale speculative read here. ++ */ ++ if (trusted_embedded_wait) ++ ProcessTrustedEmbeddedInterrupts(); ++#endif + + /* + * Check if the latch is set already first. If so, we either exit +@@ -1136,7 +1284,29 @@ WaitEventSetWait(WaitEventSet *set, long timeout, + * this file. If -1 is returned, a timeout has occurred, if 0 we have + * to retry, everything >= 1 is the number of returned events. + */ ++ /* ++ * PostgreSQL's process timer is cooperative in a trusted embedded ++ * backend. Bound waits on the owning latch by its nearest deadline. ++ */ ++#ifdef OLIPHAUNT_EMBEDDED ++ /* cur_timeout may have been reduced to zero by an already-set latch. */ ++ block_timeout = cur_timeout; ++ if (trusted_embedded_wait) ++ { ++ long embedded_timeout = ++ GetEmbeddedTimeoutDelayMilliseconds(); ++ ++ if (embedded_timeout >= 0 && ++ (block_timeout < 0 || embedded_timeout <= block_timeout)) ++ { ++ block_timeout = embedded_timeout; ++ embedded_timeout_selected = true; ++ } ++ } ++ rc = WaitEventSetWaitBlock(set, block_timeout, ++#else + rc = WaitEventSetWaitBlock(set, cur_timeout, ++#endif + occurred_events, nevents - returned_events); + + if (set->latch && +@@ -1144,7 +1314,16 @@ WaitEventSetWait(WaitEventSet *set, long timeout, + set->latch->maybe_sleeping = false; + + if (rc == -1) +- break; /* timeout occurred */ ++ { ++#ifdef OLIPHAUNT_EMBEDDED ++ if (embedded_timeout_selected) ++ { ++ /* PostgreSQL's deadline wins ties with a caller timeout. */ ++ ProcessEmbeddedTimeouts(); ++ continue; ++ } ++#endif ++ } + else + returned_events += rc; + +@@ -1157,6 +1336,9 @@ WaitEventSetWait(WaitEventSet *set, long timeout, + if (cur_timeout <= 0) + break; + } ++ ++ if (rc == -1) ++ break; /* caller timeout occurred */ + } + #ifndef WIN32 + waiting = false; +@@ -1221,6 +1403,13 @@ WaitEventSetWaitBlock(WaitEventSet *set, int cur_timeout, + returned_events < nevents; + cur_epoll_event++) + { ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++ if (cur_epoll_event->data.ptr == NULL) ++ { ++ DrainTrustedEmbeddedWakeup(); ++ continue; ++ } ++#endif + /* epoll's data pointer is set to the associated WaitEvent */ + cur_event = (WaitEvent *) cur_epoll_event->data.ptr; + +@@ -1383,6 +1572,13 @@ WaitEventSetWaitBlock(WaitEventSet *set, int cur_timeout, + returned_events < nevents; + cur_kqueue_event++) + { ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++ if (AccessWaitEvent(cur_kqueue_event) == NULL) ++ { ++ DrainTrustedEmbeddedWakeup(); ++ continue; ++ } ++#endif + /* kevent's udata points to the associated WaitEvent */ + cur_event = AccessWaitEvent(cur_kqueue_event); + +@@ -1473,11 +1669,24 @@ WaitEventSetWaitBlock(WaitEventSet *set, int cur_timeout, + { + int returned_events = 0; + int rc; ++ int poll_nevents = set->nevents; + WaitEvent *cur_event; + struct pollfd *cur_pollfd; + ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++ if (IsTrustedEmbeddedProcess()) ++ { ++ struct pollfd *wakeup_pollfd = &set->pollfds[set->nevents]; ++ ++ wakeup_pollfd->fd = trusted_embedded_wakeup_readfd; ++ wakeup_pollfd->events = POLLIN; ++ wakeup_pollfd->revents = 0; ++ poll_nevents++; ++ } ++#endif ++ + /* Sleep */ +- rc = poll(set->pollfds, set->nevents, (int) cur_timeout); ++ rc = poll(set->pollfds, poll_nevents, (int) cur_timeout); + + /* Check return code */ + if (rc < 0) +@@ -1499,6 +1708,12 @@ WaitEventSetWaitBlock(WaitEventSet *set, int cur_timeout, + return -1; + } + ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(WIN32) ++ if (IsTrustedEmbeddedProcess() && ++ set->pollfds[set->nevents].revents != 0) ++ DrainTrustedEmbeddedWakeup(); ++#endif ++ + for (cur_event = set->events, cur_pollfd = set->pollfds; + cur_event < (set->events + set->nevents) && + returned_events < nevents; +@@ -2019,6 +2234,11 @@ ResOwnerReleaseWaitEventSet(Datum res) + void + WakeupMyProc(void) + { ++#ifdef OLIPHAUNT_EMBEDDED ++ /* Direct host threads use the separately-published raw wake endpoint. */ ++ if (IsTrustedEmbeddedProcess()) ++ return; ++#endif + #if defined(WAIT_USE_SELF_PIPE) + if (waiting) + sendSelfPipeByte(); +@@ -2035,3 +2255,55 @@ WakeupOtherProc(int pid) + kill(pid, SIGURG); + } + #endif ++ ++#ifdef OLIPHAUNT_EMBEDDED ++bool ++PublishTrustedEmbeddedWakeup(OliphauntEmbeddedIO *io) ++{ ++ uint32 kind; ++ uintptr_t token; ++ ++ if (!IsTrustedEmbeddedProcess() || io == NULL || ++ io->set_interrupt_wakeup == NULL) ++ return false; ++#ifdef WIN32 ++ kind = OLIPHAUNT_EMBEDDED_WAKE_WIN32_EVENT; ++ token = (uintptr_t) pgwin32_signal_event; ++#else ++ if (trusted_embedded_wakeup_writefd < 0) ++ return false; ++ kind = OLIPHAUNT_EMBEDDED_WAKE_POSIX_FD; ++ token = (uintptr_t) trusted_embedded_wakeup_writefd; ++#endif ++ return io->set_interrupt_wakeup(io->context, kind, token) == 0; ++} ++ ++bool ++UnpublishTrustedEmbeddedWakeup(OliphauntEmbeddedIO *io) ++{ ++ return io != NULL && io->set_interrupt_wakeup != NULL && ++ io->set_interrupt_wakeup(io->context, ++ OLIPHAUNT_EMBEDDED_WAKE_NONE, 0) == 0; ++} ++ ++void ++ShutdownTrustedEmbeddedWakeup(void) ++{ ++#ifdef WIN32 ++ pgwin32_signal_shutdown_embedded(); ++#else ++ if (trusted_embedded_wakeup_readfd >= 0) ++ (void) close(trusted_embedded_wakeup_readfd); ++ if (trusted_embedded_wakeup_writefd >= 0) ++ (void) close(trusted_embedded_wakeup_writefd); ++ trusted_embedded_wakeup_readfd = -1; ++ trusted_embedded_wakeup_writefd = -1; ++ if (trusted_embedded_wakeup_reserved) ++ { ++ ReleaseExternalFD(); ++ ReleaseExternalFD(); ++ trusted_embedded_wakeup_reserved = false; ++ } ++#endif ++} ++#endif +diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c +index fd020198f3b6f7ebd122f757ed286cdac617ced1..b9718e038c89874b33ca517b10b07820224b2a5c 100644 +--- a/src/backend/tcop/postgres.c ++++ b/src/backend/tcop/postgres.c +@@ -4234,6 +4234,8 @@ typedef struct OliphauntEmbeddedLifecycle + bool cwd_restore_required; + bool proc_exit_handler_armed; + bool entrypoint_admitted; ++ bool wakeup_published; ++ OliphauntEmbeddedIO *io; + } OliphauntEmbeddedLifecycle; + + static void +@@ -4388,8 +4390,12 @@ oliphaunt_embedded_main(int argc, char *argv[], + int socket_error; + + if (argv == NULL || argv[0] == NULL || dbname == NULL || +- username == NULL || io == NULL || cwd_capture_errno == NULL) ++ username == NULL || io == NULL || cwd_capture_errno == NULL || ++ io->abi_version != OLIPHAUNT_EMBEDDED_IO_ABI_VERSION || ++ io->struct_size < sizeof(*io) || io->context == NULL || ++ io->read == NULL || io->write == NULL || io->set_timeout == NULL || ++ io->set_interrupt_wakeup == NULL) + return -1; + + *cwd_capture_errno = 0; + lifecycle = calloc(1, sizeof(*lifecycle)); +@@ -4396,6 +4402,7 @@ oliphaunt_embedded_main(int argc, char *argv[], + return -1; + lifecycle->socket_pair[0] = PGINVALID_SOCKET; + lifecycle->socket_pair[1] = PGINVALID_SOCKET; ++ lifecycle->io = io; + #ifndef WIN32 + lifecycle->original_cwd_fd = -1; + #endif +@@ -4463,6 +4470,10 @@ oliphaunt_embedded_main(int argc, char *argv[], + set_pglocale_pgservice(argv[0], PG_TEXTDOMAIN("postgres")); + + InitStandaloneProcess(argv[0]); ++ if (!PublishTrustedEmbeddedWakeup(io)) ++ ereport(FATAL, ++ (errmsg("could not publish trusted embedded wakeup endpoint: %m"))); ++ lifecycle->wakeup_published = true; + InitializeGUCOptions(); + process_postgres_switches(argc, argv, PGC_POSTMASTER, &switch_dbname); + +@@ -4542,10 +4553,16 @@ oliphaunt_embedded_main(int argc, char *argv[], + } + + PostgresMain(dbname, username, INIT_PG_TRUSTED_CLIENT); +- oliphaunt_embedded_proc_exit(0); +- lifecycle->rc = 0; + + embedded_cleanup: ++ if (lifecycle->wakeup_published) ++ { ++ if (!UnpublishTrustedEmbeddedWakeup(lifecycle->io)) ++ lifecycle->rc = lifecycle->rc == 0 ? -1 : lifecycle->rc; ++ else ++ lifecycle->wakeup_published = false; ++ } ++ + if (lifecycle->proc_exit_handler_armed && + !oliphaunt_embedded_clear_proc_exit_handler( + oliphaunt_embedded_throw_proc_exit, lifecycle)) +@@ -4571,6 +4588,11 @@ embedded_cleanup: + FreeWaitEventSet(FeBeWaitSet); + FeBeWaitSet = NULL; + } ++ if (lifecycle->entrypoint_admitted) ++ ShutdownTrustedEmbeddedLatchSupport(); ++ /* Never close a token that the host failed to forget under its mutex. */ ++ if (!lifecycle->wakeup_published) ++ ShutdownTrustedEmbeddedWakeup(); + + if (lifecycle->socket_pair[1] != PGINVALID_SOCKET) + { +@@ -4674,12 +4696,7 @@ PostgresMain(const char *dbname, const char *username, + * midst of output during who-knows-what operation... + */ + pqsignal(SIGPIPE, SIG_IGN); +-#ifndef OLIPHAUNT_EMBEDDED + pqsignal(SIGUSR1, procsignal_sigusr1_handler); +-#else +- /* ProcSignal delivery is synchronous and the host owns SIGUSR1. */ +- Assert(!IsUnderPostmaster); +-#endif + pqsignal(SIGUSR2, SIG_IGN); + pqsignal(SIGFPE, FloatExceptionHandler); + +@@ -4695,7 +4712,12 @@ PostgresMain(const char *dbname, const char *username, + BaseInit(); + + /* We need to allow SIGINT, etc during the initial transaction */ +- sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ (void) SetTrustedEmbeddedThreadSignalMask(&UnBlockSig); ++ else ++#endif ++ sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); + + /* + * Generate a random cancel key, if this is a backend serving a +@@ -5438,15 +5460,10 @@ PostgresMain(const char *dbname, const char *username, + * Whatever you had in mind to do should be set up as an + * on_proc_exit or on_shmem_exit callback, instead. Otherwise + * it will fail to be called during other backend-shutdown +- * scenarios. Embedded liboliphaunt owns the backend thread and +- * returns here so the host can join it without terminating the +- * process. ++ * scenarios. Trusted embedded mode intercepts proc_exit on the ++ * owning backend thread after all ordinary callbacks have run. + */ +-#ifdef OLIPHAUNT_EMBEDDED +- return; +-#else + proc_exit(0); +-#endif + + case PqMsg_CopyData: + case PqMsg_CopyDone: +diff --git a/src/backend/utils/init/embedded_session.c b/src/backend/utils/init/embedded_session.c +index d5060947bed0896e777355ee05377e2c672f3d35..25b215f33a1cd89b77355ebbb36e54855b5896f0 100644 +--- a/src/backend/utils/init/embedded_session.c ++++ b/src/backend/utils/init/embedded_session.c +@@ -20,7 +20,11 @@ + #include "port/atomics.h" + #include "replication/walsender.h" + #include "storage/aio.h" ++#include "storage/latch.h" ++#include "storage/waiteventset.h" ++#include "tcop/tcopprot.h" + #include "utils/guc.h" ++#include "utils/timeout.h" + + typedef enum TrustedEmbeddedSessionLifecycle + { +@@ -34,17 +38,67 @@ typedef enum TrustedEmbeddedSessionLifecycle + * deliberately not an active user session: PostgreSQL startup still owns the + * decision to attach the trusted client at the catalog-safe point. + */ +-static pg_atomic_uint32 trusted_embedded_session_lifecycle = {0}; ++pg_atomic_uint32 TrustedEmbeddedLifecycleState = {0}; ++pg_atomic_uint32 TrustedEmbeddedInterruptRequests = {0}; + + bool + PrepareTrustedEmbeddedSession(void) + { + uint32 expected = TRUSTED_EMBEDDED_SESSION_UNSELECTED; + +- return pg_atomic_compare_exchange_u32( +- &trusted_embedded_session_lifecycle, +- &expected, +- TRUSTED_EMBEDDED_SESSION_PREPARED); ++ if (!pg_atomic_compare_exchange_u32( ++ &TrustedEmbeddedLifecycleState, ++ &expected, ++ TRUSTED_EMBEDDED_SESSION_PREPARED)) ++ return false; ++ pg_atomic_write_u32(&TrustedEmbeddedInterruptRequests, 0); ++ return true; ++} ++ ++/* ++ * Host-thread entrypoint. Do not touch PostgreSQL's signal-style globals or ++ * Latch fields here: neither is a C thread-atomic interface. The atomic ++ * mailbox coalesces repeated requests and the provider wake only makes the ++ * backend thread runnable. ++ */ ++void ++RequestTrustedEmbeddedQueryCancel(void) ++{ ++ pg_atomic_fetch_or_u32(&TrustedEmbeddedInterruptRequests, ++ TRUSTED_EMBEDDED_INTERRUPT_CANCEL); ++} ++ ++/* Host-thread request to recheck PostgreSQL's current cooperative deadline. */ ++void ++RequestTrustedEmbeddedTimeoutCheck(void) ++{ ++ pg_atomic_fetch_or_u32(&TrustedEmbeddedInterruptRequests, ++ TRUSTED_EMBEDDED_INTERRUPT_TIMEOUT_RECHECK); ++} ++ ++bool ++TrustedEmbeddedInterruptPending(void) ++{ ++ return IsTrustedEmbeddedProcess() && ++ pg_atomic_read_u32(&TrustedEmbeddedInterruptRequests) != 0; ++} ++ ++/* ++ * Backend-thread interrupt ingress. This is deliberately safe before ++ * attachment so cancellation during startup retains the previous public ABI ++ * behavior. Cooperative timeout handlers run on this same owning thread. ++ */ ++void ++ProcessTrustedEmbeddedInterrupts(void) ++{ ++ uint32 requests = pg_atomic_exchange_u32( ++ &TrustedEmbeddedInterruptRequests, 0); ++ ++ if ((requests & TRUSTED_EMBEDDED_INTERRUPT_CANCEL) != 0) ++ StatementCancelHandler(SIGINT); ++ ++ if ((requests & TRUSTED_EMBEDDED_INTERRUPT_TIMEOUT_RECHECK) != 0) ++ ProcessEmbeddedTimeouts(); + } + + static bool +@@ -61,7 +115,15 @@ trusted_embedded_settings_are_safe(void) + void + ConfigurePreparedTrustedEmbeddedSession(void) + { +- if (pg_atomic_read_u32(&trusted_embedded_session_lifecycle) != ++ static const char *const unsupported_commands[] = { ++ "archive_command", ++ "restore_command", ++ "archive_cleanup_command", ++ "recovery_end_command", ++ }; ++ int command_index; ++ ++ if (pg_atomic_read_u32(&TrustedEmbeddedLifecycleState) != + TRUSTED_EMBEDDED_SESSION_PREPARED) + ereport(FATAL, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), +@@ -83,6 +145,20 @@ ConfigurePreparedTrustedEmbeddedSession(void) + PGC_S_OVERRIDE); + SetConfigOption("max_wal_senders", "0", PGC_POSTMASTER, PGC_S_OVERRIDE); + ++ for (command_index = 0; ++ command_index < lengthof(unsupported_commands); ++ command_index++) ++ { ++ const char *value = GetConfigOption(unsupported_commands[command_index], ++ true, false); ++ ++ if (value != NULL && value[0] != '\0') ++ ereport(FATAL, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("%s is not supported in a trusted embedded session", ++ unsupported_commands[command_index]))); ++ } ++ + if (!trusted_embedded_settings_are_safe()) + ereport(FATAL, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), +@@ -97,15 +173,22 @@ AttachPreparedTrustedEmbeddedSession(void) + if (!trusted_embedded_settings_are_safe()) + return false; + return pg_atomic_compare_exchange_u32( +- &trusted_embedded_session_lifecycle, ++ &TrustedEmbeddedLifecycleState, + &expected, + TRUSTED_EMBEDDED_SESSION_ATTACHED); + } + ++bool ++IsTrustedEmbeddedProcess(void) ++{ ++ return pg_atomic_read_u32(&TrustedEmbeddedLifecycleState) != ++ TRUSTED_EMBEDDED_SESSION_UNSELECTED; ++} ++ + bool + IsTrustedEmbeddedSession(void) + { +- return pg_atomic_read_u32(&trusted_embedded_session_lifecycle) == ++ return pg_atomic_read_u32(&TrustedEmbeddedLifecycleState) == + TRUSTED_EMBEDDED_SESSION_ATTACHED; + } + +diff --git a/src/backend/utils/init/miscinit.c b/src/backend/utils/init/miscinit.c +index 2000cc1487519196e6a8b7b1a22764938332325d..0132a33bf7c36d36821640262a3ebb2d07fd9016 100644 +--- a/src/backend/utils/init/miscinit.c ++++ b/src/backend/utils/init/miscinit.c +@@ -40,6 +40,9 @@ + #include "postmaster/autovacuum.h" + #include "postmaster/interrupt.h" + #include "postmaster/postmaster.h" ++#ifndef WIN32 ++#include "port/pg_pthread.h" ++#endif + #include "replication/slotsync.h" + #include "storage/fd.h" + #include "storage/ipc.h" +@@ -68,6 +71,50 @@ static List *lock_files = NIL; + + static Latch LocalLatchData; + ++#ifdef OLIPHAUNT_EMBEDDED ++/* ++ * Preserve the mask inherited from the embedding thread. PostgreSQL later ++ * restores UnBlockSig after startup and transaction error recovery, so all ++ * three masks must describe the host state rather than PostgreSQL's normal ++ * standalone signal policy. ++ */ ++static void ++InitializeTrustedEmbeddedSignalMasks(void) ++{ ++ int rc; ++ ++#ifdef WIN32 ++ rc = sigprocmask(SIG_SETMASK, NULL, &UnBlockSig); ++#else ++ rc = pthread_sigmask(SIG_SETMASK, NULL, &UnBlockSig); ++#endif ++ if (rc != 0) ++ { ++ errno = rc; ++ elog(FATAL, "could not read trusted embedded thread signal mask: %m"); ++ } ++ memcpy(&BlockSig, &UnBlockSig, sizeof(sigset_t)); ++ memcpy(&StartupBlockSig, &UnBlockSig, sizeof(sigset_t)); ++} ++ ++/* ++ * sigprocmask() is unspecified once the host process is multithreaded. Keep ++ * PostgreSQL's normal process-oriented path unchanged, but use the POSIX ++ * thread API whenever Direct restores its inherited embedding-thread mask. ++ */ ++int ++SetTrustedEmbeddedThreadSignalMask(const sigset_t *mask) ++{ ++ Assert(IsTrustedEmbeddedProcess()); ++ ++#ifdef WIN32 ++ return sigprocmask(SIG_SETMASK, mask, NULL); ++#else ++ return pthread_sigmask(SIG_SETMASK, mask, NULL); ++#endif ++} ++#endif ++ + /* ---------------------------------------------------------------- + * ignoring system indexes support stuff + * +@@ -182,7 +229,12 @@ InitStandaloneProcess(const char *argv0) + * Start our win32 signal implementation + */ + #ifdef WIN32 +- pgwin32_signal_initialize(); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ pgwin32_signal_initialize_embedded(); ++ else ++#endif ++ pgwin32_signal_initialize(); + #endif + + InitProcessGlobals(); +@@ -196,8 +248,15 @@ InitStandaloneProcess(const char *argv0) + * For consistency with InitPostmasterChild, initialize signal mask here. + * But we don't unblock SIGQUIT or provide a default handler for it. + */ +- pqinitmask(); +- sigprocmask(SIG_SETMASK, &BlockSig, NULL); ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ InitializeTrustedEmbeddedSignalMasks(); ++ else ++#endif ++ { ++ pqinitmask(); ++ sigprocmask(SIG_SETMASK, &BlockSig, NULL); ++ } + + /* Compute paths, no postmaster to inherit from */ + if (my_exec_path[0] == '\0') +@@ -238,6 +297,25 @@ InitProcessLocalLatch(void) + InitLatch(MyLatch); + } + ++#ifdef OLIPHAUNT_EMBEDDED ++/* Release process-lifetime latch state before Direct returns to its host. */ ++void ++ShutdownTrustedEmbeddedLatchSupport(void) ++{ ++ Assert(IsTrustedEmbeddedProcess()); ++ ++ ShutdownTrustedEmbeddedLatchWaitSet(); ++#ifdef WIN32 ++ if (LocalLatchData.event != NULL) ++ { ++ (void) CloseHandle(LocalLatchData.event); ++ LocalLatchData.event = NULL; ++ } ++#endif ++ MyLatch = NULL; ++} ++#endif ++ + void + SwitchBackToLocalLatch(void) + { +diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c +index 497ce5e507bbf8ed18172d566c9c3733dc41868b..5a2fe0adf18c8140a5775f3c14c018c0004a1c68 100644 +--- a/src/backend/utils/init/postinit.c ++++ b/src/backend/utils/init/postinit.c +@@ -1390,6 +1390,16 @@ ShutdownPostgres(int code, Datum arg) + static void + StatementTimeoutHandler(void) + { ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ { ++ if (ClientAuthInProgress) ++ die(SIGTERM); ++ else ++ StatementCancelHandler(SIGINT); ++ return; ++ } ++#endif + int sig = SIGINT; + + /* +@@ -1412,6 +1422,13 @@ StatementTimeoutHandler(void) + static void + LockTimeoutHandler(void) + { ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ { ++ StatementCancelHandler(SIGINT); ++ return; ++ } ++#endif + #ifdef HAVE_SETSID + /* try to signal whole process group */ + kill(-MyProcPid, SIGINT); +diff --git a/src/backend/utils/misc/timeout.c b/src/backend/utils/misc/timeout.c +index d92efa12550ae09df03c50829d5f6428dcb4f6f8..17b7582a072904a2cec0d0305f9aaa9fddbf8e31 100644 +--- a/src/backend/utils/misc/timeout.c ++++ b/src/backend/utils/misc/timeout.c +@@ -16,6 +16,9 @@ + + #include + ++#ifdef OLIPHAUNT_EMBEDDED ++#include "libpq/libpq-be.h" ++#endif + #include "miscadmin.h" + #include "storage/latch.h" + #include "utils/timeout.h" +@@ -209,6 +212,46 @@ enable_timeout(TimeoutId id, TimestampTz now, TimestampTz fin_time, + static void + schedule_alarm(TimestampTz now) + { ++#ifdef OLIPHAUNT_EMBEDDED ++ if (IsTrustedEmbeddedProcess()) ++ { ++ long timeout_ms = -1; ++ OliphauntEmbeddedIO *io = ++ MyProcPort != NULL ? MyProcPort->oliphaunt_io : NULL; ++ ++ /* ++ * The host owns SIGALRM and ITIMER_REAL only for the selected Direct ++ * entrypoint. Retain PostgreSQL's ordered deadline queue, but publish ++ * the nearest relative deadline to its private host I/O provider. ++ */ ++ if (num_active_timeouts > 0) ++ { ++ TimestampTz nearest_timeout = active_timeouts[0]->fin_time; ++ ++ enable_alarm(); ++ /* Reuse an already-published wake that is early enough. */ ++ if (signal_pending && nearest_timeout >= signal_due_at) ++ return; ++ timeout_ms = TimestampDifferenceMilliseconds(now, nearest_timeout); ++ if (io == NULL || io->set_timeout == NULL || ++ io->set_timeout(io->context, timeout_ms) != 0) ++ elog(FATAL, "could not publish trusted embedded timeout: %m"); ++ signal_due_at = nearest_timeout; ++ signal_pending = true; ++ } ++ else ++ { ++ disable_alarm(); ++ /* Preserve upstream's lazy arm/disarm behavior. */ ++ if (signal_pending) ++ return; ++ if (io == NULL || io->set_timeout == NULL || ++ io->set_timeout(io->context, -1) != 0) ++ elog(FATAL, "could not disarm trusted embedded timeout: %m"); ++ } ++ return; ++ } ++#endif + if (num_active_timeouts > 0) + { + struct itimerval timeval; +@@ -350,6 +393,42 @@ schedule_alarm(TimestampTz now) + } + + ++/* Process every timeout due at or before now, preserving priority order. */ ++static void ++process_expired_timeouts(TimestampTz now) ++{ ++ while (num_active_timeouts > 0 && ++ now >= active_timeouts[0]->fin_time) ++ { ++ timeout_params *this_timeout = active_timeouts[0]; ++ ++ remove_timeout_index(0); ++ this_timeout->indicator = true; ++ this_timeout->timeout_handler(); ++ ++ if (this_timeout->interval_in_ms > 0) ++ { ++ TimestampTz new_fin_time; ++ ++ new_fin_time = ++ TimestampTzPlusMilliseconds(this_timeout->fin_time, ++ this_timeout->interval_in_ms); ++ if (new_fin_time < now) ++ new_fin_time = ++ TimestampTzPlusMilliseconds(now, ++ this_timeout->interval_in_ms); ++ enable_timeout(this_timeout->index, now, new_fin_time, ++ this_timeout->interval_in_ms); ++ } ++ ++ /* A timeout handler such as deadlock detection need not be cheap. */ ++ now = GetCurrentTimestamp(); ++ } ++ ++ schedule_alarm(now); ++} ++ ++ + /***************************************************************************** + * Signal handler + *****************************************************************************/ +@@ -395,58 +474,7 @@ handle_sig_alarm(SIGNAL_ARGS) + disable_alarm(); + + if (num_active_timeouts > 0) +- { +- TimestampTz now = GetCurrentTimestamp(); +- +- /* While the first pending timeout has been reached ... */ +- while (num_active_timeouts > 0 && +- now >= active_timeouts[0]->fin_time) +- { +- timeout_params *this_timeout = active_timeouts[0]; +- +- /* Remove it from the active list */ +- remove_timeout_index(0); +- +- /* Mark it as fired */ +- this_timeout->indicator = true; +- +- /* And call its handler function */ +- this_timeout->timeout_handler(); +- +- /* If it should fire repeatedly, re-enable it. */ +- if (this_timeout->interval_in_ms > 0) +- { +- TimestampTz new_fin_time; +- +- /* +- * To guard against drift, schedule the next instance of +- * the timeout based on the intended firing time rather +- * than the actual firing time. But if the timeout was so +- * late that we missed an entire cycle, fall back to +- * scheduling based on the actual firing time. +- */ +- new_fin_time = +- TimestampTzPlusMilliseconds(this_timeout->fin_time, +- this_timeout->interval_in_ms); +- if (new_fin_time < now) +- new_fin_time = +- TimestampTzPlusMilliseconds(now, +- this_timeout->interval_in_ms); +- enable_timeout(this_timeout->index, now, new_fin_time, +- this_timeout->interval_in_ms); +- } +- +- /* +- * The handler might not take negligible time (CheckDeadLock +- * for instance isn't too cheap), so let's update our idea of +- * "now" after each one. +- */ +- now = GetCurrentTimestamp(); +- } +- +- /* Done firing timeouts, so reschedule next interrupt if any */ +- schedule_alarm(now); +- } ++ process_expired_timeouts(GetCurrentTimestamp()); + } + + RESUME_INTERRUPTS(); +@@ -489,9 +517,67 @@ InitializeTimeouts(void) + + all_timeouts_initialized = true; + +- /* Now establish the signal handler */ +- pqsignal(SIGALRM, handle_sig_alarm); ++ /* Only the selected Direct entrypoint leaves the host's SIGALRM untouched. */ ++#ifdef OLIPHAUNT_EMBEDDED ++ if (!IsTrustedEmbeddedProcess()) ++#endif ++ pqsignal(SIGALRM, handle_sig_alarm); ++} ++ ++#ifdef OLIPHAUNT_EMBEDDED ++/* ++ * Return the rounded-up delay to the nearest PostgreSQL timeout, or -1 if ++ * there is none. This is suitable for WaitLatch and host condition waits. ++ */ ++long ++GetEmbeddedTimeoutDelayMilliseconds(void) ++{ ++ if (!IsTrustedEmbeddedProcess() || !all_timeouts_initialized || !alarm_enabled || ++ num_active_timeouts == 0) ++ return -1; ++ ++ return TimestampDifferenceMilliseconds(GetCurrentTimestamp(), ++ active_timeouts[0]->fin_time); ++} ++ ++/* Fire due timeouts synchronously on the backend's owning thread. */ ++void ++ProcessEmbeddedTimeouts(void) ++{ ++ TimestampTz now; ++ int save_errno; ++ ++ if (!IsTrustedEmbeddedProcess() || !all_timeouts_initialized) ++ return; ++ ++ save_errno = errno; ++ now = GetCurrentTimestamp(); ++ /* The provider wake is one-shot, even when its deadline became stale. */ ++ signal_pending = false; ++ if (!alarm_enabled || num_active_timeouts == 0) ++ { ++ schedule_alarm(now); ++ errno = save_errno; ++ return; ++ } ++ if (now < active_timeouts[0]->fin_time) ++ { ++ /* A rounded or stale host expiry must republish the current deadline. */ ++ schedule_alarm(now); ++ errno = save_errno; ++ return; ++ } ++ ++ HOLD_INTERRUPTS(); ++ if (MyLatch != NULL) ++ SetLatch(MyLatch); ++ signal_pending = false; ++ disable_alarm(); ++ process_expired_timeouts(now); ++ RESUME_INTERRUPTS(); ++ errno = save_errno; + } ++#endif + + /* + * Register a timeout reason +diff --git a/src/include/libpq/libpq-be.h b/src/include/libpq/libpq-be.h +index e415b277a54aa29e190b1146218e12e89c591561..dcbdfa165b494fe940793cbdb44e03a7126cc43d 100644 +--- a/src/include/libpq/libpq-be.h ++++ b/src/include/libpq/libpq-be.h +@@ -106,11 +106,23 @@ typedef struct ClientConnectionInfo + } ClientConnectionInfo; + + #ifdef OLIPHAUNT_EMBEDDED ++#define OLIPHAUNT_EMBEDDED_IO_ABI_VERSION 1U ++ ++#define OLIPHAUNT_EMBEDDED_WAKE_NONE 0U ++#define OLIPHAUNT_EMBEDDED_WAKE_POSIX_FD 1U ++#define OLIPHAUNT_EMBEDDED_WAKE_WIN32_EVENT 2U ++ + typedef struct OliphauntEmbeddedIO + { ++ uint32 abi_version; ++ uint32 struct_size; + void *context; +- ssize_t (*read) (void *context, void *ptr, size_t len); ++ ssize_t (*read) (void *context, void *ptr, size_t len, ++ long timeout_ms); + ssize_t (*write) (void *context, const void *ptr, size_t len); ++ int (*set_timeout) (void *context, long timeout_ms); ++ int (*set_interrupt_wakeup) (void *context, uint32 kind, ++ uintptr_t token); + } OliphauntEmbeddedIO; + #endif + +diff --git a/src/include/libpq/libpq.h b/src/include/libpq/libpq.h +index aeb66ca40cf38da9f78feb66a80521323e90753b..8a9fb0b880bcc170ab849ef40b52ce086a7621ff 100644 +--- a/src/include/libpq/libpq.h ++++ b/src/include/libpq/libpq.h +@@ -77,6 +77,10 @@ extern void pq_endmsgread(void); + extern bool pq_is_reading_msg(void); + extern int pq_getmessage(StringInfo s, int maxlen); + extern int pq_getbyte(void); ++#ifdef OLIPHAUNT_EMBEDDED ++#define PQ_READ_INTERRUPTED (-2) ++extern int pq_getbyte_interruptible(void); ++#endif + extern int pq_peekbyte(void); + extern int pq_getbyte_if_available(unsigned char *c); + extern ssize_t pq_buffer_remaining_data(void); +diff --git a/src/include/miscadmin.h b/src/include/miscadmin.h +index addc3469b0ef6a0dc8254dfa8c0433bc714f8956..34bc20224b4ad1c8cff8007dfe29163ebc3589e0 100644 +--- a/src/include/miscadmin.h ++++ b/src/include/miscadmin.h +@@ -27,6 +27,9 @@ + + #include "datatype/timestamp.h" /* for TimestampTz */ + #include "pgtime.h" /* for pg_time_t */ ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(FRONTEND) ++#include "port/atomics.h" ++#endif + + + #define InvalidPid (-1) +@@ -109,7 +112,30 @@ extern PGDLLIMPORT volatile uint32 CritSectionCount; + extern void ProcessInterrupts(void); + + /* Test whether an interrupt is pending */ +-#ifndef WIN32 ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(FRONTEND) ++#define TRUSTED_EMBEDDED_INTERRUPT_CANCEL (1U << 0) ++#define TRUSTED_EMBEDDED_INTERRUPT_TIMEOUT_RECHECK (1U << 1) ++extern PGDLLIMPORT pg_atomic_uint32 TrustedEmbeddedLifecycleState; ++extern PGDLLIMPORT pg_atomic_uint32 TrustedEmbeddedInterruptRequests; ++ ++#define TRUSTED_EMBEDDED_PROCESS_SELECTED() \ ++ (unlikely(pg_atomic_read_u32( \ ++ &TrustedEmbeddedLifecycleState) != 0)) ++#define TRUSTED_EMBEDDED_INTERRUPT_PENDING() \ ++ (unlikely(pg_atomic_read_u32( \ ++ &TrustedEmbeddedInterruptRequests) != 0)) ++#ifdef WIN32 ++#define INTERRUPTS_PENDING_CONDITION() \ ++ (TRUSTED_EMBEDDED_PROCESS_SELECTED() ? \ ++ (unlikely(InterruptPending) || TRUSTED_EMBEDDED_INTERRUPT_PENDING()) : \ ++ (unlikely(UNBLOCKED_SIGNAL_QUEUE()) ? \ ++ pgwin32_dispatch_queued_signals() : (void) 0, \ ++ unlikely(InterruptPending))) ++#else ++#define INTERRUPTS_PENDING_CONDITION() \ ++ (unlikely(InterruptPending) || TRUSTED_EMBEDDED_INTERRUPT_PENDING()) ++#endif ++#elif !defined(WIN32) + #define INTERRUPTS_PENDING_CONDITION() \ + (unlikely(InterruptPending)) + #else +@@ -120,11 +146,24 @@ extern void ProcessInterrupts(void); + #endif + + /* Service interrupt, if one is pending and it's safe to service it now */ ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(FRONTEND) ++#define CHECK_FOR_INTERRUPTS() \ ++do { \ ++ if (INTERRUPTS_PENDING_CONDITION()) \ ++ { \ ++ if (IsTrustedEmbeddedProcess()) \ ++ ProcessTrustedEmbeddedInterrupts(); \ ++ if (unlikely(InterruptPending)) \ ++ ProcessInterrupts(); \ ++ } \ ++} while(0) ++#else + #define CHECK_FOR_INTERRUPTS() \ + do { \ + if (INTERRUPTS_PENDING_CONDITION()) \ + ProcessInterrupts(); \ + } while(0) ++#endif + + /* Is ProcessInterrupts() guaranteed to clear InterruptPending? */ + #define INTERRUPTS_CAN_BE_PROCESSED() \ +@@ -168,12 +207,23 @@ extern PGDLLIMPORT bool IsPostmasterEnvironment; + extern PGDLLIMPORT bool IsUnderPostmaster; + extern PGDLLIMPORT bool IsBinaryUpgrade; + +-#ifdef OLIPHAUNT_EMBEDDED ++#if defined(OLIPHAUNT_EMBEDDED) && !defined(FRONTEND) + extern bool PrepareTrustedEmbeddedSession(void); + extern void ConfigurePreparedTrustedEmbeddedSession(void); + extern bool AttachPreparedTrustedEmbeddedSession(void); ++extern bool IsTrustedEmbeddedProcess(void); + extern bool IsTrustedEmbeddedSession(void); ++extern void RequestTrustedEmbeddedQueryCancel(void); ++extern void RequestTrustedEmbeddedTimeoutCheck(void); ++extern bool TrustedEmbeddedInterruptPending(void); ++extern void ProcessTrustedEmbeddedInterrupts(void); + #else ++static inline bool ++IsTrustedEmbeddedProcess(void) ++{ ++ return false; ++} ++ + static inline bool + IsTrustedEmbeddedSession(void) + { +@@ -351,6 +401,9 @@ extern void InitStandaloneProcess(const char *argv0); + extern void InitProcessLocalLatch(void); + extern void SwitchToSharedLatch(void); + extern void SwitchBackToLocalLatch(void); ++#ifdef OLIPHAUNT_EMBEDDED ++extern void ShutdownTrustedEmbeddedLatchSupport(void); ++#endif + + /* + * MyBackendType indicates what kind of a backend this is. +diff --git a/src/include/port/win32_port.h b/src/include/port/win32_port.h +index 14f97815f810929b7481b06393bf0d0c23479837..3f6892ddd658ddf04e739254c1c97779e367683f 100644 +--- a/src/include/port/win32_port.h ++++ b/src/include/port/win32_port.h +@@ -482,6 +482,8 @@ extern PGDLLIMPORT HANDLE pgwin32_initial_signal_pipe; + #define PG_SIGNAL_COUNT 32 + + extern void pgwin32_signal_initialize(void); ++extern void pgwin32_signal_initialize_embedded(void); ++extern void pgwin32_signal_shutdown_embedded(void); + extern HANDLE pgwin32_create_signal_listener(pid_t pid); + extern void pgwin32_dispatch_queued_signals(void); + extern void pg_queue_signal(int signum); +diff --git a/src/include/storage/ipc.h b/src/include/storage/ipc.h +index 395a73a10d7fa06948ade74b7f0b81990bbe9391..8fc338bb5235df4f07904b728570a6ae05604fa0 100644 +--- a/src/include/storage/ipc.h ++++ b/src/include/storage/ipc.h +@@ -69,7 +69,6 @@ pg_noreturn extern void proc_exit(int code); + extern void shmem_exit(int code); + #ifdef OLIPHAUNT_EMBEDDED + typedef void (*oliphaunt_embedded_proc_exit_handler) (int code, void *context); +-extern void oliphaunt_embedded_proc_exit(int code); + extern bool oliphaunt_embedded_install_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, + void *context); + extern bool oliphaunt_embedded_clear_proc_exit_handler(oliphaunt_embedded_proc_exit_handler handler, +diff --git a/src/include/storage/latch.h b/src/include/storage/latch.h +index e41dc70785afe78ea5effd7509d848e5106d5041..3c13e7b3fc35d2acbd3973a0796a206a93e1e3f6 100644 +--- a/src/include/storage/latch.h ++++ b/src/include/storage/latch.h +@@ -134,7 +134,10 @@ extern void ResetLatch(Latch *latch); + extern int WaitLatch(Latch *latch, int wakeEvents, long timeout, + uint32 wait_event_info); + extern int WaitLatchOrSocket(Latch *latch, int wakeEvents, +- pgsocket sock, long timeout, uint32 wait_event_info); ++ pgsocket sock, long timeout, uint32 wait_event_info); + extern void InitializeLatchWaitSet(void); ++#ifdef OLIPHAUNT_EMBEDDED ++extern void ShutdownTrustedEmbeddedLatchWaitSet(void); ++#endif + + #endif /* LATCH_H */ +diff --git a/src/include/storage/waiteventset.h b/src/include/storage/waiteventset.h +index dd514d5299104e800ca09ea8f1b2cecc15a77671..2e62c7ec3473e1fe85d5b070cfc9978449a0cf28 100644 +--- a/src/include/storage/waiteventset.h ++++ b/src/include/storage/waiteventset.h +@@ -71,6 +71,9 @@ typedef struct WaitEvent + typedef struct WaitEventSet WaitEventSet; + + struct Latch; ++#ifdef OLIPHAUNT_EMBEDDED ++struct OliphauntEmbeddedIO; ++#endif + + /* + * prototypes for functions in waiteventset.c +@@ -94,5 +97,10 @@ extern bool WaitEventSetCanReportClosed(void); + extern void WakeupMyProc(void); + extern void WakeupOtherProc(int pid); + #endif ++#ifdef OLIPHAUNT_EMBEDDED ++extern bool PublishTrustedEmbeddedWakeup(struct OliphauntEmbeddedIO *io); ++extern bool UnpublishTrustedEmbeddedWakeup(struct OliphauntEmbeddedIO *io); ++extern void ShutdownTrustedEmbeddedWakeup(void); ++#endif + + #endif /* WAITEVENTSET_H */ +diff --git a/src/include/tcop/tcopprot.h b/src/include/tcop/tcopprot.h +index 69164e11425bcde5388e22d1434ba576eb421da0..92c5eecfb90334fe29ceeb03c786294c4af64d4d 100644 +--- a/src/include/tcop/tcopprot.h ++++ b/src/include/tcop/tcopprot.h +@@ -81,20 +81,14 @@ extern void process_postgres_switches(int argc, char *argv[], + #ifdef OLIPHAUNT_EMBEDDED + typedef struct OliphauntEmbeddedIO OliphauntEmbeddedIO; + extern int oliphaunt_embedded_main(int argc, char *argv[], +- const char *dbname, const char *username, +- OliphauntEmbeddedIO *io, int *cwd_capture_errno); +-extern void PostgresSingleUserMain(int argc, char *argv[], +- const char *username); +-extern void PostgresMain(const char *dbname, +- const char *username, +- bits32 init_postgres_flags); +-#else ++ const char *dbname, const char *username, ++ OliphauntEmbeddedIO *io, int *cwd_capture_errno); ++#endif + pg_noreturn extern void PostgresSingleUserMain(int argc, char *argv[], + const char *username); + pg_noreturn extern void PostgresMain(const char *dbname, + const char *username, + bits32 init_postgres_flags); +-#endif + extern void ResetUsage(void); + extern void ShowUsage(const char *title); + extern int check_log_duration(char *msec_str, bool was_logged); +diff --git a/src/include/utils/timeout.h b/src/include/utils/timeout.h +index 7b19beafdc950488349d09ed0e906e31d61e696c..45719dcd2980184744bf8be897108ebdd69b501c 100644 +--- a/src/include/utils/timeout.h ++++ b/src/include/utils/timeout.h +@@ -87,6 +87,11 @@ extern void disable_timeout(TimeoutId id, bool keep_indicator); + extern void disable_timeouts(const DisableTimeoutParams *timeouts, int count); + extern void disable_all_timeouts(bool keep_indicators); + ++#ifdef OLIPHAUNT_EMBEDDED ++extern void ProcessEmbeddedTimeouts(void); ++extern long GetEmbeddedTimeoutDelayMilliseconds(void); ++#endif ++ + /* accessors */ + extern bool get_timeout_active(TimeoutId id); + extern bool get_timeout_indicator(TimeoutId id, bool reset_indicator); +diff --git a/src/port/pqsignal.c b/src/port/pqsignal.c +index 146d085906352e179a416d6eefa52300270d39aa..1aa8451f7050096c22ad776e63e1d40dc8d10cae 100644 +--- a/src/port/pqsignal.c ++++ b/src/port/pqsignal.c +@@ -73,13 +73,15 @@ static volatile pqsigfunc pqsignal_handlers[PG_NSIG]; + + #if defined(OLIPHAUNT_EMBEDDED) && !defined(FRONTEND) + /* +- * Embedded backends may use the ordinary signal APIs for non-host signals. +- * Reject SIGUSR1, and bypass the port.h macros when delegating everything else. ++ * A trusted embedded backend shares its process with the host application. ++ * Signal delivery is therefore a host-owned capability, not a PostgreSQL ++ * process-control primitive. Signal zero remains useful for liveness checks ++ * because it performs no delivery and changes no host process state. + */ + int + oliphaunt_embedded_kill(pid_t pid, int signo) + { +- if (signo == SIGUSR1) ++ if (IsTrustedEmbeddedProcess() && signo != 0) + { + errno = EPERM; + return -1; +@@ -95,8 +97,9 @@ oliphaunt_embedded_kill(pid_t pid, int signo) + int + oliphaunt_embedded_raise(int signo) + { +- if (signo == SIGUSR1) ++ if (IsTrustedEmbeddedProcess()) + { ++ (void) signo; + errno = EPERM; + return -1; + } +@@ -164,7 +167,8 @@ pqsignal(int signo, pqsigfunc func) + Assert(signo < PG_NSIG); + + #if defined(OLIPHAUNT_EMBEDDED) && !defined(FRONTEND) +- if (signo == SIGUSR1) ++ /* The embedding provider owns every process-wide signal disposition. */ ++ if (IsTrustedEmbeddedProcess()) + return; + #endif + +diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c +--- a/src/backend/commands/collationcmds.c ++++ b/src/backend/commands/collationcmds.c +@@ -858,6 +858,16 @@ pg_import_system_collations(PG_FUNCTION_ARGS) + /* Load collations known to libc, using "locale -a" to enumerate them */ + #ifdef READ_LOCALE_A_OUTPUT ++#ifdef OLIPHAUNT_EMBEDDED ++ if (!oliphaunt_skip_system_collation_discovery && IsTrustedEmbeddedProcess()) ++ { ++ /* No portable in-process API enumerates libc's installed locales. */ ++ ereport(NOTICE, ++ (errmsg("skipping libc collation discovery in a trusted embedded backend"), ++ errdetail("The locale enumeration program cannot run inside the host process. ICU collation discovery is unaffected."))); ++ oliphaunt_skip_system_collation_discovery = true; ++ } ++#endif + if (!oliphaunt_skip_system_collation_discovery) + { + FILE *locale_a_handle; + char localebuf[LOCALE_NAME_BUFLEN]; +diff --git a/src/include/libpq/pqsignal.h b/src/include/libpq/pqsignal.h +--- a/src/include/libpq/pqsignal.h ++++ b/src/include/libpq/pqsignal.h +@@ -51,4 +51,8 @@ extern PGDLLIMPORT sigset_t StartupBlockSig; + + extern void pqinitmask(void); + ++#ifdef OLIPHAUNT_EMBEDDED ++extern int SetTrustedEmbeddedThreadSignalMask(const sigset_t *mask); ++#endif ++ + #endif /* PQSIGNAL_H */ diff --git a/src/native/runtime/postgres/series b/src/native/runtime/postgres/series index 6b96afc4a..80623f43a 100644 --- a/src/native/runtime/postgres/series +++ b/src/native/runtime/postgres/series @@ -19,4 +19,5 @@ src/native/runtime/patches/postgresql-18.4/0017-liboliphaunt-namespace-dynahash- src/native/runtime/patches/postgresql-18.4/0018-liboliphaunt-contain-embedded-proc-signals.patch src/native/runtime/patches/postgresql-18.4/0019-liboliphaunt-link-windows-embedded-modules-to-host.patch src/native/runtime/patches/postgresql-18.4/0020-liboliphaunt-enforce-embedded-signal-boundary.patch -src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-wake-embedded-epoll-through-self-pipe.patch +src/native/runtime/patches/postgresql-18.4/0021-liboliphaunt-model-trusted-embedded-sessions.patch +src/native/runtime/patches/postgresql-18.4/0022-liboliphaunt-preserve-host-process-boundaries.patch diff --git a/src/native/runtime/smoke/liboliphaunt_cluster_seed_smoke.c b/src/native/runtime/smoke/liboliphaunt_cluster_seed_smoke.c index 4444d7915..f49c674f7 100644 --- a/src/native/runtime/smoke/liboliphaunt_cluster_seed_smoke.c +++ b/src/native/runtime/smoke/liboliphaunt_cluster_seed_smoke.c @@ -28,16 +28,19 @@ static int contains(const OliphauntResponse *response, const char *expected) { } int main(int argc, char **argv) { - if (argc != 5) { - fprintf(stderr, "usage: %s \n", argv[0]); + if (argc != 5 && argc != 6) { + fprintf(stderr, "usage: %s [startup-guc]\n", argv[0]); return 2; } + const char *startup_args[] = {"-c", argc == 6 ? argv[5] : NULL}; const OliphauntConfig config = { .abi_version = OLIPHAUNT_ABI_VERSION, .pgdata = argv[1], .runtime_dir = argv[2], .username = "postgres", .database = "postgres", + .startup_args = argc == 6 ? startup_args : NULL, + .startup_arg_count = argc == 6 ? 2 : 0, }; OliphauntHandle *database = NULL; if (oliphaunt_init(&config, &database) != 0 || database == NULL) { diff --git a/src/native/runtime/smoke/liboliphaunt_cwd_boundary.c b/src/native/runtime/smoke/liboliphaunt_cwd_boundary.c new file mode 100644 index 000000000..7aae1c240 --- /dev/null +++ b/src/native/runtime/smoke/liboliphaunt_cwd_boundary.c @@ -0,0 +1,620 @@ +#ifndef _WIN32 +#ifndef _POSIX_C_SOURCE +#define _POSIX_C_SOURCE 200809L +#endif +#ifndef _XOPEN_SOURCE +#define _XOPEN_SOURCE 700 +#endif +#endif + +#include "../include/oliphaunt.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#ifdef _WIN32 +#error "The retained cwd-boundary consumer probe currently targets Linux only." +#else +#include +#endif + +#ifndef PATH_MAX +#define PATH_MAX 4096 +#endif + +#if defined(_MSC_VER) +#define OLIPHAUNT_SMOKE_THREAD_LOCAL __declspec(thread) +#else +#define OLIPHAUNT_SMOKE_THREAD_LOCAL _Thread_local +#endif + +static OLIPHAUNT_SMOKE_THREAD_LOCAL char + last_error_buffer[OLIPHAUNT_ERROR_CAPTURE_CAPACITY]; + +static const char *last_error_message(OliphauntHandle *handle) { + (void)oliphaunt_copy_last_error( + handle, + last_error_buffer, + sizeof(last_error_buffer)); + return last_error_buffer; +} + +static const char direct_open_marker[] = "cwd-direct-open-relative.txt"; +static const char direct_detach_marker[] = "cwd-direct-detach-relative.txt"; +static const char fatal_marker[] = "cwd-startup-fatal-relative.txt"; +static const char early_fatal_marker[] = "cwd-early-startup-fatal-relative.txt"; +static const char renamed_close_marker[] = "cwd-renamed-close-relative.txt"; +static const char missing_database[] = "oliphaunt_cwd_probe_missing_database"; + +static int fail(const char *context) { + fprintf(stderr, "cwd boundary probe failed: %s\n", context); + return 1; +} + +static int fail_errno(const char *context) { + fprintf(stderr, "cwd boundary probe failed: %s: %s\n", context, strerror(errno)); + return 1; +} + +static int current_directory(char out[PATH_MAX]) { + if (getcwd(out, PATH_MAX) == NULL) { + return fail_errno("getcwd"); + } + return 0; +} + +static bool same_directory(const char *left, const char *right) { + char resolved_left[PATH_MAX]; + char resolved_right[PATH_MAX]; + return realpath(left, resolved_left) != NULL && + realpath(right, resolved_right) != NULL && + strcmp(resolved_left, resolved_right) == 0; +} + +static bool regular_file_exists(const char *path) { + struct stat metadata; + return stat(path, &metadata) == 0 && S_ISREG(metadata.st_mode); +} + +static bool same_file_identity(const struct stat *left, const struct stat *right) { + return left->st_dev == right->st_dev && left->st_ino == right->st_ino; +} + +static int joined_path(char out[PATH_MAX], const char *directory, const char *leaf) { + int written = snprintf(out, PATH_MAX, "%s/%s", directory, leaf); + if (written < 0 || written >= PATH_MAX) { + return fail("joined path exceeds PATH_MAX"); + } + return 0; +} + +static int write_relative_marker(const char *leaf, const char *contents) { + FILE *file = fopen(leaf, "wb"); + if (file == NULL) { + return fail_errno("open relative host marker"); + } + size_t length = strlen(contents); + bool ok = fwrite(contents, 1, length, file) == length && + fflush(file) == 0 && + fsync(fileno(file)) == 0 && + fclose(file) == 0; + if (!ok) { + return fail_errno("write relative host marker"); + } + return 0; +} + +static void print_json_string(const char *value) { + putchar('"'); + for (const unsigned char *cursor = (const unsigned char *)value; *cursor != '\0'; cursor++) { + switch (*cursor) { + case '"': + fputs("\\\"", stdout); + break; + case '\\': + fputs("\\\\", stdout); + break; + case '\b': + fputs("\\b", stdout); + break; + case '\f': + fputs("\\f", stdout); + break; + case '\n': + fputs("\\n", stdout); + break; + case '\r': + fputs("\\r", stdout); + break; + case '\t': + fputs("\\t", stdout); + break; + default: + if (*cursor < 0x20) { + fprintf(stdout, "\\u%04x", (unsigned int)*cursor); + } else { + putchar((int)*cursor); + } + break; + } + } + putchar('"'); +} + +static OliphauntConfig config_for( + const char *pgdata, + const char *runtime_dir, + const char *module_dir, + const char *database) { + OliphauntConfig config = { + .abi_version = OLIPHAUNT_ABI_VERSION, + .pgdata = pgdata, + .runtime_dir = runtime_dir, + .module_dir = module_dir, + .username = "postgres", + .database = database, + .flags = 0, + .startup_args = NULL, + .startup_arg_count = 0, + }; + return config; +} + +static int run_direct(const char *pgdata, const char *runtime_dir, const char *module_dir) { + char original_cwd[PATH_MAX]; + char open_cwd[PATH_MAX]; + char detach_cwd[PATH_MAX]; + char close_cwd[PATH_MAX]; + char open_expected[PATH_MAX]; + char open_wrong[PATH_MAX]; + char detach_expected[PATH_MAX]; + char detach_wrong[PATH_MAX]; + OliphauntHandle *handle = NULL; + + if (current_directory(original_cwd) != 0 || + joined_path(open_expected, pgdata, direct_open_marker) != 0 || + joined_path(open_wrong, original_cwd, direct_open_marker) != 0 || + joined_path(detach_expected, pgdata, direct_detach_marker) != 0 || + joined_path(detach_wrong, original_cwd, direct_detach_marker) != 0) { + return 1; + } + + OliphauntConfig config = config_for(pgdata, runtime_dir, module_dir, "postgres"); + if (oliphaunt_init(&config, &handle) != 0 || handle == NULL) { + fprintf(stderr, "oliphaunt_init failed: %s\n", last_error_message(handle)); + return 1; + } + if (current_directory(open_cwd) != 0 || !same_directory(open_cwd, pgdata)) { + (void)oliphaunt_close(handle); + return fail("Direct open did not leave the host process cwd at PGDATA"); + } + if (regular_file_exists(open_wrong)) { + (void)oliphaunt_close(handle); + return fail("Direct-open marker already exists at the host sentinel"); + } + if (write_relative_marker(direct_open_marker, "written after Direct open\n") != 0 || + !regular_file_exists(open_expected) || + regular_file_exists(open_wrong)) { + (void)oliphaunt_close(handle); + return fail("relative host I/O after Direct open did not resolve under PGDATA"); + } + + if (oliphaunt_detach(handle) != 0) { + fprintf(stderr, "oliphaunt_detach failed: %s\n", last_error_message(handle)); + (void)oliphaunt_close(handle); + return 1; + } + if (current_directory(detach_cwd) != 0 || !same_directory(detach_cwd, pgdata)) { + (void)oliphaunt_close(handle); + return fail("Direct detach unexpectedly restored the process cwd"); + } + if (regular_file_exists(detach_wrong)) { + (void)oliphaunt_close(handle); + return fail("Direct-detach marker already exists at the host sentinel"); + } + if (write_relative_marker(direct_detach_marker, "written after Direct detach\n") != 0 || + !regular_file_exists(detach_expected) || + regular_file_exists(detach_wrong)) { + (void)oliphaunt_close(handle); + return fail("relative host I/O after Direct detach did not resolve under PGDATA"); + } + + if (oliphaunt_close(handle) != 0) { + return fail("terminal Direct close failed"); + } + handle = NULL; + if (current_directory(close_cwd) != 0 || !same_directory(close_cwd, original_cwd)) { + return fail("terminal Direct close did not restore the original host cwd"); + } + + fputs("{\"schema\":\"oliphaunt-native-cwd-boundary-child-v1\",\"mode\":\"direct\",", stdout); + fputs("\"originalCwd\":", stdout); + print_json_string(original_cwd); + fputs(",\"cwdAfterOpen\":", stdout); + print_json_string(open_cwd); + fputs(",\"openRelativePath\":", stdout); + print_json_string(open_expected); + fputs(",\"cwdAfterDetach\":", stdout); + print_json_string(detach_cwd); + fputs(",\"detachRelativePath\":", stdout); + print_json_string(detach_expected); + fputs(",\"cwdAfterTerminalClose\":", stdout); + print_json_string(close_cwd); + fputs(",\"openMutatesProcessCwd\":true,\"detachPreservesMutation\":true,", stdout); + fputs("\"terminalCloseRestoresCwd\":true}\n", stdout); + return 0; +} + +static int run_startup_fatal( + const char *pgdata, + const char *runtime_dir, + const char *module_dir) { + char original_cwd[PATH_MAX]; + char failure_cwd[PATH_MAX]; + char expected_marker[PATH_MAX]; + char wrong_marker[PATH_MAX]; + char error[1024]; + OliphauntHandle *handle = NULL; + + if (current_directory(original_cwd) != 0 || + joined_path(expected_marker, original_cwd, fatal_marker) != 0 || + joined_path(wrong_marker, pgdata, fatal_marker) != 0) { + return 1; + } + if (regular_file_exists(expected_marker) || regular_file_exists(wrong_marker)) { + return fail("startup-FATAL marker already exists"); + } + + OliphauntConfig config = config_for(pgdata, runtime_dir, module_dir, missing_database); + int32_t rc = oliphaunt_init(&config, &handle); + if (rc == 0 || handle != NULL) { + if (handle != NULL) { + (void)oliphaunt_close(handle); + } + return fail("missing-database startup unexpectedly succeeded"); + } + const char *last_error = last_error_message(NULL); + snprintf(error, sizeof(error), "%s", last_error != NULL ? last_error : "(null)"); + if (strstr(error, "before ReadyForQuery") == NULL) { + fprintf(stderr, "unexpected missing-database startup error: %s\n", error); + return fail("startup failure was not terminal before ReadyForQuery"); + } + if (current_directory(failure_cwd) != 0 || !same_directory(failure_cwd, original_cwd)) { + return fail("startup FATAL did not restore the original host cwd"); + } + if (write_relative_marker(fatal_marker, "written after startup FATAL\n") != 0 || + !regular_file_exists(expected_marker) || + regular_file_exists(wrong_marker)) { + return fail("relative host I/O after startup FATAL did not resolve under the original cwd"); + } + + fputs("{\"schema\":\"oliphaunt-native-cwd-boundary-child-v1\",", stdout); + fputs("\"mode\":\"startup-fatal\",\"database\":", stdout); + print_json_string(missing_database); + fputs(",\"originalCwd\":", stdout); + print_json_string(original_cwd); + fputs(",\"cwdAfterFailure\":", stdout); + print_json_string(failure_cwd); + fputs(",\"relativePathAfterFailure\":", stdout); + print_json_string(expected_marker); + fputs(",\"publicAbiError\":", stdout); + print_json_string(error); + fputs(",\"failedBeforeReadyForQuery\":true,\"startupFatalRestoresCwd\":true}\n", stdout); + return 0; +} + +static int run_early_startup_fatal( + const char *pgdata, + const char *runtime_dir, + const char *module_dir) { + /* Valid ABI syntax reaches PostgreSQL's early startup parser. */ + static const char *const startup_args[] = {"-c", "shared_buffers=not-a-size"}; + char original_cwd[PATH_MAX]; + char failure_cwd[PATH_MAX]; + char expected_marker[PATH_MAX]; + char wrong_marker[PATH_MAX]; + char error[1024]; + OliphauntHandle *handle = NULL; + + if (current_directory(original_cwd) != 0 || + joined_path(expected_marker, original_cwd, early_fatal_marker) != 0 || + joined_path(wrong_marker, pgdata, early_fatal_marker) != 0) { + return 1; + } + if (regular_file_exists(expected_marker) || regular_file_exists(wrong_marker)) { + return fail("early-startup-FATAL marker already exists"); + } + + OliphauntConfig config = config_for(pgdata, runtime_dir, module_dir, "postgres"); + config.startup_args = startup_args; + config.startup_arg_count = sizeof(startup_args) / sizeof(startup_args[0]); + int32_t rc = oliphaunt_init(&config, &handle); + if (rc == 0 || handle != NULL) { + if (handle != NULL) { + (void)oliphaunt_close(handle); + } + return fail("invalid startup GUC value unexpectedly succeeded"); + } + const char *last_error = last_error_message(NULL); + snprintf(error, sizeof(error), "%s", last_error != NULL ? last_error : "(null)"); + if (strstr(error, "before ReadyForQuery") == NULL) { + fprintf(stderr, "unexpected early startup error: %s\n", error); + return fail("early startup failure was not terminal before ReadyForQuery"); + } + if (current_directory(failure_cwd) != 0 || !same_directory(failure_cwd, original_cwd)) { + return fail("early startup FATAL changed the original host cwd"); + } + if (write_relative_marker(early_fatal_marker, "written after early startup FATAL\n") != 0 || + !regular_file_exists(expected_marker) || + regular_file_exists(wrong_marker)) { + return fail("relative host I/O after early startup FATAL escaped the original cwd"); + } + + fputs("{\"schema\":\"oliphaunt-native-cwd-boundary-child-v1\",", stdout); + fputs("\"mode\":\"early-startup-fatal\",\"injectedStartupGuc\":\"shared_buffers=not-a-size\",", stdout); + fputs("\"configuredDatabase\":\"postgres\",\"originalCwd\":", stdout); + print_json_string(original_cwd); + fputs(",\"cwdAfterFailure\":", stdout); + print_json_string(failure_cwd); + fputs(",\"relativePathAfterFailure\":", stdout); + print_json_string(expected_marker); + fputs(",\"publicAbiError\":", stdout); + print_json_string(error); + fputs(",\"failedBeforeReadyForQuery\":true,\"earlyFatalCwdPreserved\":true}\n", stdout); + return 0; +} + +static int run_unresolvable_cwd( + const char *pgdata, + const char *runtime_dir, + const char *module_dir, + const char *expected_cwd) { + char original_cwd[PATH_MAX]; + OliphauntErrorCapture error; + OliphauntHandle *handle = NULL; + + if (current_directory(original_cwd) != 0 || !same_directory(original_cwd, expected_cwd)) { + return fail("unresolvable-cwd child did not start in its exact owned sentinel"); + } + if (strcmp(original_cwd, "/") == 0 || strlen(original_cwd) < 16) { + return fail("refusing to remove an unsafe cwd sentinel"); + } + if (rmdir(expected_cwd) != 0) { + return fail_errno("remove exact owned cwd sentinel"); + } + errno = 0; + char unavailable[PATH_MAX]; + if (getcwd(unavailable, sizeof(unavailable)) != NULL || errno != ENOENT) { + return fail("deleted cwd did not make getcwd fail with ENOENT"); + } + + OliphauntConfig config = config_for(pgdata, runtime_dir, module_dir, "postgres"); + int32_t rc = oliphaunt_init_with_error(&config, &handle, &error); + if (rc == 0 || handle != NULL) { + if (handle != NULL) { + (void)oliphaunt_close(handle); + } + return fail("init accepted an unresolvable caller cwd"); + } + if (strstr(error.message, "could not retain caller working directory:") == NULL || + strstr(error.message, strerror(ENOENT)) == NULL || + strcmp(error.message, last_error_message(NULL)) != 0) { + fprintf(stderr, "unexpected cwd capture error: %s\n", error.message); + return fail("startup did not preserve the cwd capture diagnostic"); + } + if (getcwd(unavailable, sizeof(unavailable)) != NULL || errno != ENOENT) { + return fail("startup from an unresolvable cwd changed the caller cwd"); + } + + fputs("{\"schema\":\"oliphaunt-native-cwd-boundary-child-v1\",", stdout); + fputs("\"mode\":\"unresolvable-cwd\",\"deletedOriginalCwd\":", stdout); + print_json_string(original_cwd); + fputs(",\"publicAbiError\":", stdout); + print_json_string(error.message); + fputs(",\"initRejectedUnrestorableCwd\":true,\"contractSatisfied\":true}\n", stdout); + return 0; +} + +static int run_search_only_cwd( + const char *pgdata, + const char *runtime_dir, + const char *module_dir, + const char *expected_cwd) { + struct stat original_identity; + struct stat restored_identity; + OliphauntHandle *handle = NULL; + int result = 1; + + if (geteuid() == 0 || strcmp(expected_cwd, "/") == 0 || strlen(expected_cwd) < 16) { + return fail("search-only probe requires an unprivileged, owned cwd sentinel"); + } + if (!same_directory(".", expected_cwd) || stat(".", &original_identity) != 0) { + return fail("search-only child did not start in its owned cwd sentinel"); + } + if (chmod(expected_cwd, S_IXUSR) != 0) { + return fail_errno("restrict owned cwd to search permission"); + } + int fd = open(".", O_RDONLY); + if (fd >= 0) { + close(fd); + fail("fixture must deny directory reading (run as an unprivileged user)"); + goto cleanup; + } + if (errno != EACCES) { + fail_errno("verify search-only cwd denies reading"); + goto cleanup; + } + + OliphauntConfig config = config_for(pgdata, runtime_dir, module_dir, "postgres"); + if (oliphaunt_init(&config, &handle) != 0 || handle == NULL) { + fprintf(stderr, "search-only cwd init failed: %s\n", last_error_message(handle)); + goto cleanup; + } + OliphauntResponse response = {0}; + int32_t query_rc = oliphaunt_exec_simple_query(handle, "SELECT 1", strlen("SELECT 1"), &response); + bool query_ok = query_rc == 0 && response.len > 0; + oliphaunt_free_response(&response); + if (!query_ok) { + fail("query from search-only caller cwd failed"); + goto cleanup; + } + int32_t close_rc = oliphaunt_close(handle); + handle = NULL; + if (close_rc != 0 || stat(".", &restored_identity) != 0 || + !same_file_identity(&original_identity, &restored_identity)) { + fail("terminal close did not restore the search-only caller cwd"); + goto cleanup; + } + fputs("{\"schema\":\"oliphaunt-native-cwd-boundary-child-v1\"," + "\"mode\":\"search-only-cwd\",\"querySucceeded\":true," + "\"terminalCloseRestoredIdentity\":true,\"contractSatisfied\":true}\n", stdout); + result = 0; + +cleanup: + if (handle != NULL) { + (void)oliphaunt_close(handle); + } + if (chmod(expected_cwd, original_identity.st_mode & 0777) != 0) { + return fail_errno("restore owned cwd fixture permissions"); + } + return result; +} + +static int run_renamed_cwd( + const char *pgdata, + const char *runtime_dir, + const char *module_dir, + const char *expected_cwd, + const char *renamed_cwd) { + char open_cwd[PATH_MAX]; + char close_cwd[PATH_MAX]; + char restored_marker[PATH_MAX]; + char replacement_marker[PATH_MAX]; + struct stat original_identity; + struct stat renamed_identity; + struct stat replacement_identity; + struct stat restored_identity; + OliphauntHandle *handle = NULL; + + if (!same_directory(".", expected_cwd)) { + return fail("renamed-cwd child did not start in its exact owned sentinel"); + } + if (strcmp(expected_cwd, "/") == 0 || strlen(expected_cwd) < 16 || + strcmp(renamed_cwd, "/") == 0 || strlen(renamed_cwd) < 16) { + return fail("refusing unsafe renamed-cwd sentinel paths"); + } + if (stat(".", &original_identity) != 0) { + return fail_errno("stat original cwd identity"); + } + + OliphauntConfig config = config_for(pgdata, runtime_dir, module_dir, "postgres"); + if (oliphaunt_init(&config, &handle) != 0 || handle == NULL) { + fprintf(stderr, "oliphaunt_init failed: %s\n", last_error_message(handle)); + return 1; + } + if (current_directory(open_cwd) != 0 || !same_directory(open_cwd, pgdata)) { + (void)oliphaunt_close(handle); + return fail("renamed-cwd Direct open did not leave process cwd at PGDATA"); + } + + if (rename(expected_cwd, renamed_cwd) != 0) { + (void)oliphaunt_close(handle); + return fail_errno("rename exact owned cwd sentinel"); + } + if (mkdir(expected_cwd, 0700) != 0) { + (void)oliphaunt_close(handle); + return fail_errno("create replacement at original cwd pathname"); + } + if (stat(renamed_cwd, &renamed_identity) != 0 || + stat(expected_cwd, &replacement_identity) != 0) { + (void)oliphaunt_close(handle); + return fail_errno("stat renamed and replacement cwd identities"); + } + if (!same_file_identity(&original_identity, &renamed_identity) || + same_file_identity(&original_identity, &replacement_identity)) { + (void)oliphaunt_close(handle); + return fail("cwd rename/replacement did not create the intended identity discriminator"); + } + + if (oliphaunt_close(handle) != 0) { + return fail("terminal close after cwd rename/replacement failed"); + } + handle = NULL; + if (stat(".", &restored_identity) != 0 || + !same_file_identity(&restored_identity, &original_identity)) { + return fail("terminal close restored a pathname replacement instead of cwd identity"); + } + if (current_directory(close_cwd) != 0 || !same_directory(close_cwd, renamed_cwd)) { + return fail("terminal close did not resolve to the renamed original cwd"); + } + if (joined_path(restored_marker, renamed_cwd, renamed_close_marker) != 0 || + joined_path(replacement_marker, expected_cwd, renamed_close_marker) != 0 || + write_relative_marker(renamed_close_marker, "written after identity-safe Direct close\n") != 0 || + !regular_file_exists(restored_marker) || regular_file_exists(replacement_marker)) { + return fail("relative I/O after terminal close targeted the pathname replacement"); + } + + fputs("{\"schema\":\"oliphaunt-native-cwd-boundary-child-v1\",", stdout); + fputs("\"mode\":\"renamed-cwd\",\"originalPath\":", stdout); + print_json_string(expected_cwd); + fputs(",\"renamedOriginalPath\":", stdout); + print_json_string(renamed_cwd); + fputs(",\"cwdAfterOpen\":", stdout); + print_json_string(open_cwd); + fputs(",\"cwdAfterTerminalClose\":", stdout); + print_json_string(close_cwd); + fputs(",\"relativePathAfterClose\":", stdout); + print_json_string(restored_marker); + fprintf(stdout, + ",\"originalDevice\":\"%" PRIuMAX "\",\"originalInode\":\"%" PRIuMAX "\",", + (uintmax_t)original_identity.st_dev, + (uintmax_t)original_identity.st_ino); + fputs("\"pathWasReplaced\":true,\"terminalCloseRestoredIdentity\":true,", stdout); + fputs("\"relativeIoUsedRenamedOriginal\":true,\"contractSatisfied\":true}\n", stdout); + return 0; +} + +int main(int argc, char **argv) { + if (argc < 5 || argc > 7) { + fprintf(stderr, "usage: %s [owned-cwd [renamed-cwd]]\n", argv[0]); + return 2; + } + if (strcmp(argv[1], "unresolvable-cwd") == 0) { + if (argc != 6) { + return fail("unresolvable-cwd mode requires its exact owned cwd sentinel"); + } + return run_unresolvable_cwd(argv[2], argv[3], argv[4], argv[5]); + } + if (strcmp(argv[1], "renamed-cwd") == 0) { + if (argc != 7) { + return fail("renamed-cwd mode requires exact original and renamed cwd sentinels"); + } + return run_renamed_cwd(argv[2], argv[3], argv[4], argv[5], argv[6]); + } + if (strcmp(argv[1], "search-only-cwd") == 0) { + if (argc != 6) { + return fail("search-only-cwd mode requires its exact owned cwd sentinel"); + } + return run_search_only_cwd(argv[2], argv[3], argv[4], argv[5]); + } + if (argc != 5) { + return fail("unexpected exact-owned-cwd argument for this probe mode"); + } + if (strcmp(argv[1], "direct") == 0) { + return run_direct(argv[2], argv[3], argv[4]); + } + if (strcmp(argv[1], "startup-fatal") == 0) { + return run_startup_fatal(argv[2], argv[3], argv[4]); + } + if (strcmp(argv[1], "early-startup-fatal") == 0) { + return run_early_startup_fatal(argv[2], argv[3], argv[4]); + } + return fail("unknown probe mode"); +} diff --git a/src/native/runtime/smoke/liboliphaunt_generation_lifecycle.c b/src/native/runtime/smoke/liboliphaunt_generation_lifecycle.c index 14e43b97d..2ef79af2e 100644 --- a/src/native/runtime/smoke/liboliphaunt_generation_lifecycle.c +++ b/src/native/runtime/smoke/liboliphaunt_generation_lifecycle.c @@ -103,7 +103,31 @@ static void *claim_and_close_current_generation(void *data) { return NULL; } +static int verify_backend_durability_arguments(void) { + char *overrides[] = {"-c", "fsync=off", "-c", "fsync=on"}; + OliphauntHandle handle = {0}; + handle.postgres_path = "/runtime/bin/postgres"; + handle.pgdata = "/database/pgdata"; + handle.startup_args = overrides; + for (size_t count = 0; count <= 4; count += 2) { + handle.startup_arg_count = count; + OliphauntBackendArgv args = {0}; + CHECK(oliphaunt_build_backend_argv(&handle, &args) == 0, "build durability argv"); + /* The embedded default disables fsync; later caller GUCs override it. */ + bool fsync_enabled = true; + for (int i = 1; i < args.argc; i++) { + if (strcmp(args.argv[i], "-F") == 0) fsync_enabled = false; + if (strcmp(args.argv[i], "fsync=off") == 0) fsync_enabled = false; + if (strcmp(args.argv[i], "fsync=on") == 0) fsync_enabled = true; + } + CHECK(fsync_enabled == (count == 4), "default and explicit fsync override ordering"); + oliphaunt_free_backend_argv(&args); + } + return 0; +} + int main(void) { + if (verify_backend_durability_arguments() != 0) return 1; char *startup_args[] = {"-c", "search_path=public"}; OliphauntHandle resident_config; memset(&resident_config, 0, sizeof(resident_config)); @@ -127,6 +151,19 @@ int main(void) { }; CHECK(oliphaunt_config_matches_resident_runtime(&resident_config, &reopen_config), "an internally locked resident runtime must accept the same reopen mode"); + reopen_config.username = ""; + reopen_config.database = ""; + CHECK(oliphaunt_config_matches_resident_runtime(&resident_config, &reopen_config), + "empty reopen identities must use the same defaults as initial open"); + reopen_config.username = NULL; + reopen_config.database = NULL; + CHECK(oliphaunt_config_matches_resident_runtime(&resident_config, &reopen_config), + "null reopen identities must use the same defaults as initial open"); + reopen_config.username = "other"; + CHECK(!oliphaunt_config_matches_resident_runtime(&resident_config, &reopen_config), + "explicit non-default reopen identities must still be rejected"); + reopen_config.username = "postgres"; + reopen_config.database = "postgres"; reopen_config.icu_data_dir = "/selected-icu"; CHECK(!oliphaunt_config_matches_resident_runtime(&resident_config, &reopen_config), "a resident runtime must reject changing its selected ICU data"); diff --git a/src/native/runtime/smoke/liboliphaunt_signal_boundary.c b/src/native/runtime/smoke/liboliphaunt_signal_boundary.c new file mode 100644 index 000000000..7dc4b5c4e --- /dev/null +++ b/src/native/runtime/smoke/liboliphaunt_signal_boundary.c @@ -0,0 +1,1252 @@ +#ifndef _GNU_SOURCE +#define _GNU_SOURCE +#endif + +#include "../include/oliphaunt.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#if !defined(__linux__) +#error "The retained Native signal-boundary probe currently targets Linux only." +#endif + +#if defined(_MSC_VER) +#define OLIPHAUNT_SMOKE_THREAD_LOCAL __declspec(thread) +#else +#define OLIPHAUNT_SMOKE_THREAD_LOCAL _Thread_local +#endif + +static OLIPHAUNT_SMOKE_THREAD_LOCAL char + last_error_buffer[OLIPHAUNT_ERROR_CAPTURE_CAPACITY]; + +static const char *last_error_message(OliphauntHandle *handle) { + (void)oliphaunt_copy_last_error( + handle, + last_error_buffer, + sizeof(last_error_buffer)); + return last_error_buffer; +} + +#define ARRAY_LENGTH(value) (sizeof(value) / sizeof((value)[0])) +#define SENTINEL_TIMER_SECONDS 1800 +#define MAX_TRACKED_SIGNAL 128 +#define COPY_INPUT_WAIT_MILLIS 1000 +#define CPU_BOUND_SQL \ + "SELECT count(*) FROM generate_series(1, 1000000000) AS value" + +typedef struct SignalSpec { + int number; + const char *name; +} SignalSpec; + +static const SignalSpec signal_specs[] = { + {SIGHUP, "SIGHUP"}, + {SIGINT, "SIGINT"}, + {SIGTERM, "SIGTERM"}, + {SIGQUIT, "SIGQUIT"}, + {SIGPIPE, "SIGPIPE"}, + {SIGUSR1, "SIGUSR1"}, + {SIGUSR2, "SIGUSR2"}, + {SIGALRM, "SIGALRM"}, + {SIGCHLD, "SIGCHLD"}, + {SIGURG, "SIGURG"}, + {SIGFPE, "SIGFPE"}, + {SIGWINCH, "SIGWINCH"}, +}; + +typedef enum DispositionKind { + DISPOSITION_SENTINEL, + DISPOSITION_DEFAULT, + DISPOSITION_IGNORE, + DISPOSITION_OTHER, +} DispositionKind; + +typedef struct SignalObservation { + DispositionKind disposition; + bool blocked; +} SignalObservation; + +typedef struct BoundarySnapshot { + SignalObservation signals[ARRAY_LENGTH(signal_specs)]; + int64_t timer_value_micros; + int64_t timer_interval_micros; + bool timer_sentinel_preserved; + bool mask_sentinel_preserved; + int sigurg_sentinel_deliveries; +} BoundarySnapshot; + +typedef struct QueryObservation { + int abi_result; + size_t response_bytes; + bool has_error; + bool has_notice; + bool has_ready; + bool diagnostic_matched; +} QueryObservation; + +typedef struct CancelThreadState { + OliphauntHandle *handle; + const char *sql; + const char *expected_diagnostic; + QueryObservation query; + int64_t query_duration_micros; +} CancelThreadState; + +typedef struct ProtocolStreamState { + OliphauntHandle *handle; + const char *sql; + const char *expected_diagnostic; + pthread_mutex_t callback_mutex; + pthread_cond_t callback_cond; + uint8_t *response_data; + size_t response_len; + size_t response_capacity; + bool copy_input_seen; + bool finished; + QueryObservation query; + int64_t query_duration_micros; +} ProtocolStreamState; + +static volatile sig_atomic_t sentinel_deliveries[MAX_TRACKED_SIGNAL]; + +static int64_t monotonic_micros(void) { + struct timespec now; + if (clock_gettime(CLOCK_MONOTONIC, &now) != 0) { + return -1; + } + return (int64_t)now.tv_sec * INT64_C(1000000) + now.tv_nsec / 1000; +} + +static void sentinel_handler(int signo) { + if (signo > 0 && signo < MAX_TRACKED_SIGNAL) { + sentinel_deliveries[signo]++; + } +} + +static const char *disposition_name(DispositionKind disposition) { + switch (disposition) { + case DISPOSITION_SENTINEL: + return "sentinel"; + case DISPOSITION_DEFAULT: + return "default"; + case DISPOSITION_IGNORE: + return "ignore"; + case DISPOSITION_OTHER: + return "other"; + } + return "unknown"; +} + +static bool response_has_bytes( + const OliphauntResponse *response, + const char *expected) { + const size_t expected_length = strlen(expected); + if (expected_length == 0) { + return true; + } + if (response->data == NULL || response->len < expected_length) { + return false; + } + for (size_t offset = 0; offset <= response->len - expected_length; offset++) { + if (memcmp(response->data + offset, expected, expected_length) == 0) { + return true; + } + } + return false; +} + +static bool response_has_tag(const OliphauntResponse *response, uint8_t expected) { + size_t offset = 0; + while (offset + 5 <= response->len) { + const uint32_t frame_length = + ((uint32_t)response->data[offset + 1] << 24u) | + ((uint32_t)response->data[offset + 2] << 16u) | + ((uint32_t)response->data[offset + 3] << 8u) | + (uint32_t)response->data[offset + 4]; + if (frame_length < 4 || frame_length > response->len - offset - 1) { + return false; + } + if (response->data[offset] == expected) { + return true; + } + offset += (size_t)frame_length + 1u; + } + return false; +} + +static QueryObservation observe_response( + int abi_result, + const OliphauntResponse *response, + const char *expected_diagnostic) { + QueryObservation observation = { + .abi_result = abi_result, + .response_bytes = response->len, + }; + if (observation.abi_result == 0) { + observation.has_error = response_has_tag(response, 'E'); + observation.has_notice = response_has_tag(response, 'N'); + observation.has_ready = response_has_tag(response, 'Z'); + observation.diagnostic_matched = expected_diagnostic == NULL || + response_has_bytes(response, expected_diagnostic); + } + return observation; +} + +static QueryObservation execute_query( + OliphauntHandle *handle, + const char *sql, + const char *expected_diagnostic) { + OliphauntResponse response = {0}; + const int abi_result = oliphaunt_exec_simple_query( + handle, + sql, + strlen(sql), + &response); + const QueryObservation observation = observe_response( + abi_result, + &response, + expected_diagnostic); + oliphaunt_free_response(&response); + return observation; +} + +static void *cancel_query_main(void *context) { + CancelThreadState *state = (CancelThreadState *)context; + const int64_t started_micros = monotonic_micros(); + state->query = execute_query( + state->handle, + state->sql, + state->expected_diagnostic); + const int64_t finished_micros = monotonic_micros(); + state->query_duration_micros = + started_micros >= 0 && finished_micros >= started_micros + ? finished_micros - started_micros + : -1; + return NULL; +} + +static int initialize_protocol_stream_state( + ProtocolStreamState *state, + OliphauntHandle *handle, + const char *sql, + const char *expected_diagnostic) { + state->handle = handle; + state->sql = sql; + state->expected_diagnostic = expected_diagnostic; + state->query_duration_micros = -1; + if (pthread_mutex_init(&state->callback_mutex, NULL) != 0) { + return -1; + } + if (pthread_cond_init(&state->callback_cond, NULL) != 0) { + (void)pthread_mutex_destroy(&state->callback_mutex); + return -1; + } + return 0; +} + +static void destroy_protocol_stream_state(ProtocolStreamState *state) { + free(state->response_data); + state->response_data = NULL; + state->response_len = 0; + state->response_capacity = 0; + (void)pthread_cond_destroy(&state->callback_cond); + (void)pthread_mutex_destroy(&state->callback_mutex); +} + +static uint8_t *simple_query_request(const char *sql, size_t *request_len) { + const size_t sql_len = strlen(sql); + if (sql_len > (size_t)UINT32_MAX - 5u) { + return NULL; + } + const uint32_t frame_len = (uint32_t)sql_len + 5u; + *request_len = sql_len + 6u; + uint8_t *request = (uint8_t *)malloc(*request_len); + if (request == NULL) { + return NULL; + } + request[0] = 'Q'; + request[1] = (uint8_t)(frame_len >> 24u); + request[2] = (uint8_t)(frame_len >> 16u); + request[3] = (uint8_t)(frame_len >> 8u); + request[4] = (uint8_t)frame_len; + memcpy(request + 5u, sql, sql_len); + request[5u + sql_len] = '\0'; + return request; +} + +static int32_t count_unexpected_protocol_chunk( + void *context, + const uint8_t *data, + size_t len) { + size_t *callback_calls = (size_t *)context; + (void)data; + (void)len; + (*callback_calls)++; + return 0; +} + +/* + * Native accepts input only as a batch of complete frontend protocol frames. + * Prove that a valid Query frame followed by a truncated COPY-data frame is + * rejected before either public raw-protocol API can publish any of the batch + * to PostgreSQL. This is the invariant that makes an empty input queue a + * message boundary, rather than a recoverable partial-message boundary. + */ +static int expect_trailing_partial_copy_frame_rejected( + OliphauntHandle *handle, + const char *label, + const uint8_t *partial_frame, + size_t partial_frame_len, + const char *expected_error) { + size_t query_len = 0; + uint8_t *query = simple_query_request( + "COPY oliphaunt_unreachable_partial_input(value) FROM STDIN", + &query_len); + if (query == NULL || query_len > SIZE_MAX - partial_frame_len) { + free(query); + fprintf(stderr, "signal boundary probe failed: build %s request\n", label); + return -1; + } + + const size_t request_len = query_len + partial_frame_len; + uint8_t *request = (uint8_t *)malloc(request_len); + if (request == NULL) { + free(query); + fprintf(stderr, "signal boundary probe failed: allocate %s request\n", label); + return -1; + } + memcpy(request, query, query_len); + memcpy(request + query_len, partial_frame, partial_frame_len); + free(query); + + OliphauntResponse response = {0}; + if (oliphaunt_exec_protocol(handle, request, request_len, &response) == 0) { + fprintf(stderr, "signal boundary probe failed: exec accepted %s\n", label); + oliphaunt_free_response(&response); + free(request); + return -1; + } + if (response.data != NULL || response.len != 0 || + strstr(last_error_message(handle), expected_error) == NULL) { + fprintf(stderr, "signal boundary probe failed: exec %s rejection contract\n", label); + oliphaunt_free_response(&response); + free(request); + return -1; + } + + size_t callback_calls = 0; + if (oliphaunt_exec_protocol_raw_stream( + handle, + request, + request_len, + count_unexpected_protocol_chunk, + &callback_calls) == 0) { + fprintf(stderr, "signal boundary probe failed: stream accepted %s\n", label); + free(request); + return -1; + } + free(request); + if (callback_calls != 0 || + strstr(last_error_message(handle), expected_error) == NULL) { + fprintf(stderr, "signal boundary probe failed: stream %s rejection contract\n", label); + return -1; + } + return 0; +} + +static int32_t observe_protocol_stream_chunk( + void *context, + const uint8_t *data, + size_t len) { + ProtocolStreamState *state = (ProtocolStreamState *)context; + int32_t result = 0; + if (pthread_mutex_lock(&state->callback_mutex) != 0) { + return -1; + } + if (len > 0 && state->response_len > SIZE_MAX - len) { + result = -1; + } else if (len > 0 && state->response_len + len > state->response_capacity) { + size_t next_capacity = state->response_capacity == 0 + ? 1024u + : state->response_capacity; + while (next_capacity < state->response_len + len) { + if (next_capacity > SIZE_MAX / 2u) { + next_capacity = state->response_len + len; + break; + } + next_capacity *= 2u; + } + uint8_t *grown = (uint8_t *)realloc(state->response_data, next_capacity); + if (grown == NULL) { + result = -1; + } else { + state->response_data = grown; + state->response_capacity = next_capacity; + } + } + if (result == 0 && len > 0) { + memcpy(state->response_data + state->response_len, data, len); + state->response_len += len; + } + if (result == 0) { + const OliphauntResponse response = { + .data = state->response_data, + .len = state->response_len, + }; + if (response_has_tag(&response, 'G')) { + state->copy_input_seen = true; + (void)pthread_cond_broadcast(&state->callback_cond); + } + } + (void)pthread_mutex_unlock(&state->callback_mutex); + return result; +} + +static void *protocol_stream_query_main(void *context) { + ProtocolStreamState *state = (ProtocolStreamState *)context; + size_t request_len = 0; + uint8_t *request = simple_query_request(state->sql, &request_len); + const int64_t started_micros = monotonic_micros(); + int abi_result = -1; + if (request != NULL) { + abi_result = oliphaunt_exec_protocol_raw_stream( + state->handle, + request, + request_len, + observe_protocol_stream_chunk, + state); + } + const int64_t finished_micros = monotonic_micros(); + free(request); + + (void)pthread_mutex_lock(&state->callback_mutex); + const OliphauntResponse response = { + .data = state->response_data, + .len = state->response_len, + }; + state->query = observe_response( + abi_result, + &response, + state->expected_diagnostic); + state->query_duration_micros = + started_micros >= 0 && finished_micros >= started_micros + ? finished_micros - started_micros + : -1; + state->finished = true; + (void)pthread_cond_broadcast(&state->callback_cond); + (void)pthread_mutex_unlock(&state->callback_mutex); + return NULL; +} + +static int wait_for_copy_input(ProtocolStreamState *state, int timeout_millis) { + struct timespec deadline; + if (clock_gettime(CLOCK_REALTIME, &deadline) != 0) { + return -1; + } + deadline.tv_sec += timeout_millis / 1000; + deadline.tv_nsec += (long)(timeout_millis % 1000) * 1000000L; + if (deadline.tv_nsec >= 1000000000L) { + deadline.tv_sec++; + deadline.tv_nsec -= 1000000000L; + } + + if (pthread_mutex_lock(&state->callback_mutex) != 0) { + return -1; + } + int wait_result = 0; + while (!state->copy_input_seen && !state->finished && wait_result == 0) { + wait_result = pthread_cond_timedwait( + &state->callback_cond, + &state->callback_mutex, + &deadline); + } + const bool copy_input_seen = state->copy_input_seen; + (void)pthread_mutex_unlock(&state->callback_mutex); + return copy_input_seen ? 0 : -1; +} + +static int isolate_process_group(void) { + const pid_t process_id = getpid(); + if (getsid(0) != process_id || getpgrp() != process_id) { + if (setsid() < 0) { + fprintf(stderr, "signal boundary probe failed: setsid: %s\n", strerror(errno)); + return -1; + } + } + if (getsid(0) != process_id || getpgrp() != process_id) { + fprintf(stderr, "signal boundary probe failed: child is not its own session/group\n"); + return -1; + } + return 0; +} + +static int install_sentinels( + struct sigaction original_actions[ARRAY_LENGTH(signal_specs)], + size_t *installed_count, + struct itimerval *original_timer, + sigset_t *original_mask) { + struct sigaction action; + memset(&action, 0, sizeof(action)); + action.sa_handler = sentinel_handler; + action.sa_flags = SA_RESTART; + if (sigemptyset(&action.sa_mask) != 0) { + fprintf(stderr, "signal boundary probe failed: sigemptyset: %s\n", strerror(errno)); + return -1; + } + + *installed_count = 0; + for (size_t index = 0; index < ARRAY_LENGTH(signal_specs); index++) { + if (sigaction(signal_specs[index].number, NULL, &original_actions[index]) != 0 || + sigaction(signal_specs[index].number, &action, NULL) != 0) { + fprintf( + stderr, + "signal boundary probe failed: install %s sentinel: %s\n", + signal_specs[index].name, + strerror(errno)); + return -1; + } + *installed_count = index + 1; + } + + if (sigprocmask(SIG_SETMASK, NULL, original_mask) != 0) { + fprintf(stderr, "signal boundary probe failed: capture signal mask: %s\n", strerror(errno)); + return -1; + } + sigset_t sentinel_mask = *original_mask; + if (sigaddset(&sentinel_mask, SIGWINCH) != 0 || + sigdelset(&sentinel_mask, SIGURG) != 0 || + sigprocmask(SIG_SETMASK, &sentinel_mask, NULL) != 0) { + fprintf(stderr, "signal boundary probe failed: install sentinel mask: %s\n", strerror(errno)); + return -1; + } + + struct itimerval sentinel_timer; + memset(&sentinel_timer, 0, sizeof(sentinel_timer)); + sentinel_timer.it_value.tv_sec = SENTINEL_TIMER_SECONDS; + if (setitimer(ITIMER_REAL, &sentinel_timer, original_timer) != 0) { + fprintf(stderr, "signal boundary probe failed: install sentinel timer: %s\n", strerror(errno)); + return -1; + } + return 0; +} + +static int restore_host_state( + const struct sigaction original_actions[ARRAY_LENGTH(signal_specs)], + size_t installed_count, + const struct itimerval *original_timer, + const sigset_t *original_mask) { + int failed = 0; + if (setitimer(ITIMER_REAL, original_timer, NULL) != 0) { + failed = 1; + } + for (size_t index = 0; index < installed_count; index++) { + if (sigaction(signal_specs[index].number, &original_actions[index], NULL) != 0) { + failed = 1; + } + } + if (sigprocmask(SIG_SETMASK, original_mask, NULL) != 0) { + failed = 1; + } + return failed == 0 ? 0 : -1; +} + +static int64_t timeval_micros(const struct timeval *value) { + return (int64_t)value->tv_sec * INT64_C(1000000) + (int64_t)value->tv_usec; +} + +static int capture_snapshot(BoundarySnapshot *snapshot) { + sigset_t mask; + struct itimerval timer; + if (sigprocmask(SIG_SETMASK, NULL, &mask) != 0 || + getitimer(ITIMER_REAL, &timer) != 0) { + fprintf(stderr, "signal boundary probe failed: capture process state: %s\n", strerror(errno)); + return -1; + } + + for (size_t index = 0; index < ARRAY_LENGTH(signal_specs); index++) { + struct sigaction action; + if (sigaction(signal_specs[index].number, NULL, &action) != 0) { + fprintf( + stderr, + "signal boundary probe failed: inspect %s: %s\n", + signal_specs[index].name, + strerror(errno)); + return -1; + } + if (action.sa_handler == sentinel_handler) { + snapshot->signals[index].disposition = DISPOSITION_SENTINEL; + } else if (action.sa_handler == SIG_DFL) { + snapshot->signals[index].disposition = DISPOSITION_DEFAULT; + } else if (action.sa_handler == SIG_IGN) { + snapshot->signals[index].disposition = DISPOSITION_IGNORE; + } else { + snapshot->signals[index].disposition = DISPOSITION_OTHER; + } + snapshot->signals[index].blocked = + sigismember(&mask, signal_specs[index].number) == 1; + } + + snapshot->timer_value_micros = timeval_micros(&timer.it_value); + snapshot->timer_interval_micros = timeval_micros(&timer.it_interval); + snapshot->timer_sentinel_preserved = + snapshot->timer_interval_micros == 0 && + snapshot->timer_value_micros > INT64_C(1500000000); + snapshot->mask_sentinel_preserved = + sigismember(&mask, SIGWINCH) == 1 && + sigismember(&mask, SIGURG) == 0; + snapshot->sigurg_sentinel_deliveries = (int)sentinel_deliveries[SIGURG]; + return 0; +} + +/* + * A caller-thread mask snapshot cannot detect PostgreSQL unblocking a signal + * on its backend pthread. Send a process-directed blocked signal and require + * it to remain pending until this caller consumes it synchronously. + */ +static int verify_blocked_signal_routing(const char *phase) { + const sig_atomic_t deliveries_before = sentinel_deliveries[SIGWINCH]; + if (kill(getpid(), SIGWINCH) != 0) { + fprintf(stderr, "signal boundary probe failed: %s send SIGWINCH: %s\n", + phase, strerror(errno)); + return -1; + } + struct timespec settle = {.tv_sec = 0, .tv_nsec = 10000000}; + while (nanosleep(&settle, &settle) != 0 && errno == EINTR) { + } + sigset_t pending; + if (sigpending(&pending) != 0 || sigismember(&pending, SIGWINCH) != 1 || + sentinel_deliveries[SIGWINCH] != deliveries_before) { + fprintf(stderr, + "signal boundary probe failed: %s backend thread accepted blocked SIGWINCH\n", + phase); + return -1; + } + sigset_t only_sigwinch; + if (sigemptyset(&only_sigwinch) != 0 || + sigaddset(&only_sigwinch, SIGWINCH) != 0) { + return -1; + } + const struct timespec no_wait = {.tv_sec = 0, .tv_nsec = 0}; + if (sigtimedwait(&only_sigwinch, NULL, &no_wait) != SIGWINCH || + sentinel_deliveries[SIGWINCH] != deliveries_before) { + fprintf(stderr, + "signal boundary probe failed: %s could not consume pending SIGWINCH\n", + phase); + return -1; + } + return 0; +} + +static void print_snapshot(const char *name, const BoundarySnapshot *snapshot, bool comma) { + printf("\"%s\":{\"signals\":{", name); + for (size_t index = 0; index < ARRAY_LENGTH(signal_specs); index++) { + printf( + "%s\"%s\":{\"disposition\":\"%s\",\"blocked\":%s}", + index == 0 ? "" : ",", + signal_specs[index].name, + disposition_name(snapshot->signals[index].disposition), + snapshot->signals[index].blocked ? "true" : "false"); + } + printf( + "},\"timer\":{\"valueMicros\":%lld,\"intervalMicros\":%lld," + "\"sentinelPreserved\":%s},\"maskSentinelPreserved\":%s," + "\"sigurgSentinelDeliveries\":%d}%s", + (long long)snapshot->timer_value_micros, + (long long)snapshot->timer_interval_micros, + snapshot->timer_sentinel_preserved ? "true" : "false", + snapshot->mask_sentinel_preserved ? "true" : "false", + snapshot->sigurg_sentinel_deliveries, + comma ? "," : ""); +} + +static void print_query_observation(const QueryObservation *observation) { + printf( + "{\"abiResult\":%d,\"responseBytes\":%zu,\"hasError\":%s," + "\"hasNotice\":%s,\"hasReady\":%s,\"diagnosticMatched\":%s}", + observation->abi_result, + observation->response_bytes, + observation->has_error ? "true" : "false", + observation->has_notice ? "true" : "false", + observation->has_ready ? "true" : "false", + observation->diagnostic_matched ? "true" : "false"); +} + +static int query_is_success(const QueryObservation *observation) { + return observation->abi_result == 0 && + !observation->has_error && + observation->has_ready; +} + +int main(int argc, char **argv) { + if (argc != 4) { + fprintf(stderr, "usage: %s \n", argv[0]); + return 2; + } + if (isolate_process_group() != 0) { + return 1; + } + + struct sigaction original_actions[ARRAY_LENGTH(signal_specs)]; + struct itimerval original_timer; + sigset_t original_mask; + size_t installed_count = 0; + bool sentinels_installed = false; + OliphauntHandle *handle = NULL; + int result = 1; + + BoundarySnapshot installed = {0}; + BoundarySnapshot after_init = {0}; + BoundarySnapshot after_timeout = {0}; + BoundarySnapshot after_cancel = {0}; + BoundarySnapshot after_cpu_timeout = {0}; + BoundarySnapshot after_cpu_cancel = {0}; + BoundarySnapshot after_copy_input_timeout = {0}; + BoundarySnapshot after_copy_input_cancel = {0}; + BoundarySnapshot after_supervisor_sql = {0}; + BoundarySnapshot after_close = {0}; + + QueryObservation input_validation_recovery = {0}; + QueryObservation timeout_query = {0}; + QueryObservation timeout_recovery = {0}; + QueryObservation cancel_recovery = {0}; + QueryObservation cpu_timeout_query = {0}; + QueryObservation cpu_timeout_recovery = {0}; + QueryObservation cpu_cancel_setup = {0}; + QueryObservation cpu_cancel_recovery = {0}; + QueryObservation copy_input_timeout_setup = {0}; + QueryObservation copy_input_timeout_recovery = {0}; + QueryObservation copy_input_cancel_setup = {0}; + QueryObservation copy_input_cancel_recovery = {0}; + QueryObservation reload_query = {0}; + QueryObservation rotate_query = {0}; + QueryObservation promote_query = {0}; + QueryObservation final_recovery = {0}; + CancelThreadState cancel_state = {0}; + int cancel_result = -1; + int64_t cancel_call_duration_micros = -1; + int64_t cancel_total_duration_micros = -1; + bool cancel_thread_created = false; + bool cancel_thread_joined = false; + int64_t cpu_timeout_duration_micros = -1; + CancelThreadState cpu_cancel_state = {0}; + int cpu_cancel_result = -1; + int64_t cpu_cancel_call_duration_micros = -1; + int64_t cpu_cancel_total_duration_micros = -1; + bool cpu_cancel_thread_created = false; + bool cpu_cancel_thread_joined = false; + ProtocolStreamState copy_input_timeout_state = {0}; + bool copy_input_timeout_state_initialized = false; + ProtocolStreamState copy_input_cancel_state = {0}; + bool copy_input_cancel_state_initialized = false; + pthread_t copy_input_cancel_thread; + bool copy_input_cancel_thread_created = false; + bool copy_input_cancel_thread_joined = false; + bool copy_input_observed_before_cancel = false; + int copy_input_cancel_result = -1; + int64_t copy_input_cancel_call_duration_micros = -1; + int64_t copy_input_cancel_total_duration_micros = -1; + int close_result = -1; + int blocked_signal_routing_checks = 0; + + if (install_sentinels( + original_actions, + &installed_count, + &original_timer, + &original_mask) != 0) { + goto cleanup; + } + sentinels_installed = true; + if (capture_snapshot(&installed) != 0) { + goto cleanup; + } + + OliphauntConfig config = { + .abi_version = OLIPHAUNT_ABI_VERSION, + .pgdata = argv[1], + .runtime_dir = argv[2], + .module_dir = argv[3], + .username = "postgres", + .database = "postgres", + .flags = 0, + .startup_args = NULL, + .startup_arg_count = 0, + }; + if (oliphaunt_init(&config, &handle) != 0 || handle == NULL) { + fprintf( + stderr, + "signal boundary probe failed: oliphaunt_init: %s\n", + last_error_message(handle)); + goto cleanup; + } + if (capture_snapshot(&after_init) != 0 || + verify_blocked_signal_routing("after init") != 0) { + goto cleanup; + } + blocked_signal_routing_checks++; + + /* + * These suffixes would leave COPY waiting after consuming respectively + * part of the next message length and part of its declared body. Native + * must reject the whole concatenated batch before the preceding Query can + * reach PostgreSQL. + */ + static const uint8_t partial_copy_length[] = {'d', 0, 0}; + static const uint8_t partial_copy_body[] = {'d', 0, 0, 0, 8, '1'}; + if (expect_trailing_partial_copy_frame_rejected( + handle, + "COPY-data partial length", + partial_copy_length, + sizeof(partial_copy_length), + "truncated message header") != 0 || + expect_trailing_partial_copy_frame_rejected( + handle, + "COPY-data partial body", + partial_copy_body, + sizeof(partial_copy_body), + "truncated message body") != 0) { + goto cleanup; + } + input_validation_recovery = execute_query(handle, "SELECT 0", NULL); + if (!query_is_success(&input_validation_recovery)) { + fprintf(stderr, "signal boundary probe failed: input validation recovery\n"); + goto cleanup; + } + + timeout_query = execute_query( + handle, + "SET statement_timeout = '40ms'; SELECT pg_sleep(0.2)", + "canceling statement due to statement timeout"); + if (timeout_query.abi_result != 0 || !timeout_query.has_error || + !timeout_query.has_ready || !timeout_query.diagnostic_matched) { + fprintf(stderr, "signal boundary probe failed: statement_timeout contract\n"); + goto cleanup; + } + timeout_recovery = execute_query( + handle, + "SET statement_timeout = 0; SELECT 1", + NULL); + if (!query_is_success(&timeout_recovery) || capture_snapshot(&after_timeout) != 0 || + verify_blocked_signal_routing("after timeout recovery") != 0) { + fprintf(stderr, "signal boundary probe failed: timeout recovery\n"); + goto cleanup; + } + blocked_signal_routing_checks++; + + cancel_state.handle = handle; + cancel_state.sql = "SELECT pg_sleep(2)"; + cancel_state.expected_diagnostic = "canceling statement due to user request"; + pthread_t cancel_thread; + const int64_t cancel_total_started_micros = monotonic_micros(); + if (pthread_create(&cancel_thread, NULL, cancel_query_main, &cancel_state) != 0) { + fprintf(stderr, "signal boundary probe failed: create cancel query thread\n"); + goto cleanup; + } + cancel_thread_created = true; + struct timespec cancel_delay = {.tv_sec = 0, .tv_nsec = 100000000}; + while (nanosleep(&cancel_delay, &cancel_delay) != 0 && errno == EINTR) { + } + const int64_t cancel_call_started_micros = monotonic_micros(); + cancel_result = oliphaunt_cancel(handle); + const int64_t cancel_call_finished_micros = monotonic_micros(); + if (cancel_call_started_micros >= 0 && + cancel_call_finished_micros >= cancel_call_started_micros) { + cancel_call_duration_micros = + cancel_call_finished_micros - cancel_call_started_micros; + } + if (pthread_join(cancel_thread, NULL) != 0) { + fprintf(stderr, "signal boundary probe failed: join cancel query thread\n"); + goto cleanup; + } + cancel_thread_joined = true; + const int64_t cancel_total_finished_micros = monotonic_micros(); + if (cancel_total_started_micros >= 0 && + cancel_total_finished_micros >= cancel_total_started_micros) { + cancel_total_duration_micros = + cancel_total_finished_micros - cancel_total_started_micros; + } + if (cancel_result != 0 || cancel_state.query.abi_result != 0 || + !cancel_state.query.has_error || !cancel_state.query.has_ready || + !cancel_state.query.diagnostic_matched) { + fprintf(stderr, "signal boundary probe failed: public ABI cancel contract\n"); + goto cleanup; + } + cancel_recovery = execute_query(handle, "SELECT 2", NULL); + if (!query_is_success(&cancel_recovery) || capture_snapshot(&after_cancel) != 0 || + verify_blocked_signal_routing("after cancel recovery") != 0) { + fprintf(stderr, "signal boundary probe failed: cancel recovery\n"); + goto cleanup; + } + blocked_signal_routing_checks++; + + const int64_t cpu_timeout_started_micros = monotonic_micros(); + cpu_timeout_query = execute_query( + handle, + "SET statement_timeout = '40ms'; " CPU_BOUND_SQL, + "canceling statement due to statement timeout"); + const int64_t cpu_timeout_finished_micros = monotonic_micros(); + if (cpu_timeout_started_micros >= 0 && + cpu_timeout_finished_micros >= cpu_timeout_started_micros) { + cpu_timeout_duration_micros = + cpu_timeout_finished_micros - cpu_timeout_started_micros; + } + if (cpu_timeout_query.abi_result != 0 || !cpu_timeout_query.has_error || + !cpu_timeout_query.has_ready || !cpu_timeout_query.diagnostic_matched) { + fprintf(stderr, "signal boundary probe failed: CPU-bound statement_timeout contract\n"); + goto cleanup; + } + cpu_timeout_recovery = execute_query( + handle, + "SET statement_timeout = 0; SELECT 11", + NULL); + if (!query_is_success(&cpu_timeout_recovery) || + capture_snapshot(&after_cpu_timeout) != 0) { + fprintf(stderr, "signal boundary probe failed: CPU-bound timeout recovery\n"); + goto cleanup; + } + + cpu_cancel_setup = execute_query(handle, "SET statement_timeout = '2s'", NULL); + if (!query_is_success(&cpu_cancel_setup)) { + fprintf(stderr, "signal boundary probe failed: CPU-bound cancel guard setup\n"); + goto cleanup; + } + cpu_cancel_state.handle = handle; + cpu_cancel_state.sql = CPU_BOUND_SQL; + cpu_cancel_state.expected_diagnostic = "canceling statement due to user request"; + pthread_t cpu_cancel_thread; + const int64_t cpu_cancel_total_started_micros = monotonic_micros(); + if (pthread_create( + &cpu_cancel_thread, + NULL, + cancel_query_main, + &cpu_cancel_state) != 0) { + fprintf(stderr, "signal boundary probe failed: create CPU-bound cancel query thread\n"); + goto cleanup; + } + cpu_cancel_thread_created = true; + struct timespec cpu_cancel_delay = {.tv_sec = 0, .tv_nsec = 100000000}; + while (nanosleep(&cpu_cancel_delay, &cpu_cancel_delay) != 0 && errno == EINTR) { + } + const int64_t cpu_cancel_call_started_micros = monotonic_micros(); + cpu_cancel_result = oliphaunt_cancel(handle); + const int64_t cpu_cancel_call_finished_micros = monotonic_micros(); + if (cpu_cancel_call_started_micros >= 0 && + cpu_cancel_call_finished_micros >= cpu_cancel_call_started_micros) { + cpu_cancel_call_duration_micros = + cpu_cancel_call_finished_micros - cpu_cancel_call_started_micros; + } + if (pthread_join(cpu_cancel_thread, NULL) != 0) { + fprintf(stderr, "signal boundary probe failed: join CPU-bound cancel query thread\n"); + goto cleanup; + } + cpu_cancel_thread_joined = true; + const int64_t cpu_cancel_total_finished_micros = monotonic_micros(); + if (cpu_cancel_total_started_micros >= 0 && + cpu_cancel_total_finished_micros >= cpu_cancel_total_started_micros) { + cpu_cancel_total_duration_micros = + cpu_cancel_total_finished_micros - cpu_cancel_total_started_micros; + } + if (cpu_cancel_result != 0 || cpu_cancel_state.query.abi_result != 0 || + !cpu_cancel_state.query.has_error || !cpu_cancel_state.query.has_ready || + !cpu_cancel_state.query.diagnostic_matched) { + fprintf(stderr, "signal boundary probe failed: CPU-bound public ABI cancel contract\n"); + goto cleanup; + } + cpu_cancel_recovery = execute_query( + handle, + "SET statement_timeout = 0; SELECT 12", + NULL); + if (!query_is_success(&cpu_cancel_recovery) || + capture_snapshot(&after_cpu_cancel) != 0) { + fprintf(stderr, "signal boundary probe failed: CPU-bound cancel recovery\n"); + goto cleanup; + } + + copy_input_timeout_setup = execute_query( + handle, + "SET statement_timeout = '40ms'; " + "CREATE TEMP TABLE oliphaunt_copy_input_timeout(value integer)", + NULL); + if (!query_is_success(©_input_timeout_setup)) { + fprintf(stderr, "signal boundary probe failed: COPY-input timeout setup\n"); + goto cleanup; + } + if (initialize_protocol_stream_state( + ©_input_timeout_state, + handle, + "COPY oliphaunt_copy_input_timeout(value) FROM STDIN", + "canceling statement due to statement timeout") != 0) { + fprintf(stderr, "signal boundary probe failed: initialize COPY-input timeout state\n"); + goto cleanup; + } + copy_input_timeout_state_initialized = true; + (void)protocol_stream_query_main(©_input_timeout_state); + if (!copy_input_timeout_state.copy_input_seen || + copy_input_timeout_state.query.abi_result != 0 || + !copy_input_timeout_state.query.has_error || + !copy_input_timeout_state.query.has_ready || + !copy_input_timeout_state.query.diagnostic_matched) { + fprintf( + stderr, + "signal boundary probe failed: COPY-input statement_timeout contract\n"); + goto cleanup; + } + copy_input_timeout_recovery = execute_query( + handle, + "SET statement_timeout = 0; SELECT 21", + NULL); + if (!query_is_success(©_input_timeout_recovery) || + capture_snapshot(&after_copy_input_timeout) != 0) { + fprintf(stderr, "signal boundary probe failed: COPY-input timeout recovery\n"); + goto cleanup; + } + + copy_input_cancel_setup = execute_query( + handle, + "SET statement_timeout = '2s'; " + "CREATE TEMP TABLE oliphaunt_copy_input_cancel(value integer)", + NULL); + if (!query_is_success(©_input_cancel_setup)) { + fprintf(stderr, "signal boundary probe failed: COPY-input cancel guard setup\n"); + goto cleanup; + } + if (initialize_protocol_stream_state( + ©_input_cancel_state, + handle, + "COPY oliphaunt_copy_input_cancel(value) FROM STDIN", + "canceling statement due to user request") != 0) { + fprintf(stderr, "signal boundary probe failed: initialize COPY-input cancel state\n"); + goto cleanup; + } + copy_input_cancel_state_initialized = true; + const int64_t copy_input_cancel_total_started_micros = monotonic_micros(); + if (pthread_create( + ©_input_cancel_thread, + NULL, + protocol_stream_query_main, + ©_input_cancel_state) != 0) { + fprintf(stderr, "signal boundary probe failed: create COPY-input cancel thread\n"); + goto cleanup; + } + copy_input_cancel_thread_created = true; + copy_input_observed_before_cancel = + wait_for_copy_input(©_input_cancel_state, COPY_INPUT_WAIT_MILLIS) == 0; + const int64_t copy_input_cancel_call_started_micros = monotonic_micros(); + copy_input_cancel_result = oliphaunt_cancel(handle); + const int64_t copy_input_cancel_call_finished_micros = monotonic_micros(); + if (copy_input_cancel_call_started_micros >= 0 && + copy_input_cancel_call_finished_micros >= copy_input_cancel_call_started_micros) { + copy_input_cancel_call_duration_micros = + copy_input_cancel_call_finished_micros - copy_input_cancel_call_started_micros; + } + if (pthread_join(copy_input_cancel_thread, NULL) != 0) { + fprintf(stderr, "signal boundary probe failed: join COPY-input cancel thread\n"); + goto cleanup; + } + copy_input_cancel_thread_joined = true; + const int64_t copy_input_cancel_total_finished_micros = monotonic_micros(); + if (copy_input_cancel_total_started_micros >= 0 && + copy_input_cancel_total_finished_micros >= copy_input_cancel_total_started_micros) { + copy_input_cancel_total_duration_micros = + copy_input_cancel_total_finished_micros - + copy_input_cancel_total_started_micros; + } + if (!copy_input_observed_before_cancel || copy_input_cancel_result != 0 || + copy_input_cancel_state.query.abi_result != 0 || + !copy_input_cancel_state.query.has_error || + !copy_input_cancel_state.query.has_ready || + !copy_input_cancel_state.query.diagnostic_matched) { + fprintf(stderr, "signal boundary probe failed: COPY-input cancel contract\n"); + goto cleanup; + } + copy_input_cancel_recovery = execute_query( + handle, + "SET statement_timeout = 0; SELECT 22", + NULL); + if (!query_is_success(©_input_cancel_recovery) || + capture_snapshot(&after_copy_input_cancel) != 0 || + verify_blocked_signal_routing("after COPY cancel recovery") != 0) { + fprintf(stderr, "signal boundary probe failed: COPY-input cancel recovery\n"); + goto cleanup; + } + blocked_signal_routing_checks++; + + reload_query = execute_query( + handle, + "SELECT pg_reload_conf()", + "configuration reload is not supported in a trusted embedded session"); + rotate_query = execute_query( + handle, + "SELECT pg_rotate_logfile()", + "log rotation is not supported in a trusted embedded session"); + promote_query = execute_query( + handle, + "SELECT pg_promote(false, 1)", + "standby promotion is not supported in a trusted embedded session"); + if (reload_query.abi_result != 0 || !reload_query.has_error || + !reload_query.has_ready || !reload_query.diagnostic_matched || + rotate_query.abi_result != 0 || rotate_query.has_error || + !rotate_query.has_notice || !rotate_query.has_ready || + !rotate_query.diagnostic_matched || + promote_query.abi_result != 0 || !promote_query.has_error || + !promote_query.has_ready || !promote_query.diagnostic_matched) { + fprintf(stderr, "signal boundary probe failed: supervisor-only SQL contract\n"); + goto cleanup; + } + final_recovery = execute_query(handle, "SELECT 3", NULL); + if (!query_is_success(&final_recovery) || + capture_snapshot(&after_supervisor_sql) != 0) { + fprintf(stderr, "signal boundary probe failed: supervisor SQL recovery\n"); + goto cleanup; + } + + close_result = oliphaunt_close(handle); + handle = NULL; + if (close_result != 0 || capture_snapshot(&after_close) != 0) { + fprintf(stderr, "signal boundary probe failed: oliphaunt_close\n"); + goto cleanup; + } + result = 0; + +cleanup: + if (cancel_thread_created && !cancel_thread_joined) { + (void)pthread_join(cancel_thread, NULL); + } + if (cpu_cancel_thread_created && !cpu_cancel_thread_joined) { + (void)pthread_join(cpu_cancel_thread, NULL); + } + if (copy_input_cancel_thread_created && !copy_input_cancel_thread_joined) { + (void)pthread_join(copy_input_cancel_thread, NULL); + } + if (handle != NULL) { + (void)oliphaunt_close(handle); + } + int restoration_result = -1; + if (sentinels_installed) { + restoration_result = restore_host_state( + original_actions, + installed_count, + &original_timer, + &original_mask); + } + if (copy_input_timeout_state_initialized) { + destroy_protocol_stream_state(©_input_timeout_state); + } + if (copy_input_cancel_state_initialized) { + destroy_protocol_stream_state(©_input_cancel_state); + } + if (result != 0) { + return result; + } + if (restoration_result != 0) { + fprintf(stderr, "signal boundary probe failed: restore child host state\n"); + return 1; + } + + printf( + "{\"schema\":\"oliphaunt-native-signal-boundary-child-v1\"," + "\"privateProcessSession\":true,\"phases\":{"); + print_snapshot("installed", &installed, true); + print_snapshot("afterInit", &after_init, true); + print_snapshot("afterTimeout", &after_timeout, true); + print_snapshot("afterCancel", &after_cancel, true); + print_snapshot("afterCpuTimeout", &after_cpu_timeout, true); + print_snapshot("afterCpuCancel", &after_cpu_cancel, true); + print_snapshot("afterCopyInputTimeout", &after_copy_input_timeout, true); + print_snapshot("afterCopyInputCancel", &after_copy_input_cancel, true); + print_snapshot("afterSupervisorSql", &after_supervisor_sql, true); + print_snapshot("afterClose", &after_close, false); + printf( + "},\"operations\":{\"inputValidation\":{" + "\"partialCopyLengthRejectedBeforePublish\":true," + "\"partialCopyBodyRejectedBeforePublish\":true,\"recovery\":"); + print_query_observation(&input_validation_recovery); + printf("},\"timeout\":{\"query\":"); + print_query_observation(&timeout_query); + printf(",\"recovery\":"); + print_query_observation(&timeout_recovery); + printf( + "},\"cancel\":{\"cancelResult\":%d,\"queryDurationMicros\":%lld," + "\"cancelCallDurationMicros\":%lld,\"totalDurationMicros\":%lld,\"query\":", + cancel_result, + (long long)cancel_state.query_duration_micros, + (long long)cancel_call_duration_micros, + (long long)cancel_total_duration_micros); + print_query_observation(&cancel_state.query); + printf(",\"recovery\":"); + print_query_observation(&cancel_recovery); + printf( + "},\"cpuTimeout\":{\"queryDurationMicros\":%lld,\"query\":", + (long long)cpu_timeout_duration_micros); + print_query_observation(&cpu_timeout_query); + printf(",\"recovery\":"); + print_query_observation(&cpu_timeout_recovery); + printf( + "},\"cpuCancel\":{\"cancelResult\":%d,\"queryDurationMicros\":%lld," + "\"cancelCallDurationMicros\":%lld,\"totalDurationMicros\":%lld,\"setup\":", + cpu_cancel_result, + (long long)cpu_cancel_state.query_duration_micros, + (long long)cpu_cancel_call_duration_micros, + (long long)cpu_cancel_total_duration_micros); + print_query_observation(&cpu_cancel_setup); + printf(",\"query\":"); + print_query_observation(&cpu_cancel_state.query); + printf(",\"recovery\":"); + print_query_observation(&cpu_cancel_recovery); + printf( + "},\"copyInputTimeout\":{\"copyInObserved\":%s," + "\"queryDurationMicros\":%lld,\"setup\":", + copy_input_timeout_state.copy_input_seen ? "true" : "false", + (long long)copy_input_timeout_state.query_duration_micros); + print_query_observation(©_input_timeout_setup); + printf(",\"query\":"); + print_query_observation(©_input_timeout_state.query); + printf(",\"recovery\":"); + print_query_observation(©_input_timeout_recovery); + printf( + "},\"copyInputCancel\":{\"copyInObservedBeforeCancel\":%s," + "\"cancelResult\":%d,\"queryDurationMicros\":%lld," + "\"cancelCallDurationMicros\":%lld,\"totalDurationMicros\":%lld," + "\"setup\":", + copy_input_observed_before_cancel ? "true" : "false", + copy_input_cancel_result, + (long long)copy_input_cancel_state.query_duration_micros, + (long long)copy_input_cancel_call_duration_micros, + (long long)copy_input_cancel_total_duration_micros); + print_query_observation(©_input_cancel_setup); + printf(",\"query\":"); + print_query_observation(©_input_cancel_state.query); + printf(",\"recovery\":"); + print_query_observation(©_input_cancel_recovery); + printf("},\"supervisorSql\":{\"reload\":"); + print_query_observation(&reload_query); + printf(",\"rotate\":"); + print_query_observation(&rotate_query); + printf(",\"promote\":"); + print_query_observation(&promote_query); + printf(",\"recovery\":"); + print_query_observation(&final_recovery); + printf("}},\"sentinelDeliveries\":{"); + for (size_t index = 0; index < ARRAY_LENGTH(signal_specs); index++) { + printf( + "%s\"%s\":%d", + index == 0 ? "" : ",", + signal_specs[index].name, + (int)sentinel_deliveries[signal_specs[index].number]); + } + printf( + "},\"blockedSignalRoutingChecks\":%d,\"closeResult\":%d," + "\"childHostStateRestoredAfterObservation\":true}\n", + blocked_signal_routing_checks, + close_result); + return 0; +} diff --git a/src/native/runtime/smoke/liboliphaunt_smoke.c b/src/native/runtime/smoke/liboliphaunt_smoke.c index 076f087fd..52ca9dc7a 100644 --- a/src/native/runtime/smoke/liboliphaunt_smoke.c +++ b/src/native/runtime/smoke/liboliphaunt_smoke.c @@ -25,6 +25,7 @@ #else #include #include +#include #include #include #endif @@ -56,6 +57,55 @@ static const char *last_error_message(OliphauntHandle *handle) { return last_error_buffer; } +#ifndef _WIN32 +static volatile sig_atomic_t host_sigusr1_sentinel_calls = 0; + +static void host_sigusr1_sentinel(int signo) { + if (signo == SIGUSR1) { + host_sigusr1_sentinel_calls++; + } +} + +static int install_host_sigusr1_sentinel(struct sigaction *previous) { + struct sigaction action; + + memset(&action, 0, sizeof(action)); + action.sa_handler = host_sigusr1_sentinel; + if (sigemptyset(&action.sa_mask) != 0 || sigaction(SIGUSR1, &action, previous) != 0) { + fprintf(stderr, "could not install host SIGUSR1 sentinel: %s\n", strerror(errno)); + return 1; + } + return 0; +} + +static int verify_host_sigusr1_sentinel(const struct sigaction *previous) { + struct sigaction current; + + if (sigaction(SIGUSR1, NULL, ¤t) != 0) { + fprintf(stderr, "could not inspect host SIGUSR1 sentinel: %s\n", strerror(errno)); + return 1; + } + if (current.sa_handler != host_sigusr1_sentinel) { + fprintf(stderr, "embedded PostgreSQL replaced the host SIGUSR1 handler\n"); + return 1; + } + if (host_sigusr1_sentinel_calls != 0) { + fprintf(stderr, "embedded PostgreSQL emitted %d host SIGUSR1 signal(s)\n", + (int)host_sigusr1_sentinel_calls); + return 1; + } + if ((raise)(SIGUSR1) != 0 || host_sigusr1_sentinel_calls != 1) { + fprintf(stderr, "host SIGUSR1 sentinel was not callable after embedded shutdown\n"); + return 1; + } + if (sigaction(SIGUSR1, previous, NULL) != 0) { + fprintf(stderr, "could not restore host SIGUSR1 handler: %s\n", strerror(errno)); + return 1; + } + return 0; +} +#endif + Datum liboliphaunt_smoke_static_answer(PG_FUNCTION_ARGS) { (void)fcinfo; PG_RETURN_INT32(2718); @@ -1386,6 +1436,32 @@ static int exec_plpgsql_smoke(OliphauntHandle *db) { "31415"); } +static int exec_event_trigger_smoke(OliphauntHandle *db) { + return exec_simple_query_expect_bytes( + db, + "DROP EVENT TRIGGER IF EXISTS liboliphaunt_ddl_end; " + "DROP FUNCTION IF EXISTS liboliphaunt_record_event() CASCADE; " + "DROP TABLE IF EXISTS liboliphaunt_event_on; " + "DROP TABLE IF EXISTS liboliphaunt_event_off; " + "DROP TABLE IF EXISTS liboliphaunt_event_log; " + "CREATE TABLE liboliphaunt_event_log(tag text NOT NULL); " + "CREATE FUNCTION liboliphaunt_record_event() RETURNS event_trigger " + "LANGUAGE plpgsql AS $$ BEGIN " + "INSERT INTO liboliphaunt_event_log VALUES (TG_TAG); " + "END $$; " + "CREATE EVENT TRIGGER liboliphaunt_ddl_end ON ddl_command_end " + "EXECUTE FUNCTION liboliphaunt_record_event(); " + "SET event_triggers = off; " + "CREATE TABLE liboliphaunt_event_off(value integer); " + "SET event_triggers = on; " + "CREATE TABLE liboliphaunt_event_on(value integer); " + "SELECT count(*)::text || ':' || min(tag) FROM liboliphaunt_event_log; " + "DROP EVENT TRIGGER liboliphaunt_ddl_end; " + "DROP FUNCTION liboliphaunt_record_event(); " + "DROP TABLE liboliphaunt_event_on, liboliphaunt_event_off, liboliphaunt_event_log", + "1:CREATE TABLE"); +} + static int file_exists(const char *path) { struct stat st; return stat(path, &st) == 0 && S_ISREG(st.st_mode); @@ -2741,6 +2817,14 @@ static int run_cycle(const char *pgdata, const char *runtime_dir) { return 1; } + if (exec_simple_query_expect_bytes( + db, + "SELECT 'native-default-fsync-' || current_setting('fsync')", + "native-default-fsync-off") != 0) { + oliphaunt_close(db); + return 1; + } + if (exec_query_expect_bytes( db, "SELECT CASE WHEN " @@ -2817,6 +2901,11 @@ static int run_cycle(const char *pgdata, const char *runtime_dir) { return 1; } + if (exec_event_trigger_smoke(db) != 0) { + oliphaunt_close(db); + return 1; + } + if (exec_static_extension_registry_smoke(db) != 0) { oliphaunt_close(db); return 1; @@ -2961,6 +3050,13 @@ int main(int argc, char **argv) { return 1; } +#ifndef _WIN32 + struct sigaction previous_sigusr1; + if (install_host_sigusr1_sentinel(&previous_sigusr1) != 0) { + return 1; + } +#endif + const char *host_pgdata = "/tmp/oliphaunt-host-pgdata-sentinel"; if (set_pgdata_env_for_smoke(host_pgdata) != 0) { return 1; @@ -2971,6 +3067,11 @@ int main(int argc, char **argv) { if (expect_pgdata_env("oliphaunt_close", host_pgdata) != 0) { return 1; } +#ifndef _WIN32 + if (verify_host_sigusr1_sentinel(&previous_sigusr1) != 0) { + return 1; + } +#endif if (expect_terminal_shutdown_reopen_rejected(argv[1], argv[2]) != 0) { return 1; } diff --git a/src/native/runtime/smoke/liboliphaunt_stream_queue.c b/src/native/runtime/smoke/liboliphaunt_stream_queue.c index 834207666..b80fb9433 100644 --- a/src/native/runtime/smoke/liboliphaunt_stream_queue.c +++ b/src/native/runtime/smoke/liboliphaunt_stream_queue.c @@ -6,6 +6,11 @@ enum { QUEUE_LIMIT = 65536, INPUT_SIZE = QUEUE_LIMIT * 9 + 17 }; static unsigned char input[INPUT_SIZE]; +/* This queue-only fixture has no PostgreSQL backend or armed deadline. */ +void RequestTrustedEmbeddedTimeoutCheck(void) { + assert(!"an unarmed queue fixture must not notify PostgreSQL of a timeout"); +} + static void *produce(void *context) { OliphauntHandle *handle = context; pthread_mutex_lock(&handle->mutex); diff --git a/src/native/runtime/src/liboliphaunt_archive_tar.c b/src/native/runtime/src/liboliphaunt_archive_tar.c index 5eb6c143b..9841f9b29 100644 --- a/src/native/runtime/src/liboliphaunt_archive_tar.c +++ b/src/native/runtime/src/liboliphaunt_archive_tar.c @@ -25,6 +25,9 @@ #define O_BINARY 0 #endif +/* File-read scratch space; the complete returned archive still grows in memory. */ +#define ARCHIVE_FILE_READ_CHUNK_BYTES (64 * 1024) + static int buffer_reserve(OliphauntByteBuffer *buffer, size_t additional) { if (additional > SIZE_MAX - buffer->len) { return -1; @@ -553,7 +556,7 @@ static int tar_append_file_contents(OliphauntByteBuffer *archive, OliphauntHandl set_error(handle, message); return -1; } - uint8_t chunk[64 * 1024]; + uint8_t chunk[ARCHIVE_FILE_READ_CHUNK_BYTES]; size_t remaining = size; while (remaining > 0) { size_t take = remaining < sizeof(chunk) ? remaining : sizeof(chunk); diff --git a/src/native/runtime/src/liboliphaunt_config.c b/src/native/runtime/src/liboliphaunt_config.c index 4bc32298f..72d36ee0f 100644 --- a/src/native/runtime/src/liboliphaunt_config.c +++ b/src/native/runtime/src/liboliphaunt_config.c @@ -25,7 +25,7 @@ static bool config_string_matches( const char *actual, const char *requested, const char *fallback) { - const char *expected = requested != NULL ? requested : fallback; + const char *expected = requested != NULL && requested[0] != '\0' ? requested : fallback; return strcmp(actual != NULL ? actual : "", expected != NULL ? expected : "") == 0; } diff --git a/src/native/runtime/src/liboliphaunt_internal.h b/src/native/runtime/src/liboliphaunt_internal.h index 92e75fe52..7bcc6e00c 100644 --- a/src/native/runtime/src/liboliphaunt_internal.h +++ b/src/native/runtime/src/liboliphaunt_internal.h @@ -8,10 +8,16 @@ #include #include #include +#include #define OLIPHAUNT_ICU_DATA_DIR_ENV "OLIPHAUNT_ICU_DATA_DIR" #define OLIPHAUNT_EMBEDDED_MODULE_DIR_ENV "OLIPHAUNT_EMBEDDED_MODULE_DIR" #define OLIPHAUNT_ERROR_CAPACITY 1024 +#define OLIPHAUNT_EMBEDDED_IO_ABI_VERSION 1U + +#define OLIPHAUNT_EMBEDDED_WAKE_NONE 0U +#define OLIPHAUNT_EMBEDDED_WAKE_POSIX_FD 1U +#define OLIPHAUNT_EMBEDDED_WAKE_WIN32_EVENT 2U /* * Every fallible public C operation installs one of these stack-owned scopes. @@ -44,9 +50,13 @@ void oliphaunt_error_capture_current( bool failed); typedef struct OliphauntEmbeddedIO { + uint32_t abi_version; + uint32_t struct_size; void *context; - ssize_t (*read)(void *context, void *ptr, size_t len); + ssize_t (*read)(void *context, void *ptr, size_t len, long timeout_ms); ssize_t (*write)(void *context, const void *ptr, size_t len); + int (*set_timeout)(void *context, long timeout_ms); + int (*set_interrupt_wakeup)(void *context, uint32_t kind, uintptr_t token); } OliphauntEmbeddedIO; typedef struct OliphauntOutputChunk { @@ -163,6 +173,7 @@ struct OliphauntHandle { bool thread_started; bool backend_exited; int backend_status; + int cwd_capture_errno; pthread_mutex_t mutex; pthread_mutex_t error_mutex; @@ -179,6 +190,16 @@ struct OliphauntHandle { size_t input_off; size_t input_cap; + /* PostgreSQL's nearest cooperative timeout, guarded by mutex. */ + bool embedded_timeout_armed; + struct timespec embedded_timeout_deadline; + uint64_t embedded_timeout_generation; + uint64_t embedded_timeout_notified_generation; + + /* PostgreSQL-owned raw wake endpoint, guarded by mutex. */ + uint32_t interrupt_wakeup_kind; + uintptr_t interrupt_wakeup_token; + unsigned char *output; size_t output_len; size_t output_cap; @@ -311,8 +332,11 @@ uint64_t oliphaunt_elapsed_ns(uint64_t started_ns); void oliphaunt_reset_trace_locked(OliphauntHandle *handle, size_t request_len); void oliphaunt_print_trace_locked(OliphauntHandle *handle, uint64_t total_ns); -ssize_t oliphaunt_embedded_read(void *context, void *ptr, size_t len); +ssize_t oliphaunt_embedded_read(void *context, void *ptr, size_t len, long timeout_ms); ssize_t oliphaunt_embedded_write(void *context, const void *ptr, size_t len); +int oliphaunt_embedded_set_timeout(void *context, long timeout_ms); +int oliphaunt_embedded_set_interrupt_wakeup(void *context, uint32_t kind, uintptr_t token); +int oliphaunt_wake_backend_locked(OliphauntHandle *handle); int oliphaunt_set_input_locked(OliphauntHandle *handle, const void *buf, size_t len); int oliphaunt_startup_timeout_ms(void); int oliphaunt_wait_for_ready_locked(OliphauntHandle *handle, int timeout_ms); diff --git a/src/native/runtime/src/liboliphaunt_native.c b/src/native/runtime/src/liboliphaunt_native.c index 6b6dd6e09..b7ab45cdc 100644 --- a/src/native/runtime/src/liboliphaunt_native.c +++ b/src/native/runtime/src/liboliphaunt_native.c @@ -6,7 +6,6 @@ #include "liboliphaunt_internal.h" #include -#include #include #include #include @@ -24,14 +23,10 @@ extern int oliphaunt_embedded_main( char **argv, const char *dbname, const char *username, - OliphauntEmbeddedIO *io); + OliphauntEmbeddedIO *io, + int *cwd_capture_errno); -typedef struct Latch Latch; - -extern volatile sig_atomic_t InterruptPending; -extern volatile sig_atomic_t QueryCancelPending; -extern Latch *MyLatch; -extern void SetLatch(Latch *latch); +extern void RequestTrustedEmbeddedQueryCancel(void); static int32_t close_unpublished_handle(OliphauntHandle *handle); @@ -388,16 +383,19 @@ static void *backend_thread_main(void *arg) { return NULL; } + int cwd_capture_errno = 0; int rc = oliphaunt_embedded_main( backend_argv.argc, backend_argv.argv, handle->database, handle->username, - &handle->io); + &handle->io, + &cwd_capture_errno); restore_backend_runtime_env(handle); oliphaunt_free_backend_argv(&backend_argv); pthread_mutex_lock(&handle->mutex); + handle->cwd_capture_errno = cwd_capture_errno; handle->backend_status = rc; handle->backend_exited = true; handle->closing = true; @@ -414,9 +412,13 @@ static int start_backend(OliphauntHandle *handle) { return -1; } + handle->io.abi_version = OLIPHAUNT_EMBEDDED_IO_ABI_VERSION; + handle->io.struct_size = (uint32_t)sizeof(handle->io); handle->io.context = handle; handle->io.read = oliphaunt_embedded_read; handle->io.write = oliphaunt_embedded_write; + handle->io.set_timeout = oliphaunt_embedded_set_timeout; + handle->io.set_interrupt_wakeup = oliphaunt_embedded_set_interrupt_wakeup; pthread_attr_t attr; int rc = pthread_attr_init(&attr); @@ -452,6 +454,17 @@ static int start_backend(OliphauntHandle *handle) { pthread_mutex_lock(&handle->mutex); rc = oliphaunt_wait_for_ready_locked(handle, oliphaunt_startup_timeout_ms()); + if (rc != 0 && handle->cwd_capture_errno != 0) { + /* Capture happens before PostgreSQL can emit a protocol error. Keep + * its cause confined to startup, not ordinary query readiness waits. */ + char message[OLIPHAUNT_ERROR_CAPACITY]; + snprintf( + message, + sizeof(message), + "could not retain caller working directory: %s", + strerror(handle->cwd_capture_errno)); + set_error(handle, message); + } if (rc == 0) { handle->output_len = 0; handle->output_scan_off = 0; @@ -630,6 +643,22 @@ int32_t oliphaunt_init_with_error( return run_init_operation(config, out, error, true); } +static int32_t reset_session_command(OliphauntHandle *handle, const char *sql) { + OliphauntResponse response = {0}; + int32_t rc = oliphaunt_exec_simple_query(handle, sql, strlen(sql), &response); + if (rc == 0) { + bool tag_matches = false; + if (!oliphaunt_response_confirms_command( + response.data, response.len, sql, &tag_matches) || + !tag_matches || handle->transaction_status != 'I') { + set_error(handle, "native session reset was not confirmed; logical handle remains active"); + rc = -1; + } + } + oliphaunt_free_response(&response); + return rc; +} + static int32_t oliphaunt_detach_impl(OliphauntHandle *handle) { if (handle == NULL) { return 0; @@ -658,19 +687,13 @@ static int32_t oliphaunt_detach_impl(OliphauntHandle *handle) { if (can_reset) { if (in_transaction) { - OliphauntResponse response = {0}; - static const char rollback_sql[] = "ROLLBACK"; - int32_t rc = oliphaunt_exec_simple_query(handle, rollback_sql, sizeof(rollback_sql) - 1, &response); - oliphaunt_free_response(&response); + int32_t rc = reset_session_command(handle, "ROLLBACK"); if (rc != 0) { return rc; } } - OliphauntResponse response = {0}; - static const char discard_sql[] = "DISCARD ALL"; - int32_t rc = oliphaunt_exec_simple_query(handle, discard_sql, sizeof(discard_sql) - 1, &response); - oliphaunt_free_response(&response); + int32_t rc = reset_session_command(handle, "DISCARD ALL"); if (rc != 0) { return rc; } @@ -883,15 +906,12 @@ static int32_t oliphaunt_cancel_impl(OliphauntHandle *handle) { return -1; } - InterruptPending = true; - QueryCancelPending = true; - if (MyLatch != NULL) { - SetLatch(MyLatch); - } + RequestTrustedEmbeddedQueryCancel(); + int wake_rc = oliphaunt_wake_backend_locked(handle); pthread_cond_broadcast(&handle->input_cond); pthread_cond_broadcast(&handle->output_cond); pthread_mutex_unlock(&handle->mutex); - return 0; + return wake_rc; } int32_t oliphaunt_cancel(OliphauntHandle *handle) { diff --git a/src/native/runtime/src/liboliphaunt_protocol.c b/src/native/runtime/src/liboliphaunt_protocol.c index f7aa60f2c..133876bf5 100644 --- a/src/native/runtime/src/liboliphaunt_protocol.c +++ b/src/native/runtime/src/liboliphaunt_protocol.c @@ -13,9 +13,23 @@ #include #include #include +#ifndef _WIN32 +#include +#endif #define DEFAULT_STARTUP_TIMEOUT_MS 60000 +/* Hard queue bound, not preallocation or a total-response limit. Producers + * split larger writes and wait for space as the consumer drains the queue. */ #define DEFAULT_STREAM_QUEUE_MAX_BYTES (4 * 1024 * 1024) +/* First buffered response allocation; grows on demand, unlike the queue bound. */ +#define INITIAL_BUFFERED_OUTPUT_BYTES (8 * 1024) + +extern void RequestTrustedEmbeddedTimeoutCheck(void); +extern bool TrustedEmbeddedInterruptPending(void); + +static int wait_for_output_locked( + OliphauntHandle *handle, + const struct timespec *caller_deadline); static int validate_frontend_protocol_frames(OliphauntHandle *handle, const uint8_t *request, size_t request_len) { if (request_len == 0) { @@ -52,7 +66,7 @@ static int append_output_locked(OliphauntHandle *handle, const void *buf, size_t } size_t required = handle->output_len + len; if (required > handle->output_cap) { - size_t next = handle->output_cap ? handle->output_cap : 8192; + size_t next = handle->output_cap ? handle->output_cap : INITIAL_BUFFERED_OUTPUT_BYTES; while (next < required) { next *= 2; } @@ -179,12 +193,8 @@ static int wait_for_stream_queue_room_locked(OliphauntHandle *handle, size_t len errno = EPIPE; return -1; } - int rc = pthread_cond_wait(&handle->output_cond, &handle->mutex); - if (rc != 0) { - char message[OLIPHAUNT_ERROR_CAPACITY]; - snprintf(message, sizeof(message), "stream queue wait failed: %d", rc); - set_error(handle, message); - errno = rc; + int rc = wait_for_output_locked(handle, NULL); + if (rc < 0) { return -1; } } @@ -333,7 +343,7 @@ int oliphaunt_startup_timeout_ms(void) { return parsed > 0 ? parsed : DEFAULT_STARTUP_TIMEOUT_MS; } -static void add_ms_to_timespec(struct timespec *ts, int ms) { +static void add_ms_to_timespec(struct timespec *ts, long ms) { ts->tv_sec += ms / 1000; ts->tv_nsec += (long)(ms % 1000) * 1000000L; if (ts->tv_nsec >= 1000000000L) { @@ -342,11 +352,141 @@ static void add_ms_to_timespec(struct timespec *ts, int ms) { } } +static int compare_timespec(const struct timespec *left, const struct timespec *right) { + if (left->tv_sec < right->tv_sec) { + return -1; + } + if (left->tv_sec > right->tv_sec) { + return 1; + } + if (left->tv_nsec < right->tv_nsec) { + return -1; + } + if (left->tv_nsec > right->tv_nsec) { + return 1; + } + return 0; +} + +static int realtime_now_locked(OliphauntHandle *handle, struct timespec *now) { + if (clock_gettime(CLOCK_REALTIME, now) == 0) { + return 0; + } + int saved_errno = errno; + char message[OLIPHAUNT_ERROR_CAPACITY]; + snprintf(message, sizeof(message), "clock_gettime(CLOCK_REALTIME) failed: %d", saved_errno); + set_error(handle, message); + errno = saved_errno; + return -1; +} + +static int notify_embedded_timeout_locked(OliphauntHandle *handle) { + if (!handle->embedded_timeout_armed || + handle->embedded_timeout_notified_generation == handle->embedded_timeout_generation) { + return 0; + } + + /* Publish the notification before waking the owning backend thread. */ + handle->embedded_timeout_notified_generation = handle->embedded_timeout_generation; + RequestTrustedEmbeddedTimeoutCheck(); + return oliphaunt_wake_backend_locked(handle) == 0 ? 1 : -1; +} + +static int notify_due_embedded_timeout_locked( + OliphauntHandle *handle, + const struct timespec *now) { + if (!handle->embedded_timeout_armed || + compare_timespec(now, &handle->embedded_timeout_deadline) < 0) { + return 0; + } + return notify_embedded_timeout_locked(handle); +} + +static int check_embedded_timeout_deadline_locked(OliphauntHandle *handle) { + if (!handle->embedded_timeout_armed || + handle->embedded_timeout_notified_generation == handle->embedded_timeout_generation) { + return 0; + } + + struct timespec now; + if (realtime_now_locked(handle, &now) != 0) { + return -1; + } + return notify_due_embedded_timeout_locked(handle, &now) < 0 ? -1 : 0; +} + +/* + * Wait for backend output while also relaying PostgreSQL's nearest cooperative + * timeout. caller_deadline is an independent API-level deadline (currently + * startup only); returning 1 means that deadline expired, not a PG timeout. + */ +static int wait_for_output_locked( + OliphauntHandle *handle, + const struct timespec *caller_deadline) { + bool pg_deadline_pending = + handle->embedded_timeout_armed && + handle->embedded_timeout_notified_generation != handle->embedded_timeout_generation; + struct timespec now; + + if (caller_deadline != NULL || pg_deadline_pending) { + if (realtime_now_locked(handle, &now) != 0) { + return -1; + } + int pg_timeout_notification = + pg_deadline_pending ? notify_due_embedded_timeout_locked(handle, &now) : 0; + if (pg_timeout_notification < 0) { + return -1; + } + if (caller_deadline != NULL && compare_timespec(&now, caller_deadline) >= 0) { + return 1; + } + if (pg_timeout_notification > 0) { + return 0; + } + } + + struct timespec wait_deadline_storage; + const struct timespec *wait_deadline = caller_deadline; + if (pg_deadline_pending && + (wait_deadline == NULL || + compare_timespec(&handle->embedded_timeout_deadline, wait_deadline) < 0)) { + wait_deadline_storage = handle->embedded_timeout_deadline; + wait_deadline = &wait_deadline_storage; + } + + int rc = wait_deadline != NULL + ? pthread_cond_timedwait(&handle->output_cond, &handle->mutex, wait_deadline) + : pthread_cond_wait(&handle->output_cond, &handle->mutex); + if (rc == 0) { + return 0; + } + if (rc != ETIMEDOUT) { + char message[OLIPHAUNT_ERROR_CAPACITY]; + snprintf(message, sizeof(message), "pthread wait failed: %d", rc); + set_error(handle, message); + errno = rc; + return -1; + } + + if (realtime_now_locked(handle, &now) != 0) { + return -1; + } + if (notify_due_embedded_timeout_locked(handle, &now) < 0) { + return -1; + } + if (caller_deadline != NULL && compare_timespec(&now, caller_deadline) >= 0) { + return 1; + } + return 0; +} + int oliphaunt_wait_for_ready_locked(OliphauntHandle *handle, int timeout_ms) { bool has_timeout = timeout_ms > 0; struct timespec deadline; if (has_timeout) { - clock_gettime(CLOCK_REALTIME, &deadline); + if (realtime_now_locked(handle, &deadline) != 0) { + return -1; + } add_ms_to_timespec(&deadline, timeout_ms); } @@ -369,10 +509,8 @@ int oliphaunt_wait_for_ready_locked(OliphauntHandle *handle, int timeout_ms) { set_error(handle, "native backend is closing before ReadyForQuery"); return -1; } - int rc = has_timeout - ? pthread_cond_timedwait(&handle->output_cond, &handle->mutex, &deadline) - : pthread_cond_wait(&handle->output_cond, &handle->mutex); - if (has_timeout && rc == ETIMEDOUT) { + int rc = wait_for_output_locked(handle, has_timeout ? &deadline : NULL); + if (rc > 0) { char message[OLIPHAUNT_ERROR_CAPACITY]; snprintf( message, @@ -382,21 +520,175 @@ int oliphaunt_wait_for_ready_locked(OliphauntHandle *handle, int timeout_ms) { set_error(handle, message); return -1; } - if (rc != 0) { - char message[OLIPHAUNT_ERROR_CAPACITY]; - snprintf(message, sizeof(message), "pthread wait failed: %d", rc); - set_error(handle, message); + if (rc < 0) { return -1; } } return 0; } -ssize_t oliphaunt_embedded_read(void *context, void *ptr, size_t len) { +int oliphaunt_embedded_set_timeout(void *context, long timeout_ms) { + OliphauntHandle *handle = (OliphauntHandle *)context; + struct timespec deadline = {0, 0}; + + if (timeout_ms >= 0) { + if (clock_gettime(CLOCK_REALTIME, &deadline) != 0) { + return -1; + } + add_ms_to_timespec(&deadline, timeout_ms); + } + + pthread_mutex_lock(&handle->mutex); + if (handle->interrupt_wakeup_kind == OLIPHAUNT_EMBEDDED_WAKE_NONE) { + pthread_mutex_unlock(&handle->mutex); + errno = ENODEV; + return -1; + } + handle->embedded_timeout_generation++; + if (handle->embedded_timeout_generation == 0) { + handle->embedded_timeout_generation = 1; + } + handle->embedded_timeout_armed = timeout_ms >= 0; + if (handle->embedded_timeout_armed) { + handle->embedded_timeout_deadline = deadline; + } + pthread_cond_broadcast(&handle->output_cond); + pthread_mutex_unlock(&handle->mutex); + return 0; +} + +int oliphaunt_embedded_set_interrupt_wakeup(void *context, uint32_t kind, uintptr_t token) { + OliphauntHandle *handle = (OliphauntHandle *)context; + + if ((kind == OLIPHAUNT_EMBEDDED_WAKE_NONE && token != 0) || +#ifdef _WIN32 + (kind != OLIPHAUNT_EMBEDDED_WAKE_NONE && + (kind != OLIPHAUNT_EMBEDDED_WAKE_WIN32_EVENT || token == 0)) +#else + (kind != OLIPHAUNT_EMBEDDED_WAKE_NONE && + (kind != OLIPHAUNT_EMBEDDED_WAKE_POSIX_FD || token > INT_MAX)) +#endif + ) { + errno = EINVAL; + return -1; + } + + pthread_mutex_lock(&handle->mutex); + if (kind != OLIPHAUNT_EMBEDDED_WAKE_NONE && + handle->interrupt_wakeup_kind != OLIPHAUNT_EMBEDDED_WAKE_NONE && + (handle->interrupt_wakeup_kind != kind || + handle->interrupt_wakeup_token != token)) { + pthread_mutex_unlock(&handle->mutex); + errno = EBUSY; + return -1; + } + handle->interrupt_wakeup_kind = kind; + handle->interrupt_wakeup_token = token; + pthread_mutex_unlock(&handle->mutex); + return 0; +} + +int oliphaunt_wake_backend_locked(OliphauntHandle *handle) { + int saved_errno = errno; + + if (handle->interrupt_wakeup_kind == OLIPHAUNT_EMBEDDED_WAKE_NONE) { + set_error(handle, "trusted embedded wakeup endpoint is not published"); + errno = ENODEV; + return -1; + } +#ifdef _WIN32 + if (handle->interrupt_wakeup_kind != OLIPHAUNT_EMBEDDED_WAKE_WIN32_EVENT || + !SetEvent((HANDLE)handle->interrupt_wakeup_token)) { + set_error(handle, "failed to signal trusted embedded Windows wakeup event"); + errno = EIO; + return -1; + } +#else + if (handle->interrupt_wakeup_kind != OLIPHAUNT_EMBEDDED_WAKE_POSIX_FD) { + set_error(handle, "trusted embedded wakeup endpoint has the wrong platform kind"); + errno = EINVAL; + return -1; + } + char byte = 0; + for (;;) { + ssize_t rc = write((int)handle->interrupt_wakeup_token, &byte, 1); + if (rc == 1) { + break; + } + if (rc < 0 && errno == EINTR) { + continue; + } + if (rc < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) { + break; + } + char message[OLIPHAUNT_ERROR_CAPACITY]; + snprintf(message, sizeof(message), "failed to signal trusted embedded wakeup pipe: %d", errno); + set_error(handle, message); + return -1; + } +#endif + errno = saved_errno; + return 0; +} + +ssize_t oliphaunt_embedded_read(void *context, void *ptr, size_t len, long timeout_ms) { OliphauntHandle *handle = (OliphauntHandle *)context; pthread_mutex_lock(&handle->mutex); - while (handle->input_off >= handle->input_len && !handle->closing) { - pthread_cond_wait(&handle->input_cond, &handle->mutex); + if (handle->input_off >= handle->input_len && !handle->closing) { + if (TrustedEmbeddedInterruptPending()) { + pthread_mutex_unlock(&handle->mutex); + errno = EINTR; + return -1; + } + + const bool has_timeout = timeout_ms >= 0; + struct timespec deadline; + + if (has_timeout) { + /* A spurious wake must not extend PostgreSQL's published delay. */ + if (clock_gettime(CLOCK_REALTIME, &deadline) != 0) { + pthread_mutex_unlock(&handle->mutex); + errno = EINTR; + return -1; + } + add_ms_to_timespec(&deadline, timeout_ms); + } + + while (handle->input_off >= handle->input_len && + !handle->closing && + !TrustedEmbeddedInterruptPending()) { + int wait_rc = has_timeout + ? pthread_cond_timedwait( + &handle->input_cond, + &handle->mutex, + &deadline) + : pthread_cond_wait(&handle->input_cond, &handle->mutex); + + if (wait_rc == 0) { + /* Recheck every predicate after a real or spurious wake. */ + continue; + } + if (wait_rc == ETIMEDOUT) { + if (handle->embedded_timeout_armed) { + handle->embedded_timeout_notified_generation = + handle->embedded_timeout_generation; + } + /* The timed-read value itself is authoritative. */ + RequestTrustedEmbeddedTimeoutCheck(); + (void)oliphaunt_wake_backend_locked(handle); + } + pthread_mutex_unlock(&handle->mutex); + errno = wait_rc == ETIMEDOUT ? EINTR : wait_rc; + return -1; + } + + if (handle->input_off >= handle->input_len && + !handle->closing && + TrustedEmbeddedInterruptPending()) { + pthread_mutex_unlock(&handle->mutex); + errno = EINTR; + return -1; + } } if (handle->input_off >= handle->input_len && handle->closing) { @@ -942,6 +1234,16 @@ static int32_t oliphaunt_exec_protocol_raw_stream_impl( while (status == 0) { OliphauntOutputChunk *chunk = NULL; while ((chunk = pop_stream_chunk_locked(handle)) != NULL) { + if (check_embedded_timeout_deadline_locked(handle) != 0) { + free_stream_chunk(chunk); + status = -1; + break; + } + /* + * A user callback runs without the handle mutex and may block. + * There is deliberately no watchdog thread, so check the PG + * deadline immediately before and after this unbounded region. + */ pthread_mutex_unlock(&handle->mutex); if (!callback_failed) { int32_t callback_rc = callback(callback_context, chunk->data, chunk->len); @@ -965,9 +1267,15 @@ static int32_t oliphaunt_exec_protocol_raw_stream_impl( } handle->stream_copy_input = false; } + if (check_embedded_timeout_deadline_locked(handle) != 0) { + status = -1; + break; + } + } + if (status != 0) { + break; } - if (status != 0) break; if (handle->stream_failed) { /* Embedded writes report queue/scanner failures from PostgreSQL's * backend thread. Snapshot that shared error before releasing the @@ -997,11 +1305,8 @@ static int32_t oliphaunt_exec_protocol_raw_stream_impl( break; } - int rc = pthread_cond_wait(&handle->output_cond, &handle->mutex); - if (rc != 0) { - char message[OLIPHAUNT_ERROR_CAPACITY]; - snprintf(message, sizeof(message), "pthread wait failed: %d", rc); - set_error(handle, message); + int rc = wait_for_output_locked(handle, NULL); + if (rc < 0) { status = -1; break; } @@ -1009,6 +1314,11 @@ static int32_t oliphaunt_exec_protocol_raw_stream_impl( OliphauntOutputChunk *chunk = NULL; while ((chunk = pop_stream_chunk_locked(handle)) != NULL) { + if (check_embedded_timeout_deadline_locked(handle) != 0) { + free_stream_chunk(chunk); + status = -1; + break; + } pthread_mutex_unlock(&handle->mutex); if (!callback_failed) { int32_t callback_rc = callback(callback_context, chunk->data, chunk->len); @@ -1018,6 +1328,10 @@ static int32_t oliphaunt_exec_protocol_raw_stream_impl( } free_stream_chunk(chunk); pthread_mutex_lock(&handle->mutex); + if (check_embedded_timeout_deadline_locked(handle) != 0) { + status = -1; + break; + } } handle->streaming = false; diff --git a/src/native/runtime/src/liboliphaunt_runtime.c b/src/native/runtime/src/liboliphaunt_runtime.c index 8b1a0f269..eef657aed 100644 --- a/src/native/runtime/src/liboliphaunt_runtime.c +++ b/src/native/runtime/src/liboliphaunt_runtime.c @@ -8,6 +8,8 @@ #include #include +/* Native backend thread stack, not Wasmer's execution stack or a SQL memory + * budget. OLIPHAUNT_STACK_BYTES adjusts this allocation before thread creation. */ #define DEFAULT_BACKEND_STACK_BYTES (8 * 1024 * 1024) static const char *const DEFAULT_BACKEND_ARGS[] = { diff --git a/src/native/runtime/tools/run-host-c-smoke.sh b/src/native/runtime/tools/run-host-c-smoke.sh index b59c16cdb..f51559b0a 100755 --- a/src/native/runtime/tools/run-host-c-smoke.sh +++ b/src/native/runtime/tools/run-host-c-smoke.sh @@ -103,11 +103,13 @@ printf 'liboliphaunt host target: %s\nliboliphaunt work root: %s\n' "$target" "$ msvc_env_file='' cluster_root='' +process_root='' remove_icu=0 smoke_failure_root='' cleanup() { [ -z "$msvc_env_file" ] || rm -f "$msvc_env_file" [ -z "$cluster_root" ] || rm -rf "$cluster_root" + [ -z "$process_root" ] || rm -rf "$process_root" [ "$remove_icu" = 0 ] || rm -rf "$install_dir/share/icu" [ -z "$smoke_failure_root" ] || printf 'native smoke root: %s\n' "$smoke_failure_root" >&2 return 0 @@ -207,6 +209,27 @@ require_file "$archive_fixture" for _attempt in 1 2; do "$bin_dir/liboliphaunt_smoke$exe_suffix" "$(native_path "$smoke_root/pgdata")" "$(native_path "$install_dir")" "$(native_path "$archive_fixture")" done +# Process-global signal and cwd probes require Linux and run in separate +# processes because terminal Direct close consumes that process's backend. +if [ "$platform" = linux ]; then + module_dir="${OLIPHAUNT_EMBEDDED_MODULE_DIR:-$library_dir/modules}" + compile smoke liboliphaunt_signal_boundary + compile smoke liboliphaunt_cwd_boundary + process_root="$(mktemp -d "$work_root/process-boundary.XXXXXX")" + mkdir "$process_root/database" + cp -R "$smoke_root/pgdata" "$process_root/database/pgdata" + bun src/native/runtime/tools/native-smoke-data.mts managed-root "$process_root/database" + "$bin_dir/liboliphaunt_signal_boundary" "$process_root/database/pgdata" "$install_dir" "$module_dir" + for mode in direct startup-fatal early-startup-fatal unresolvable-cwd search-only-cwd renamed-cwd; do + probe_cwd="$(mktemp -d "$process_root/cwd.XXXXXX")" + cwd_args=() + case "$mode" in + unresolvable-cwd | search-only-cwd) cwd_args=("$probe_cwd") ;; + renamed-cwd) cwd_args=("$probe_cwd" "$probe_cwd.renamed") ;; + esac + (cd "$probe_cwd" && "$bin_dir/liboliphaunt_cwd_boundary" "$mode" "$process_root/database/pgdata" "$install_dir" "$module_dir" ${cwd_args[@]+"${cwd_args[@]}"}) + done +fi smoke_failure_root='' [ -n "$root_arg" ] || rm -rf "$smoke_root" [ "$cluster_seeds" = 1 ] || exit 0 @@ -244,4 +267,11 @@ for _attempt in 1 2; do "$bin_dir/liboliphaunt_cluster_seed_smoke$exe_suffix" "$(native_path "$cluster_root/standard-icu-import/pgdata")" "$(native_path "$install_dir")" \ "SELECT pg_import_system_collations('pg_catalog'); SELECT CASE WHEN EXISTS (SELECT 1 FROM pg_collation WHERE collname LIKE '%-x-icu') THEN 'OLIPHAUNT_ICU_IMPORT_OK' ELSE 'OLIPHAUNT_ICU_IMPORT_MISSING' END" OLIPHAUNT_ICU_IMPORT_OK done +for fsync in on off; do + durability_args=() + [ "$fsync" != on ] || durability_args=(fsync=on) + ICU_DATA=/ambient/unverified-icu OLIPHAUNT_INTERNAL_SKIP_SYSTEM_COLLATION_DISCOVERY=1 \ + "$bin_dir/liboliphaunt_cluster_seed_smoke$exe_suffix" "$(native_path "$cluster_root/standard-icu-import/pgdata")" "$(native_path "$install_dir")" \ + "SELECT 'native-fsync-' || current_setting('fsync')" "native-fsync-$fsync" ${durability_args[@]+"${durability_args[@]}"} +done printf 'native standard and ICU cluster seeds passed open, catalog, close, and reopen qualification\n' >&2 diff --git a/src/native/runtime/tools/test-generation-lifecycle.sh b/src/native/runtime/tools/test-generation-lifecycle.sh index 4fdd41242..fe028782c 100755 --- a/src/native/runtime/tools/test-generation-lifecycle.sh +++ b/src/native/runtime/tools/test-generation-lifecycle.sh @@ -31,6 +31,7 @@ cc \ "$source_root/src/liboliphaunt_backup_state.c" \ "$source_root/src/liboliphaunt_config.c" \ "$source_root/src/liboliphaunt_process.c" \ + "$source_root/src/liboliphaunt_runtime.c" \ "$source_root/smoke/liboliphaunt_generation_lifecycle.c" \ ${platform_lib:+"$platform_lib"} \ -o "$work_root/generation-lifecycle" diff --git a/src/native/runtime/tools/test-session-reset.mjs b/src/native/runtime/tools/test-session-reset.mjs new file mode 100644 index 000000000..643a0ce9c --- /dev/null +++ b/src/native/runtime/tools/test-session-reset.mjs @@ -0,0 +1,103 @@ +#!/usr/bin/env node +// Compile the actual session-reset helper with fault-injected protocol replies. +import assert from 'node:assert/strict'; +import { execFileSync } from 'node:child_process'; +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +if (process.platform === 'win32') { + console.log('SKIP: session-reset fixture requires a POSIX C compiler'); + process.exit(0); +} + +const native = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); +const host = readFileSync(path.join(native, 'src/liboliphaunt_native.c'), 'utf8'); +function between(source, start, end) { + const begin = source.indexOf(start); + const finish = source.indexOf(end, begin + start.length); + assert(begin >= 0 && finish > begin, `missing source boundary: ${start}`); + return source.slice(begin, finish); +} +const reset = between( + host, + 'static int32_t reset_session_command(', + 'static int32_t oliphaunt_detach_impl(', +); +const scratch = mkdtempSync(path.join(os.tmpdir(), 'oliphaunt-native-cleanup-')); +function check(name, source, extra = []) { + const file = path.join(scratch, `${name}.c`); + const executable = path.join(scratch, name); + writeFileSync(file, source); + execFileSync( + process.env.CC || 'cc', + [ + '-std=c11', + '-D_POSIX_C_SOURCE=200809L', + '-O2', + '-Wall', + '-Wextra', + '-Werror', + '-pthread', + '-I', + path.join(native, 'src'), + file, + ...extra, + '-o', + executable, + ], + { stdio: 'inherit' }, + ); + execFileSync(executable, [], { stdio: 'inherit' }); +} +try { + check( + 'session-reset', + ` +#include "liboliphaunt_internal.h" +#include +#include +#include +static int failure; +void oliphaunt_set_error(OliphauntHandle *h, const char *s) { + snprintf(h->last_error, sizeof(h->last_error), "%s", s); +} +int32_t oliphaunt_exec_simple_query(OliphauntHandle *h, const char *sql, + size_t len, OliphauntResponse *out) { + if (failure == 1) { oliphaunt_set_error(h, "transport failed"); return -1; } + size_t body = len + 1; + out->len = 5 + body + 6; + out->data = calloc(1, out->len); + assert(out->data); + out->data[0] = failure == 2 ? 'E' : 'C'; + out->data[4] = (unsigned char)(body + 4); + memcpy(out->data + 5, sql, body); + if (failure == 3) out->data[5] = '?'; + memcpy(out->data + 5 + body, "Z\\0\\0\\0\\5I", 6); + h->transaction_status = failure == 4 ? 'E' : 'I'; + return 0; +} +void oliphaunt_free_response(OliphauntResponse *response) { + free(response->data); response->data = NULL; response->len = 0; +} +${reset} +int main(void) { + OliphauntHandle h = {0}; + const char *commands[] = {"ROLLBACK", "DISCARD ALL"}; + for (size_t i = 0; i < 2; i++) { + for (failure = 0; failure <= 4; failure++) { + h.last_error[0] = 0; + assert((reset_session_command(&h, commands[i]) == 0) == (failure == 0)); + if (failure) assert(h.last_error[0]); + if (failure == 1) assert(strcmp(h.last_error, "transport failed") == 0); + } + } + puts("native reset command validation passed"); +} +`, + [path.join(native, 'src/liboliphaunt_backup_state.c')], + ); +} finally { + rmSync(scratch, { recursive: true, force: true }); +} diff --git a/src/native/sdks/rust/README.md b/src/native/sdks/rust/README.md index 1fa813383..36134fd1b 100644 --- a/src/native/sdks/rust/README.md +++ b/src/native/sdks/rust/README.md @@ -29,6 +29,8 @@ fn main() -> Result<(), Box> { Default storage is a disposable temporary directory. Direct mode stays bound to its first root and configuration for the lifetime of the process, even after closing a handle. Use the quickstart's persistent-storage example for application data; run it as an alternative to this disposable example. Always close database handles explicitly. +Direct and broker modes retain the native `fsync=off` default. Persistent storage alone does not provide crash safety; set `.startup_guc("fsync", "on")` when it is required. Changing the default is tracked separately from patch consolidation. + ## Build your integration - [Guide](https://oliphaunt.dev/docs/sdk/rust/guide): parameters, transactions, extensions, backups, and shutdown. diff --git a/src/native/sdks/rust/tests/native_smoke.rs b/src/native/sdks/rust/tests/native_smoke.rs index d2feee679..aa5cd8b60 100644 --- a/src/native/sdks/rust/tests/native_smoke.rs +++ b/src/native/sdks/rust/tests/native_smoke.rs @@ -515,6 +515,116 @@ fn verify_direct_database(root: &Path) -> Result<(), Box> Ok(database.close()?) } +#[test] +fn broker_preserves_configured_identity_and_session_policy_when_available() { + if std::env::var_os("LIBOLIPHAUNT_PATH").is_none() + || std::env::var_os("OLIPHAUNT_BROKER").is_none() + { + eprintln!("skipping installed broker identity smoke: native library/broker is unset"); + return; + } + let root = unique_root("native-broker-identity"); + let result = std::panic::catch_unwind(|| { + let mut bootstrap = DirectOliphaunt::builder() + .broker() + .storage(DatabaseStorage::Directory(root.clone())) + .open() + .unwrap(); + bootstrap + .execute("CREATE DATABASE broker_identity") + .unwrap(); + bootstrap.close().unwrap(); + let open = |username: &str| { + DirectOliphaunt::builder() + .broker() + .storage(DatabaseStorage::Directory(root.clone())) + .username(username) + .database("broker_identity") + .open() + }; + let mut admin = open("postgres").unwrap(); + for sql in [ + "CREATE ROLE broker_reader LOGIN", + "CREATE ROLE broker_no_login NOLOGIN", + "ALTER ROLE broker_reader SET application_name = 'configured-broker-role'", + "ALTER DATABASE broker_identity SET default_statistics_target = 137", + "CREATE TABLE broker_login_events(who name)", + "CREATE FUNCTION broker_login_event() RETURNS event_trigger LANGUAGE plpgsql SECURITY DEFINER SET search_path = pg_catalog, public AS $$ BEGIN INSERT INTO public.broker_login_events VALUES (session_user); END $$", + "CREATE EVENT TRIGGER broker_login ON login EXECUTE FUNCTION broker_login_event()", + "GRANT SELECT ON broker_login_events TO broker_reader", + ] { + admin.execute(sql).unwrap(); + } + admin.close().unwrap(); + + for expected_logins in ["1", "2"] { + let mut database = open("broker_reader").unwrap(); + let row = database.query("SELECT current_user::text AS current_role, session_user::text AS session_role, current_database() AS database, (system_user IS NULL)::text AS no_auth_identity, current_setting('application_name') AS application_name, current_setting('default_statistics_target') AS statistics_target, (SELECT count(*)::text FROM broker_login_events WHERE who = session_user) AS logins").unwrap(); + assert_eq!( + row.get_text(0, "current_role").unwrap(), + Some("broker_reader") + ); + assert_eq!( + row.get_text(0, "session_role").unwrap(), + Some("broker_reader") + ); + assert_eq!(row.get_text(0, "no_auth_identity").unwrap(), Some("true")); + assert_eq!( + row.get_text(0, "database").unwrap(), + Some("broker_identity") + ); + assert_eq!(row.get_text(0, "statistics_target").unwrap(), Some("137")); + assert_eq!( + row.get_text(0, "application_name").unwrap(), + Some("configured-broker-role") + ); + assert_eq!(row.get_text(0, "logins").unwrap(), Some(expected_logins)); + database + .execute("SET ROLE postgres") + .expect_err("configured role must not acquire superuser rights"); + database + .execute("SET application_name = 'session-local-change'") + .unwrap(); + database.execute("DISCARD ALL").unwrap(); + assert_eq!( + database + .query("SELECT current_setting('application_name') AS value") + .unwrap() + .get_text(0, "value") + .unwrap(), + Some("configured-broker-role") + ); + database.close().unwrap(); + } + + for username in ["broker_no_login", "broker_missing_role"] { + assert!( + open(username).is_err(), + "broker accepted invalid startup role {username}" + ); + } + let mut opted_out = DirectOliphaunt::builder() + .broker() + .storage(DatabaseStorage::Directory(root.clone())) + .startup_guc("fsync", "off") + .open() + .unwrap(); + assert_eq!( + opted_out + .query("SELECT current_setting('fsync') AS value") + .unwrap() + .get_text(0, "value") + .unwrap(), + Some("off") + ); + opted_out.close().unwrap(); + }); + let _ = std::fs::remove_dir_all(root); + if let Err(payload) = result { + std::panic::resume_unwind(payload); + } +} + #[test] fn descriptorless_nonempty_root_is_rejected_without_mutation_when_available() { if std::env::var_os("LIBOLIPHAUNT_PATH").is_none() { diff --git a/src/query/rust/src/lib.rs b/src/query/rust/src/lib.rs index c8a82ae64..506f87dc5 100644 --- a/src/query/rust/src/lib.rs +++ b/src/query/rust/src/lib.rs @@ -2606,7 +2606,8 @@ fn command_tag_row_count(tag: &str) -> Option { parts.last().or(Some(command))?.parse().ok() } -fn read_backend_message(bytes: &[u8]) -> Result<(u8, &[u8], &[u8])> { +/// Split one complete backend frame from a PostgreSQL protocol response. +pub fn read_backend_message(bytes: &[u8]) -> Result<(u8, &[u8], &[u8])> { if bytes.len() < 5 { return Err(protocol("truncated backend message header")); } diff --git a/src/query/rust/src/wire.rs b/src/query/rust/src/wire.rs index ff040fea8..19bf89516 100644 --- a/src/query/rust/src/wire.rs +++ b/src/query/rust/src/wire.rs @@ -8,6 +8,8 @@ pub const SSL_REQUEST_CODE: i32 = 80_877_103; pub const GSSENC_REQUEST_CODE: i32 = 80_877_104; pub const CANCEL_REQUEST_CODE: i32 = 80_877_102; pub const PROTOCOL_3: i32 = 196_608; +// Admission limit for one complete frontend frame, including its header; no +// up-front allocation. Separate from read batches and the buffered-output cap. pub const MAX_FRONTEND_MESSAGE: usize = 128 * 1024 * 1024; #[derive(Default)] diff --git a/src/third-party/postgres/apply-series.sh b/src/third-party/postgres/apply-series.sh index 8b518777c..9a225f0e1 100644 --- a/src/third-party/postgres/apply-series.sh +++ b/src/third-party/postgres/apply-series.sh @@ -1,7 +1,9 @@ #!/usr/bin/env bash set -euo pipefail repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +[ "$#" -eq 2 ] || { echo "usage: $0 SOURCE_DIRECTORY PATCH_SERIES" >&2; exit 2; } source_dir="${1:?PostgreSQL source directory required}" +source_dir="$(cd "$source_dir" && pwd -P)" series="${2:?ordered patch series required}" while IFS= read -r entry || [ -n "$entry" ]; do case "$entry" in @@ -12,10 +14,8 @@ while IFS= read -r entry || [ -n "$entry" ]; do [ -f "$patch" ] && [ ! -L "$patch" ] || { echo "missing regular PostgreSQL patch: $patch" >&2; exit 2; } - if [ "${3:-}" = "--context-fuzz" ]; then - # The existing embedded WASIX series contains upstream-context offsets. - (cd "$source_dir" && patch --batch --forward --no-backup-if-mismatch -p1 < "$patch") - else + # Generated source trees may be nested inside the Oliphaunt checkout. + # Do not discover that enclosing Git worktree and silently skip hunks. + GIT_CEILING_DIRECTORIES="$(dirname "$source_dir")" \ git -C "$source_dir" apply --whitespace=error-all "$patch" - fi done < "$series" diff --git a/src/third-party/postgres/apply-series.test.mjs b/src/third-party/postgres/apply-series.test.mjs new file mode 100644 index 000000000..a76ce4be6 --- /dev/null +++ b/src/third-party/postgres/apply-series.test.mjs @@ -0,0 +1,40 @@ +import assert from 'node:assert/strict'; +import { spawnSync } from 'node:child_process'; +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import path from 'node:path'; +import test from 'node:test'; +import { fileURLToPath } from 'node:url'; + +const root = fileURLToPath(new URL('../../../', import.meta.url)); + +test('strict series applies inside an enclosing checkout and rejects stale context', { + skip: process.platform === 'win32', +}, () => { + // The source must be inside the real checkout to reproduce Git discovery. + const fixture = mkdtempSync(path.join(root, '.patch-series-test-')); + try { + const source = path.join(fixture, 'source'); + mkdirSync(source); + writeFileSync(path.join(source, 'probe'), 'before\n'); + const patch = path.join(fixture, 'change.patch'); + writeFileSync( + patch, + 'diff --git a/probe b/probe\n--- a/probe\n+++ b/probe\n@@ -1 +1 @@\n-before\n+after\n', + ); + const series = path.join(fixture, 'series'); + writeFileSync(series, `${path.relative(root, patch)}\n`); + const apply = () => + spawnSync( + 'bash', + [path.join(root, 'src/third-party/postgres/apply-series.sh'), source, series], + { encoding: 'utf8' }, + ); + const first = apply(); + assert.equal(first.status, 0, first.stderr); + assert.equal(readFileSync(path.join(source, 'probe'), 'utf8'), 'after\n'); + assert.notEqual(apply().status, 0, 'reapplying stale context must fail'); + assert.equal(readFileSync(path.join(source, 'probe'), 'utf8'), 'after\n'); + } finally { + rmSync(fixture, { recursive: true, force: true }); + } +}); diff --git a/src/third-party/postgres/patches/wasix/0014-oliphaunt-wasix-speed-up-hash-bytes-unaligned-loads.patch b/src/third-party/postgres/patches/wasix/0014-oliphaunt-wasix-speed-up-hash-bytes-unaligned-loads.patch deleted file mode 100644 index dd304486a..000000000 --- a/src/third-party/postgres/patches/wasix/0014-oliphaunt-wasix-speed-up-hash-bytes-unaligned-loads.patch +++ /dev/null @@ -1,183 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: speed up hash_bytes unaligned loads - -PostgreSQL's unaligned little-endian hash_bytes() path builds each uint32 from -four byte loads and shifts. On WASIX this is visible in guest CPU profiles for -hash-table-heavy query paths, and WebAssembly supports unaligned 32-bit loads. - -Use a tiny memcpy-based load helper for WASIX only. This keeps the C semantics -defined for unaligned input while giving clang/LLVM the pattern it can lower to -a single little-endian wasm load. Big-endian and non-WASIX builds keep the -upstream byte-by-byte path. ---- - src/common/hashfn.c | 114 ++++++++++++++++++++++++++++++++++++++++++++-------- - 1 file changed, 97 insertions(+), 17 deletions(-) - -diff --git a/src/common/hashfn.c b/src/common/hashfn.c -index 8a6bd816ff..31c8ac9ee8 100644 ---- a/src/common/hashfn.c -+++ b/src/common/hashfn.c -@@ -47,6 +47,17 @@ - - #define rot(x,k) pg_rotate_left32(x, k) - -+#if defined(__wasi__) && !defined(WORDS_BIGENDIAN) -+static inline uint32 -+oliphaunt_wasix_hash_load32(const unsigned char *ptr) -+{ -+ uint32 value; -+ -+ memcpy(&value, ptr, sizeof(value)); -+ return value; -+} -+#endif -+ - /*---------- - * mix -- mix 3 32-bit values reversibly. - * -@@ -265,9 +276,15 @@ hash_bytes(const unsigned char *k, int keylen) - b += (k[7] + ((uint32) k[6] << 8) + ((uint32) k[5] << 16) + ((uint32) k[4] << 24)); - c += (k[11] + ((uint32) k[10] << 8) + ((uint32) k[9] << 16) + ((uint32) k[8] << 24)); - #else /* !WORDS_BIGENDIAN */ -+#ifdef __wasi__ -+ a += oliphaunt_wasix_hash_load32(k); -+ b += oliphaunt_wasix_hash_load32(k + 4); -+ c += oliphaunt_wasix_hash_load32(k + 8); -+#else - a += (k[0] + ((uint32) k[1] << 8) + ((uint32) k[2] << 16) + ((uint32) k[3] << 24)); - b += (k[4] + ((uint32) k[5] << 8) + ((uint32) k[6] << 16) + ((uint32) k[7] << 24)); - c += (k[8] + ((uint32) k[9] << 8) + ((uint32) k[10] << 16) + ((uint32) k[11] << 24)); -+#endif - #endif /* WORDS_BIGENDIAN */ - mix(a, b, c); - k += 12; -@@ -314,6 +331,46 @@ hash_bytes(const unsigned char *k, int keylen) - /* case 0: nothing left to add */ - } - #else /* !WORDS_BIGENDIAN */ -+#ifdef __wasi__ -+ switch (len) -+ { -+ case 11: -+ c += ((uint32) k[10] << 24); -+ /* fall through */ -+ case 10: -+ c += ((uint32) k[9] << 16); -+ /* fall through */ -+ case 9: -+ c += ((uint32) k[8] << 8); -+ /* fall through */ -+ case 8: -+ /* the lowest byte of c is reserved for the length */ -+ b += oliphaunt_wasix_hash_load32(k + 4); -+ a += oliphaunt_wasix_hash_load32(k); -+ break; -+ case 7: -+ b += ((uint32) k[6] << 16); -+ /* fall through */ -+ case 6: -+ b += ((uint32) k[5] << 8); -+ /* fall through */ -+ case 5: -+ b += k[4]; -+ /* fall through */ -+ case 4: -+ a += oliphaunt_wasix_hash_load32(k); -+ break; -+ case 3: -+ a += ((uint32) k[2] << 16); -+ /* fall through */ -+ case 2: -+ a += ((uint32) k[1] << 8); -+ /* fall through */ -+ case 1: -+ a += k[0]; -+ /* case 0: nothing left to add */ -+ } -+#else - switch (len) - { - case 11: -@@ -351,6 +408,7 @@ hash_bytes(const unsigned char *k, int keylen) - a += k[0]; - /* case 0: nothing left to add */ - } -+#endif - #endif /* WORDS_BIGENDIAN */ - } - -@@ -504,9 +562,15 @@ hash_bytes_extended(const unsigned char *k, int keylen, uint64 seed) - b += (k[7] + ((uint32) k[6] << 8) + ((uint32) k[5] << 16) + ((uint32) k[4] << 24)); - c += (k[11] + ((uint32) k[10] << 8) + ((uint32) k[9] << 16) + ((uint32) k[8] << 24)); - #else /* !WORDS_BIGENDIAN */ -+#ifdef __wasi__ -+ a += oliphaunt_wasix_hash_load32(k); -+ b += oliphaunt_wasix_hash_load32(k + 4); -+ c += oliphaunt_wasix_hash_load32(k + 8); -+#else - a += (k[0] + ((uint32) k[1] << 8) + ((uint32) k[2] << 16) + ((uint32) k[3] << 24)); - b += (k[4] + ((uint32) k[5] << 8) + ((uint32) k[6] << 16) + ((uint32) k[7] << 24)); - c += (k[8] + ((uint32) k[9] << 8) + ((uint32) k[10] << 16) + ((uint32) k[11] << 24)); -+#endif - #endif /* WORDS_BIGENDIAN */ - mix(a, b, c); - k += 12; -@@ -553,6 +617,46 @@ hash_bytes_extended(const unsigned char *k, int keylen, uint64 seed) - /* case 0: nothing left to add */ - } - #else /* !WORDS_BIGENDIAN */ -+#ifdef __wasi__ -+ switch (len) -+ { -+ case 11: -+ c += ((uint32) k[10] << 24); -+ /* fall through */ -+ case 10: -+ c += ((uint32) k[9] << 16); -+ /* fall through */ -+ case 9: -+ c += ((uint32) k[8] << 8); -+ /* fall through */ -+ case 8: -+ /* the lowest byte of c is reserved for the length */ -+ b += oliphaunt_wasix_hash_load32(k + 4); -+ a += oliphaunt_wasix_hash_load32(k); -+ break; -+ case 7: -+ b += ((uint32) k[6] << 16); -+ /* fall through */ -+ case 6: -+ b += ((uint32) k[5] << 8); -+ /* fall through */ -+ case 5: -+ b += k[4]; -+ /* fall through */ -+ case 4: -+ a += oliphaunt_wasix_hash_load32(k); -+ break; -+ case 3: -+ a += ((uint32) k[2] << 16); -+ /* fall through */ -+ case 2: -+ a += ((uint32) k[1] << 8); -+ /* fall through */ -+ case 1: -+ a += k[0]; -+ /* case 0: nothing left to add */ -+ } -+#else - switch (len) - { - case 11: -@@ -590,6 +694,7 @@ hash_bytes_extended(const unsigned char *k, int keylen, uint64 seed) - a += k[0]; - /* case 0: nothing left to add */ - } -+#endif - #endif /* WORDS_BIGENDIAN */ - } - --- -2.49.0 diff --git a/src/third-party/postgres/patches/wasix/0015-oliphaunt-wasix-add-top-xid-current-transaction-fast-path.patch b/src/third-party/postgres/patches/wasix/0015-oliphaunt-wasix-add-top-xid-current-transaction-fast-path.patch deleted file mode 100644 index 3337c2189..000000000 --- a/src/third-party/postgres/patches/wasix/0015-oliphaunt-wasix-add-top-xid-current-transaction-fast-path.patch +++ /dev/null @@ -1,43 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: add top xid current transaction fast path - -TransactionIdIsCurrentTransactionId() is a hot visibility helper. In the -common single-backend top-level transaction case, once the top XID comparison -has failed there are no other current XIDs to discover unless parallel-worker -state, a parent subtransaction stack, or subcommitted child XIDs exist. - -Short-circuit that ordinary case before entering the generic stack walk. The -existing parallel and subtransaction paths are preserved for every case that -can still have another current XID. ---- - src/backend/access/transam/xact.c | 13 +++++++++++++ - 1 file changed, 13 insertions(+) - -diff --git a/src/backend/access/transam/xact.c b/src/backend/access/transam/xact.c -index b885513e3e..7026efae20 100644 ---- a/src/backend/access/transam/xact.c -+++ b/src/backend/access/transam/xact.c -@@ -961,6 +961,19 @@ TransactionIdIsCurrentTransactionId(TransactionId xid) - if (TransactionIdEquals(xid, GetTopTransactionIdIfAny())) - return true; - -+ /* -+ * Fast path for the common top-level transaction case. Once the top XID -+ * comparison failed, there are no other current XIDs to find unless we are -+ * in a parallel worker, a subtransaction stack, or have subcommitted -+ * children. -+ */ -+ s = CurrentTransactionState; -+ if (nParallelCurrentXids == 0 && -+ s->parent == NULL && -+ s->state != TRANS_ABORT && -+ s->nChildXids == 0) -+ return false; -+ - /* - * In parallel workers, the XIDs we must consider as current are stored in - * ParallelCurrentXids rather than the transaction-state stack. Note that --- -2.49.0 diff --git a/src/third-party/postgres/patches/wasix/0016-oliphaunt-wasix-add-btree-int4-compare-fast-path.patch b/src/third-party/postgres/patches/wasix/0016-oliphaunt-wasix-add-btree-int4-compare-fast-path.patch deleted file mode 100644 index af2d60f6a..000000000 --- a/src/third-party/postgres/patches/wasix/0016-oliphaunt-wasix-add-btree-int4-compare-fast-path.patch +++ /dev/null @@ -1,78 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: add btree int4 compare fast path - -WASIX makes PostgreSQL's fmgr indirect-call trampoline comparatively visible in -btree-heavy point lookup and update paths. The built-in int4 btree order proc -is a pure three-way integer comparison, so the embedded WASIX runtime can avoid -that trampoline when the relation metadata proves that the comparison is the -standard integer btree family with int4 input on both sides. - -Keep the optimization narrow: it is WASIX plus OLIPHAUNT_WASM_SINGLE_USER only, -still obtains tuple values through index_getattr(), requires the built-in -integer btree family and int4 opclass input type, requires int4/InvalidOid scan -subtype and InvalidOid collation, and falls back to PostgreSQL's upstream -FunctionCall2Coll path for every other case. ---- - src/backend/access/nbtree/nbtsearch.c | 35 +++++++++++++++++++++++++++++++---- - 1 file changed, 31 insertions(+), 4 deletions(-) - -diff --git a/src/backend/access/nbtree/nbtsearch.c b/src/backend/access/nbtree/nbtsearch.c -index 3ad6f1f3b4..49c18b831e 100644 ---- a/src/backend/access/nbtree/nbtsearch.c -+++ b/src/backend/access/nbtree/nbtsearch.c -@@ -16,7 +16,9 @@ - #include "postgres.h" - - #include "access/nbtree.h" - #include "access/relscan.h" -+#include "catalog/pg_opfamily_d.h" -+#include "catalog/pg_type_d.h" - #include "access/xact.h" - #include "miscadmin.h" - #include "pgstat.h" -@@ -767,10 +769,37 @@ _bt_compare(Relation rel, - * to flip the sign of the comparison result. (Unless it's a DESC - * column, in which case we *don't* flip the sign.) - */ -- result = DatumGetInt32(FunctionCall2Coll(&scankey->sk_func, -- scankey->sk_collation, -- datum, -- scankey->sk_argument)); -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) -+ if (rel->rd_opfamily[i - 1] == INTEGER_BTREE_FAM_OID && -+ rel->rd_opcintype[i - 1] == INT4OID && -+ (scankey->sk_subtype == INT4OID || -+ scankey->sk_subtype == InvalidOid) && -+ scankey->sk_collation == InvalidOid) -+ { -+ int32 index_value = DatumGetInt32(datum); -+ int32 scan_value = DatumGetInt32(scankey->sk_argument); -+ -+ /* -+ * Avoid the fmgr trampoline only for PostgreSQL's built-in -+ * int4 btree comparison. The result is still "index datum -+ * compared with scan argument"; the existing DESC handling -+ * below preserves upstream _bt_compare() polarity. -+ */ -+ if (index_value > scan_value) -+ result = 1; -+ else if (index_value == scan_value) -+ result = 0; -+ else -+ result = -1; -+ } -+ else -+#endif -+ { -+ result = DatumGetInt32(FunctionCall2Coll(&scankey->sk_func, -+ scankey->sk_collation, -+ datum, -+ scankey->sk_argument)); -+ } - - if (!(scankey->sk_flags & SK_BT_DESC)) - INVERT_COMPARE_RESULT(result); --- -2.49.0 diff --git a/src/third-party/postgres/patches/wasix/0018-oliphaunt-wasix-avoid-pg-dump-executequery-lto-collision.patch b/src/third-party/postgres/patches/wasix/0018-oliphaunt-wasix-avoid-pg-dump-executequery-lto-collision.patch index e5a6f2541..2a67c5113 100644 --- a/src/third-party/postgres/patches/wasix/0018-oliphaunt-wasix-avoid-pg-dump-executequery-lto-collision.patch +++ b/src/third-party/postgres/patches/wasix/0018-oliphaunt-wasix-avoid-pg-dump-executequery-lto-collision.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: avoid pg_dump executeQuery LTO collision diff --git a/src/third-party/postgres/patches/wasix/0024-oliphaunt-wasix-add-like-literal-substring-fast-path.patch b/src/third-party/postgres/patches/wasix/0024-oliphaunt-wasix-add-like-literal-substring-fast-path.patch deleted file mode 100644 index d11b0e3ae..000000000 --- a/src/third-party/postgres/patches/wasix/0024-oliphaunt-wasix-add-like-literal-substring-fast-path.patch +++ /dev/null @@ -1,96 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: add like literal substring fast path - -The benchmark suite has simple `%literal%` LIKE predicates where the pattern -contains no wildcard other than the leading and trailing `%`. PostgreSQL's -generic LIKE matcher still enters the recursive wildcard engine for those -cases. Add a narrow single-byte fast path for deterministic locales that -reduces the check to `memchr` plus `memcmp`, while preserving upstream behavior -for lower-case/case-insensitive match variants, non-deterministic collations, -escapes, `_`, inner `%`, and every other pattern shape. ---- - src/backend/utils/adt/like.c | 45 ++++++++++++++++++++++++++++++++++++++ - src/backend/utils/adt/like_match.c | 10 +++++++++ - 2 files changed, 55 insertions(+) - -diff --git a/src/backend/utils/adt/like.c b/src/backend/utils/adt/like.c -index 36d2797000..84be33c000 100644 ---- a/src/backend/utils/adt/like.c -+++ b/src/backend/utils/adt/like.c -@@ -101,6 +101,51 @@ SB_lower_char(unsigned char c, pg_locale_t locale) - return tolower_l(c, locale->info.lt); - } - -+static inline int -+MatchTextLiteralSubstring(const char *t, int tlen, const char *p, int plen, -+ pg_locale_t locale) -+{ -+ const char *needle; -+ int needle_len; -+ const char *pos; -+ const char *end; -+ unsigned char first; -+ -+ if (locale && !locale->deterministic) -+ return LIKE_ABORT; -+ if (plen < 2 || p[0] != '%' || p[plen - 1] != '%') -+ return LIKE_ABORT; -+ -+ needle = p + 1; -+ needle_len = plen - 2; -+ for (int i = 0; i < needle_len; i++) -+ { -+ if (needle[i] == '%' || needle[i] == '_' || needle[i] == '\\') -+ return LIKE_ABORT; -+ } -+ -+ if (needle_len == 0) -+ return LIKE_TRUE; -+ if (tlen < needle_len) -+ return LIKE_FALSE; -+ -+ first = (unsigned char) needle[0]; -+ pos = t; -+ end = t + tlen - needle_len + 1; -+ while (pos < end) -+ { -+ const char *hit = memchr(pos, first, end - pos); -+ -+ if (hit == NULL) -+ return LIKE_FALSE; -+ if (memcmp(hit, needle, needle_len) == 0) -+ return LIKE_TRUE; -+ pos = hit + 1; -+ } -+ -+ return LIKE_FALSE; -+} -+ - - #define NextByte(p, plen) ((p)++, (plen)--) - -diff --git a/src/backend/utils/adt/like_match.c b/src/backend/utils/adt/like_match.c -index 0e3b2bc000..7b93dd4000 100644 ---- a/src/backend/utils/adt/like_match.c -+++ b/src/backend/utils/adt/like_match.c -@@ -86,6 +86,16 @@ MatchText(const char *t, int tlen, const char *p, int plen, pg_locale_t locale) - /* Since this function recurses, it could be driven to stack overflow */ - check_stack_depth(); - -+#ifndef MATCH_LOWER -+ { -+ int fast_match; -+ -+ fast_match = MatchTextLiteralSubstring(t, tlen, p, plen, locale); -+ if (fast_match != LIKE_ABORT) -+ return fast_match; -+ } -+#endif -+ - /* - * In this loop, we advance by char when matching wildcards (and thus on - * recursive entry to this function we are properly char-synced). On other --- -2.50.0 diff --git a/src/third-party/postgres/patches/wasix/0026-oliphaunt-wasix-add-first-int4-leaf-compare-fast-path.patch b/src/third-party/postgres/patches/wasix/0026-oliphaunt-wasix-add-first-int4-leaf-compare-fast-path.patch deleted file mode 100644 index f99c40747..000000000 --- a/src/third-party/postgres/patches/wasix/0026-oliphaunt-wasix-add-first-int4-leaf-compare-fast-path.patch +++ /dev/null @@ -1,113 +0,0 @@ -From 0000000000000000000000000000000000000026 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: add first int4 leaf compare fast path - -The btree compare diagnostic shows that `_bt_compare` dominates the indexed -update miss after the broad executor/storage split. The existing WASIX int4 -fast path avoids the fmgr comparator trampoline, but it still obtains the first -attribute through `index_getattr()`. - -Add a narrower leaf-page shortcut for the normal benchmark shape: single-column -non-null int4 btree tuples using the built-in integer opfamily. The first int4 -datum is read directly from the index tuple data area. Unequal comparisons -return immediately; equal comparisons skip the attribute loop but still fall -through to PostgreSQL's existing heap-TID/truncated-key tie-break logic. Pivot -tuples, posting tuples, null bitmaps, DESC keys, non-int4 opclasses, and -non-WASIX builds continue through the upstream path. ---- - src/backend/access/nbtree/nbtsearch.c | 65 ++++++++++++++++++++++++++++++++--- - 1 file changed, 61 insertions(+), 4 deletions(-) - -diff --git a/src/backend/access/nbtree/nbtsearch.c b/src/backend/access/nbtree/nbtsearch.c ---- a/src/backend/access/nbtree/nbtsearch.c -+++ b/src/backend/access/nbtree/nbtsearch.c -@@ -667,6 +667,56 @@ _bt_binsrch_posting(BTScanInsert key, Page page, OffsetNumber offnum) - return low; - } - -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) -+/* -+ * Fast path for the dominant single-column int4 leaf-page compare shape. -+ * -+ * The existing int4 shortcut still pays index_getattr() to reach the first -+ * fixed-width datum. For a normal non-null leaf tuple, the first int4 datum -+ * starts directly at the index tuple data offset. Equal keys deliberately -+ * return true as result zero: the caller skips the attribute loop but still -+ * executes PostgreSQL's existing heap-TID and truncated-key tie-break logic. -+ */ -+static inline bool -+_bt_oliphaunt_fast_first_int4_leaf_compare(Relation rel, BTScanInsert key, -+ IndexTuple itup, int ntupatts, -+ int32 *result) -+{ -+ ScanKey scankey; -+ int32 index_value; -+ int32 scan_value; -+ -+ if (key->keysz != 1 || ntupatts < 1) -+ return false; -+ if (BTreeTupleIsPivot(itup) || BTreeTupleIsPosting(itup) || -+ IndexTupleHasNulls(itup)) -+ return false; -+ -+ scankey = key->scankeys; -+ if (scankey->sk_attno != 1 || -+ (scankey->sk_flags & (SK_ISNULL | SK_BT_DESC)) != 0 || -+ rel->rd_opfamily[0] != INTEGER_BTREE_FAM_OID || -+ rel->rd_opcintype[0] != INT4OID || -+ (scankey->sk_subtype != INT4OID && -+ scankey->sk_subtype != InvalidOid) || -+ scankey->sk_collation != InvalidOid) -+ return false; -+ -+ memcpy(&index_value, -+ (char *) itup + IndexInfoFindDataOffset(itup->t_info), -+ sizeof(index_value)); -+ scan_value = DatumGetInt32(scankey->sk_argument); -+ -+ if (index_value > scan_value) -+ *result = -1; -+ else if (index_value < scan_value) -+ *result = 1; -+ else -+ *result = 0; -+ return true; -+} -+#endif -+ - /*---------- - * _bt_compare() -- Compare insertion-type scankey to tuple on a page. - * -@@ -708,6 +758,7 @@ _bt_compare(Relation rel, - int ncmpkey; - int ntupatts; - int32 result; -+ bool key_equal_fast = false; - - Assert(_bt_check_natts(rel, key->heapkeyspace, page, offnum)); - Assert(key->keysz <= IndexRelationGetNumberOfKeyAttributes(rel)); -@@ -740,7 +791,20 @@ _bt_compare(Relation rel, - Assert(key->heapkeyspace || ncmpkey == key->keysz); - Assert(!BTreeTupleIsPosting(itup) || key->allequalimage); - scankey = key->scankeys; -- for (int i = 1; i <= ncmpkey; i++) -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) -+ if (P_ISLEAF(opaque) && -+ _bt_oliphaunt_fast_first_int4_leaf_compare(rel, key, itup, ntupatts, -+ &result)) -+ { -+ if (result != 0) -+ { -+ return result; -+ } -+ key_equal_fast = true; -+ scankey++; -+ } -+#endif -+ for (int i = key_equal_fast ? 2 : 1; i <= ncmpkey; i++) - { - Datum datum; - bool isNull; --- -2.39.5 diff --git a/src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-buffer-strong-random.patch b/src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-buffer-strong-random.patch deleted file mode 100644 index 6ee8616e7..000000000 --- a/src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-buffer-strong-random.patch +++ /dev/null @@ -1,135 +0,0 @@ -From 0000000000000000000000000000000000000037 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Mon, 17 Aug 2026 00:00:00 +0000 -Subject: [PATCH] oliphaunt-wasix: buffer strong random - -PostgreSQL otherwise opens the virtual /dev/urandom device for every strong -random request. Concurrent WASIX backends can transiently exhaust their guest -descriptor budget while creating cancel keys, and the embedded backend causes -one host entropy request for every 8 or 16 bytes. - -Use the pinned WASIX libc getrandom interface directly in every WASIX build. -Keep a small guest-side reservoir only for the explicitly single-backend -build; concurrent backends retain no state that a fork could duplicate. -Errors and short reads remain visible through pg_strong_random's existing -boolean contract. Non-WASIX PostgreSQL builds retain upstream behavior. ---- - src/port/pg_strong_random.c | 103 +++++++++++++++++++++++++++++++++++++ - 1 file changed, 103 insertions(+) - -diff --git a/src/port/pg_strong_random.c b/src/port/pg_strong_random.c -index ea6780d..c0f42f4 100644 ---- a/src/port/pg_strong_random.c -+++ b/src/port/pg_strong_random.c -@@ -92,6 +92,109 @@ pg_strong_random(void *buf, size_t len) - return false; - } - -+#elif defined(__wasi__) -+ -+#include -+ -+static bool -+wasix_strong_random_fill(void *buf, size_t len) -+{ -+ unsigned char *p = buf; -+ -+ while (len > 0) -+ { -+ ssize_t res; -+ -+ res = getrandom(p, len, 0); -+ if (res < 0) -+ { -+ if (errno == EINTR) -+ continue; -+ return false; -+ } -+ if (res == 0) -+ return false; -+ p += res; -+ len -= res; -+ } -+ -+ return true; -+} -+ -+#if defined(OLIPHAUNT_WASM_SINGLE_USER) -+ -+/* -+ * WASIX random_get crosses the guest/host boundary. Keep a small reservoir -+ * in the single-backend guest so UUID-sized requests do not each require a -+ * host call. The runtime forbids backend fork/thread creation, so this state -+ * cannot be duplicated by a concurrent PostgreSQL process. -+ */ -+#define WASIX_STRONG_RANDOM_POOL_SIZE 4096 -+ -+static unsigned char wasix_strong_random_pool[WASIX_STRONG_RANDOM_POOL_SIZE]; -+static size_t wasix_strong_random_used = WASIX_STRONG_RANDOM_POOL_SIZE; -+ -+static bool -+wasix_strong_random_refill(void) -+{ -+ if (!wasix_strong_random_fill(wasix_strong_random_pool, -+ sizeof(wasix_strong_random_pool))) -+ { -+ wasix_strong_random_used = sizeof(wasix_strong_random_pool); -+ return false; -+ } -+ wasix_strong_random_used = 0; -+ return true; -+} -+ -+void -+pg_strong_random_init(void) -+{ -+ /* Discard unused bytes whenever PostgreSQL initializes the source. */ -+ wasix_strong_random_used = sizeof(wasix_strong_random_pool); -+} -+ -+bool -+pg_strong_random(void *buf, size_t len) -+{ -+ unsigned char *p = buf; -+ -+ while (len > 0) -+ { -+ size_t available; -+ size_t copy_len; -+ -+ if (wasix_strong_random_used == sizeof(wasix_strong_random_pool) && -+ !wasix_strong_random_refill()) -+ return false; -+ -+ available = sizeof(wasix_strong_random_pool) - wasix_strong_random_used; -+ copy_len = Min(len, available); -+ memcpy(p, wasix_strong_random_pool + wasix_strong_random_used, copy_len); -+ wasix_strong_random_used += copy_len; -+ p += copy_len; -+ len -= copy_len; -+ } -+ -+ return true; -+} -+ -+#else -+ -+void -+pg_strong_random_init(void) -+{ -+ /* No guest-side state in a process that may fork. */ -+} -+ -+bool -+pg_strong_random(void *buf, size_t len) -+{ -+ return wasix_strong_random_fill(buf, len); -+} -+ -+#endif -+ - #elif WIN32 - - #include --- -2.51.0 diff --git a/src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-use-checked-getrandom.patch b/src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-use-checked-getrandom.patch new file mode 100644 index 000000000..cabd8ee96 --- /dev/null +++ b/src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-use-checked-getrandom.patch @@ -0,0 +1,68 @@ +From 0000000000000000000000000000000000000037 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Mon, 17 Aug 2026 00:00:00 +0000 +Subject: [PATCH] oliphaunt-wasix: use checked getrandom + +PostgreSQL otherwise opens the virtual /dev/urandom device for every strong +random request. That adds virtual descriptor pressure. The pinned Wasmer +source's RandomFile also discards entropy-provider errors before returning +zero-filled bytes; the host stack fixes device users independently. + +Use the pinned WASIX libc getrandom interface directly. Retry interrupted +reads, complete short reads, and report errors or zero-length progress through +pg_strong_random's existing boolean contract. Keep no guest entropy reservoir: +fork, clone, snapshot, and re-entry cannot duplicate buffered bytes. Batching +must earn its place independently before adding lifecycle-sensitive state. +Non-WASIX PostgreSQL builds retain upstream behavior. +--- + src/port/pg_strong_random.c | 35 +++++++++++++++++++++++++++++++++++ + 1 file changed, 35 insertions(+) + +diff --git a/src/port/pg_strong_random.c b/src/port/pg_strong_random.c +index ea6780d..df61fbd 100644 +--- a/src/port/pg_strong_random.c ++++ b/src/port/pg_strong_random.c +@@ -92,6 +92,41 @@ pg_strong_random(void *buf, size_t len) + return false; + } + ++#elif defined(__wasi__) ++ ++#include ++ ++void ++pg_strong_random_init(void) ++{ ++ /* No guest-side state to initialize or duplicate. */ ++} ++ ++bool ++pg_strong_random(void *buf, size_t len) ++{ ++ unsigned char *p = buf; ++ ++ while (len > 0) ++ { ++ ssize_t res; ++ ++ res = getrandom(p, len, 0); ++ if (res < 0) ++ { ++ if (errno == EINTR) ++ continue; ++ return false; ++ } ++ if (res == 0) ++ return false; ++ p += res; ++ len -= res; ++ } ++ ++ return true; ++} ++ + #elif WIN32 + + #include +-- +2.51.0 diff --git a/src/wasix/browser-host/build-provenance.mts b/src/wasix/browser-host/build-provenance.mts index ee54b2b9e..8611ca941 100644 --- a/src/wasix/browser-host/build-provenance.mts +++ b/src/wasix/browser-host/build-provenance.mts @@ -8,10 +8,16 @@ const repositoryRoot = resolve(hostDirectory, '../../..'); const sourceManifestPath = 'src/wasix/browser-host/source.toml'; const buildScriptPath = 'src/wasix/browser-host/build-sdk.sh'; const provenanceScriptPath = 'src/wasix/browser-host/build-provenance.mts'; +const protocolTransportContractPath = 'src/wasix/runtime/protocol-contract/contract.json'; const safePatchName = /^\d{4}-(?:wasmer-(?:(?:js|wasix)-)?|virtual-(?:fs|mio)-)[a-z0-9-]+\.patch$/u; export async function loadHostBuildContract() { const source = await readFile(resolve(repositoryRoot, sourceManifestPath), 'utf8'); + const protocolTransportContractBytes = await readFile( + resolve(repositoryRoot, protocolTransportContractPath), + ); + const protocolTransportContract = JSON.parse(protocolTransportContractBytes.toString('utf8')); + validateProtocolTransportContract(protocolTransportContract); const patchSeries = tomlStringArray(source, 'patches', 'series'); if (patchSeries.length === 0 || new Set(patchSeries).size !== patchSeries.length) { throw new Error('WASIX host patch series must be non-empty and unique'); @@ -32,6 +38,8 @@ export async function loadHostBuildContract() { 'src/third-party/tools/fetch-sources.sh', 'src/third-party/tools/source-fetch-core.mts', 'src/third-party/tools/source-archive.mts', + protocolTransportContractPath, + 'src/wasix/browser-host/protocol-contract.generated.rs', ]); const digests = []; for (const input of inputs) { @@ -42,8 +50,24 @@ export async function loadHostBuildContract() { const provenance = deepFreeze({ wasmerJsCommit: tomlString(source, 'wasmer-js', 'commit'), wasmerWasixVersion: tomlString(source, 'wasmer-wasix', 'version'), + virtualFsVersion: tomlString(source, 'virtual-fs', 'version'), inputsSha256: sha256(digests.join('')), - guestConcurrency: 'denied-for-oliphaunt-single-backend', + guestConcurrency: 'typed-single-program-host-policy', + clockDispatch: 'server-direct-js-16ms-or-1024-reads-shared-setter-fallback-tools-canonical', + fdClose: 'typed-filesystem-durability-policy', + syncFilesystemBridge: 'realm-local-fresh-owned-js-transfer', + toolProtocolWrite: 'owned-js-copy-before-callback', + protocolTransport: { + schema: protocolTransportContract.schema, + contractSha256: sha256(protocolTransportContractBytes), + modes: Object.fromEntries( + protocolTransportContract.modes.map(({ name, value }) => [name, value]), + ), + bufferedOutputLimitBytes: protocolTransportContract.bufferedOutput.limitBytes, + callbackChunkMaxBytes: protocolTransportContract.streamedOutput.callbackChunkMaxBytes, + flushWasmResult: protocolTransportContract.flush.wasmResult, + }, + randomDevice: 'virtual-fs-checked-getrandom', optimization: { cargoProfile: 'release', rustOptLevel: 3, @@ -54,6 +78,30 @@ export async function loadHostBuildContract() { return Object.freeze({ inputs, patchSeries: Object.freeze(patchSeries), provenance }); } +function validateProtocolTransportContract(contract) { + const modes = contract?.modes; + if ( + contract?.schema !== 'oliphaunt-wasix-postgres-protocol-transport-contract-v1' || + !Array.isArray(modes) || + modes.length !== 4 || + modes.some( + (mode) => + typeof mode?.name !== 'string' || !Number.isSafeInteger(mode.value) || mode.value < 0, + ) || + new Set(modes.map(({ name }) => name)).size !== modes.length || + new Set(modes.map(({ value }) => value)).size !== modes.length || + !Number.isSafeInteger(contract.bufferedOutput?.limitBytes) || + contract.bufferedOutput.limitBytes <= 0 || + !Number.isSafeInteger(contract.streamedOutput?.callbackChunkMaxBytes) || + contract.streamedOutput.callbackChunkMaxBytes <= 0 || + contract.flush?.wasmResult !== 'i32' + ) { + throw new Error( + `invalid PostgreSQL protocol transport contract: ${protocolTransportContractPath}`, + ); + } +} + function tomlString(source, section, key) { const body = tomlSection(source, section); const match = body.match(new RegExp(`^\\s*${escapeRegExp(key)}\\s*=\\s*"([^"]+)"\\s*$`, 'mu')); diff --git a/src/wasix/browser-host/build-sdk.sh b/src/wasix/browser-host/build-sdk.sh index f83c13bce..bab50bcb6 100755 --- a/src/wasix/browser-host/build-sdk.sh +++ b/src/wasix/browser-host/build-sdk.sh @@ -45,7 +45,12 @@ if ! command -v bun >/dev/null 2>&1; then exit 1 fi patch_series="$(bun "$provenance_script" --patch-series)" +node "$repo_root/src/wasix/runtime/protocol-contract/generate.mjs" --check input_hash="$(bun "$provenance_script" --inputs-sha256)" +if [[ ! "$input_hash" =~ ^[0-9a-f]{64}$ ]]; then + echo "wasix-ts host build: invalid source identity" >&2 + exit 1 +fi patch_command="patch" if command -v gpatch >/dev/null 2>&1; then @@ -135,9 +140,11 @@ while IFS= read -r patch_name; do exit 1 ;; esac - "$patch_command" --batch --forward -d "$patch_dir" -p1 < "$patch_file" + "$patch_command" --batch --forward --fuzz=0 -d "$patch_dir" -p1 < "$patch_file" done <<< "$patch_series" +install -m 0644 "$host_dir/protocol-contract.generated.rs" "$wasmer_js_dir/src/protocol_contract.rs" + # The pinned source commit's npm lock predates its package metadata. Patch only # the missing root metadata and dependencies, then install the integrity-pinned # graph without allowing the package manager to rewrite it. diff --git a/src/wasix/browser-host/clock.test.mts b/src/wasix/browser-host/clock.test.mts new file mode 100644 index 000000000..1ef6d98db --- /dev/null +++ b/src/wasix/browser-host/clock.test.mts @@ -0,0 +1,114 @@ +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import test from 'node:test'; +import { runInNewContext } from 'node:vm'; + +// Execute the actual inline JavaScript shipped by the source patch, not a +// second implementation of its clock arithmetic. +const additions = readFileSync( + new URL('./patches/0013-wasmer-wasix-fast-single-backend-clock.patch', import.meta.url), + 'utf8', +) + .split('\n') + .filter((line) => line.startsWith('+') && !line.startsWith('+++')) + .map((line) => line.slice(1)) + .join('\n'); +const javascript = [...additions.matchAll(/inline_js = r#"([\s\S]*?)"#/gu)] + .map((match) => match[1].replaceAll('export function ', 'function ')) + .join('\n'); + +test('canonical monotonic time shares an origin across workers', () => { + const readClock = (timeOrigin: number, now: number) => + runInNewContext(`${javascript}; oliphaunt_monotonic_time_ms()`, { + performance: { timeOrigin, now: () => now }, + }); + assert.equal(readClock(1_000, 100), readClock(1_050, 50)); + assert.equal(readClock(1_050, 51), 1_101); +}); + +test('direct clock preserves the canonical epoch and bounds fallback intervals', () => { + let now = 100; + const memory = new WebAssembly.Memory({ initial: 1, maximum: 2 }); + let calls = 0; + const fallback = (clockId: number, _precision: bigint, pointer: number) => { + calls++; + if (clockId > 1 || pointer > memory.buffer.byteLength - 8) return 28; + new DataView(memory.buffer).setBigUint64(pointer, 9_000_000_000n, true); + return 0; + }; + const clock = runInNewContext(`${javascript}; oliphauntDirectClockImport(memory, fallback)`, { + memory, + fallback, + Date: { now: () => now }, + performance: { now: () => now }, + }); + assert.equal(clock(1, 0n, 0), 0); + now++; + assert.equal(clock(1, 0n, 0), 0); + assert.equal(new DataView(memory.buffer).getBigUint64(0, true), 9_001_000_000n); + assert.equal(calls, 1); + now += 16; + clock(1, 0n, 0); + assert.equal(calls, 2); + for (let i = 0; i < 1024; i++) clock(1, 0n, 0); + assert.equal(calls, 2); + clock(1, 0n, 0); + assert.equal(calls, 3); + memory.grow(1); + assert.equal(clock(1, 0n, 65_536), 0); + assert.equal(new DataView(memory.buffer).getBigUint64(65_536, true), 9_000_000_000n); + assert.equal(clock(1, 0n, memory.buffer.byteLength), 28); + assert.equal(clock(2, 0n, 0), 28); +}); + +test('linking a setter preserves fast reads; calling it switches all shared-memory readers', () => { + const factories = runInNewContext( + `${javascript}; ({read: oliphauntDirectClockImport, set: oliphauntClockSetImport})`, + { Date: { now: () => 100 }, performance: { now: () => 100 } }, + ); + for (const clockId of [0, 1]) { + for (const errno of [0, 28]) { + const memory = new WebAssembly.Memory({ initial: 1, maximum: 2, shared: true }); + let calls = 0; + let time = 100_000_000n; + const fallback = (_id: number, _precision: bigint, pointer: number) => { + calls++; + new DataView(memory.buffer).setBigUint64(pointer, time, true); + return 0; + }; + const main = factories.read(memory, fallback); + const side = factories.read(memory, fallback); + main(clockId, 0n, 0); + side(clockId, 0n, 8); + const setter = factories.set(memory, (_id: number, value: bigint) => { + // WASIX can run pending work before updating its clock offset. + const before = calls; + main(clockId, 0n, 0); + assert.equal(calls, before + 1); + if (errno === 0) time = value; + return errno; + }); + main(clockId, 0n, 0); + side(clockId, 0n, 8); + assert.equal(calls, 2, 'importing the setter must not disable fast reads'); + assert.equal(setter(clockId, 5_000_000_000n), errno); + const late = factories.read(memory, fallback); + memory.grow(1); + for (const read of [main, side, late]) { + const before = calls; + read(clockId, 0n, 65_536); + assert.equal(calls, before + 1); + assert.equal(new DataView(memory.buffer).getBigUint64(65_536, true), time); + } + const other = new WebAssembly.Memory({ initial: 1 }); + const independent = factories.read(other, () => { + calls++; + return 0; + }); + const before = calls; + independent(clockId, 0n, 0); + independent(clockId, 0n, 0); + assert.equal(calls, before + 1, 'unrelated databases keep their fast path'); + } + } +}); diff --git a/src/wasix/browser-host/moon.yml b/src/wasix/browser-host/moon.yml index 95d5ed198..295835261 100644 --- a/src/wasix/browser-host/moon.yml +++ b/src/wasix/browser-host/moon.yml @@ -6,6 +6,12 @@ layer: "library" stack: "frontend" tasks: + test: + tags: ["quality", "unit"] + command: "bun test src/wasix/browser-host/clock.test.mts" + inputs: ["clock.test.mts", "patches/0013-wasmer-wasix-fast-single-backend-clock.patch"] + options: + runFromWorkspaceRoot: true build: tags: ["artifact", "build", "requires-rust"] command: "bash src/wasix/browser-host/build-sdk.sh" @@ -17,6 +23,11 @@ tasks: - "/tools/dev/curl-platform-flags.sh" - "/tools/dev/acquisition.sh" - "/src/third-party/tools/{fetch-sources.sh,source-fetch-core.mts,source-archive.mts}" + - "/src/wasix/runtime/protocol-contract/contract.json" + - "/src/wasix/runtime/protocol-contract/generate.mjs" + - "/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h" + - "/src/wasix/sdks/rust/src/oliphaunt/protocol_limits_generated.rs" + - "/src/wasix/sdks/ts/src/protocol-limits.generated.ts" - "@group(cargo-workspace)" outputs: - "/target/oliphaunt-wasix-ts/host/wasmer-sdk/**/*" diff --git a/src/wasix/browser-host/patches/0001-wasmer-js-run-configured-wasix-process.patch b/src/wasix/browser-host/patches/0001-wasmer-js-run-configured-wasix-process.patch index 1416e7e80..c6ee8f4cd 100644 --- a/src/wasix/browser-host/patches/0001-wasmer-js-run-configured-wasix-process.patch +++ b/src/wasix/browser-host/patches/0001-wasmer-js-run-configured-wasix-process.patch @@ -1,23 +1,32 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: execute the configured WASIX process + +Use the configured WASIX environment and propagate nonzero process exits. +Oliphaunt needs its mounts, program arguments and host policy honored when +running PostgreSQL and frontend tools. This is a general runner correction; +the surrounding Oliphaunt policy remains a downstream integration. + diff --git a/Cargo.toml b/Cargo.toml -index b21b6eb..4a26370 100644 +index b21b6eb..65d86be 100644 --- a/Cargo.toml +++ b/Cargo.toml -@@ -126,7 +126,7 @@ wasmer-types = { version = "4.2.5", default-features = false } - +@@ -127,7 +127,7 @@ wasm-bindgen-test = "0.3" + [profile.release] lto = true -opt-level = 'z' +opt-level = 3 - + [package.metadata.wasm-pack.profile.release.wasm-bindgen] debug-js-glue = false @@ -135,9 +135,13 @@ demangle-name-section = false dwarf-debug-info = false - + [package.metadata.wasm-pack.profile.release] -wasm-opt = ["--enable-threads", "--enable-bulk-memory", "-Oz"] +wasm-opt = ["--enable-threads", "--enable-bulk-memory", "-O3"] - + [patch.crates-io] +wasmer = { path = "../wasmer-6.1.0" } +wasmer-wasix = { path = "../wasmer-wasix-0.601.0" } @@ -42,18 +51,18 @@ index 6e25018..99ee44e 100644 -use wasmer_wasix::runners::wasi::{WasiRunner, RuntimeOrEngine}; +use wasmer_wasix::runtime::task_manager::VirtualTaskManagerExt; +use wasmer_wasix::{WasiEnvBuilder, WasiError, WasiRuntimeError}; - + use crate::{instance::ExitCondition, utils::Error, Instance, RunOptions}; - + @@ -43,9 +42,9 @@ async fn run_wasix_inner(wasm_module: WasmModule, config: RunOptions) -> Result< let (stdin, stdout, stderr) = config.configure_builder(&mut builder)?; - + let (exit_code_tx, exit_code_rx) = oneshot::channel(); - let mut runner = WasiRunner::new(); - + let module: wasmer::Module = wasm_module.to_module(&*runtime).await?; + let module_bytes = module.serialize().unwrap(); - + // Note: The WasiEnvBuilder::run() method blocks, so we need to run it on // the thread pool. @@ -54,8 +53,26 @@ async fn run_wasix_inner(wasm_module: WasmModule, config: RunOptions) -> Result< diff --git a/src/wasix/browser-host/patches/0002-wasmer-wasix-add-0702-compatibility-imports.patch b/src/wasix/browser-host/patches/0002-wasmer-wasix-add-0702-compatibility-imports.patch index 3f15508cd..064fe0ced 100644 --- a/src/wasix/browser-host/patches/0002-wasmer-wasix-add-0702-compatibility-imports.patch +++ b/src/wasix/browser-host/patches/0002-wasmer-wasix-add-0702-compatibility-imports.patch @@ -1,3 +1,12 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: provide explicit compatibility imports + +Resolve newer WASIX guest imports against the pinned host ABI. Unsupported +fork/context operations return explicit failures rather than pretending that +this older host implements them. This compatibility bridge is version-specific; +upstream should adopt the actual ABI implementations rather than these stubs. + diff --git a/src/lib.rs b/src/lib.rs index 00615d4..5f50af3 100644 --- a/src/lib.rs diff --git a/src/wasix/browser-host/patches/0003-wasmer-js-install-browser-runtime-devices.patch b/src/wasix/browser-host/patches/0003-wasmer-js-install-browser-runtime-devices.patch index d8c6a2400..9217a0576 100644 --- a/src/wasix/browser-host/patches/0003-wasmer-js-install-browser-runtime-devices.patch +++ b/src/wasix/browser-host/patches/0003-wasmer-js-install-browser-runtime-devices.patch @@ -1,5 +1,14 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: install browser runtime devices + +Install the filesystem devices needed by PostgreSQL and frontend tools in +the configured browser environment. Host-selected execution, clock and close +policies are explicit; they must not be selected by untrusted guest variables. +This is the downstream device/profile composition point. + diff --git a/src/options.rs b/src/options.rs -index f44a05f..3233b8e 100644 +index 9fb3958..d074351 100644 --- a/src/options.rs +++ b/src/options.rs @@ -1,8 +1,12 @@ @@ -9,16 +18,16 @@ index f44a05f..3233b8e 100644 + path::{Path, PathBuf}, + sync::Arc, +}; - + use anyhow::Context; use js_sys::Array; -use virtual_fs::TmpFileSystem; +use virtual_fs::{random_file::RandomFile, TmpFileSystem}; use wasm_bindgen::{prelude::wasm_bindgen, JsCast, JsValue, UnwrapThrowExt}; use wasmer_wasix::WasiEnvBuilder; - -@@ -221,6 +225,13 @@ impl RunOptions { - + +@@ -226,6 +230,13 @@ impl RunOptions { + pub(crate) fn filesystem(&self) -> Result { let root = TmpFileSystem::new(); + for path in ["/dev", "/dev/shm"] { @@ -28,6 +37,6 @@ index f44a05f..3233b8e 100644 + root.new_open_options_ext() + .insert_device_file(PathBuf::from("/dev/urandom"), Box::::default()) + .context("Unable to create /dev/urandom")?; - + for (dest, fs) in self.mounted_directories()? { tracing::trace!(%dest, ?fs, "Mounting directory"); diff --git a/src/wasix/browser-host/patches/0005-wasmer-js-use-object-wasm-init.patch b/src/wasix/browser-host/patches/0005-wasmer-js-use-object-wasm-init.patch index fee8e7ed7..7114c8ba4 100644 --- a/src/wasix/browser-host/patches/0005-wasmer-js-use-object-wasm-init.patch +++ b/src/wasix/browser-host/patches/0005-wasmer-js-use-object-wasm-init.patch @@ -1,3 +1,12 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: use object-form Wasm initialization + +Use the initialization argument shape generated by the pinned wasm-bindgen +version. This removes a compatibility warning without changing the module or +memory supplied by the caller; it belongs with the corresponding upstream +wasm-bindgen update. + diff --git a/src-js/index.ts b/src-js/index.ts index e0c303f..8da1d59 100644 --- a/src-js/index.ts diff --git a/src/wasix/browser-host/patches/0006-wasmer-js-reuse-precompiled-wasix-module.patch b/src/wasix/browser-host/patches/0006-wasmer-js-reuse-precompiled-wasix-module.patch index 029615755..ba690ab2d 100644 --- a/src/wasix/browser-host/patches/0006-wasmer-js-reuse-precompiled-wasix-module.patch +++ b/src/wasix/browser-host/patches/0006-wasmer-js-reuse-precompiled-wasix-module.patch @@ -1,3 +1,13 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: reuse compiled modules and finish after pool idle + +Accept an already compiled WebAssembly module alongside original bytes so +repeated process launches do not compile identical code again. Wait for the +worker pool to become idle before publishing process completion, including +nested tasks. Reuse and scheduler lifecycle changes should be reviewed as +separate general-purpose upstream changes. + diff --git a/src/options.rs b/src/options.rs index d074351f..967ac166 100644 --- a/src/options.rs diff --git a/src/wasix/browser-host/patches/0007-wasmer-js-run-oliphaunt-direct.patch b/src/wasix/browser-host/patches/0007-wasmer-js-run-oliphaunt-direct.patch index 73f4ed108..f1496b391 100644 --- a/src/wasix/browser-host/patches/0007-wasmer-js-run-oliphaunt-direct.patch +++ b/src/wasix/browser-host/patches/0007-wasmer-js-run-oliphaunt-direct.patch @@ -1,3 +1,13 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: drive the embedded PostgreSQL guest directly + +Add a same-realm synchronous driver for the explicit embedded guest exports. +It preserves typed startup/recovery outcomes and owns the guest lifecycle, +removing worker message transport from browser-main execution. This is an +Oliphaunt-specific ABI adapter, not a generic PostgreSQL or WASIX interface; +the series adds bounded transport and fail-closed phases on top. + diff --git a/src/lib.rs b/src/lib.rs index 7b22da5..6803f8e 100644 --- a/src/lib.rs @@ -19,12 +29,12 @@ index 7b22da5..6803f8e 100644 run::run_wasix, utils::StringOrBytes, diff --git a/src/options.rs b/src/options.rs -index 967ac16..7b93ca4 100644 +index 967ac16..d56c81d 100644 --- a/src/options.rs +++ b/src/options.rs @@ -186,6 +186,19 @@ extern "C" { } - + impl RunOptions { + pub(crate) fn configure_direct_builder( + &self, @@ -54,13 +64,13 @@ index 967ac16..7b93ca4 100644 - builder.add_env(key, value); - } + self.configure_common_builder(builder)?; - + let stdin = match self.read_stdin() { Some(stdin) => { @@ -226,11 +233,29 @@ impl RunOptions { let (stderr_file, stderr) = crate::streams::output_pipe(); builder.set_stderr(Box::new(stderr_file)); - + + Ok((stdin, stdout, stderr)) + } + @@ -82,18 +92,18 @@ index 967ac16..7b93ca4 100644 let fs = self.filesystem()?; builder.set_fs(Box::new(fs)); builder.add_preopen_dir("/")?; - + - Ok((stdin, stdout, stderr)) + Ok(()) } - + pub(crate) fn filesystem(&self) -> Result { diff --git a/src/postgres_direct.rs b/src/postgres_direct.rs new file mode 100644 -index 0000000..e0f5463 +index 0000000..c45000a --- /dev/null +++ b/src/postgres_direct.rs -@@ -0,0 +1,567 @@ +@@ -0,0 +1,565 @@ +use std::sync::Arc; + +use anyhow::{ensure, Context}; @@ -107,6 +117,24 @@ index 0000000..e0f5463 +const DEFAULT_PROGRAM_NAME: &str = "/bin/postgres"; +const OLIPHAUNT_EXIT_ALIVE: i32 = 99; + ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++enum MainLoopOutcome { ++ Processed, ++ Recovered, ++ InputEnded, ++} ++ ++impl MainLoopOutcome { ++ fn from_i32(value: i32) -> Option { ++ match value { ++ 0 => Some(Self::Processed), ++ 1 => Some(Self::Recovered), ++ 2 => Some(Self::InputEnded), ++ _ => None, ++ } ++ } ++} ++ +/// Instantiate the integrated Oliphaunt/PostgreSQL guest in this JS realm. +/// +/// The returned driver is synchronous by design: every guest export runs on @@ -135,7 +163,6 @@ index 0000000..e0f5463 + output_reset: TypedFunction<(), i32>, + output_len: TypedFunction<(), i32>, + output_read: TypedFunction<(i32, i32), i32>, -+ set_force_host_error_recovery: Option>, + set_active: TypedFunction, + wasi_start: TypedFunction<(), ()>, + start_oliphaunt: TypedFunction<(), ()>, @@ -145,9 +172,8 @@ index 0000000..e0f5463 + send_conn_data: TypedFunction<(), ()>, + pq_flush: TypedFunction<(), ()>, + pq_buffer_remaining_data: TypedFunction<(), i32>, -+ main_loop: TypedFunction<(), ()>, ++ main_loop: TypedFunction<(), i32>, + send_ready: TypedFunction<(), ()>, -+ recover_error: TypedFunction<(), ()>, + backend_started: bool, + protocol_started: bool, + closed: bool, @@ -225,32 +251,24 @@ index 0000000..e0f5463 + self.push_input(&payload)?; + let max_attempts = (payload.len() / 5).saturating_add(2).max(1); + let mut attempts = 0usize; -+ let mut recovered_protocol_error = false; + while self.protocol_input_remaining()? > 0 { + attempts += 1; + ensure!( + attempts <= max_attempts, + "PostgreSQL direct protocol pump did not drain input" + ); -+ if self.main_loop.call(&mut self.store).is_err() { -+ // Match the native host: the exported recovery boundary is -+ // authoritative even when a JS engine reports an unfamiliar -+ // trap spelling. -+ self.recover_protocol_error(payload.len())?; -+ recovered_protocol_error = true; ++ match self.call_main_loop()? { ++ MainLoopOutcome::Processed | MainLoopOutcome::Recovered => {} ++ MainLoopOutcome::InputEnded => { ++ return Err(self.terminal_main_loop_outcome( ++ "PostgresMainLoopOnce reported input end while dispatching buffered protocol input", ++ )); ++ } + } + } -+ self.send_ready -+ .call(&mut self.store) -+ .context("PostgresSendReadyForQueryIfNecessary")?; -+ self.pq_flush -+ .call(&mut self.store) -+ .context("oliphaunt_wasix_pq_flush after protocol input")?; -+ let output = self.take_output()?; -+ if !recovered_protocol_error && protocol_response_contains_error(&output) { -+ self.recover_non_trapping_protocol_error()?; -+ } -+ Ok(output) ++ self.send_ready_after_main_loop()?; ++ self.flush_after_main_loop()?; ++ self.take_output() + } + + fn close_inner(&mut self) -> anyhow::Result<()> { @@ -310,11 +328,6 @@ index 0000000..e0f5463 + let output_reset = typed_export(&mut store, &instance, "oliphaunt_wasix_output_reset")?; + let output_len = typed_export(&mut store, &instance, "oliphaunt_wasix_output_len")?; + let output_read = typed_export(&mut store, &instance, "oliphaunt_wasix_output_read")?; -+ let set_force_host_error_recovery = optional_typed_export( -+ &mut store, -+ &instance, -+ "oliphaunt_wasix_set_force_host_error_recovery", -+ )?; + let set_active = typed_export(&mut store, &instance, "oliphaunt_wasix_set_active")?; + let wasi_start = typed_export(&mut store, &instance, "_start")?; + let start_oliphaunt = typed_export(&mut store, &instance, "oliphaunt_wasix_start")?; @@ -332,7 +345,6 @@ index 0000000..e0f5463 + &instance, + "PostgresSendReadyForQueryIfNecessary", + )?; -+ let recover_error = typed_export(&mut store, &instance, "PostgresMainLongJmp")?; + + let mut direct = Self { + _runtime: runtime, @@ -347,7 +359,6 @@ index 0000000..e0f5463 + output_reset, + output_len, + output_read, -+ set_force_host_error_recovery, + set_active, + wasi_start, + start_oliphaunt, @@ -359,7 +370,6 @@ index 0000000..e0f5463 + pq_buffer_remaining_data, + main_loop, + send_ready, -+ recover_error, + backend_started: false, + protocol_started: false, + closed: false, @@ -373,18 +383,71 @@ index 0000000..e0f5463 + Ok(()) + } + ++ fn poison_main_loop(&mut self) { ++ // A trap, invalid typed outcome, or failed post-step export can leave ++ // arbitrary guest state behind. Do not call another guest export, ++ // including the ordinary close sequence. ++ self.closed = true; ++ self.backend_started = false; ++ self.protocol_started = false; ++ } ++ ++ fn terminal_main_loop_outcome(&mut self, failure: impl Into) -> anyhow::Error { ++ let failure = failure.into(); ++ self.poison_main_loop(); ++ anyhow::anyhow!("{failure}; the direct instance is closed") ++ } ++ ++ fn terminal_main_loop_error(&mut self, error: wasmer::RuntimeError) -> anyhow::Error { ++ self.poison_main_loop(); ++ anyhow::Error::from(error).context( ++ "PostgresMainLoopOnce trapped instead of returning a typed outcome; \ ++ the direct instance is closed", ++ ) ++ } ++ ++ fn call_main_loop(&mut self) -> anyhow::Result { ++ let status = match self.main_loop.call(&mut self.store) { ++ Ok(status) => status, ++ Err(error) => return Err(self.terminal_main_loop_error(error)), ++ }; ++ MainLoopOutcome::from_i32(status).ok_or_else(|| { ++ self.terminal_main_loop_outcome(format!( ++ "PostgresMainLoopOnce returned invalid typed outcome {status}" ++ )) ++ }) ++ } ++ ++ fn send_ready_after_main_loop(&mut self) -> anyhow::Result<()> { ++ match self.send_ready.call(&mut self.store) { ++ Ok(()) => Ok(()), ++ Err(error) => { ++ self.poison_main_loop(); ++ Err(anyhow::Error::from(error).context( ++ "PostgresSendReadyForQueryIfNecessary trapped after a typed main-loop outcome; \ ++ the direct instance is closed", ++ )) ++ } ++ } ++ } ++ ++ fn flush_after_main_loop(&mut self) -> anyhow::Result<()> { ++ match self.pq_flush.call(&mut self.store) { ++ Ok(()) => Ok(()), ++ Err(error) => { ++ self.poison_main_loop(); ++ Err(anyhow::Error::from(error).context( ++ "oliphaunt_wasix_pq_flush trapped after a typed main-loop outcome; \ ++ the direct instance is closed", ++ )) ++ } ++ } ++ } ++ + fn start_backend(&mut self) -> anyhow::Result<()> { + if self.backend_started { + return Ok(()); + } -+ if let Some(set_force) = &self.set_force_host_error_recovery { -+ // Wasmer JS surfaces PostgreSQL's wasm EH at the export boundary; -+ // retain nested guest unwinding and recover the top-level jump via -+ // its dedicated exit(100) path. -+ set_force -+ .call(&mut self.store, 0) -+ .context("oliphaunt_wasix_set_force_host_error_recovery(0)")?; -+ } + self.set_active + .call(&mut self.store, 1) + .context("oliphaunt_wasix_set_active(1)")?; @@ -513,41 +576,6 @@ index 0000000..e0f5463 + .context("pq_buffer_remaining_data") + } + -+ fn recover_protocol_error(&mut self, payload_len: usize) -> anyhow::Result<()> { -+ self.recover_error -+ .call(&mut self.store) -+ .context("PostgresMainLongJmp after protocol trap")?; -+ let max_attempts = (payload_len / 5).saturating_add(2).max(1); -+ let mut attempts = 0usize; -+ while self.protocol_input_remaining()? > 0 { -+ attempts += 1; -+ ensure!( -+ attempts <= max_attempts, -+ "protocol recovery did not drain input" -+ ); -+ if self.main_loop.call(&mut self.store).is_err() { -+ self.recover_error -+ .call(&mut self.store) -+ .context("PostgresMainLongJmp while draining protocol error")?; -+ } -+ } -+ Ok(()) -+ } -+ -+ fn recover_non_trapping_protocol_error(&mut self) -> anyhow::Result<()> { -+ self.recover_error -+ .call(&mut self.store) -+ .context("PostgresMainLongJmp after backend ErrorResponse")?; -+ self.send_ready -+ .call(&mut self.store) -+ .context("PostgresSendReadyForQueryIfNecessary after backend ErrorResponse")?; -+ self.pq_flush -+ .call(&mut self.store) -+ .context("oliphaunt_wasix_pq_flush after backend ErrorResponse")?; -+ let _ = self.take_output()?; -+ Ok(()) -+ } -+ + fn startup_error(&mut self, error: anyhow::Error) -> Error { + let _ = self.pq_flush.call(&mut self.store); + let protocol = self.take_output().unwrap_or_default(); @@ -641,26 +669,6 @@ index 0000000..e0f5463 + _ => None, + }) +} -+ -+fn protocol_response_contains_error(response: &[u8]) -> bool { -+ let mut cursor = 0usize; -+ while cursor + 5 <= response.len() { -+ let tag = response[cursor]; -+ let len = i32::from_be_bytes(response[cursor + 1..cursor + 5].try_into().unwrap()); -+ if len < 4 { -+ return false; -+ } -+ let total = 1usize.saturating_add(len as usize); -+ if cursor + total > response.len() { -+ return false; -+ } -+ if tag == b'E' { -+ return true; -+ } -+ cursor += total; -+ } -+ false -+} diff --git a/src/runtime.rs b/src/runtime.rs index d71638f..cb106d7 100644 --- a/src/runtime.rs @@ -668,7 +676,7 @@ index d71638f..cb106d7 100644 @@ -83,7 +83,7 @@ impl Runtime { Ok(rt) } - + - pub(crate) fn with_task_manager(&self, task_manager: Arc) -> Self { + pub(crate) fn with_task_manager(&self, task_manager: Arc) -> Self { let mut runtime = self.clone(); @@ -737,14 +745,14 @@ index 2e9b0ee..76f8695 100644 @@ -31,6 +31,7 @@ //! [`Worker`]: thread_pool_worker::ThreadPoolWorker //! [`Scheduler`]: scheduler::Scheduler - + +mod caller_realm; mod interop; mod post_message_payload; mod scheduler; @@ -42,6 +43,7 @@ mod worker_handle; mod worker_message; - + pub(crate) use self::{ + caller_realm::CallerRealmTaskManager, post_message_payload::{AsyncJob, BlockingJob, Notification, PostMessagePayload}, diff --git a/src/wasix/browser-host/patches/0008-wasmer-instantiate-js-modules-async.patch b/src/wasix/browser-host/patches/0008-wasmer-instantiate-js-modules-async.patch index ee888d33e..5d96b7765 100644 --- a/src/wasix/browser-host/patches/0008-wasmer-instantiate-js-modules-async.patch +++ b/src/wasix/browser-host/patches/0008-wasmer-instantiate-js-modules-async.patch @@ -1,3 +1,12 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer: instantiate JavaScript modules asynchronously + +Expose asynchronous module instantiation for the JavaScript backend. Large +PostgreSQL modules exceed browser main-thread synchronous instantiation limits; +non-JavaScript backends keep their existing instantiation semantics. This is +a general engine API addition suitable for a separate upstream proposal. + diff --git a/Cargo.toml b/Cargo.toml --- a/Cargo.toml +++ b/Cargo.toml diff --git a/src/wasix/browser-host/patches/0009-wasmer-wasix-instantiate-main-module-async.patch b/src/wasix/browser-host/patches/0009-wasmer-wasix-instantiate-main-module-async.patch index a35c89014..0d9706c62 100644 --- a/src/wasix/browser-host/patches/0009-wasmer-wasix-instantiate-main-module-async.patch +++ b/src/wasix/browser-host/patches/0009-wasmer-wasix-instantiate-main-module-async.patch @@ -1,3 +1,12 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: propagate asynchronous main-module instantiation + +Thread the asynchronous engine instantiation API through the WASIX builder, +environment and linker while preserving the synchronous entrypoint. This is +the host-side companion to the engine API patch, needed to start large browser +guests without falling back to worker-only placement. + diff --git a/src/state/builder.rs b/src/state/builder.rs --- a/src/state/builder.rs +++ b/src/state/builder.rs diff --git a/src/wasix/browser-host/patches/0010-wasmer-js-refresh-npm-lock.patch b/src/wasix/browser-host/patches/0010-wasmer-js-refresh-npm-lock.patch index 050643ba2..6ab8fbf9a 100644 --- a/src/wasix/browser-host/patches/0010-wasmer-js-refresh-npm-lock.patch +++ b/src/wasix/browser-host/patches/0010-wasmer-js-refresh-npm-lock.patch @@ -1,3 +1,11 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: align the pinned npm lock with package metadata + +Supply the missing root metadata and dependency entries in the pinned source +lock so npm ci can install without rewriting it. This is reproducibility repair +for this exact upstream revision, not a runtime optimization. + diff --git a/package-lock.json b/package-lock.json index ef54a7d..83d263d 100644 --- a/package-lock.json diff --git a/src/wasix/browser-host/patches/0011-wasmer-wasix-deny-single-backend-guest-spawn.patch b/src/wasix/browser-host/patches/0011-wasmer-wasix-deny-single-backend-guest-spawn.patch index 6a3285bfe..f21814408 100644 --- a/src/wasix/browser-host/patches/0011-wasmer-wasix-deny-single-backend-guest-spawn.patch +++ b/src/wasix/browser-host/patches/0011-wasmer-wasix-deny-single-backend-guest-spawn.patch @@ -1,35 +1,270 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: make execution and I/O policy host-owned + +Represent single-program execution, clock selection and close behavior as a +typed host policy. Deny guest process/thread creation before side effects and +retain control-plane limits as an independent defense. Guest environment +variables must not grant capabilities. The generic policy mechanism and +Oliphaunt's selected profile should be separated for upstream review. + +diff --git a/src/capabilities.rs b/src/capabilities.rs +index e3c3b0ab..cbff2665 100644 +--- a/src/capabilities.rs ++++ b/src/capabilities.rs +@@ -2,6 +2,73 @@ use std::time::Duration; + + use crate::http::HttpClientCapabilityV1; + ++/// Defines how much guest-controlled execution a WASI environment permits. ++#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)] ++pub enum WasiGuestExecutionMode { ++ /// Retain the normal WASIX process and thread behavior. ++ #[default] ++ General, ++ /// Keep the initial program as the environment's only process and thread. ++ SingleProgram, ++} ++ ++/// Selects the implementation used for `clock_time_get`. ++#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)] ++pub enum WasiClockTimeGetMode { ++ /// Use the complete Rust syscall for every clock read. ++ #[default] ++ Canonical, ++ /// Use the JavaScript clock import with bounded canonical fallbacks. ++ DirectJs, ++} ++ ++/// Describes whether a filesystem needs WASIX to flush before closing a file. ++#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)] ++pub enum WasiFdClosePolicy { ++ /// Preserve the upstream flush-before-close behavior. ++ #[default] ++ FlushBeforeClose, ++ /// Writes complete synchronously, so POSIX close must not become fsync. ++ WritesCompleteSynchronously, ++} ++ ++/// Immutable host-selected execution policy for one WASI environment. ++/// ++/// This policy is deliberately separate from guest environment variables and ++/// from mergeable [`Capabilities`]. The builder fixes it before instantiation. ++#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)] ++pub struct WasiHostPolicy { ++ guest_execution: WasiGuestExecutionMode, ++ clock_time_get: WasiClockTimeGetMode, ++ fd_close: WasiFdClosePolicy, ++} ++ ++impl WasiHostPolicy { ++ pub const fn new( ++ guest_execution: WasiGuestExecutionMode, ++ clock_time_get: WasiClockTimeGetMode, ++ fd_close: WasiFdClosePolicy, ++ ) -> Self { ++ Self { ++ guest_execution, ++ clock_time_get, ++ fd_close, ++ } ++ } ++ ++ pub const fn guest_execution(self) -> WasiGuestExecutionMode { ++ self.guest_execution ++ } ++ ++ pub const fn clock_time_get(self) -> WasiClockTimeGetMode { ++ self.clock_time_get ++ } ++ ++ pub const fn fd_close(self) -> WasiFdClosePolicy { ++ self.fd_close ++ } ++} ++ + /// Defines capabilities for a Wasi environment. + #[derive(Clone, Debug, PartialEq, Eq, Hash)] + pub struct Capabilities { +diff --git a/src/os/task/control_plane.rs b/src/os/task/control_plane.rs +index 6f88ee67..c1a82b77 100644 +--- a/src/os/task/control_plane.rs ++++ b/src/os/task/control_plane.rs +@@ -120,9 +120,9 @@ impl WasiControlPlane { + pub(crate) fn register_task(&self) -> Result { + let count = self.state.task_count.fetch_add(1, Ordering::SeqCst); + if let Some(max) = self.state.config.max_task_count { +- if count > max { ++ if count >= max { + self.state.task_count.fetch_sub(1, Ordering::SeqCst); +- return Err(ControlPlaneError::TaskLimitReached { max: count }); ++ return Err(ControlPlaneError::TaskLimitReached { max }); + } + } + Ok(TaskCountGuard(self.state.task_count.clone())) +@@ -234,6 +234,11 @@ mod tests { + .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) + .unwrap(); + ++ assert_eq!( ++ p1.new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ .err(), ++ Some(ControlPlaneError::TaskLimitReached { max: 2 }) ++ ); + assert_eq!( + p.new_process(ModuleHash::random()).unwrap_err(), + ControlPlaneError::TaskLimitReached { max: 2 } +diff --git a/src/state/builder.rs b/src/state/builder.rs +index 768e427d..d57e1913 100644 +--- a/src/state/builder.rs ++++ b/src/state/builder.rs +@@ -17,7 +17,7 @@ use wasmer_config::package::PackageId; + use crate::journal::{DynJournal, DynReadableJournal, SnapshotTrigger}; + use crate::{ + bin_factory::{BinFactory, BinaryPackage}, +- capabilities::Capabilities, ++ capabilities::{Capabilities, WasiGuestExecutionMode, WasiHostPolicy}, + fs::{WasiFs, WasiFsRoot, WasiInodes}, + os::task::control_plane::{ControlPlaneConfig, ControlPlaneError, WasiControlPlane}, + state::WasiState, +@@ -84,6 +84,8 @@ pub struct WasiEnvBuilder { + + pub(super) capabilites: Capabilities, + ++ pub(super) host_policy: WasiHostPolicy, ++ + #[cfg(feature = "journal")] + pub(super) snapshot_on: Vec, + +@@ -759,6 +761,16 @@ impl WasiEnvBuilder { + self.capabilites = capabilities; + } + ++ /// Selects immutable host-side behavior before the environment is built. ++ pub fn host_policy(mut self, host_policy: WasiHostPolicy) -> Self { ++ self.set_host_policy(host_policy); ++ self ++ } ++ ++ pub fn set_host_policy(&mut self, host_policy: WasiHostPolicy) { ++ self.host_policy = host_policy; ++ } ++ + #[cfg(feature = "journal")] + pub fn add_snapshot_trigger(&mut self, on: SnapshotTrigger) { + self.snapshot_on.push(on); +@@ -966,9 +978,17 @@ impl WasiEnvBuilder { + let bin_factory = BinFactory::new(runtime.clone()); + + let capabilities = self.capabilites; ++ let host_policy = self.host_policy; ++ ++ let max_task_count = ++ if host_policy.guest_execution() == WasiGuestExecutionMode::SingleProgram { ++ Some(1) ++ } else { ++ capabilities.threading.max_threads ++ }; + + let plane_config = ControlPlaneConfig { +- max_task_count: capabilities.threading.max_threads, ++ max_task_count, + enable_asynchronous_threading: capabilities.threading.enable_asynchronous_threading, + enable_exponential_cpu_backoff: capabilities.threading.enable_exponential_cpu_backoff, + }; +@@ -982,6 +1002,7 @@ impl WasiEnvBuilder { + control_plane, + bin_factory, + capabilities, ++ host_policy, + memory_ty: None, + process: None, + thread: None, +diff --git a/src/state/env.rs b/src/state/env.rs +index a9701e1e..77d8a892 100644 +--- a/src/state/env.rs ++++ b/src/state/env.rs +@@ -27,7 +27,7 @@ use webc::metadata::annotations::Wasi; + use crate::journal::{DynJournal, JournalEffector, SnapshotTrigger}; + use crate::{ + bin_factory::{BinFactory, BinaryPackage, BinaryPackageCommand}, +- capabilities::Capabilities, ++ capabilities::{Capabilities, WasiHostPolicy}, + fs::{WasiFsRoot, WasiInodes}, + import_object_for_all_wasi_versions, + os::task::{ +@@ -54,6 +54,7 @@ pub struct WasiEnvInit { + pub mapped_commands: HashMap, + pub bin_factory: BinFactory, + pub capabilities: Capabilities, ++ pub(crate) host_policy: WasiHostPolicy, + + pub control_plane: WasiControlPlane, + pub memory_ty: Option, +@@ -110,6 +111,7 @@ impl WasiEnvInit { + mapped_commands: self.mapped_commands.clone(), + bin_factory: self.bin_factory.clone(), + capabilities: self.capabilities.clone(), ++ host_policy: self.host_policy, + control_plane: self.control_plane.clone(), + memory_ty: None, + process: None, +@@ -139,6 +141,8 @@ pub struct WasiEnv { + pub vfork: Option, + /// Seed used to rotate around the events returned by `poll_oneoff` + pub poll_seed: u64, ++ /// Immutable behavior selected by the host before instantiation. ++ pub(crate) host_policy: WasiHostPolicy, + /// Shared state of the WASI system. Manages all the data that the + /// executing WASI program can see. + pub(crate) state: Arc, +@@ -193,6 +197,7 @@ impl Clone for WasiEnv { + control_plane: self.control_plane.clone(), + process: self.process.clone(), + poll_seed: self.poll_seed, ++ host_policy: self.host_policy, + thread: self.thread.clone(), + layout: self.layout.clone(), + vfork: self.vfork.clone(), +@@ -237,6 +242,7 @@ impl WasiEnv { + layout: self.layout.clone(), + vfork: None, + poll_seed: 0, ++ host_policy: self.host_policy, + bin_factory, + state, + inner: Default::default(), +@@ -375,6 +381,7 @@ impl WasiEnv { + layout, + vfork: None, + poll_seed: 0, ++ host_policy: init.host_policy, + state: Arc::new(init.state), + inner: Default::default(), + owned_handles: Vec::new(), diff --git a/src/syscalls/wasix/mod.rs b/src/syscalls/wasix/mod.rs -index 0ec461d..74c1c61 100644 +index 0ec461de..d6edc59b 100644 --- a/src/syscalls/wasix/mod.rs +++ b/src/syscalls/wasix/mod.rs -@@ -173,3 +173,18 @@ pub use tty_get::*; +@@ -173,3 +173,9 @@ pub use tty_get::*; pub use tty_set::*; - + use tracing::{debug_span, field, instrument, trace_span, Span}; + -+use crate::WasiEnv; ++use crate::{capabilities::WasiGuestExecutionMode, WasiEnv}; + -+const OLIPHAUNT_SINGLE_BACKEND_ENV: &[u8] = b"OLIPHAUNT_WASIX_SINGLE_BACKEND=1"; -+ -+fn oliphaunt_single_backend_requested(env: &WasiEnv) -> bool { -+ env.state -+ .envs -+ .lock() -+ .map(|envs| { -+ envs.iter() -+ .any(|entry| entry.as_slice() == OLIPHAUNT_SINGLE_BACKEND_ENV) -+ }) -+ .unwrap_or(true) ++fn single_program_requested(env: &WasiEnv) -> bool { ++ env.host_policy.guest_execution() == WasiGuestExecutionMode::SingleProgram +} diff --git a/src/syscalls/wasix/proc_exec3.rs b/src/syscalls/wasix/proc_exec3.rs -index e72cdd7..b6c29b7 100644 +index e72cdd7c..7362c34f 100644 --- a/src/syscalls/wasix/proc_exec3.rs +++ b/src/syscalls/wasix/proc_exec3.rs @@ -34,6 +34,11 @@ pub fn proc_exec3( ) -> Result { WasiEnv::do_pending_operations(&mut ctx)?; - -+ if oliphaunt_single_backend_requested(ctx.data()) { + ++ if single_program_requested(ctx.data()) { + warn!("guest process replacement denied for the Oliphaunt single-backend pgwire runtime"); + return Ok(Errno::Notcapable); + } @@ -38,14 +273,14 @@ index e72cdd7..b6c29b7 100644 if let Some(exit_code) = unsafe { handle_rewind::(&mut ctx) } { // We should never get here as the process will be termined diff --git a/src/syscalls/wasix/proc_fork.rs b/src/syscalls/wasix/proc_fork.rs -index f2b17e5..f7fead0 100644 +index f2b17e51..3c953a8e 100644 --- a/src/syscalls/wasix/proc_fork.rs +++ b/src/syscalls/wasix/proc_fork.rs @@ -27,6 +27,11 @@ pub fn proc_fork( ) -> Result { WasiEnv::do_pending_operations(&mut ctx)?; - -+ if oliphaunt_single_backend_requested(ctx.data()) { + ++ if single_program_requested(ctx.data()) { + warn!("guest process fork denied for the Oliphaunt single-backend pgwire runtime"); + return Ok(Errno::Notcapable); + } @@ -54,14 +289,14 @@ index f2b17e5..f7fead0 100644 warn!("process forking not supported for dynamically linked modules"); Errno::Notsup diff --git a/src/syscalls/wasix/proc_spawn.rs b/src/syscalls/wasix/proc_spawn.rs -index 958b9b0..fff03a8 100644 +index 958b9b03..48d7b0d6 100644 --- a/src/syscalls/wasix/proc_spawn.rs +++ b/src/syscalls/wasix/proc_spawn.rs @@ -42,6 +42,11 @@ pub fn proc_spawn( ) -> Result { WasiEnv::do_pending_operations(&mut ctx)?; - -+ if oliphaunt_single_backend_requested(ctx.data()) { + ++ if single_program_requested(ctx.data()) { + warn!("guest process creation denied for the Oliphaunt single-backend pgwire runtime"); + return Ok(Errno::Notcapable); + } @@ -70,14 +305,14 @@ index 958b9b0..fff03a8 100644 let control_plane = &env.control_plane; let memory = unsafe { env.memory_view(&ctx) }; diff --git a/src/syscalls/wasix/proc_spawn2.rs b/src/syscalls/wasix/proc_spawn2.rs -index b907118..2922514 100644 +index b9071184..4d26e950 100644 --- a/src/syscalls/wasix/proc_spawn2.rs +++ b/src/syscalls/wasix/proc_spawn2.rs @@ -44,6 +44,11 @@ pub fn proc_spawn2( ) -> Result { WasiEnv::do_pending_operations(&mut ctx)?; - -+ if oliphaunt_single_backend_requested(ctx.data()) { + ++ if single_program_requested(ctx.data()) { + warn!("guest process creation denied for the Oliphaunt single-backend pgwire runtime"); + return Ok(Errno::Notcapable); + } @@ -86,17 +321,15 @@ index b907118..2922514 100644 let memory = unsafe { ctx.data().memory_view(&ctx) }; let mut name = unsafe { get_input_str_ok!(&memory, name, name_len) }; diff --git a/src/syscalls/wasix/thread_spawn.rs b/src/syscalls/wasix/thread_spawn.rs -index 3598b70..b3fc666 100644 +index 87c4dfd5..8526b057 100644 --- a/src/syscalls/wasix/thread_spawn.rs +++ b/src/syscalls/wasix/thread_spawn.rs -@@ -57,6 +57,13 @@ pub fn thread_spawn_internal_from_wasi( +@@ -57,6 +57,11 @@ pub fn thread_spawn_internal_from_wasi( ctx: &mut FunctionEnvMut<'_, WasiEnv>, start_ptr: WasmPtr, M>, ) -> Result { -+ if oliphaunt_single_backend_requested(ctx.data()) { -+ warn!( -+ "guest thread creation denied for the Oliphaunt single-backend pgwire runtime" -+ ); ++ if single_program_requested(ctx.data()) { ++ warn!("guest thread creation denied for the Oliphaunt single-backend pgwire runtime"); + return Err(Errno::Notcapable); + } + diff --git a/src/wasix/browser-host/patches/0012-wasmer-js-remove-retired-wasm32-wasi-target.patch b/src/wasix/browser-host/patches/0012-wasmer-js-remove-retired-wasm32-wasi-target.patch index 9f164d0ba..37beec6ce 100644 --- a/src/wasix/browser-host/patches/0012-wasmer-js-remove-retired-wasm32-wasi-target.patch +++ b/src/wasix/browser-host/patches/0012-wasmer-js-remove-retired-wasm32-wasi-target.patch @@ -1,3 +1,11 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: remove the retired Rust target requirement + +Remove the obsolete wasm32-wasi target from the pinned toolchain manifest. +The browser SDK builds wasm32-unknown-unknown, so installing the retired target +is unnecessary and can prevent the otherwise valid source build. + diff --git a/rust-toolchain.toml b/rust-toolchain.toml --- a/rust-toolchain.toml +++ b/rust-toolchain.toml diff --git a/src/wasix/browser-host/patches/0013-wasmer-wasix-fast-single-backend-clock.patch b/src/wasix/browser-host/patches/0013-wasmer-wasix-fast-single-backend-clock.patch index 3ac9a97d0..100ae5124 100644 --- a/src/wasix/browser-host/patches/0013-wasmer-wasix-fast-single-backend-clock.patch +++ b/src/wasix/browser-host/patches/0013-wasmer-wasix-fast-single-backend-clock.patch @@ -1,145 +1,370 @@ -diff --git a/src/state/env.rs b/src/state/env.rs -index a9701e1..6895971 100644 ---- a/src/state/env.rs -+++ b/src/state/env.rs -@@ -139,6 +139,14 @@ pub struct WasiEnv { - pub vfork: Option, - /// Seed used to rotate around the events returned by `poll_oneoff` - pub poll_seed: u64, -+ /// Cached Oliphaunt runtime profile. This avoids locking and scanning the -+ /// guest environment on high-frequency syscall paths. -+ pub(crate) oliphaunt_single_backend: bool, -+ /// Counts fast clock reads so pending signals are still serviced at a -+ /// bounded interval without paying that cost on every timing sample. -+ pub(crate) oliphaunt_fast_clock_calls: u16, -+ /// Whether this environment has installed a synthetic WASI clock offset. -+ pub(crate) oliphaunt_clock_offset_active: bool, - /// Shared state of the WASI system. Manages all the data that the - /// executing WASI program can see. - pub(crate) state: Arc, -@@ -193,6 +201,9 @@ impl Clone for WasiEnv { - control_plane: self.control_plane.clone(), - process: self.process.clone(), - poll_seed: self.poll_seed, -+ oliphaunt_single_backend: self.oliphaunt_single_backend, -+ oliphaunt_fast_clock_calls: 0, -+ oliphaunt_clock_offset_active: self.oliphaunt_clock_offset_active, - thread: self.thread.clone(), - layout: self.layout.clone(), - vfork: self.vfork.clone(), -@@ -237,6 +248,9 @@ impl WasiEnv { - layout: self.layout.clone(), - vfork: None, - poll_seed: 0, -+ oliphaunt_single_backend: self.oliphaunt_single_backend, -+ oliphaunt_fast_clock_calls: 0, -+ oliphaunt_clock_offset_active: self.oliphaunt_clock_offset_active, - bin_factory, - state, - inner: Default::default(), -@@ -348,6 +362,18 @@ impl WasiEnv { - init: WasiEnvInit, - module_hash: ModuleHash, - ) -> Result { -+ const OLIPHAUNT_SINGLE_BACKEND_ENV: &[u8] = -+ b"OLIPHAUNT_WASIX_SINGLE_BACKEND=1"; -+ -+ let oliphaunt_single_backend = init -+ .state -+ .envs -+ .lock() -+ .map(|envs| { -+ envs.iter() -+ .any(|entry| entry.as_slice() == OLIPHAUNT_SINGLE_BACKEND_ENV) +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: bound direct JavaScript clock dispatch + +Install an opt-in JavaScript clock provider for single-program WASIX guests. +Realtime and monotonic state remain independent, and each selected domain +returns to the canonical syscall after either 16 observed milliseconds or +1,024 direct reads. Clock setters disable direct reads for every module using +the same memory before calling WASIX; merely importing a setter is harmless. +Unsupported clocks and invalid memory retain canonical behavior. Journaling +and CPU backoff remain incompatible with the direct provider. + +Also correct the canonical wasm32 JavaScript platform clock implementation so +realtime and monotonic clocks use their respective domains and unsupported CPU +clocks report Notsup. + +diff --git a/src/lib.rs b/src/lib.rs +index 287f2a4b..a68a6630 100644 +--- a/src/lib.rs ++++ b/src/lib.rs +@@ -92,6 +92,233 @@ pub use virtual_net::{ + }; + use wasmer_wasix_types::wasi::{Errno, ExitCode}; + ++#[cfg(all(feature = "js", target_arch = "wasm32"))] ++use wasm_bindgen::{prelude::wasm_bindgen, JsValue}; ++ ++#[cfg(all(feature = "js", target_arch = "wasm32"))] ++#[wasm_bindgen(inline_js = r#" ++const clockStates = new WeakMap(); ++function clockState(memory) { ++ let state = clockStates.get(memory); ++ if (state === undefined) { ++ state = { canonical: false }; ++ clockStates.set(memory, state); ++ } ++ return state; ++} ++export function oliphauntDirectClockImport(memory, fallback) { ++ const state = clockState(memory); ++ const MAX_CANONICAL_INTERVAL_MILLIS = 16; ++ // Coarsened or stalled clock sources must not suppress ++ // canonical checkpoints forever. ++ const MAX_DIRECT_READS_BETWEEN_FALLBACKS = 1024; ++ const MAX_CANONICAL_NANOSECONDS = 9223372036854775807n; ++ let lastRealtimeFallbackMillis; ++ let lastMonotonicFallbackMillis; ++ let realtimeDirectReadsSinceFallback = 0; ++ let monotonicDirectReadsSinceFallback = 0; ++ let monotonicAnchor; ++ let buffer = memory.buffer; ++ let view = new DataView(buffer); ++ ++ function sampleRealtimeMillis() { ++ const millis = Date.now(); ++ return Number.isFinite(millis) && millis >= 0 ? millis : undefined; ++ } ++ ++ function samplePerformanceMillis() { ++ const performanceObject = globalThis.performance; ++ if (performanceObject === undefined || typeof performanceObject.now !== "function") { ++ return undefined; ++ } ++ const millis = performanceObject.now(); ++ return Number.isFinite(millis) && millis >= 0 ? millis : undefined; ++ } ++ ++ function refreshView(timePointer) { ++ let pointer; ++ if (typeof timePointer === "bigint") { ++ if (timePointer < 0n || timePointer > BigInt(Number.MAX_SAFE_INTEGER)) { ++ return undefined; ++ } ++ pointer = Number(timePointer); ++ } else if (typeof timePointer === "number" && Number.isInteger(timePointer)) { ++ pointer = timePointer >>> 0; ++ } else { ++ return undefined; ++ } ++ ++ if (buffer !== memory.buffer) { ++ buffer = memory.buffer; ++ view = new DataView(buffer); ++ } ++ return pointer <= buffer.byteLength - 8 ? pointer : undefined; ++ } ++ ++ function callFallback(clockId, precision, timePointer) { ++ const errno = fallback(clockId, precision, timePointer); ++ if (clockId === 0) { ++ realtimeDirectReadsSinceFallback = 0; ++ lastRealtimeFallbackMillis = ++ errno === 0 ? sampleRealtimeMillis() : undefined; ++ } else if (clockId === 1) { ++ monotonicDirectReadsSinceFallback = 0; ++ const sampledMonotonicMillis = ++ errno === 0 ? samplePerformanceMillis() : undefined; ++ lastMonotonicFallbackMillis = sampledMonotonicMillis; ++ const pointer = errno === 0 ? refreshView(timePointer) : undefined; ++ monotonicAnchor = ++ sampledMonotonicMillis !== undefined && pointer !== undefined ++ ? { ++ millis: sampledMonotonicMillis, ++ nanoseconds: view.getBigUint64(pointer, true), ++ } ++ : undefined; ++ } ++ return errno; ++ } ++ ++ return function clock_time_get(clockId, precision, timePointer) { ++ if (state.canonical) return fallback(clockId, precision, timePointer); ++ let nanoseconds; ++ if (clockId === 0) { ++ const realtimeMillis = sampleRealtimeMillis(); ++ if (realtimeMillis === undefined || ++ lastRealtimeFallbackMillis === undefined || ++ realtimeMillis < lastRealtimeFallbackMillis || ++ realtimeDirectReadsSinceFallback >= MAX_DIRECT_READS_BETWEEN_FALLBACKS || ++ realtimeMillis - lastRealtimeFallbackMillis >= MAX_CANONICAL_INTERVAL_MILLIS) { ++ return callFallback(clockId, precision, timePointer); ++ } ++ nanoseconds = BigInt(Math.trunc(realtimeMillis)) * 1000000n; ++ } else if (clockId === 1) { ++ const monotonicMillis = samplePerformanceMillis(); ++ if (monotonicMillis === undefined || ++ lastMonotonicFallbackMillis === undefined || ++ monotonicMillis < lastMonotonicFallbackMillis || ++ monotonicDirectReadsSinceFallback >= MAX_DIRECT_READS_BETWEEN_FALLBACKS || ++ monotonicMillis - lastMonotonicFallbackMillis >= MAX_CANONICAL_INTERVAL_MILLIS || ++ monotonicAnchor === undefined || ++ monotonicMillis < monotonicAnchor.millis) { ++ return callFallback(clockId, precision, timePointer); ++ } ++ nanoseconds = monotonicAnchor.nanoseconds + ++ BigInt(Math.trunc((monotonicMillis - monotonicAnchor.millis) * 1000000)); ++ } else { ++ return callFallback(clockId, precision, timePointer); ++ } ++ ++ if (nanoseconds > MAX_CANONICAL_NANOSECONDS) { ++ return callFallback(clockId, precision, timePointer); ++ } ++ ++ const pointer = refreshView(timePointer); ++ if (pointer === undefined) { ++ return callFallback(clockId, precision, timePointer); ++ } ++ view.setBigUint64(pointer, nanoseconds, true); ++ if (clockId === 0) { ++ realtimeDirectReadsSinceFallback += 1; ++ } else { ++ monotonicDirectReadsSinceFallback += 1; ++ } ++ return 0; ++ }; ++} ++ ++export function oliphauntClockSetImport(memory, fallback) { ++ const state = clockState(memory); ++ return function clock_time_set(clockId, time) { ++ // Pending WASIX work may reenter another module. Switch all readers ++ // first, and stay canonical even when the setter returns an error. ++ state.canonical = true; ++ return fallback(clockId, time); ++ }; ++} ++"#)] ++extern "C" { ++ #[wasm_bindgen(js_name = oliphauntDirectClockImport)] ++ fn oliphaunt_direct_clock_import(memory: &JsValue, fallback: &JsValue) -> js_sys::Function; ++ #[wasm_bindgen(js_name = oliphauntClockSetImport)] ++ fn oliphaunt_clock_set_import(memory: &JsValue, fallback: &JsValue) -> js_sys::Function; ++} ++ ++#[cfg(all(feature = "js", target_arch = "wasm32"))] ++pub(crate) fn install_oliphaunt_direct_clock( ++ store: &mut impl AsStoreMut, ++ module: &wasmer::Module, ++ env: &FunctionEnv, ++ memory: Option<&wasmer::Memory>, ++ imports: &mut Imports, ++) -> Result<(), String> { ++ use wasmer::js::AsJs; ++ ++ let (host_policy, has_backoff, has_journal) = { ++ let store_ref = store.as_store_ref(); ++ let env = env.as_ref(&store_ref); ++ ( ++ env.host_policy, ++ env.enable_exponential_cpu_backoff.is_some(), ++ env.enable_journal, ++ ) ++ }; ++ if host_policy.clock_time_get() != capabilities::WasiClockTimeGetMode::DirectJs { ++ return Ok(()); ++ } ++ if host_policy.guest_execution() != capabilities::WasiGuestExecutionMode::SingleProgram { ++ return Err("direct JavaScript clock requires single-program guest execution".into()); ++ } ++ if has_backoff { ++ return Err("direct JavaScript clock is incompatible with CPU backoff".into()); ++ } ++ if has_journal { ++ return Err("direct JavaScript clock is incompatible with journaling".into()); ++ } ++ ++ const CLOCK_NAMESPACES: [&str; 4] = [ ++ "wasi_snapshot_preview1", ++ "wasi_unstable", ++ "wasix_32v1", ++ "wasix_64v1", ++ ]; ++ let clock_imports = module ++ .imports() ++ .filter(|import| { ++ CLOCK_NAMESPACES.contains(&import.module()) ++ && matches!(import.name(), "clock_time_get" | "clock_time_set") ++ }) ++ .map(|import| (import.module().to_string(), import.name().to_string())) ++ .collect::>(); ++ if clock_imports.is_empty() { ++ return Ok(()); ++ } ++ let memory = memory.ok_or_else(|| { ++ "direct JavaScript clock requires memory to be available before instantiation".to_string() ++ })?; ++ ++ let raw_memory = memory.as_jsvalue(&store.as_store_ref()); ++ for (namespace, name) in clock_imports { ++ let fallback = imports ++ .get_export(&namespace, &name) ++ .and_then(|export| match export { ++ wasmer::Extern::Function(function) => Some(function), ++ _ => None, + }) -+ .unwrap_or(true); - let process = if let Some(p) = init.process { - p - } else { -@@ -375,6 +401,9 @@ impl WasiEnv { - layout, - vfork: None, - poll_seed: 0, -+ oliphaunt_single_backend, -+ oliphaunt_fast_clock_calls: 0, -+ oliphaunt_clock_offset_active: false, - state: Arc::new(init.state), - inner: Default::default(), - owned_handles: Vec::new(), -diff --git a/src/syscalls/wasi/clock_time_get.rs b/src/syscalls/wasi/clock_time_get.rs -index 434bd95..4400a63 100644 ---- a/src/syscalls/wasi/clock_time_get.rs -+++ b/src/syscalls/wasi/clock_time_get.rs -@@ -28,20 +28,31 @@ pub fn clock_time_get( - precision: Timestamp, - time: WasmPtr, - ) -> Result { -- WasiEnv::do_pending_operations(&mut ctx)?; -- -- ctx = wasi_try_ok!(maybe_backoff::(ctx)?); -+ let oliphaunt_fast_path = ctx.data().oliphaunt_single_backend; -+ if oliphaunt_fast_path { -+ let check_pending = { -+ let env = ctx.data_mut(); -+ env.oliphaunt_fast_clock_calls = env.oliphaunt_fast_clock_calls.wrapping_add(1); -+ env.oliphaunt_fast_clock_calls & 0x03ff == 0 ++ .ok_or_else(|| format!("missing {namespace}.{name} canonical fallback"))?; ++ let function_type = fallback.ty(&store.as_store_ref()); ++ let raw_fallback = fallback.as_jsvalue(&store.as_store_ref()); ++ let raw_direct = if name == "clock_time_set" { ++ oliphaunt_clock_set_import(&raw_memory, &raw_fallback) ++ } else { ++ oliphaunt_direct_clock_import(&raw_memory, &raw_fallback) + }; -+ if check_pending { -+ WasiEnv::do_pending_operations(&mut ctx)?; -+ } -+ } else { -+ WasiEnv::do_pending_operations(&mut ctx)?; -+ ctx = wasi_try_ok!(maybe_backoff::(ctx)?); ++ let direct = wasmer::Function::from_jsvalue(store, &function_type, raw_direct.as_ref()) ++ .map_err(|error| format!("failed to install {namespace}.{name}: {error:?}"))?; ++ imports.define(&namespace, &name, direct); + } ++ Ok(()) ++} ++ + pub use crate::{ + fs::{default_fs_backing, Fd, WasiFs, WasiInodes, VIRTUAL_ROOT_FD}, + os::{ +diff --git a/src/state/env.rs b/src/state/env.rs +index 77d8a892..11842968 100644 +--- a/src/state/env.rs ++++ b/src/state/env.rs +@@ -546,6 +546,20 @@ impl WasiEnv { + None + }; - let env = ctx.data(); - let memory = unsafe { env.memory_view(&ctx) }; ++ #[cfg(all(feature = "js", target_arch = "wasm32"))] ++ crate::install_oliphaunt_direct_clock( ++ &mut store, ++ &module, ++ &func_env.env, ++ imported_memory.as_ref(), ++ &mut import_object, ++ ) ++ .map_err(|error| { ++ WasiThreadError::InstanceCreateFailed(Box::new(wasmer::InstantiationError::Link( ++ wasmer::LinkError::Resource(error), ++ ))) ++ })?; ++ + // Construct the instance. + let instance_result = if async_instantiation { + #[cfg(all(feature = "js", target_arch = "wasm32"))] +diff --git a/src/state/linker.rs b/src/state/linker.rs +index 4f509100..e462986d 100644 +--- a/src/state/linker.rs ++++ b/src/state/linker.rs +@@ -1180,6 +1180,20 @@ impl Linker { + &well_known_imports, + )?; - let mut t_out = wasi_try_ok!(platform_clock_time_get(clock_id, precision)); -- { -+ if !oliphaunt_fast_path || env.oliphaunt_clock_offset_active { - let guard = env.state.clock_offset.lock().unwrap(); - if let Some(offset) = guard.get(&clock_id) { - t_out += *offset; - } -- }; -+ } - wasi_try_mem_ok!(time.write(&memory, t_out as Timestamp)); - Ok(Errno::Success) - } -diff --git a/src/syscalls/wasi/clock_time_set.rs b/src/syscalls/wasi/clock_time_set.rs -index 2c1354e..db9996f 100644 ---- a/src/syscalls/wasi/clock_time_set.rs -+++ b/src/syscalls/wasi/clock_time_set.rs -@@ -47,5 +47,7 @@ pub fn clock_time_set_internal( ++ #[cfg(all(feature = "js", target_arch = "wasm32"))] ++ crate::install_oliphaunt_direct_clock( ++ store, ++ main_module, ++ &func_env.env, ++ Some(&memory), ++ &mut imports, ++ ) ++ .map_err(|error| { ++ LinkError::InstantiationError(InstantiationError::Link(wasmer::LinkError::Resource( ++ error, ++ ))) ++ })?; ++ + // TODO: figure out which way is faster (stubs in main or stubs in sides), + // use that ordering. My *guess* is that, since main exports all the libc + // functions and those are called frequently by basically any code, then giving +@@ -1428,6 +1442,20 @@ impl Linker { + &mut pending_resolutions, + )?; - let mut guard = env.state.clock_offset.lock().unwrap(); - guard.insert(clock_id, t_offset); -+ drop(guard); -+ ctx.data_mut().oliphaunt_clock_offset_active = true; - Errno::Success - } -diff --git a/src/syscalls/wasix/mod.rs b/src/syscalls/wasix/mod.rs -index ac01c1f..d11763e 100644 ---- a/src/syscalls/wasix/mod.rs -+++ b/src/syscalls/wasix/mod.rs -@@ -176,15 +176,6 @@ use tracing::{debug_span, field, instrument, trace_span, Span}; ++ #[cfg(all(feature = "js", target_arch = "wasm32"))] ++ crate::install_oliphaunt_direct_clock( ++ store, ++ &main_module, ++ &func_env.env, ++ Some(&memory), ++ &mut imports, ++ ) ++ .map_err(|error| { ++ LinkError::InstantiationError(InstantiationError::Link(wasmer::LinkError::Resource( ++ error, ++ ))) ++ })?; ++ + let main_instance = Instance::new(store, &main_module, &imports)?; - use crate::WasiEnv; + instance_group.main_instance = Some(main_instance.clone()); +@@ -2647,6 +2675,20 @@ impl InstanceGroupState { + &well_known_imports, + )?; --const OLIPHAUNT_SINGLE_BACKEND_ENV: &[u8] = b"OLIPHAUNT_WASIX_SINGLE_BACKEND=1"; -- - fn oliphaunt_single_backend_requested(env: &WasiEnv) -> bool { -- env.state -- .envs -- .lock() -- .map(|envs| { -- envs.iter() -- .any(|entry| entry.as_slice() == OLIPHAUNT_SINGLE_BACKEND_ENV) -- }) -- .unwrap_or(true) -+ env.oliphaunt_single_backend - } ++ #[cfg(all(feature = "js", target_arch = "wasm32"))] ++ crate::install_oliphaunt_direct_clock( ++ store, ++ &module, ++ env, ++ Some(&self.memory), ++ &mut imports, ++ ) ++ .map_err(|error| { ++ LinkError::InstantiationError(InstantiationError::Link(wasmer::LinkError::Resource( ++ error, ++ ))) ++ })?; ++ + let instance = Instance::new(store, &module, &imports)?; + + let instance_handles = WasiModuleInstanceHandles::new( +@@ -2757,6 +2799,20 @@ impl InstanceGroupState { + pending_resolutions, + )?; + ++ #[cfg(all(feature = "js", target_arch = "wasm32"))] ++ crate::install_oliphaunt_direct_clock( ++ store, ++ &dl_module.module, ++ env, ++ Some(&self.memory), ++ &mut imports, ++ ) ++ .map_err(|error| { ++ LinkError::InstantiationError(InstantiationError::Link(wasmer::LinkError::Resource( ++ error, ++ ))) ++ })?; ++ + let instance = Instance::new(store, &dl_module.module, &imports)?; + + // This is a non-main instance of a side module, so it needs a new TLS area diff --git a/src/syscalls/wasm.rs b/src/syscalls/wasm.rs index dd731f0..1cac381 100644 --- a/src/syscalls/wasm.rs @@ -152,7 +377,7 @@ index dd731f0..1cac381 100644 use wasmer::WasmRef; use crate::syscalls::types::{ -@@ -8,15 +7,33 @@ use crate::syscalls::types::{ +@@ -8,15 +6,33 @@ use crate::syscalls::types::{ *, }; @@ -183,14 +408,14 @@ index dd731f0..1cac381 100644 - Snapshot0Clockid::Realtime => 1, - Snapshot0Clockid::ProcessCputimeId => 1, - Snapshot0Clockid::ThreadCputimeId => 1, -+ Snapshot0Clockid::Monotonic => 1_000_000, -+ Snapshot0Clockid::Realtime => 1_000_000, -+ Snapshot0Clockid::ProcessCputimeId => 1_000_000, -+ Snapshot0Clockid::ThreadCputimeId => 1_000_000, ++ Snapshot0Clockid::Monotonic | Snapshot0Clockid::Realtime => 1_000_000, ++ Snapshot0Clockid::ProcessCputimeId | Snapshot0Clockid::ThreadCputimeId => { ++ return Err(Errno::Notsup) ++ } _ => return Err(Errno::Inval), }; Ok(t_out) -@@ -24,10 +29,21 @@ pub fn platform_clock_res_get( +@@ -24,10 +40,22 @@ pub fn platform_clock_res_get( pub fn platform_clock_time_get( clock_id: Snapshot0Clockid, @@ -211,9 +436,10 @@ index dd731f0..1cac381 100644 + .checked_mul(1_000_000) + .ok_or(Errno::Overflow) + } -+ Snapshot0Clockid::Monotonic -+ | Snapshot0Clockid::ProcessCputimeId -+ | Snapshot0Clockid::ThreadCputimeId => monotonic_time_ns(), ++ Snapshot0Clockid::Monotonic => monotonic_time_ns(), ++ Snapshot0Clockid::ProcessCputimeId | Snapshot0Clockid::ThreadCputimeId => { ++ Err(Errno::Notsup) ++ } + _ => Err(Errno::Inval), + } } diff --git a/src/wasix/browser-host/patches/0014-wasmer-js-track-directory-mutations.patch b/src/wasix/browser-host/patches/0014-wasmer-js-track-directory-mutations.patch index 2006a8a95..b645bcf54 100644 --- a/src/wasix/browser-host/patches/0014-wasmer-js-track-directory-mutations.patch +++ b/src/wasix/browser-host/patches/0014-wasmer-js-track-directory-mutations.patch @@ -1,3 +1,13 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: expose directory mutation tracking + +Track changed filesystem paths so persistent stores publish mutations instead +of rescanning an entire database after every operation. Keep untracked mounts +available for immutable runtime data. This interface is downstream storage +integration; upstream value depends on a general dirty-path API and ownership +contract, not the PostgreSQL benchmark alone. + diff --git a/src/fs/directory.rs b/src/fs/directory.rs index 34af7da..51fafa5 100644 --- a/src/fs/directory.rs diff --git a/src/wasix/browser-host/patches/0015-wasmer-js-add-sync-filesystem-bridge.patch b/src/wasix/browser-host/patches/0015-wasmer-js-add-sync-filesystem-bridge.patch index 50212600d..e67f11450 100644 --- a/src/wasix/browser-host/patches/0015-wasmer-js-add-sync-filesystem-bridge.patch +++ b/src/wasix/browser-host/patches/0015-wasmer-js-add-sync-filesystem-bridge.patch @@ -1,3 +1,13 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: bridge synchronous filesystem operations safely + +Adapt the browser's synchronous storage provider to virtual-fs, preserving +provider errors and filesystem semantics. Each cross-language transfer uses a +fresh owned JS buffer so callback retention or memory growth cannot alias guest +memory. The bridge protocol is Oliphaunt-specific; its ownership guarantees +are required correctness, not an optional copy-elision experiment. + diff --git a/src/fs/directory.rs b/src/fs/directory.rs index 7c28ec4..da777a4 100644 --- a/src/fs/directory.rs @@ -40,16 +50,22 @@ index 5053993..058a657 100644 +pub(crate) use self::sync_bridge::SyncBridgeFileSystem; diff --git a/src/fs/sync_bridge.rs b/src/fs/sync_bridge.rs new file mode 100644 -index 0000000..0dad6b9 +index 0000000..b9c3c12 --- /dev/null +++ b/src/fs/sync_bridge.rs -@@ -0,0 +1,640 @@ +@@ -0,0 +1,724 @@ +use std::{ ++ cell::RefCell, ++ collections::HashMap, + fmt, + io::{self, SeekFrom}, + path::{Path, PathBuf}, + pin::Pin, -+ sync::{Arc, Mutex}, ++ rc::Rc, ++ sync::{ ++ atomic::{AtomicU32, Ordering}, ++ Arc, Mutex, ++ }, + task::{Context, Poll}, +}; + @@ -99,6 +115,20 @@ index 0000000..0dad6b9 +const FLAG_APPEND: i32 = 1 << 4; +const FLAG_TRUNCATE: i32 = 1 << 5; + ++static NEXT_BACKEND_ID: AtomicU32 = AtomicU32::new(1); ++ ++thread_local! { ++ /// JavaScript references must remain in the realm that created them. ++ /// Rust filesystem objects carry only the corresponding numeric key. ++ static REALM_BACKENDS: RefCell>> = ++ RefCell::new(HashMap::new()); ++} ++ ++struct RealmBackend { ++ backend: JsValue, ++ request: Function, ++} ++ +/// A synchronous virtual filesystem delegated to a JavaScript object in the +/// caller realm. The backend owns OPFS access handles while Rust owns the +/// syscall-facing filesystem. Calls are deliberately single-flight because @@ -241,22 +271,18 @@ index 0000000..0dad6b9 +} + +struct Backend { -+ backend: JsValue, -+ request: Function, ++ id: u32, ++ origin_thread: u32, + capacity: usize, + gate: Mutex<()>, +} + -+// wasm-bindgen 0.2.101 marks JS references realm-local. This filesystem is -+// constructed only for Oliphaunt's enforced single-backend worker, and `gate` -+// serializes every call. The references never cross a JavaScript realm. -+unsafe impl Send for Backend {} -+unsafe impl Sync for Backend {} -+ +impl fmt::Debug for Backend { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter + .debug_struct("Backend") ++ .field("id", &self.id) ++ .field("origin_thread", &self.origin_thread) + .field("capacity", &self.capacity()) + .finish() + } @@ -271,9 +297,23 @@ index 0000000..0dad6b9 + .map_err(|_| FsError::InvalidInput)? + .dyn_into::() + .map_err(|_| FsError::InvalidInput)?; ++ let origin_thread = wasmer::js::current_thread_id(); ++ let id = REALM_BACKENDS ++ .try_with(|backends| { ++ let mut backends = backends.try_borrow_mut().map_err(|_| FsError::Lock)?; ++ let id = loop { ++ let candidate = NEXT_BACKEND_ID.fetch_add(1, Ordering::Relaxed); ++ if candidate != 0 && !backends.contains_key(&candidate) { ++ break candidate; ++ } ++ }; ++ backends.insert(id, Rc::new(RealmBackend { backend, request })); ++ Ok::(id) ++ }) ++ .map_err(|_| FsError::Lock)??; + Ok(Self { -+ backend, -+ request, ++ id, ++ origin_thread, + capacity, + gate: Mutex::new(()), + }) @@ -283,6 +323,21 @@ index 0000000..0dad6b9 + self.capacity + } + ++ fn realm_backend(&self) -> virtual_fs::Result> { ++ if wasmer::js::current_thread_id() != self.origin_thread { ++ return Err(FsError::Lock); ++ } ++ REALM_BACKENDS ++ .try_with(|backends| { ++ let backends = backends.try_borrow().map_err(|_| FsError::Lock)?; ++ backends ++ .get(&self.id) ++ .map(Rc::clone) ++ .ok_or(FsError::InvalidFd) ++ }) ++ .map_err(|_| FsError::Lock)? ++ } ++ + fn request( + &self, + opcode: i32, @@ -330,27 +385,36 @@ index 0000000..0dad6b9 + arg1: u64, + flags: i32, + ) -> virtual_fs::Result { -+ let _guard = self.gate.lock().map_err(|_| FsError::Lock)?; ++ // A JavaScript callback can synchronously reenter this filesystem. ++ // Never wait for a lock that the current call already holds. ++ let _guard = self.gate.try_lock().map_err(|_| FsError::Lock)?; + if payload.len() > self.capacity() || output.len() > self.capacity() { + return Err(FsError::StorageFull); + } ++ if !payload.is_empty() && !output.is_empty() { ++ return Err(FsError::InvalidInput); ++ } ++ let realm_backend = self.realm_backend()?; ++ let transfer_length = payload.len().max(output.len()); ++ let transfer = Uint8Array::new_with_length( ++ u32::try_from(transfer_length).map_err(|_| FsError::StorageFull)?, ++ ); ++ if !payload.is_empty() { ++ transfer.copy_from(payload); ++ } + let arguments = Array::new(); + arguments.push(&JsValue::from(opcode)); + arguments.push(&JsValue::from_str(path)); -+ // Construct all Rust-owned arguments before exposing a view into Wasm -+ // memory. The backend call is synchronous and cannot retain the view. -+ let view = if output.is_empty() { -+ unsafe { Uint8Array::view(payload) } -+ } else { -+ unsafe { Uint8Array::view_mut_raw(output.as_mut_ptr(), output.len()) } -+ }; -+ arguments.push(view.as_ref()); ++ // Each callback receives an exact-sized JavaScript-owned allocation. ++ // It may synchronously mutate, retain, or transfer the bytes without ++ // aliasing Rust memory or observing storage from another request. ++ arguments.push(transfer.as_ref()); + arguments.push(&JsValue::from_f64(arg0 as f64)); + arguments.push(&JsValue::from_f64(arg1 as f64)); + arguments.push(&JsValue::from(flags)); -+ let raw = self ++ let raw = realm_backend + .request -+ .apply(&self.backend, &arguments) ++ .apply(&realm_backend.backend, &arguments) + .map_err(|_| FsError::IOError)?; + let values = Array::from(&raw); + if values.length() != 4 { @@ -364,6 +428,20 @@ index 0000000..0dad6b9 + if response_len > output.len() { + return Err(FsError::InvalidData); + } ++ if response_len > 0 { ++ // A callback may transfer the allocation after performing a ++ // side-effecting operation. Detachment is only an error when a ++ // successful response still has bytes that Rust must copy. ++ if transfer.length() as usize != transfer_length { ++ return Err(FsError::InvalidData); ++ } ++ transfer ++ .subarray( ++ 0, ++ u32::try_from(response_len).map_err(|_| FsError::InvalidData)?, ++ ) ++ .copy_to(&mut output[..response_len]); ++ } + Ok(ResponseHead { + length: response_len, + value0, @@ -372,6 +450,22 @@ index 0000000..0dad6b9 + } +} + ++impl Drop for Backend { ++ fn drop(&mut self) { ++ // A filesystem may be released on a different worker because the ++ // virtual-fs traits require Send + Sync. Never touch foreign-realm JS ++ // references; in that exceptional case the realm-local entry leaks. ++ if wasmer::js::current_thread_id() != self.origin_thread { ++ return; ++ } ++ let _ = REALM_BACKENDS.try_with(|backends| { ++ if let Ok(mut backends) = backends.try_borrow_mut() { ++ backends.remove(&self.id); ++ } ++ }); ++ } ++} ++ +struct ResponseHead { + length: usize, + value0: u64, diff --git a/src/wasix/browser-host/patches/0016-wasmer-wasix-direct-single-backend-clock.patch b/src/wasix/browser-host/patches/0016-wasmer-wasix-direct-single-backend-clock.patch deleted file mode 100644 index a80f8d8e1..000000000 --- a/src/wasix/browser-host/patches/0016-wasmer-wasix-direct-single-backend-clock.patch +++ /dev/null @@ -1,304 +0,0 @@ -diff --git a/src/lib.rs b/src/lib.rs ---- a/src/lib.rs -+++ b/src/lib.rs -@@ -92,6 +92,178 @@ pub use crate::{ - }; - use wasmer_wasix_types::wasi::{Errno, ExitCode}; - -+#[cfg(all(feature = "js", target_arch = "wasm32"))] -+use wasm_bindgen::{prelude::wasm_bindgen, JsValue}; -+ -+#[cfg(all(feature = "js", target_arch = "wasm32"))] -+#[wasm_bindgen(inline_js = r#" -+export function oliphauntFastClockImport(memory, fallback) { -+ let lastFallbackWallMillis; -+ const monotonicAnchors = new Map(); -+ let buffer = memory.buffer; -+ let view = new DataView(buffer); -+ function refreshView(timePointer) { -+ const pointer = timePointer >>> 0; -+ if (buffer !== memory.buffer) { -+ buffer = memory.buffer; -+ view = new DataView(buffer); -+ } -+ return pointer + 8 <= buffer.byteLength ? pointer : undefined; -+ } -+ function callFallback(clockId, precision, timePointer) { -+ const errno = fallback(clockId, precision, timePointer); -+ if (errno === 0) { -+ const wallMillis = Date.now(); -+ if (Number.isFinite(wallMillis) && wallMillis >= 0) { -+ lastFallbackWallMillis = wallMillis; -+ } -+ } -+ return errno; -+ } -+ function fallbackAndCalibrate(clockId, precision, timePointer, monotonicMillis) { -+ const errno = callFallback(clockId, precision, timePointer); -+ if (errno !== 0 || monotonicMillis === undefined) { -+ return errno; -+ } -+ const pointer = refreshView(timePointer); -+ if (pointer === undefined) { -+ return errno; -+ } -+ const sampledMillis = globalThis.performance.now(); -+ if (!Number.isFinite(sampledMillis) || sampledMillis < monotonicMillis) { -+ monotonicAnchors.delete(clockId); -+ return errno; -+ } -+ monotonicAnchors.set(clockId, { -+ millis: sampledMillis, -+ nanoseconds: view.getBigUint64(pointer, true), -+ }); -+ return errno; -+ } -+ return function clock_time_get(clockId, precision, timePointer) { -+ let millis; -+ let nanoseconds; -+ if (clockId === 0) { -+ millis = Date.now(); -+ if (!Number.isFinite(millis) || millis < 0) { -+ return callFallback(clockId, precision, timePointer); -+ } -+ nanoseconds = BigInt(Math.trunc(millis)) * 1000000n; -+ } else if (clockId === 1) { -+ if (globalThis.performance === undefined) { -+ return callFallback(clockId, precision, timePointer); -+ } -+ millis = globalThis.performance.now(); -+ if (!Number.isFinite(millis) || millis < 0) { -+ return callFallback(clockId, precision, timePointer); -+ } -+ const anchor = monotonicAnchors.get(clockId); -+ if (anchor === undefined || millis < anchor.millis || millis - anchor.millis >= 16) { -+ return fallbackAndCalibrate(clockId, precision, timePointer, millis); -+ } -+ nanoseconds = anchor.nanoseconds + BigInt(Math.round((millis - anchor.millis) * 1000000)); -+ } else { -+ return callFallback(clockId, precision, timePointer); -+ } -+ -+ // Service Wasmer's pending signal/timer work on a real-time bound, -+ // independent of how quickly or slowly the guest reads its clock. -+ // Keep scheduling in one clock domain. Date.now() rollback forces an -+ // immediate canonical fallback, which also re-establishes the bound. -+ const wallMillis = Date.now(); -+ if (!Number.isFinite(wallMillis) || wallMillis < 0 || -+ lastFallbackWallMillis === undefined || -+ wallMillis < lastFallbackWallMillis || -+ wallMillis - lastFallbackWallMillis >= 16) { -+ return callFallback(clockId, precision, timePointer); -+ } -+ -+ const pointer = refreshView(timePointer); -+ if (pointer === undefined) { -+ return callFallback(clockId, precision, timePointer); -+ } -+ view.setBigUint64(pointer, nanoseconds, true); -+ return 0; -+ }; -+} -+"#)] -+extern "C" { -+ #[wasm_bindgen(js_name = oliphauntFastClockImport)] -+ fn oliphaunt_fast_clock_import( -+ memory: &JsValue, -+ fallback: &JsValue, -+ ) -> js_sys::Function; -+} -+ -+#[cfg(all(feature = "js", target_arch = "wasm32"))] -+pub(crate) fn install_oliphaunt_fast_clock( -+ store: &mut impl AsStoreMut, -+ module: &wasmer::Module, -+ env: &FunctionEnv, -+ memory: &wasmer::Memory, -+ imports: &mut Imports, -+) -> Result<(), String> { -+ use wasmer::js::AsJs; -+ -+ if !env.as_ref(&store.as_store_ref()).oliphaunt_single_backend { -+ return Ok(()); -+ } -+ -+ const CLOCK_NAMESPACES: [&str; 2] = ["wasi_snapshot_preview1", "wasi_unstable"]; -+ if module.imports().any(|import| { -+ CLOCK_NAMESPACES.contains(&import.module()) && import.name() == "clock_time_set" -+ }) { -+ return Ok(()); -+ } -+ -+ let namespaces = module -+ .imports() -+ .filter(|import| { -+ CLOCK_NAMESPACES.contains(&import.module()) && import.name() == "clock_time_get" -+ }) -+ .map(|import| import.module().to_string()) -+ .collect::>(); -+ if namespaces.is_empty() { -+ return Ok(()); -+ } -+ -+ let raw_memory = memory.as_jsvalue(&store.as_store_ref()); -+ for namespace in namespaces { -+ let fallback = imports -+ .get_export(&namespace, "clock_time_get") -+ .and_then(|export| match export { -+ wasmer::Extern::Function(function) => Some(function), -+ _ => None, -+ }) -+ .ok_or_else(|| format!("missing {namespace}.clock_time_get fallback"))?; -+ let function_type = fallback.ty(&store.as_store_ref()); -+ let raw_fallback = fallback.as_jsvalue(&store.as_store_ref()); -+ let raw_fast = oliphaunt_fast_clock_import(&raw_memory, &raw_fallback); -+ let fast = wasmer::Function::from_jsvalue( -+ store, -+ &function_type, -+ raw_fast.as_ref(), -+ ) -+ .map_err(|error| format!("{error:?}"))?; -+ imports.define(&namespace, "clock_time_get", fast); -+ } -+ env.as_mut(&mut store.as_store_mut()) -+ .oliphaunt_direct_clock_active = true; -+ Ok(()) -+} -+ -+#[cfg(all(feature = "js", target_arch = "wasm32"))] -+pub fn oliphaunt_direct_memory( -+ env: &WasiFunctionEnv, -+ store: &impl wasmer::AsStoreRef, -+) -> Option { -+ use wasmer::js::AsJs; -+ -+ env.data(store) -+ .try_memory() -+ .map(|memory| memory.as_jsvalue(store)) -+} -+ - pub use crate::{ - fs::{default_fs_backing, Fd, WasiFs, WasiInodes, VIRTUAL_ROOT_FD}, - os::{ -diff --git a/src/state/env.rs b/src/state/env.rs ---- a/src/state/env.rs -+++ b/src/state/env.rs -@@ -145,6 +145,8 @@ pub struct WasiEnv { - /// Counts fast clock reads so pending signals are still serviced at a - /// bounded interval without paying that cost on every timing sample. - pub(crate) oliphaunt_fast_clock_calls: u16, -+ /// Whether the guest clock import uses the direct JavaScript wrapper. -+ pub(crate) oliphaunt_direct_clock_active: bool, - /// Whether this environment has installed a synthetic WASI clock offset. - pub(crate) oliphaunt_clock_offset_active: bool, - /// Shared state of the WASI system. Manages all the data that the -@@ -203,6 +205,7 @@ impl Clone for WasiEnv { - poll_seed: self.poll_seed, - oliphaunt_single_backend: self.oliphaunt_single_backend, - oliphaunt_fast_clock_calls: 0, -+ oliphaunt_direct_clock_active: false, - oliphaunt_clock_offset_active: self.oliphaunt_clock_offset_active, - thread: self.thread.clone(), - layout: self.layout.clone(), -@@ -250,6 +253,7 @@ impl WasiEnv { - poll_seed: 0, - oliphaunt_single_backend: self.oliphaunt_single_backend, - oliphaunt_fast_clock_calls: 0, -+ oliphaunt_direct_clock_active: false, - oliphaunt_clock_offset_active: self.oliphaunt_clock_offset_active, - bin_factory, - state, -@@ -403,6 +407,7 @@ impl WasiEnv { - poll_seed: 0, - oliphaunt_single_backend, - oliphaunt_fast_clock_calls: 0, -+ oliphaunt_direct_clock_active: false, - oliphaunt_clock_offset_active: false, - state: Arc::new(init.state), - inner: Default::default(), -@@ -568,6 +573,22 @@ impl WasiEnv { - None - }; - -+ #[cfg(all(feature = "js", target_arch = "wasm32"))] -+ if let Some(memory) = imported_memory.as_ref() { -+ crate::install_oliphaunt_fast_clock( -+ &mut store, -+ &module, -+ &func_env.env, -+ memory, -+ &mut import_object, -+ ) -+ .map_err(|error| { -+ WasiThreadError::InstanceCreateFailed(Box::new( -+ wasmer::InstantiationError::Link(wasmer::LinkError::Resource(error)), -+ )) -+ })?; -+ } -+ - // Construct the instance. - let instance_result = if async_instantiation { - #[cfg(all(feature = "js", target_arch = "wasm32"))] -diff --git a/src/state/linker.rs b/src/state/linker.rs ---- a/src/state/linker.rs -+++ b/src/state/linker.rs -@@ -1180,6 +1180,20 @@ impl Linker { - &well_known_imports, - )?; - -+ #[cfg(all(feature = "js", target_arch = "wasm32"))] -+ crate::install_oliphaunt_fast_clock( -+ store, -+ main_module, -+ &func_env.env, -+ &memory, -+ &mut imports, -+ ) -+ .map_err(|error| { -+ LinkError::InstantiationError(InstantiationError::Link( -+ wasmer::LinkError::Resource(error), -+ )) -+ })?; -+ - // TODO: figure out which way is faster (stubs in main or stubs in sides), - // use that ordering. My *guess* is that, since main exports all the libc - // functions and those are called frequently by basically any code, then giving -@@ -1428,6 +1442,20 @@ impl Linker { - &mut pending_resolutions, - )?; - -+ #[cfg(all(feature = "js", target_arch = "wasm32"))] -+ crate::install_oliphaunt_fast_clock( -+ store, -+ &main_module, -+ &func_env.env, -+ &memory, -+ &mut imports, -+ ) -+ .map_err(|error| { -+ LinkError::InstantiationError(InstantiationError::Link( -+ wasmer::LinkError::Resource(error), -+ )) -+ })?; -+ - let main_instance = Instance::new(store, &main_module, &imports)?; - - instance_group.main_instance = Some(main_instance.clone()); -diff --git a/src/syscalls/wasi/clock_time_get.rs b/src/syscalls/wasi/clock_time_get.rs ---- a/src/syscalls/wasi/clock_time_get.rs -+++ b/src/syscalls/wasi/clock_time_get.rs -@@ -32,8 +32,12 @@ pub fn clock_time_get( - if oliphaunt_fast_path { - let check_pending = { - let env = ctx.data_mut(); -- env.oliphaunt_fast_clock_calls = env.oliphaunt_fast_clock_calls.wrapping_add(1); -- env.oliphaunt_fast_clock_calls & 0x03ff == 0 -+ if env.oliphaunt_direct_clock_active { -+ true -+ } else { -+ env.oliphaunt_fast_clock_calls = env.oliphaunt_fast_clock_calls.wrapping_add(1); -+ env.oliphaunt_fast_clock_calls & 0x03ff == 0 -+ } - }; - if check_pending { - WasiEnv::do_pending_operations(&mut ctx)?; diff --git a/src/wasix/browser-host/patches/0016-wasmer-wasix-expose-direct-memory.patch b/src/wasix/browser-host/patches/0016-wasmer-wasix-expose-direct-memory.patch new file mode 100644 index 000000000..0e161d8e0 --- /dev/null +++ b/src/wasix/browser-host/patches/0016-wasmer-wasix-expose-direct-memory.patch @@ -0,0 +1,36 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: expose the direct instance memory handle + +Expose the instance memory needed by the direct JS protocol adapter without +changing clock policy or guest execution. This small downstream accessor is +paired with the direct PostgreSQL host; upstream should prefer an appropriately +scoped general memory accessor. + +diff --git a/src/lib.rs b/src/lib.rs +index 1064c373..4333c83c 100644 +--- a/src/lib.rs ++++ b/src/lib.rs +@@ -279,6 +279,22 @@ pub(crate) fn install_oliphaunt_direct_clock( + Ok(()) + } + ++/// Return the JavaScript WebAssembly memory backing a WASIX environment. ++/// ++/// Oliphaunt's direct pgwire adapter uses this to transfer protocol bytes ++/// without routing them through an unrelated clock-import optimization. ++#[cfg(all(feature = "js", target_arch = "wasm32"))] ++pub fn oliphaunt_direct_memory( ++ env: &WasiFunctionEnv, ++ store: &impl wasmer::AsStoreRef, ++) -> Option { ++ use wasmer::js::AsJs; ++ ++ env.data(store) ++ .try_memory() ++ .map(|memory| memory.as_jsvalue(store)) ++} ++ + pub use crate::{ + fs::{default_fs_backing, Fd, WasiFs, WasiInodes, VIRTUAL_ROOT_FD}, + os::{ diff --git a/src/wasix/browser-host/patches/0017-wasmer-js-direct-pgwire-memory-bridge.patch b/src/wasix/browser-host/patches/0017-wasmer-js-direct-pgwire-memory-bridge.patch index a560fd1bd..8e3abc79b 100644 --- a/src/wasix/browser-host/patches/0017-wasmer-js-direct-pgwire-memory-bridge.patch +++ b/src/wasix/browser-host/patches/0017-wasmer-js-direct-pgwire-memory-bridge.patch @@ -1,8 +1,17 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: copy direct protocol bytes through guest memory + +Use the direct guest input/output exports with JavaScript typed arrays to +avoid redundant Rust staging. Returned protocol bytes remain owned copies; +no borrowed view may escape across guest reentry or memory growth. The copy +boundary is part of the downstream ABI and must remain independently safe. + diff --git a/src/postgres_direct.rs b/src/postgres_direct.rs -index b381faf5..87e202c5 100644 +index c45000a..67a600f 100644 --- a/src/postgres_direct.rs +++ b/src/postgres_direct.rs -@@ -2,15 +2,44 @@ use std::sync::Arc; +@@ -2,9 +2,11 @@ use std::sync::Arc; use anyhow::{ensure, Context}; use js_sys::{Uint8Array, WebAssembly}; @@ -16,8 +25,9 @@ index b381faf5..87e202c5 100644 use crate::{runtime::Runtime, tasks::CallerRealmTaskManager, utils::Error, RunOptions}; - const DEFAULT_PROGRAM_NAME: &str = "/bin/postgres"; - const OLIPHAUNT_EXIT_ALIVE: i32 = 99; +@@ -29,6 +31,32 @@ impl MainLoopOutcome { + } + } +#[wasm_bindgen(inline_js = r#" +export function oliphauntCopyToGuest(memory, pointer, input) { @@ -45,17 +55,17 @@ index b381faf5..87e202c5 100644 + length: u32, + ) -> Result; +} -+ /// Instantiate the integrated Oliphaunt/PostgreSQL guest in this JS realm. /// /// The returned driver is synchronous by design: every guest export runs on -@@ -31,14 +60,15 @@ pub struct OliphauntDirectInstance { +@@ -48,15 +76,15 @@ pub struct OliphauntDirectInstance { + _runtime: Arc, store: Store, _instance: WasmerInstance, - env: WasiFunctionEnv, -+ _env: WasiFunctionEnv, - malloc: TypedFunction, - free: TypedFunction, ++ _env: WasiFunctionEnv, + guest_memory: JsValue, input_reset: TypedFunction<(), i32>, - input_write: TypedFunction<(i32, i32), i32>, @@ -66,11 +76,10 @@ index b381faf5..87e202c5 100644 output_len: TypedFunction<(), i32>, - output_read: TypedFunction<(i32, i32), i32>, + output_data: TypedFunction<(), i32>, -+ output_contains_error: TypedFunction<(), i32>, - set_force_host_error_recovery: Option>, set_active: TypedFunction, wasi_start: TypedFunction<(), ()>, -@@ -62,8 +92,8 @@ impl OliphauntDirectInstance { + start_oliphaunt: TypedFunction<(), ()>, +@@ -78,8 +106,8 @@ impl OliphauntDirectInstance { /// Start PostgreSQL and process one frontend startup packet. #[wasm_bindgen(js_name = startup)] pub fn startup(&mut self, packet: Uint8Array) -> Result { @@ -81,7 +90,7 @@ index b381faf5..87e202c5 100644 Err(error) => Err(self.startup_error(error)), } } -@@ -71,9 +101,7 @@ impl OliphauntDirectInstance { +@@ -87,9 +115,7 @@ impl OliphauntDirectInstance { /// Execute raw PostgreSQL frontend-protocol bytes synchronously. #[wasm_bindgen(js_name = execProtocolRaw)] pub fn exec_protocol_raw(&mut self, input: Uint8Array) -> Result { @@ -92,7 +101,7 @@ index b381faf5..87e202c5 100644 } /// Shut down the embedded lifecycle. This does not consume the JS object. -@@ -83,7 +111,7 @@ impl OliphauntDirectInstance { +@@ -99,7 +125,7 @@ impl OliphauntDirectInstance { } impl OliphauntDirectInstance { @@ -101,7 +110,7 @@ index b381faf5..87e202c5 100644 self.ensure_open()?; ensure!( !self.protocol_started, -@@ -91,7 +119,7 @@ impl OliphauntDirectInstance { +@@ -107,7 +133,7 @@ impl OliphauntDirectInstance { ); self.start_backend()?; self.reset_io()?; @@ -110,7 +119,7 @@ index b381faf5..87e202c5 100644 let port = self .get_port -@@ -112,22 +140,23 @@ impl OliphauntDirectInstance { +@@ -128,22 +154,23 @@ impl OliphauntDirectInstance { .context("oliphaunt_wasix_pq_flush after startup")?; self.protocol_started = true; } @@ -138,34 +147,18 @@ index b381faf5..87e202c5 100644 + let payload_len = payload.length() as usize; + let max_attempts = (payload_len / 5).saturating_add(2).max(1); let mut attempts = 0usize; - let mut recovered_protocol_error = false; while self.protocol_input_remaining()? > 0 { -@@ -140,7 +169,7 @@ impl OliphauntDirectInstance { - // Match the native host: the exported recovery boundary is - // authoritative even when a JS engine reports an unfamiliar - // trap spelling. -- self.recover_protocol_error(payload.len())?; -+ self.recover_protocol_error(payload_len)?; - recovered_protocol_error = true; - } + attempts += 1; +@@ -162,7 +189,7 @@ impl OliphauntDirectInstance { } -@@ -150,8 +179,13 @@ impl OliphauntDirectInstance { - self.pq_flush - .call(&mut self.store) - .context("oliphaunt_wasix_pq_flush after protocol input")?; -- let output = self.take_output()?; -- if !recovered_protocol_error && protocol_response_contains_error(&output) { -+ let output_contains_error = self -+ .output_contains_error -+ .call(&mut self.store) -+ .context("oliphaunt_wasix_output_contains_error")? -+ != 0; -+ let output = self.take_output_js()?; -+ if !recovered_protocol_error && output_contains_error { - self.recover_non_trapping_protocol_error()?; - } - Ok(output) -@@ -196,6 +230,8 @@ impl OliphauntDirectInstance { + self.send_ready_after_main_loop()?; + self.flush_after_main_loop()?; +- self.take_output() ++ self.take_output_js() + } + + fn close_inner(&mut self) -> anyhow::Result<()> { +@@ -204,6 +231,8 @@ impl OliphauntDirectInstance { .instantiate_async(module, &mut store) .await .context("instantiate Oliphaunt direct WASIX module")?; @@ -174,7 +167,7 @@ index b381faf5..87e202c5 100644 seed_exported_c_string( &mut store, &instance, -@@ -204,16 +240,19 @@ impl OliphauntDirectInstance { +@@ -212,16 +241,14 @@ impl OliphauntDirectInstance { DEFAULT_PROGRAM_NAME, )?; @@ -191,21 +184,17 @@ index b381faf5..87e202c5 100644 let output_len = typed_export(&mut store, &instance, "oliphaunt_wasix_output_len")?; - let output_read = typed_export(&mut store, &instance, "oliphaunt_wasix_output_read")?; + let output_data = typed_export(&mut store, &instance, "oliphaunt_wasix_output_data")?; -+ let output_contains_error = typed_export( -+ &mut store, -+ &instance, -+ "oliphaunt_wasix_output_contains_error", -+ )?; - let set_force_host_error_recovery = optional_typed_export( - &mut store, - &instance, -@@ -243,14 +282,15 @@ impl OliphauntDirectInstance { + let set_active = typed_export(&mut store, &instance, "oliphaunt_wasix_set_active")?; + let wasi_start = typed_export(&mut store, &instance, "_start")?; + let start_oliphaunt = typed_export(&mut store, &instance, "oliphaunt_wasix_start")?; +@@ -244,15 +271,15 @@ impl OliphauntDirectInstance { + _runtime: runtime, store, _instance: instance, - env, -+ _env: env, - malloc, - free, ++ _env: env, + guest_memory, input_reset, - input_write, @@ -216,11 +205,10 @@ index b381faf5..87e202c5 100644 output_len, - output_read, + output_data, -+ output_contains_error, - set_force_host_error_recovery, set_active, wasi_start, -@@ -322,68 +362,42 @@ impl OliphauntDirectInstance { + start_oliphaunt, +@@ -375,68 +402,42 @@ impl OliphauntDirectInstance { Ok(()) } @@ -314,7 +302,7 @@ index b381faf5..87e202c5 100644 ensure!( self.output_reset .call(&mut self.store) -@@ -391,16 +405,7 @@ impl OliphauntDirectInstance { +@@ -444,16 +445,7 @@ impl OliphauntDirectInstance { == 0, "oliphaunt_wasix_output_reset after read failed" ); @@ -332,14 +320,7 @@ index b381faf5..87e202c5 100644 } fn protocol_input_remaining(&mut self) -> anyhow::Result { -@@ -448,14 +453,16 @@ impl OliphauntDirectInstance { - self.pq_flush - .call(&mut self.store) - .context("oliphaunt_wasix_pq_flush after backend ErrorResponse")?; -- let _ = self.take_output()?; -+ let _ = self.take_output_js()?; - Ok(()) - } +@@ -472,8 +464,10 @@ impl OliphauntDirectInstance { fn startup_error(&mut self, error: anyhow::Error) -> Error { let _ = self.pq_flush.call(&mut self.store); @@ -352,7 +333,7 @@ index b381faf5..87e202c5 100644 return Error::from(error); } -@@ -463,7 +470,7 @@ impl OliphauntDirectInstance { +@@ -481,7 +475,7 @@ impl OliphauntDirectInstance { let _ = js_sys::Reflect::set( &js_error, &wasm_bindgen::JsValue::from_str("protocolResponse"), @@ -361,27 +342,3 @@ index b381faf5..87e202c5 100644 ); let _ = js_sys::Reflect::set( &js_error, -@@ -545,23 +552,3 @@ fn runtime_exit_code(error: &wasmer::RuntimeError) -> Option { - _ => None, - }) - } -- --fn protocol_response_contains_error(response: &[u8]) -> bool { -- let mut cursor = 0usize; -- while cursor + 5 <= response.len() { -- let tag = response[cursor]; -- let len = i32::from_be_bytes(response[cursor + 1..cursor + 5].try_into().unwrap()); -- if len < 4 { -- return false; -- } -- let total = 1usize.saturating_add(len as usize); -- if cursor + total > response.len() { -- return false; -- } -- if tag == b'E' { -- return true; -- } -- cursor += total; -- } -- false --} diff --git a/src/wasix/browser-host/patches/0018-wasmer-js-bound-direct-stderr.patch b/src/wasix/browser-host/patches/0018-wasmer-js-bound-direct-stderr.patch index d34f7af51..de2a4a9e3 100644 --- a/src/wasix/browser-host/patches/0018-wasmer-js-bound-direct-stderr.patch +++ b/src/wasix/browser-host/patches/0018-wasmer-js-bound-direct-stderr.patch @@ -1,6 +1,17 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: bound direct guest stderr diagnostics + +Retain a bounded tail of guest stderr and attach it to direct execution errors. +This preserves useful startup/trap diagnostics without an unbounded output +buffer or asynchronous stream. Truncation here affects diagnostic stderr only, +not PostgreSQL protocol or captured frontend-tool output. + +diff --git a/src/options.rs b/src/options.rs +index d56c81d..d2f8d00 100644 --- a/src/options.rs +++ b/src/options.rs -@@ -189,12 +189,13 @@ +@@ -189,12 +189,13 @@ impl RunOptions { pub(crate) fn configure_direct_builder( &self, builder: &mut WasiEnvBuilder, @@ -15,6 +26,8 @@ Ok(()) } +diff --git a/src/postgres_direct.rs b/src/postgres_direct.rs +index 67a600f..2e090d3 100644 --- a/src/postgres_direct.rs +++ b/src/postgres_direct.rs @@ -1,7 +1,15 @@ @@ -34,7 +47,7 @@ use wasm_bindgen::{prelude::wasm_bindgen, JsValue}; use wasmer::{Instance as WasmerInstance, Store, TypedFunction, Value, WasmTypeList}; use wasmer_wasix::{ -@@ -12,6 +20,133 @@ +@@ -12,6 +20,133 @@ use crate::{runtime::Runtime, tasks::CallerRealmTaskManager, utils::Error, RunOp const DEFAULT_PROGRAM_NAME: &str = "/bin/postgres"; const OLIPHAUNT_EXIT_ALIVE: i32 = 99; @@ -166,9 +179,9 @@ + } +} - #[wasm_bindgen(inline_js = r#" - export function oliphauntCopyToGuest(memory, pointer, input) { -@@ -50,7 +185,10 @@ + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + enum MainLoopOutcome { +@@ -67,7 +202,10 @@ pub async fn instantiate_oliphaunt_direct( module_bytes: Uint8Array, options: RunOptions, ) -> Result { @@ -180,7 +193,7 @@ } #[wasm_bindgen] -@@ -60,6 +198,7 @@ +@@ -77,6 +215,7 @@ pub struct OliphauntDirectInstance { store: Store, _instance: WasmerInstance, _env: WasiFunctionEnv, @@ -188,7 +201,7 @@ guest_memory: JsValue, input_reset: TypedFunction<(), i32>, input_reserve: TypedFunction, -@@ -101,12 +240,18 @@ +@@ -115,12 +254,18 @@ impl OliphauntDirectInstance { /// Execute raw PostgreSQL frontend-protocol bytes synchronously. #[wasm_bindgen(js_name = execProtocolRaw)] pub fn exec_protocol_raw(&mut self, input: Uint8Array) -> Result { @@ -209,7 +222,7 @@ } } -@@ -212,6 +357,7 @@ +@@ -213,6 +358,7 @@ impl OliphauntDirectInstance { module: WebAssembly::Module, module_bytes: Uint8Array, options: RunOptions, @@ -217,7 +230,7 @@ ) -> Result { // Direct execution intentionally has no configurable networking or // worker-backed runtime. RunOptions is reused for args/env/fs mounts. -@@ -222,7 +368,7 @@ +@@ -223,7 +369,7 @@ impl OliphauntDirectInstance { .as_string() .unwrap_or_else(|| DEFAULT_PROGRAM_NAME.to_owned()); let mut builder = WasiEnvBuilder::new(program_name).runtime(runtime.clone()); @@ -226,7 +239,7 @@ let module = wasmer::Module::from((module, module_bytes.to_vec())); let mut store = Store::new(runtime.engine()); -@@ -282,6 +428,7 @@ +@@ -272,6 +418,7 @@ impl OliphauntDirectInstance { store, _instance: instance, _env: env, @@ -234,7 +247,7 @@ guest_memory, input_reset, input_reserve, -@@ -458,6 +605,7 @@ +@@ -463,6 +610,7 @@ impl OliphauntDirectInstance { } fn startup_error(&mut self, error: anyhow::Error) -> Error { diff --git a/src/wasix/browser-host/patches/0019-wasmer-parse-exception-reference-types.patch b/src/wasix/browser-host/patches/0019-wasmer-parse-exception-reference-types.patch index 6a6137364..85ab6a135 100644 --- a/src/wasix/browser-host/patches/0019-wasmer-parse-exception-reference-types.patch +++ b/src/wasix/browser-host/patches/0019-wasmer-parse-exception-reference-types.patch @@ -1,7 +1,17 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer: recognize exception reference types in JS metadata + +Map nullable and non-null exception reference types to the engine exception +reference representation. PostgreSQL's Wasm exception-handling build otherwise +fails module metadata conversion before execution. This is a narrow upstream +engine compatibility fix, not a relaxation of module validation. + diff --git a/src/utils/polyfill.rs b/src/utils/polyfill.rs +index 8b54de4..6660bb3 100644 --- a/src/utils/polyfill.rs +++ b/src/utils/polyfill.rs -@@ -371,6 +371,8 @@ pub fn wpreftype_to_type(ty: wasmparser::RefType) -> WasmResult { +@@ -372,6 +372,8 @@ pub fn wpreftype_to_type(ty: wasmparser::RefType) -> WasmResult { Ok(Type::ExternRef) } else if ty.is_func_ref() { Ok(Type::FuncRef) diff --git a/src/wasix/browser-host/patches/0020-wasmer-wasix-preserve-posix-close-durability.patch b/src/wasix/browser-host/patches/0020-wasmer-wasix-preserve-posix-close-durability.patch index 631c3017a..cd53f8ef3 100644 --- a/src/wasix/browser-host/patches/0020-wasmer-wasix-preserve-posix-close-durability.patch +++ b/src/wasix/browser-host/patches/0020-wasmer-wasix-preserve-posix-close-durability.patch @@ -1,8 +1,24 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: distinguish close policy from explicit durability + +Keep the historical flush-before-close default, but let an explicit host +policy select POSIX close semantics when the filesystem writes synchronously. +Do not turn every descriptor close into a durability barrier; PostgreSQL's +explicit sync operations remain authoritative. This policy must never be +selected merely from a guest-controlled environment variable. + diff --git a/src/syscalls/wasi/fd_close.rs b/src/syscalls/wasi/fd_close.rs -index 647eec3..40756fe 100644 +index 41d27119..e97c6eeb 100644 --- a/src/syscalls/wasi/fd_close.rs +++ b/src/syscalls/wasi/fd_close.rs -@@ -30,13 +30,18 @@ pub fn fd_close(mut ctx: FunctionEnvMut<'_, WasiEnv>, fd: WasiFd) -> Result, fd: WasiFd) -> Result {} - Err(e) => { - return Ok(e); -+ if !env.oliphaunt_single_backend { ++ if env.host_policy.fd_close() == WasiFdClosePolicy::FlushBeforeClose { + // HACK: upstream uses tokio files to back WASI file handles. Since + // tokio does writes in the background, it may miss writes if the file + // is closed without flushing first. Hence, flush those hosts here. diff --git a/src/wasix/browser-host/patches/0021-wasmer-js-stream-direct-pgwire.patch b/src/wasix/browser-host/patches/0021-wasmer-js-stream-direct-pgwire.patch index a33026d70..4f34ecf85 100644 --- a/src/wasix/browser-host/patches/0021-wasmer-js-stream-direct-pgwire.patch +++ b/src/wasix/browser-host/patches/0021-wasmer-js-stream-direct-pgwire.patch @@ -1,16 +1,27 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: stream direct protocol and capture tool output + +Add bounded protocol callbacks, direct tool execution and complete in-memory +stdout/stderr capture. The transport follows the shared guest contract and +preserves owned callback bytes; capture failure fails instead of returning a +partial successful result. These are coupled downstream protocol/tool adapters, +not a claim that the bundled patch is ready for upstream as one change. + diff --git a/src/lib.rs b/src/lib.rs -index 6803f8e..0708da4 100644 +index 6803f8e..d8bee81 100644 --- a/src/lib.rs +++ b/src/lib.rs -@@ -18,6 +18,7 @@ mod run; +@@ -18,6 +18,8 @@ mod run; mod runtime; mod streams; mod tasks; ++mod protocol_contract; +mod tool_direct; mod utils; mod wasmer; mod ws; -@@ -33,6 +34,9 @@ pub use crate::{ +@@ -33,6 +35,9 @@ pub use crate::{ postgres_direct::{instantiate_oliphaunt_direct, OliphauntDirectInstance}, registry::RegistryConfig, run::run_wasix, @@ -21,10 +32,20 @@ index 6803f8e..0708da4 100644 wasmer::Wasmer, }; diff --git a/src/options.rs b/src/options.rs -index d2f8d00..8c0ff2f 100644 +index d2f8d00..4283c82 100644 --- a/src/options.rs +++ b/src/options.rs -@@ -12,6 +12,8 @@ use wasmer_wasix::WasiEnvBuilder; +@@ -8,10 +8,17 @@ use anyhow::Context; + use js_sys::Array; + use virtual_fs::{random_file::RandomFile, TmpFileSystem}; + use wasm_bindgen::{prelude::wasm_bindgen, JsCast, JsValue, UnwrapThrowExt}; +-use wasmer_wasix::WasiEnvBuilder; ++use wasmer_wasix::{ ++ capabilities::{ ++ WasiClockTimeGetMode, WasiFdClosePolicy, WasiGuestExecutionMode, WasiHostPolicy, ++ }, ++ WasiEnvBuilder, ++}; use crate::{runtime::Runtime, utils::Error, Directory, DirectoryInit, JsRuntime, StringOrBytes}; @@ -33,7 +54,7 @@ index d2f8d00..8c0ff2f 100644 #[wasm_bindgen] extern "C" { /// A proxy for `Option<&Runtime>`. -@@ -189,12 +191,14 @@ impl RunOptions { +@@ -189,12 +196,19 @@ impl RunOptions { pub(crate) fn configure_direct_builder( &self, builder: &mut WasiEnvBuilder, @@ -42,16 +63,22 @@ index d2f8d00..8c0ff2f 100644 stderr: Box, ) -> Result<(), Error> { - self.configure_common_builder(builder)?; -+ self.configure_common_builder(builder, None)?; - +- - builder.set_stdin(Box::::default()); - builder.set_stdout(Box::::default()); ++ self.configure_common_builder(builder, None)?; ++ builder.set_host_policy(WasiHostPolicy::new( ++ WasiGuestExecutionMode::SingleProgram, ++ WasiClockTimeGetMode::DirectJs, ++ WasiFdClosePolicy::WritesCompleteSynchronously, ++ )); ++ + builder.set_stdin(stdin); + builder.set_stdout(stdout); builder.set_stderr(stderr); Ok(()) -@@ -213,33 +217,60 @@ impl RunOptions { +@@ -213,33 +227,64 @@ impl RunOptions { ), Error, > { @@ -68,7 +95,11 @@ index d2f8d00..8c0ff2f 100644 + stderr: Box, + ) -> Result<(), Error> { + self.configure_common_builder(builder, Some(protocol))?; -+ builder.add_env("OLIPHAUNT_WASIX_SINGLE_BACKEND", "1"); ++ builder.set_host_policy(WasiHostPolicy::new( ++ WasiGuestExecutionMode::SingleProgram, ++ WasiClockTimeGetMode::Canonical, ++ WasiFdClosePolicy::WritesCompleteSynchronously, ++ )); + builder.add_env("OLIPHAUNT_DIRECT_PGWIRE", OLIPHAUNT_PROTOCOL_DEVICE); + match self.read_stdin() { + Some(input) => builder.set_stdin(Box::new(virtual_fs::StaticFile::new(input))), @@ -122,7 +153,7 @@ index d2f8d00..8c0ff2f 100644 ) -> Result<(), Error> { for arg in self.parse_args()? { builder.add_arg(arg); -@@ -253,6 +284,11 @@ impl RunOptions { +@@ -253,6 +298,11 @@ impl RunOptions { builder.set_current_dir(cwd); } let fs = self.filesystem()?; @@ -135,7 +166,7 @@ index d2f8d00..8c0ff2f 100644 builder.add_preopen_dir("/")?; diff --git a/src/postgres_direct.rs b/src/postgres_direct.rs -index 17f74b6..b78334d 100644 +index 2e090d3..3c38b13 100644 --- a/src/postgres_direct.rs +++ b/src/postgres_direct.rs @@ -1,4 +1,5 @@ @@ -144,27 +175,25 @@ index 17f74b6..b78334d 100644 collections::VecDeque, io::{self, SeekFrom}, pin::Pin, -@@ -10 +11 @@ use anyhow::{ensure, Context}; +@@ -7,10 +8,10 @@ use std::{ + }; + + use anyhow::{ensure, Context}; -use js_sys::{Uint8Array, WebAssembly}; +use js_sys::{Function, Uint8Array, WebAssembly}; -@@ -13 +14 @@ use virtual_fs::VirtualFile; + use tokio::io::{AsyncRead, AsyncSeek, AsyncWrite, ReadBuf}; + use virtual_fs::VirtualFile; -use wasm_bindgen::{prelude::wasm_bindgen, JsValue}; +use wasm_bindgen::{prelude::wasm_bindgen, JsCast, JsValue}; -@@ -21,6 +22,292 @@ use crate::{runtime::Runtime, tasks::CallerRealmTaskManager, utils::Error, RunOp + use wasmer::{Instance as WasmerInstance, Store, TypedFunction, Value, WasmTypeList}; + use wasmer_wasix::{ + oliphaunt_direct_memory, Runtime as _, WasiEnvBuilder, WasiError, WasiFunctionEnv, +@@ -21,6 +22,275 @@ use crate::{runtime::Runtime, tasks::CallerRealmTaskManager, utils::Error, RunOp const DEFAULT_PROGRAM_NAME: &str = "/bin/postgres"; const OLIPHAUNT_EXIT_ALIVE: i32 = 99; const STDERR_LIMIT_BYTES: usize = 16 * 1024; -+const PROTOCOL_CHUNK_BYTES: usize = 64 * 1024; -+const PROTOCOL_BUFFERED: i32 = 0; -+const PROTOCOL_HYBRID: i32 = 2; -+const PROTOCOL_STREAM_COMPLETE: u32 = 0; -+const PROTOCOL_STREAM_CALLBACK_ABORTED: u32 = 1; -+ -+#[derive(Debug)] -+enum ProtocolStreamOutcome { -+ Complete, -+ CallbackAborted(String), -+} ++use crate::protocol_contract::{PROTOCOL_CALLBACK_CHUNK_BYTES as PROTOCOL_CHUNK_BYTES, ++ PROTOCOL_BUFFERED, PROTOCOL_HYBRID}; + +thread_local! { + static PROTOCOL_CALLBACK: RefCell> = const { RefCell::new(None) }; @@ -327,24 +356,20 @@ index 17f74b6..b78334d 100644 + Ok(()) + } + -+ fn finish(&self) -> Option { ++ fn finish(&self) -> anyhow::Result<()> { + PROTOCOL_CALLBACK.with(|active| *active.borrow_mut() = None); -+ self ++ match self + .failure + .lock() + .expect("protocol stdout lock poisoned") + .take() ++ { ++ Some(failure) => anyhow::bail!("protocol stream callback failed: {failure}"), ++ None => Ok(()), ++ } + } + + fn deliver(&self, input: &[u8]) -> io::Result<()> { -+ if self -+ .failure -+ .lock() -+ .expect("protocol stdout lock poisoned") -+ .is_some() -+ { -+ return Ok(()); -+ } + let callback = PROTOCOL_CALLBACK + .with(|active| active.borrow().clone()) + .ok_or_else(|| { @@ -356,11 +381,7 @@ index 17f74b6..b78334d 100644 + let failure = format!("{error:?}"); + *self.failure.lock().expect("protocol stdout lock poisoned") = + Some(failure.clone()); -+ // The callback result is reported only after the guest has -+ // continued to ReadyForQuery. Stop delivery, but accept and -+ // discard subsequent writes so PostgreSQL can recover without -+ // invoking the failed consumer again. -+ return Ok(()); ++ return Err(io::Error::other(failure)); + } + } + Ok(()) @@ -443,7 +464,7 @@ index 17f74b6..b78334d 100644 #[derive(Clone, Debug, Default)] struct BoundedStderr { -@@ -186,9 +473,18 @@ pub async fn instantiate_oliphaunt_direct( +@@ -203,9 +473,18 @@ pub async fn instantiate_oliphaunt_direct( options: RunOptions, ) -> Result { let stderr = BoundedStderr::default(); @@ -465,7 +486,7 @@ index 17f74b6..b78334d 100644 } #[wasm_bindgen] -@@ -198,6 +494,8 @@ pub struct OliphauntDirectInstance { +@@ -215,6 +494,8 @@ pub struct OliphauntDirectInstance { store: Store, _instance: WasmerInstance, _env: WasiFunctionEnv, @@ -474,25 +495,29 @@ index 17f74b6..b78334d 100644 stderr: BoundedStderr, guest_memory: JsValue, input_reset: TypedFunction<(), i32>, -@@ -208,6 +506,7 @@ pub struct OliphauntDirectInstance { +@@ -224,6 +505,7 @@ pub struct OliphauntDirectInstance { + output_reset: TypedFunction<(), i32>, output_len: TypedFunction<(), i32>, output_data: TypedFunction<(), i32>, - output_contains_error: TypedFunction<(), i32>, + set_protocol_transport: TypedFunction, - set_force_host_error_recovery: Option>, set_active: TypedFunction, wasi_start: TypedFunction<(), ()>, -@@ -240,12 +539,48 @@ impl OliphauntDirectInstance { + start_oliphaunt: TypedFunction<(), ()>, +@@ -254,12 +536,48 @@ impl OliphauntDirectInstance { /// Execute raw PostgreSQL frontend-protocol bytes synchronously. #[wasm_bindgen(js_name = execProtocolRaw)] pub fn exec_protocol_raw(&mut self, input: Uint8Array) -> Result { +- match self.exec_protocol_raw_inner(&input) { ++ if let Err(error) = self.ensure_open() { ++ return Err(Error::from(self.stderr.attach(error))); ++ } + if let Err(error) = self + .set_protocol_transport + .call(&mut self.store, PROTOCOL_BUFFERED) + { + return Err(Error::from(self.stderr.attach(error.into()))); + } - match self.exec_protocol_raw_inner(&input) { ++ match self.exec_protocol_raw_inner(&input, false) { Ok(output) => Ok(output), Err(error) => Err(Error::from(self.stderr.attach(error))), } @@ -504,12 +529,9 @@ index 17f74b6..b78334d 100644 + &mut self, + input: Uint8Array, + on_chunk: Function, -+ ) -> Result { ++ ) -> Result<(), Error> { + match self.exec_protocol_stream_inner(&input, on_chunk) { -+ Ok(ProtocolStreamOutcome::Complete) => Ok(PROTOCOL_STREAM_COMPLETE), -+ Ok(ProtocolStreamOutcome::CallbackAborted(_)) => { -+ Ok(PROTOCOL_STREAM_CALLBACK_ABORTED) -+ } ++ Ok(()) => Ok(()), + Err(error) => Err(Error::from(self.stderr.attach(error))), + } + } @@ -531,7 +553,7 @@ index 17f74b6..b78334d 100644 /// Shut down the embedded lifecycle. This does not consume the JS object. pub fn close(&mut self) -> Result<(), Error> { match self.close_inner() { -@@ -256,6 +591,65 @@ impl OliphauntDirectInstance { +@@ -270,6 +588,64 @@ impl OliphauntDirectInstance { } impl OliphauntDirectInstance { @@ -541,15 +563,9 @@ index 17f74b6..b78334d 100644 + on_read: Function, + on_chunk: Function, + ) -> anyhow::Result<()> { ++ self.ensure_open()?; + self.protocol_stdin.begin(on_read)?; -+ let result = self.exec_protocol_stream_inner(payload, on_chunk).and_then(|outcome| { -+ match outcome { -+ ProtocolStreamOutcome::Complete => Ok(()), -+ ProtocolStreamOutcome::CallbackAborted(failure) => { -+ anyhow::bail!("protocol stream callback failed: {failure}") -+ } -+ } -+ }); ++ let result = self.exec_protocol_stream_inner(payload, on_chunk); + self.protocol_stdin.finish(); + result + } @@ -558,7 +574,8 @@ index 17f74b6..b78334d 100644 + &mut self, + payload: &Uint8Array, + on_chunk: Function, -+ ) -> anyhow::Result { ++ ) -> anyhow::Result<()> { ++ self.ensure_open()?; + self.protocol_stdout.begin(on_chunk.clone())?; + let previous = match self + .set_protocol_transport @@ -571,33 +588,82 @@ index 17f74b6..b78334d 100644 + return Err(error); + } + }; -+ let execution = self.exec_protocol_raw_inner(payload); -+ let restore = self -+ .set_protocol_transport -+ .call(&mut self.store, previous) -+ .context("restore protocol transport"); -+ let callback_failure = self.protocol_stdout.finish(); ++ let execution = self.exec_protocol_raw_inner(payload, true); ++ let restore = if self.closed { ++ // A terminal main-loop outcome poisons guest state. Never invoke ++ // another guest export merely to restore the transport selector. ++ Ok(()) ++ } else { ++ self.set_protocol_transport ++ .call(&mut self.store, previous) ++ .context("restore protocol transport") ++ .map(|_| ()) ++ }; ++ let callback = self.protocol_stdout.finish(); + + let output = execution?; + restore?; -+ if let Some(failure) = callback_failure { -+ return Ok(ProtocolStreamOutcome::CallbackAborted(failure)); -+ } ++ callback?; + let length = output.length() as usize; + for start in (0..length).step_by(PROTOCOL_CHUNK_BYTES) { + let end = start.saturating_add(PROTOCOL_CHUNK_BYTES).min(length); + let chunk = output.slice(start as u32, end as u32); -+ if let Err(error) = on_chunk.call1(&JsValue::UNDEFINED, &chunk) { -+ return Ok(ProtocolStreamOutcome::CallbackAborted(format!("{error:?}"))); -+ } ++ on_chunk ++ .call1(&JsValue::UNDEFINED, &chunk) ++ .map_err(|error| anyhow::anyhow!("protocol stream callback failed: {error:?}"))?; + } -+ Ok(ProtocolStreamOutcome::Complete) ++ Ok(()) + } + fn startup_inner(&mut self, packet: &Uint8Array) -> anyhow::Result { self.ensure_open()?; ensure!( -@@ -357,6 +751,8 @@ impl OliphauntDirectInstance { +@@ -302,7 +678,11 @@ impl OliphauntDirectInstance { + self.take_output_js() + } + +- fn exec_protocol_raw_inner(&mut self, payload: &Uint8Array) -> anyhow::Result { ++ fn exec_protocol_raw_inner( ++ &mut self, ++ payload: &Uint8Array, ++ streaming: bool, ++ ) -> anyhow::Result { + self.ensure_open()?; + ensure!( + self.protocol_started, +@@ -317,6 +697,7 @@ impl OliphauntDirectInstance { + let payload_len = payload.length() as usize; + let max_attempts = (payload_len / 5).saturating_add(2).max(1); + let mut attempts = 0usize; ++ let mut input_ended = false; + while self.protocol_input_remaining()? > 0 { + attempts += 1; + ensure!( +@@ -325,6 +706,10 @@ impl OliphauntDirectInstance { + ); + match self.call_main_loop()? { + MainLoopOutcome::Processed | MainLoopOutcome::Recovered => {} ++ MainLoopOutcome::InputEnded if streaming => { ++ input_ended = true; ++ break; ++ } + MainLoopOutcome::InputEnded => { + return Err(self.terminal_main_loop_outcome( + "PostgresMainLoopOnce reported input end while dispatching buffered protocol input", +@@ -332,8 +717,10 @@ impl OliphauntDirectInstance { + } + } + } +- self.send_ready_after_main_loop()?; +- self.flush_after_main_loop()?; ++ if !input_ended { ++ self.send_ready_after_main_loop()?; ++ self.flush_after_main_loop()?; ++ } + self.take_output_js() + } + +@@ -358,6 +745,8 @@ impl OliphauntDirectInstance { module: WebAssembly::Module, module_bytes: Uint8Array, options: RunOptions, @@ -606,7 +672,7 @@ index 17f74b6..b78334d 100644 stderr: BoundedStderr, ) -> Result { // Direct execution intentionally has no configurable networking or -@@ -368,7 +764,12 @@ impl OliphauntDirectInstance { +@@ -369,7 +758,12 @@ impl OliphauntDirectInstance { .as_string() .unwrap_or_else(|| DEFAULT_PROGRAM_NAME.to_owned()); let mut builder = WasiEnvBuilder::new(program_name).runtime(runtime.clone()); @@ -620,19 +686,19 @@ index 17f74b6..b78334d 100644 let module = wasmer::Module::from((module, module_bytes.to_vec())); let mut store = Store::new(runtime.engine()); -@@ -399,6 +800,11 @@ impl OliphauntDirectInstance { - &instance, - "oliphaunt_wasix_output_contains_error", - )?; +@@ -395,6 +789,11 @@ impl OliphauntDirectInstance { + let output_reset = typed_export(&mut store, &instance, "oliphaunt_wasix_output_reset")?; + let output_len = typed_export(&mut store, &instance, "oliphaunt_wasix_output_len")?; + let output_data = typed_export(&mut store, &instance, "oliphaunt_wasix_output_data")?; + let set_protocol_transport = typed_export( + &mut store, + &instance, + "oliphaunt_wasix_set_protocol_transport", + )?; - let set_force_host_error_recovery = optional_typed_export( - &mut store, - &instance, -@@ -428,6 +834,8 @@ impl OliphauntDirectInstance { + let set_active = typed_export(&mut store, &instance, "oliphaunt_wasix_set_active")?; + let wasi_start = typed_export(&mut store, &instance, "_start")?; + let start_oliphaunt = typed_export(&mut store, &instance, "oliphaunt_wasix_start")?; +@@ -418,6 +817,8 @@ impl OliphauntDirectInstance { store, _instance: instance, _env: env, @@ -641,22 +707,24 @@ index 17f74b6..b78334d 100644 stderr, guest_memory, input_reset, -@@ -438,6 +846,7 @@ impl OliphauntDirectInstance { +@@ -427,6 +828,7 @@ impl OliphauntDirectInstance { + output_reset, output_len, output_data, - output_contains_error, + set_protocol_transport, - set_force_host_error_recovery, set_active, wasi_start, + start_oliphaunt, diff --git a/src/tool_direct.rs b/src/tool_direct.rs new file mode 100644 -index 0000000..74719c7 +index 0000000..fc92239 --- /dev/null +++ b/src/tool_direct.rs -@@ -0,0 +1,432 @@ +@@ -0,0 +1,588 @@ +use std::{ + cell::RefCell, ++ error::Error as StdError, ++ fmt, + io::{self, SeekFrom}, + pin::Pin, + sync::{Arc, Mutex}, @@ -680,7 +748,7 @@ index 0000000..74719c7 +}; + +const DEFAULT_PROGRAM_NAME: &str = "wasm"; -+const PROTOCOL_CHUNK_BYTES: usize = 64 * 1024; ++use crate::protocol_contract::PROTOCOL_CALLBACK_CHUNK_BYTES as PROTOCOL_CHUNK_BYTES; + +#[wasm_bindgen] +extern "C" { @@ -854,10 +922,10 @@ index 0000000..74719c7 + Err(error) => return Poll::Ready(Err(error)), + }; + let length = input.len().min(PROTOCOL_CHUNK_BYTES); -+ // SAFETY: this hidden host ABI accepts only Oliphaunt's internal -+ // callbacks. They synchronously copy, never mutate, and never retain -+ // the view beyond this call, before Rust can reuse the input slice. -+ let chunk = unsafe { Uint8Array::view(&input[..length]) }; ++ // Copy the Rust-owned input into JavaScript-owned storage before the ++ // callback. It may retain or mutate this independent chunk without ++ // aliasing memory that Rust can reuse after this call returns. ++ let chunk = Uint8Array::from(&input[..length]); + match callbacks.write.call1(&JsValue::UNDEFINED, &chunk) { + Ok(_) => Poll::Ready(Ok(length)), + Err(error) => Poll::Ready(Err(io::Error::other(format!( @@ -923,16 +991,176 @@ index 0000000..74719c7 + } +} + -+/// Fresh append-only process output. Clones share the allocation so the host -+/// can take ownership of the bytes after `_start` without a second full copy. -+#[derive(Clone, Debug, Default)] ++#[derive(Clone, Copy, Debug)] ++enum CaptureStream { ++ Stdout, ++ Stderr, ++} ++ ++impl fmt::Display for CaptureStream { ++ fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { ++ formatter.write_str(match self { ++ Self::Stdout => "stdout", ++ Self::Stderr => "stderr", ++ }) ++ } ++} ++ ++#[derive(Clone, Copy, Debug)] ++enum CaptureFailure { ++ AllocationFailed(CaptureStream), ++ LockPoisoned, ++} ++ ++impl CaptureFailure { ++ fn io_error(self) -> io::Error { ++ let kind = match self { ++ Self::AllocationFailed(_) => io::ErrorKind::OutOfMemory, ++ Self::LockPoisoned => io::ErrorKind::Other, ++ }; ++ io::Error::new(kind, self) ++ } ++} ++ ++impl fmt::Display for CaptureFailure { ++ fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { ++ match self { ++ Self::AllocationFailed(stream) => { ++ write!(formatter, "Oliphaunt WASIX tool {stream} capture allocation failed") ++ } ++ Self::LockPoisoned => { ++ formatter.write_str("Oliphaunt WASIX tool output capture lock was poisoned") ++ } ++ } ++ } ++} ++ ++impl StdError for CaptureFailure {} ++ ++#[derive(Debug, Default)] ++struct CaptureState { ++ stdout: Vec, ++ stderr: Vec, ++ failure: Option, ++} ++ ++impl CaptureState { ++ fn output(&self, stream: CaptureStream) -> &Vec { ++ match stream { ++ CaptureStream::Stdout => &self.stdout, ++ CaptureStream::Stderr => &self.stderr, ++ } ++ } ++ ++ fn output_mut(&mut self, stream: CaptureStream) -> &mut Vec { ++ match stream { ++ CaptureStream::Stdout => &mut self.stdout, ++ CaptureStream::Stderr => &mut self.stderr, ++ } ++ } ++ ++ fn discard_output(&mut self) { ++ self.stdout = Vec::new(); ++ self.stderr = Vec::new(); ++ } ++ ++ fn fail(&mut self, failure: CaptureFailure) -> io::Error { ++ let failure = *self.failure.get_or_insert(failure); ++ self.discard_output(); ++ failure.io_error() ++ } ++ ++ fn poison_lock(&mut self) -> CaptureFailure { ++ self.failure = Some(CaptureFailure::LockPoisoned); ++ self.discard_output(); ++ CaptureFailure::LockPoisoned ++ } ++ ++ fn append(&mut self, stream: CaptureStream, input: &[u8]) -> io::Result { ++ if let Some(failure) = self.failure { ++ return Err(failure.io_error()); ++ } ++ if self.output_mut(stream).try_reserve(input.len()).is_err() { ++ return Err(self.fail(CaptureFailure::AllocationFailed(stream))); ++ } ++ self.output_mut(stream).extend_from_slice(input); ++ Ok(input.len()) ++ } ++} ++ ++/// Complete in-memory stdout/stderr capture for a fresh tool process. ++/// The handle refuses to publish either stream after any capture failure. ++#[derive(Clone, Debug)] +struct CaptureFile { -+ bytes: Arc>>, ++ state: Arc>, ++ stream: CaptureStream, ++} ++ ++#[derive(Clone, Debug)] ++struct CaptureHandle { ++ state: Arc>, +} + +impl CaptureFile { -+ fn take(&self) -> Vec { -+ std::mem::take(&mut *self.bytes.lock().unwrap_or_else(|error| error.into_inner())) ++ fn pair() -> (Self, Self, CaptureHandle) { ++ let state = Arc::new(Mutex::new(CaptureState::default())); ++ ( ++ Self { ++ state: state.clone(), ++ stream: CaptureStream::Stdout, ++ }, ++ Self { ++ state: state.clone(), ++ stream: CaptureStream::Stderr, ++ }, ++ CaptureHandle { state }, ++ ) ++ } ++ ++ fn with_state( ++ &self, ++ operation: impl FnOnce(&mut CaptureState) -> io::Result, ++ ) -> io::Result { ++ let mut state = match self.state.lock() { ++ Ok(state) => state, ++ Err(error) => { ++ let mut state = error.into_inner(); ++ return Err(state.poison_lock().io_error()); ++ } ++ }; ++ operation(&mut state) ++ } ++} ++ ++impl CaptureHandle { ++ fn failure(&self) -> Option { ++ match self.state.lock() { ++ Ok(state) => state.failure, ++ Err(error) => { ++ let mut state = error.into_inner(); ++ state.poison_lock(); ++ state.failure ++ } ++ } ++ } ++ ++ fn finish(&self) -> anyhow::Result<(Vec, Vec)> { ++ let mut state = match self.state.lock() { ++ Ok(state) => state, ++ Err(error) => { ++ let mut state = error.into_inner(); ++ let failure = state.poison_lock(); ++ return Err(anyhow::Error::new(failure)); ++ } ++ }; ++ if let Some(failure) = state.failure { ++ state.discard_output(); ++ return Err(anyhow::Error::new(failure)); ++ } ++ Ok(( ++ std::mem::take(&mut state.stdout), ++ std::mem::take(&mut state.stderr), ++ )) + } +} + @@ -952,11 +1180,8 @@ index 0000000..74719c7 + _cx: &mut TaskContext<'_>, + input: &[u8], + ) -> Poll> { -+ self.bytes -+ .lock() -+ .unwrap_or_else(|error| error.into_inner()) -+ .extend_from_slice(input); -+ Poll::Ready(Ok(input.len())) ++ let stream = self.stream; ++ Poll::Ready(self.with_state(|state| state.append(stream, input))) + } + + fn poll_flush(self: Pin<&mut Self>, _cx: &mut TaskContext<'_>) -> Poll> { @@ -974,11 +1199,8 @@ index 0000000..74719c7 + } + + fn poll_complete(self: Pin<&mut Self>, _cx: &mut TaskContext<'_>) -> Poll> { -+ Poll::Ready(Ok(self -+ .bytes -+ .lock() -+ .unwrap_or_else(|error| error.into_inner()) -+ .len() as u64)) ++ let stream = self.stream; ++ Poll::Ready(self.with_state(|state| Ok(state.output(stream).len() as u64))) + } +} + @@ -996,19 +1218,15 @@ index 0000000..74719c7 + } + + fn size(&self) -> u64 { -+ self.bytes -+ .lock() -+ .unwrap_or_else(|error| error.into_inner()) -+ .len() as u64 ++ let stream = self.stream; ++ match self.with_state(|state| Ok(state.output(stream).len() as u64)) { ++ Ok(size) => size, ++ Err(_) => 0, ++ } + } + -+ fn set_len(&mut self, new_size: u64) -> virtual_fs::Result<()> { -+ let new_size = usize::try_from(new_size).map_err(|_| virtual_fs::FsError::UnknownError)?; -+ self.bytes -+ .lock() -+ .unwrap_or_else(|error| error.into_inner()) -+ .resize(new_size, 0); -+ Ok(()) ++ fn set_len(&mut self, _new_size: u64) -> virtual_fs::Result<()> { ++ Err(virtual_fs::FsError::PermissionDenied) + } + + fn unlink(&mut self) -> virtual_fs::Result<()> { @@ -1029,6 +1247,7 @@ index 0000000..74719c7 + +/// Run an Oliphaunt frontend tool synchronously in the caller's JavaScript +/// realm, while routing its private PGWire device through blocking callbacks. ++/// The protocol write callback receives an owned JavaScript copy that it may retain. +#[wasm_bindgen(js_name = runOliphauntToolDirect)] +pub async fn run_oliphaunt_tool_direct( + prepared: &OliphauntPreparedTool, @@ -1041,8 +1260,7 @@ index 0000000..74719c7 + .program() + .as_string() + .unwrap_or_else(|| DEFAULT_PROGRAM_NAME.to_owned()); -+ let stdout = CaptureFile::default(); -+ let stderr = CaptureFile::default(); ++ let (stdout, stderr, capture) = CaptureFile::pair(); + let mut builder = WasiEnvBuilder::new(program_name).runtime(prepared.runtime.clone()); + options.configure_tool_direct_builder( + &mut builder, @@ -1068,15 +1286,19 @@ index 0000000..74719c7 + Err(error) => match runtime_exit_code(&error) { + Some(code) => code, + None => { ++ if let Some(failure) = capture.failure() { ++ return Err(Error::from(anyhow::Error::new(failure))); ++ } + return Err(Error::from( + anyhow::Error::from(error).context("run Oliphaunt WASIX tool"), -+ )) ++ )); + } + }, + } + }; + -+ Ok(tool_output(code, stdout.take(), stderr.take())) ++ let (stdout, stderr) = capture.finish().map_err(Error::from)?; ++ Ok(tool_output(code, stdout, stderr)) +} + +fn runtime_exit_code(error: &wasmer::RuntimeError) -> Option { diff --git a/src/wasix/browser-host/patches/0023-wasmer-js-prepare-trusted-embedded-session.patch b/src/wasix/browser-host/patches/0023-wasmer-js-prepare-trusted-embedded-session.patch new file mode 100644 index 000000000..6031d4168 --- /dev/null +++ b/src/wasix/browser-host/patches/0023-wasmer-js-prepare-trusted-embedded-session.patch @@ -0,0 +1,57 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: prepare trusted embedded PostgreSQL sessions + +Require and invoke the guest's explicit trusted-embedded-session entrypoint +before client initialization. The embedded host has already selected the +session identity; replaying server authentication/bootstrap paths here can +undo that boundary. This requires the matching PostgreSQL guest patch and is +an Oliphaunt-specific trusted-host ABI, not a generic authentication bypass. + +diff --git a/src/postgres_direct.rs b/src/postgres_direct.rs +index 3c38b13..7b14a8f 100644 +--- a/src/postgres_direct.rs ++++ b/src/postgres_direct.rs +@@ -508,6 +508,7 @@ pub struct OliphauntDirectInstance { + output_data: TypedFunction<(), i32>, + set_protocol_transport: TypedFunction, + set_active: TypedFunction, ++ prepare_trusted_embedded_session: TypedFunction<(), i32>, + wasi_start: TypedFunction<(), ()>, + start_oliphaunt: TypedFunction<(), ()>, + run_atexit_funcs: Option>, +@@ -796,6 +797,11 @@ impl OliphauntDirectInstance { + "oliphaunt_wasix_set_protocol_transport", + )?; + let set_active = typed_export(&mut store, &instance, "oliphaunt_wasix_set_active")?; ++ let prepare_trusted_embedded_session = typed_export( ++ &mut store, ++ &instance, ++ "oliphaunt_wasix_prepare_trusted_embedded_session", ++ )?; + let wasi_start = typed_export(&mut store, &instance, "_start")?; + let start_oliphaunt = typed_export(&mut store, &instance, "oliphaunt_wasix_start")?; + let run_atexit_funcs = +@@ -831,6 +837,7 @@ impl OliphauntDirectInstance { + output_data, + set_protocol_transport, + set_active, ++ prepare_trusted_embedded_session, + wasi_start, + start_oliphaunt, + run_atexit_funcs, +@@ -922,6 +929,14 @@ impl OliphauntDirectInstance { + self.set_active + .call(&mut self.store, 1) + .context("oliphaunt_wasix_set_active(1)")?; ++ let prepare_status = self ++ .prepare_trusted_embedded_session ++ .call(&mut self.store) ++ .context("oliphaunt_wasix_prepare_trusted_embedded_session")?; ++ ensure!( ++ prepare_status == 0, ++ "oliphaunt_wasix_prepare_trusted_embedded_session rejected after PostgreSQL startup began" ++ ); + match self.wasi_start.call(&mut self.store) { + Ok(()) => {} + Err(error) if runtime_exit_code(&error) == Some(OLIPHAUNT_EXIT_ALIVE) => {} diff --git a/src/wasix/browser-host/patches/0024-virtual-fs-propagate-random-errors.patch b/src/wasix/browser-host/patches/0024-virtual-fs-propagate-random-errors.patch new file mode 100644 index 000000000..f05b8bee7 --- /dev/null +++ b/src/wasix/browser-host/patches/0024-virtual-fs-propagate-random-errors.patch @@ -0,0 +1,32 @@ +From 0000000000000000000000000000000000000024 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Wed, 26 Aug 2026 00:00:00 +0000 +Subject: [PATCH] virtual-fs: propagate random device errors + +RandomFile currently ignores getrandom failures and then reports a successful +read of the zero-initialized staging buffer. Predictable bytes must never be +reported as entropy. + +Return the provider error as an I/O error without advancing the destination. +The caller can then preserve its own fail-closed randomness contract. +--- + src/random_file.rs | 4 +++- + 1 file changed, 3 insertions(+), 1 deletion(-) + +diff --git a/src/random_file.rs b/src/random_file.rs +index 2667d16..ad880c3 100644 +--- a/src/random_file.rs ++++ b/src/random_file.rs +@@ -54,7 +54,9 @@ impl AsyncRead for RandomFile { + buf: &mut tokio::io::ReadBuf<'_>, + ) -> Poll> { + let mut data = vec![0u8; buf.remaining()]; +- getrandom::getrandom(&mut data).ok(); ++ if let Err(error) = getrandom::getrandom(&mut data) { ++ return Poll::Ready(Err(error.into())); ++ } + buf.put_slice(&data[..]); + Poll::Ready(Ok(())) + } +-- +2.51.0 diff --git a/src/wasix/browser-host/patches/0025-wasmer-js-fail-closed-direct-guest-phases.patch b/src/wasix/browser-host/patches/0025-wasmer-js-fail-closed-direct-guest-phases.patch new file mode 100644 index 000000000..e9ee5a51a --- /dev/null +++ b/src/wasix/browser-host/patches/0025-wasmer-js-fail-closed-direct-guest-phases.patch @@ -0,0 +1,1017 @@ +From 0000000000000000000000000000000000000025 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Thu, 27 Aug 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: fail closed across direct guest phases + +Treat every error after entering a direct protocol guest phase as terminal. +Centralize the lifecycle transition so transport selection, input bridging, +the protocol pump, post-step exports, output extraction, and transport restore +all poison before returning. Cleanup errors are terminal too, while repeated +close remains idempotent and never re-enters a poisoned guest. + +Validate bridge return values and preserve ProcessStartupPacket rejection as a +cleanup-capable terminal-for-work state. Handle _start rejection atomically: +cache and validate the pending v1 outcome descriptor before _start, classify +only Exit98 as rejection, then copy a complete ErrorResponse-bearing protocol +stream directly from guest memory without invoking another guest export. +Malformed outcomes and every other _start failure poison generically. + +Keep buffered-input streaming (mode 3) distinct from full duplex (mode 2), +require an empty guest output buffer after streamed execution, and validate +every flush status before output extraction. Keep buffered-output copies within +the signed-i32 bridge range and make callback failure sticky so a guest retry +cannot call JavaScript again after the first callback exception. Treat a +poisoned callback state lock as an equally sticky terminal failure without +panicking. Preserve that causal callback failure ahead of the derived guest +flush failure so the public error retains the JavaScript callback diagnostic. + +Keep the causal error at the exported boundary; do not label unrelated failures +as an output-size problem. The signed bridge length is the only output ceiling. +--- + +--- a/src/postgres_direct.rs ++++ b/src/postgres_direct.rs +@@ -20,10 +20,18 @@ + use crate::{runtime::Runtime, tasks::CallerRealmTaskManager, utils::Error, RunOptions}; + + const DEFAULT_PROGRAM_NAME: &str = "/bin/postgres"; ++const OLIPHAUNT_EXIT_STARTUP_REJECTED: i32 = 98; + const OLIPHAUNT_EXIT_ALIVE: i32 = 99; + const STDERR_LIMIT_BYTES: usize = 16 * 1024; + use crate::protocol_contract::{PROTOCOL_CALLBACK_CHUNK_BYTES as PROTOCOL_CHUNK_BYTES, + PROTOCOL_BUFFERED, PROTOCOL_HYBRID}; ++use crate::protocol_contract::PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT; ++const PROCESS_STARTUP_OK: i32 = 0; ++const PROCESS_STARTUP_ERROR: i32 = -1; ++const STARTUP_OUTCOME_VERSION: u32 = 1; ++const STARTUP_OUTCOME_DESCRIPTOR_BYTES: u32 = 32; ++const STARTUP_OUTCOME_PENDING: u32 = 0; ++const STARTUP_OUTCOME_REJECTED: u32 = 1; + + thread_local! { + static PROTOCOL_CALLBACK: RefCell> = const { RefCell::new(None) }; +@@ -167,40 +175,105 @@ + } + } + ++#[derive(Clone, Debug)] ++enum ProtocolStdoutFailure { ++ Callback(String), ++ LockPoisoned, ++} ++ ++impl ProtocolStdoutFailure { ++ fn message(&self) -> String { ++ match self { ++ Self::Callback(failure) => format!("protocol stream callback failed: {failure}"), ++ Self::LockPoisoned => "protocol stream state lock was poisoned".to_string(), ++ } ++ } ++ ++ fn into_anyhow(self) -> anyhow::Error { ++ anyhow::anyhow!(self.message()) ++ } ++ ++ fn into_io(self) -> io::Error { ++ io::Error::other(self.message()) ++ } ++} ++ ++#[derive(Debug, Default)] ++struct ProtocolStdoutState { ++ failure: Option, ++} ++ ++impl ProtocolStdoutState { ++ fn poison_lock(&mut self) -> ProtocolStdoutFailure { ++ self.failure = Some(ProtocolStdoutFailure::LockPoisoned); ++ ProtocolStdoutFailure::LockPoisoned ++ } ++ ++ fn record_callback_failure(&mut self, failure: String) -> ProtocolStdoutFailure { ++ match &self.failure { ++ Some(failure) => failure.clone(), ++ None => { ++ let failure = ProtocolStdoutFailure::Callback(failure); ++ self.failure = Some(failure.clone()); ++ failure ++ } ++ } ++ } ++} ++ + #[derive(Clone, Debug, Default)] + struct ProtocolStdout { +- failure: Arc>>, ++ state: Arc>, + } + + impl ProtocolStdout { + fn begin(&self, callback: Function) -> anyhow::Result<()> { + PROTOCOL_CALLBACK.with(|active| { +- let mut active = active.borrow_mut(); + ensure!( +- active.is_none(), ++ active.borrow().is_none(), + "protocol stream callback is already active" + ); +- *active = Some(callback); + Ok(()) + })?; +- *self.failure.lock().expect("protocol stdout lock poisoned") = None; ++ self.with_state(|state| state.failure = None) ++ .map_err(ProtocolStdoutFailure::into_anyhow)?; ++ PROTOCOL_CALLBACK.with(|active| *active.borrow_mut() = Some(callback)); + Ok(()) + } + + fn finish(&self) -> anyhow::Result<()> { + PROTOCOL_CALLBACK.with(|active| *active.borrow_mut() = None); +- match self +- .failure +- .lock() +- .expect("protocol stdout lock poisoned") +- .take() +- { +- Some(failure) => anyhow::bail!("protocol stream callback failed: {failure}"), ++ let failure = match self.with_state(|state| state.failure.clone()) { ++ Ok(failure) => failure, ++ Err(failure) => Some(failure), ++ }; ++ match failure { ++ Some(failure) => Err(failure.into_anyhow()), + None => Ok(()), + } + } + ++ fn with_state( ++ &self, ++ operation: impl FnOnce(&mut ProtocolStdoutState) -> T, ++ ) -> Result { ++ match self.state.lock() { ++ Ok(mut state) => Ok(operation(&mut state)), ++ Err(poisoned) => { ++ let mut state = poisoned.into_inner(); ++ Err(state.poison_lock()) ++ } ++ } ++ } ++ + fn deliver(&self, input: &[u8]) -> io::Result<()> { ++ let failure = match self.with_state(|state| state.failure.clone()) { ++ Ok(failure) => failure, ++ Err(failure) => Some(failure), ++ }; ++ if let Some(failure) = failure { ++ return Err(failure.into_io()); ++ } + let callback = PROTOCOL_CALLBACK + .with(|active| active.borrow().clone()) + .ok_or_else(|| { +@@ -209,10 +282,12 @@ + for input in input.chunks(PROTOCOL_CHUNK_BYTES) { + let chunk = Uint8Array::from(input); + if let Err(error) = callback.call1(&JsValue::UNDEFINED, &chunk) { +- let failure = format!("{error:?}"); +- *self.failure.lock().expect("protocol stdout lock poisoned") = +- Some(failure.clone()); +- return Err(io::Error::other(failure)); ++ let failure = match self ++ .with_state(|state| state.record_callback_failure(format!("{error:?}"))) ++ { ++ Ok(failure) | Err(failure) => failure, ++ }; ++ return Err(failure.into_io()); + } + } + Ok(()) +@@ -426,6 +501,34 @@ + InputEnded, + } + ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++enum DirectInstanceState { ++ Open, ++ StartupPhase, ++ GuestPhase, ++ StartupRejected, ++ Closing, ++ Poisoned, ++ Closed, ++} ++ ++enum BackendStartOutcome { ++ Alive, ++ Rejected(Vec), ++} ++ ++enum DirectStartupOutcome { ++ Processed { output: Uint8Array, status: i32 }, ++ Rejected(Vec), ++} ++ ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++struct StartupOutcomeDescriptor { ++ kind: u32, ++ data_pointer: u64, ++ data_length: u64, ++} ++ + impl MainLoopOutcome { + fn from_i32(value: i32) -> Option { + match value { +@@ -509,19 +612,20 @@ + set_protocol_transport: TypedFunction, + set_active: TypedFunction, + prepare_trusted_embedded_session: TypedFunction<(), i32>, ++ startup_outcome: TypedFunction<(), i32>, + wasi_start: TypedFunction<(), ()>, + start_oliphaunt: TypedFunction<(), ()>, + run_atexit_funcs: Option>, + get_port: TypedFunction<(), i32>, + process_startup: TypedFunction<(i32, i32, i32), i32>, + send_conn_data: TypedFunction<(), ()>, +- pq_flush: TypedFunction<(), ()>, ++ pq_flush: TypedFunction<(), i32>, + pq_buffer_remaining_data: TypedFunction<(), i32>, + main_loop: TypedFunction<(), i32>, + send_ready: TypedFunction<(), ()>, + backend_started: bool, + protocol_started: bool, +- closed: bool, ++ state: DirectInstanceState, + } + + #[wasm_bindgen] +@@ -530,26 +634,39 @@ + #[wasm_bindgen(js_name = startup)] + pub fn startup(&mut self, packet: Uint8Array) -> Result { + match self.startup_inner(&packet) { +- Ok(output) => Ok(output), +- Err(error) => Err(self.startup_error(error)), ++ Ok(DirectStartupOutcome::Processed { output, .. }) => Ok(output), ++ Ok(DirectStartupOutcome::Rejected(protocol)) => Err(self.startup_rejection_error( ++ anyhow::anyhow!("_start rejected Oliphaunt single-user backend"), ++ protocol, ++ )), ++ Err(error) if self.state == DirectInstanceState::StartupPhase => { ++ self.poison_guest_phase(); ++ Err(Error::from(self.stderr.attach(error))) ++ } ++ Err(error) => Err(Error::from(self.stderr.attach(error))), + } + } + + /// Execute raw PostgreSQL frontend-protocol bytes synchronously. + #[wasm_bindgen(js_name = execProtocolRaw)] + pub fn exec_protocol_raw(&mut self, input: Uint8Array) -> Result { +- if let Err(error) = self.ensure_open() { ++ if let Err(error) = self.ensure_protocol_started() { + return Err(Error::from(self.stderr.attach(error))); + } +- if let Err(error) = self +- .set_protocol_transport +- .call(&mut self.store, PROTOCOL_BUFFERED) +- { +- return Err(Error::from(self.stderr.attach(error.into()))); +- } +- match self.exec_protocol_raw_inner(&input, false) { ++ let result = self.run_guest_phase("buffered protocol exchange", |direct| { ++ let previous = direct ++ .set_protocol_transport ++ .call(&mut direct.store, PROTOCOL_BUFFERED) ++ .context("enable buffered protocol transport")?; ++ ensure!( ++ previous == PROTOCOL_BUFFERED, ++ "enable buffered protocol transport returned previous mode {previous}, expected {PROTOCOL_BUFFERED}" ++ ); ++ direct.exec_protocol_raw_inner(&input, false) ++ }); ++ match result { + Ok(output) => Ok(output), + Err(error) => Err(Error::from(self.stderr.attach(error))), + } + } + +@@ -560,7 +677,13 @@ + input: Uint8Array, + on_chunk: Function, +- ) -> Result<(), Error> { ++ ) -> Result { +- match self.exec_protocol_stream_inner(&input, on_chunk) { ++ if let Err(error) = self.ensure_protocol_started() { ++ return Err(Error::from(self.stderr.attach(error))); ++ } ++ let result = self.run_guest_phase("streaming protocol exchange", |direct| { ++ direct.exec_protocol_stream_inner(&input, on_chunk) ++ }); ++ match result { +- Ok(()) => Ok(()), ++ Ok(()) => Ok(0), + Err(error) => Err(Error::from(self.stderr.attach(error))), + } +@@ -574,7 +697,13 @@ + on_read: Function, + on_chunk: Function, + ) -> Result<(), Error> { +- match self.exec_protocol_duplex_inner(&input, on_read, on_chunk) { ++ if let Err(error) = self.ensure_protocol_started() { ++ return Err(Error::from(self.stderr.attach(error))); ++ } ++ let result = self.run_guest_phase("duplex protocol exchange", |direct| { ++ direct.exec_protocol_duplex_inner(&input, on_read, on_chunk) ++ }); ++ match result { + Ok(()) => Ok(()), + Err(error) => Err(Error::from(self.stderr.attach(error))), + } +@@ -596,11 +725,20 @@ + on_read: Function, + on_chunk: Function, + ) -> anyhow::Result<()> { +- self.ensure_open()?; + self.protocol_stdin.begin(on_read)?; +- let result = self.exec_protocol_stream_inner(payload, on_chunk); ++ let result = ++ self.exec_protocol_callback_inner(payload, on_chunk.clone(), PROTOCOL_HYBRID, true); + self.protocol_stdin.finish(); +- result ++ let output = result?; ++ let length = output.length() as usize; ++ for start in (0..length).step_by(PROTOCOL_CHUNK_BYTES) { ++ let end = start.saturating_add(PROTOCOL_CHUNK_BYTES).min(length); ++ let chunk = output.slice(start as u32, end as u32); ++ on_chunk ++ .call1(&JsValue::UNDEFINED, &chunk) ++ .map_err(|error| anyhow::anyhow!("protocol stream callback failed: {error:?}"))?; ++ } ++ Ok(()) + } + + fn exec_protocol_stream_inner( +@@ -608,53 +746,107 @@ + payload: &Uint8Array, + on_chunk: Function, + ) -> anyhow::Result<()> { +- self.ensure_open()?; ++ let output = self.exec_protocol_callback_inner( ++ payload, ++ on_chunk, ++ PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT, ++ false, ++ )?; ++ ensure!( ++ output.length() == 0, ++ "buffered-input/streamed-output transport left {} buffered protocol bytes", ++ output.length() ++ ); ++ Ok(()) ++ } ++ ++ fn exec_protocol_callback_inner( ++ &mut self, ++ payload: &Uint8Array, ++ on_chunk: Function, ++ transport: i32, ++ streaming_input: bool, ++ ) -> anyhow::Result { + self.protocol_stdout.begin(on_chunk.clone())?; + let previous = match self + .set_protocol_transport +- .call(&mut self.store, PROTOCOL_HYBRID) +- .context("enable hybrid protocol transport") ++ .call(&mut self.store, transport) ++ .with_context(|| format!("enable protocol transport mode {transport}")) + { +- Ok(previous) => previous, ++ Ok(previous) if previous == PROTOCOL_BUFFERED => previous, ++ Ok(previous) => { ++ let _ = self.protocol_stdout.finish(); ++ anyhow::bail!( ++ "enable protocol transport mode {transport} returned previous mode {previous}, expected {PROTOCOL_BUFFERED}" ++ ); ++ } + Err(error) => { + let _ = self.protocol_stdout.finish(); + return Err(error); + } + }; +- let execution = self.exec_protocol_raw_inner(payload, true); +- let restore = if self.closed { +- // A terminal main-loop outcome poisons guest state. Never invoke +- // another guest export merely to restore the transport selector. ++ let execution = self.exec_protocol_raw_inner(payload, streaming_input); ++ let restore = if execution.is_err() { ++ // Once any guest-facing step fails, guest state is unknown. The ++ // outer phase boundary will poison the instance; do not re-enter ++ // the guest merely to restore the transport selector. + Ok(()) + } else { + self.set_protocol_transport + .call(&mut self.store, previous) +- .context("restore protocol transport") +- .map(|_| ()) ++ .with_context(|| format!("restore protocol transport from mode {transport}")) ++ .and_then(|replaced| { ++ ensure!( ++ replaced == transport, ++ "restore protocol transport replaced mode {replaced}, expected {transport}" ++ ); ++ Ok(()) ++ }) + }; + let callback = self.protocol_stdout.finish(); + +- let output = execution?; +- restore?; + callback?; ++ let output = execution?; ++ restore?; +- let length = output.length() as usize; +- for start in (0..length).step_by(PROTOCOL_CHUNK_BYTES) { +- let end = start.saturating_add(PROTOCOL_CHUNK_BYTES).min(length); +- let chunk = output.slice(start as u32, end as u32); +- on_chunk +- .call1(&JsValue::UNDEFINED, &chunk) +- .map_err(|error| anyhow::anyhow!("protocol stream callback failed: {error:?}"))?; +- } +- Ok(()) ++ Ok(output) + } + +- fn startup_inner(&mut self, packet: &Uint8Array) -> anyhow::Result { +- self.ensure_open()?; ++ fn startup_inner(&mut self, packet: &Uint8Array) -> anyhow::Result { ++ match self.state { ++ DirectInstanceState::Open => {} ++ DirectInstanceState::StartupRejected => { ++ anyhow::bail!( ++ "Oliphaunt direct startup was already rejected; the direct instance cannot be reused" ++ ) ++ } ++ _ => self.ensure_open()?, ++ } + ensure!( + !self.protocol_started, + "Oliphaunt direct startup already completed" + ); +- self.start_backend()?; ++ self.state = DirectInstanceState::StartupPhase; ++ let outcome = self.startup_guest_inner(packet)?; ++ match outcome { ++ DirectStartupOutcome::Processed { status, .. } => { ++ self.state = if status == PROCESS_STARTUP_ERROR { ++ DirectInstanceState::StartupRejected ++ } else { ++ DirectInstanceState::Open ++ }; ++ } ++ DirectStartupOutcome::Rejected(_) => self.poison_guest_phase(), ++ } ++ Ok(outcome) ++ } ++ ++ fn startup_guest_inner(&mut self, packet: &Uint8Array) -> anyhow::Result { ++ match self.start_backend()? { ++ BackendStartOutcome::Alive => {} ++ BackendStartOutcome::Rejected(protocol) => { ++ return Ok(DirectStartupOutcome::Rejected(protocol)); ++ } ++ } + self.reset_io()?; + self.push_input_js(packet)?; + +@@ -667,17 +859,28 @@ + .process_startup + .call(&mut self.store, port, 1, 1) + .context("ProcessStartupPacket")?; +- self.pq_flush.call(&mut self.store).ok(); +- if status == 0 { ++ ensure!( ++ status == PROCESS_STARTUP_OK || status == PROCESS_STARTUP_ERROR, ++ "ProcessStartupPacket returned invalid status {status}" ++ ); ++ if status == PROCESS_STARTUP_OK { + self.send_conn_data + .call(&mut self.store) + .context("oliphaunt_wasix_send_conn_data")?; +- self.pq_flush +- .call(&mut self.store) +- .context("oliphaunt_wasix_pq_flush after startup")?; + self.protocol_started = true; + } +- self.take_output_js() ++ let flush_status = self ++ .pq_flush ++ .call(&mut self.store) ++ .context("oliphaunt_wasix_pq_flush after startup")?; ++ ensure!( ++ flush_status == 0, ++ "oliphaunt_wasix_pq_flush after startup returned {flush_status}" ++ ); ++ Ok(DirectStartupOutcome::Processed { ++ output: self.take_output_js()?, ++ status, ++ }) + } + + fn exec_protocol_raw_inner( +@@ -685,7 +888,10 @@ + payload: &Uint8Array, + streaming: bool, + ) -> anyhow::Result { +- self.ensure_open()?; ++ ensure!( ++ self.state == DirectInstanceState::GuestPhase, ++ "direct protocol exchange did not enter its guest phase" ++ ); + ensure!( + self.protocol_started, + "Oliphaunt direct startup has not completed" +@@ -727,21 +933,48 @@ + } + + fn close_inner(&mut self) -> anyhow::Result<()> { +- if self.closed { +- return Ok(()); +- } +- self.set_active +- .call(&mut self.store, 0) +- .context("oliphaunt_wasix_set_active(0)")?; +- if let Some(run_atexit_funcs) = &self.run_atexit_funcs { +- run_atexit_funcs +- .call(&mut self.store) +- .context("oliphaunt_wasix_run_atexit_funcs")?; ++ match self.state { ++ DirectInstanceState::Poisoned | DirectInstanceState::Closed => return Ok(()), ++ DirectInstanceState::StartupPhase ++ | DirectInstanceState::GuestPhase ++ | DirectInstanceState::Closing => { ++ self.poison_guest_phase(); ++ anyhow::bail!( ++ "Oliphaunt direct instance cannot close during an active guest phase; the direct instance is closed" ++ ); ++ } ++ DirectInstanceState::Open | DirectInstanceState::StartupRejected => {} + } +- self.closed = true; ++ self.state = DirectInstanceState::Closing; ++ let result = (|| { ++ let expected_active = i32::from(self.backend_started); ++ let previous_active = self ++ .set_active ++ .call(&mut self.store, 0) ++ .context("oliphaunt_wasix_set_active(0)")?; ++ ensure!( ++ previous_active == expected_active, ++ "oliphaunt_wasix_set_active(0) returned previous state {previous_active}, expected {expected_active}" ++ ); ++ if let Some(run_atexit_funcs) = &self.run_atexit_funcs { ++ run_atexit_funcs ++ .call(&mut self.store) ++ .context("oliphaunt_wasix_run_atexit_funcs")?; ++ } ++ Ok(()) ++ })(); + self.backend_started = false; + self.protocol_started = false; +- Ok(()) ++ match result { ++ Ok(()) => { ++ self.state = DirectInstanceState::Closed; ++ Ok(()) ++ } ++ Err(error) => { ++ self.state = DirectInstanceState::Poisoned; ++ Err(error.context("direct instance cleanup failed; the direct instance is closed")) ++ } ++ } + } + async fn instantiate( + module: WebAssembly::Module, +@@ -802,6 +1035,8 @@ + &instance, + "oliphaunt_wasix_prepare_trusted_embedded_session", + )?; ++ let startup_outcome = ++ typed_export(&mut store, &instance, "oliphaunt_wasix_startup_outcome_v1")?; + let wasi_start = typed_export(&mut store, &instance, "_start")?; + let start_oliphaunt = typed_export(&mut store, &instance, "oliphaunt_wasix_start")?; + let run_atexit_funcs = +@@ -838,6 +1073,7 @@ + set_protocol_transport, + set_active, + prepare_trusted_embedded_session, ++ startup_outcome, + wasi_start, + start_oliphaunt, + run_atexit_funcs, +@@ -850,38 +1086,79 @@ + send_ready, + backend_started: false, + protocol_started: false, +- closed: false, ++ state: DirectInstanceState::Open, + }; + direct.reset_io()?; + Ok(direct) + } + + fn ensure_open(&self) -> anyhow::Result<()> { +- ensure!(!self.closed, "Oliphaunt direct instance is closed"); ++ ensure!( ++ self.state == DirectInstanceState::Open, ++ "Oliphaunt direct instance is closed" ++ ); + Ok(()) + } + +- fn poison_main_loop(&mut self) { +- // A trap, invalid typed outcome, or failed post-step export can leave +- // arbitrary guest state behind. Do not call another guest export, +- // including the ordinary close sequence. +- self.closed = true; ++ fn ensure_protocol_started(&self) -> anyhow::Result<()> { ++ match self.state { ++ DirectInstanceState::Open => {} ++ DirectInstanceState::StartupRejected => { ++ anyhow::bail!( ++ "Oliphaunt direct startup was rejected; the direct instance cannot be reused" ++ ) ++ } ++ _ => self.ensure_open()?, ++ } ++ ensure!( ++ self.protocol_started, ++ "Oliphaunt direct startup has not completed" ++ ); ++ Ok(()) ++ } ++ ++ fn run_guest_phase( ++ &mut self, ++ phase: &'static str, ++ operation: impl FnOnce(&mut Self) -> anyhow::Result, ++ ) -> anyhow::Result { ++ self.ensure_open()?; ++ self.state = DirectInstanceState::GuestPhase; ++ match operation(self) { ++ Ok(value) if self.state == DirectInstanceState::GuestPhase => { ++ self.state = DirectInstanceState::Open; ++ Ok(value) ++ } ++ Ok(_) => anyhow::bail!( ++ "{phase} reached a terminal lifecycle state; the direct instance is closed" ++ ), ++ Err(error) => { ++ if self.state == DirectInstanceState::GuestPhase { ++ self.poison_guest_phase(); ++ } ++ Err(error.context(format!( ++ "{phase} failed after entering the guest; the direct instance is closed" ++ ))) ++ } ++ } ++ } ++ ++ fn poison_guest_phase(&mut self) { ++ // Any failure after a guest phase starts can leave arbitrary guest ++ // state behind. No later operation, including close, may call another ++ // guest export. ++ self.state = DirectInstanceState::Poisoned; + self.backend_started = false; + self.protocol_started = false; + } + +- fn terminal_main_loop_outcome(&mut self, failure: impl Into) -> anyhow::Error { +- let failure = failure.into(); +- self.poison_main_loop(); +- anyhow::anyhow!("{failure}; the direct instance is closed") ++ fn terminal_main_loop_outcome(&self, failure: impl Into) -> anyhow::Error { ++ anyhow::anyhow!(failure.into()) + } + +- fn terminal_main_loop_error(&mut self, error: wasmer::RuntimeError) -> anyhow::Error { +- self.poison_main_loop(); +- anyhow::Error::from(error).context( +- "PostgresMainLoopOnce trapped instead of returning a typed outcome; \ +- the direct instance is closed", +- ) ++ fn terminal_main_loop_error(&self, error: wasmer::RuntimeError) -> anyhow::Error { ++ anyhow::Error::from(error) ++ .context("PostgresMainLoopOnce trapped instead of returning a typed outcome") + } + + fn call_main_loop(&mut self) -> anyhow::Result { +@@ -897,38 +1174,35 @@ + } + + fn send_ready_after_main_loop(&mut self) -> anyhow::Result<()> { +- match self.send_ready.call(&mut self.store) { +- Ok(()) => Ok(()), +- Err(error) => { +- self.poison_main_loop(); +- Err(anyhow::Error::from(error).context( +- "PostgresSendReadyForQueryIfNecessary trapped after a typed main-loop outcome; \ +- the direct instance is closed", +- )) +- } +- } ++ self.send_ready ++ .call(&mut self.store) ++ .context("PostgresSendReadyForQueryIfNecessary trapped after a typed main-loop outcome") + } + + fn flush_after_main_loop(&mut self) -> anyhow::Result<()> { +- match self.pq_flush.call(&mut self.store) { +- Ok(()) => Ok(()), +- Err(error) => { +- self.poison_main_loop(); +- Err(anyhow::Error::from(error).context( +- "oliphaunt_wasix_pq_flush trapped after a typed main-loop outcome; \ +- the direct instance is closed", +- )) +- } +- } ++ let status = self ++ .pq_flush ++ .call(&mut self.store) ++ .context("oliphaunt_wasix_pq_flush trapped after a typed main-loop outcome")?; ++ ensure!( ++ status == 0, ++ "protocol output flush returned {status} after a typed main-loop outcome" ++ ); ++ Ok(()) + } + +- fn start_backend(&mut self) -> anyhow::Result<()> { ++ fn start_backend(&mut self) -> anyhow::Result { + if self.backend_started { +- return Ok(()); ++ return Ok(BackendStartOutcome::Alive); + } +- self.set_active ++ let previous_active = self ++ .set_active + .call(&mut self.store, 1) + .context("oliphaunt_wasix_set_active(1)")?; ++ ensure!( ++ previous_active == 0, ++ "oliphaunt_wasix_set_active(1) returned previous state {previous_active}, expected 0" ++ ); + let prepare_status = self + .prepare_trusted_embedded_session + .call(&mut self.store) +@@ -937,16 +1211,107 @@ + prepare_status == 0, + "oliphaunt_wasix_prepare_trusted_embedded_session rejected after PostgreSQL startup began" + ); ++ let startup_outcome_pointer = self ++ .startup_outcome ++ .call(&mut self.store) ++ .context("oliphaunt_wasix_startup_outcome_v1 before _start")?; ++ let pending = self.read_startup_outcome_descriptor(startup_outcome_pointer)?; ++ ensure!( ++ pending.kind == STARTUP_OUTCOME_PENDING, ++ "startup outcome was not pending before _start" ++ ); ++ ensure!( ++ pending.data_pointer == 0 && pending.data_length == 0, ++ "pending startup outcome carried protocol data" ++ ); + match self.wasi_start.call(&mut self.store) { +- Ok(()) => {} + Err(error) if runtime_exit_code(&error) == Some(OLIPHAUNT_EXIT_ALIVE) => {} ++ Err(error) if runtime_exit_code(&error) == Some(OLIPHAUNT_EXIT_STARTUP_REJECTED) => { ++ let protocol = self ++ .read_startup_rejection(startup_outcome_pointer) ++ .context("read rejected _start outcome")?; ++ return Ok(BackendStartOutcome::Rejected(protocol)); ++ } ++ Ok(()) => anyhow::bail!("_start returned without an Oliphaunt lifecycle exit"), + Err(error) => return Err(error).context("_start Oliphaunt single-user backend"), + } + self.start_oliphaunt + .call(&mut self.store) + .context("oliphaunt_wasix_start")?; + self.backend_started = true; +- Ok(()) ++ Ok(BackendStartOutcome::Alive) ++ } ++ ++ fn read_startup_outcome_descriptor( ++ &self, ++ descriptor_pointer: i32, ++ ) -> anyhow::Result { ++ ensure!( ++ descriptor_pointer != 0, ++ "oliphaunt_wasix_startup_outcome_v1 returned null" ++ ); ++ let bytes = oliphaunt_copy_from_guest( ++ &self.guest_memory, ++ descriptor_pointer as u32, ++ STARTUP_OUTCOME_DESCRIPTOR_BYTES, ++ ) ++ .map_err(|error| anyhow::anyhow!("copy startup outcome descriptor: {error:?}"))? ++ .to_vec(); ++ ensure!( ++ bytes.len() == STARTUP_OUTCOME_DESCRIPTOR_BYTES as usize, ++ "short startup outcome descriptor" ++ ); ++ let version = read_u32_le(&bytes, 0); ++ let size = read_u32_le(&bytes, 4); ++ let kind = read_u32_le(&bytes, 8); ++ let reserved = read_u32_le(&bytes, 12); ++ ensure!( ++ version == STARTUP_OUTCOME_VERSION, ++ "unsupported startup outcome version {version}" ++ ); ++ ensure!( ++ size == STARTUP_OUTCOME_DESCRIPTOR_BYTES, ++ "invalid startup outcome descriptor size {size}" ++ ); ++ ensure!(reserved == 0, "startup outcome reserved field was nonzero"); ++ Ok(StartupOutcomeDescriptor { ++ kind, ++ data_pointer: read_u64_le(&bytes, 16), ++ data_length: read_u64_le(&bytes, 24), ++ }) ++ } ++ ++ fn read_startup_rejection(&self, descriptor_pointer: i32) -> anyhow::Result> { ++ let descriptor = self.read_startup_outcome_descriptor(descriptor_pointer)?; ++ ensure!( ++ descriptor.kind == STARTUP_OUTCOME_REJECTED, ++ "_start Exit98 did not publish a rejected startup outcome" ++ ); ++ ensure!( ++ (1..=1_048_576).contains(&descriptor.data_length) && descriptor.data_pointer > 0, ++ "rejected startup outcome carried empty or oversized protocol data" ++ ); ++ let end = descriptor ++ .data_pointer ++ .checked_add(descriptor.data_length) ++ .context("startup rejection protocol bounds overflow")?; ++ ensure!( ++ end <= u32::MAX as u64 + 1, ++ "startup rejection protocol exceeds Wasm32 memory bounds" ++ ); ++ let pointer = u32::try_from(descriptor.data_pointer) ++ .context("startup rejection protocol pointer exceeds u32")?; ++ let length = u32::try_from(descriptor.data_length) ++ .context("startup rejection protocol length exceeds u32")?; ++ let protocol = oliphaunt_copy_from_guest(&self.guest_memory, pointer, length) ++ .map_err(|error| anyhow::anyhow!("copy startup rejection protocol: {error:?}"))? ++ .to_vec(); ++ ensure!( ++ protocol.len() == length as usize, ++ "short startup rejection protocol copy" ++ ); ++ validate_startup_rejection_protocol(&protocol)?; ++ Ok(protocol) + } + + fn reset_io(&mut self) -> anyhow::Result<()> { +@@ -1002,7 +1367,9 @@ + .context("oliphaunt_wasix_output_data")?; + ensure!(ptr > 0, "oliphaunt_wasix_output_data returned null"); + let output = oliphaunt_copy_from_guest(&self.guest_memory, ptr as u32, len as u32) +- .map_err(|error| anyhow::anyhow!("copy direct protocol output from guest: {error:?}")); ++ .map_err(|error| { ++ anyhow::anyhow!("copy direct protocol output from guest: {error:?}") ++ })?; + ensure!( + self.output_reset + .call(&mut self.store) +@@ -1010,7 +1377,7 @@ + == 0, + "oliphaunt_wasix_output_reset after read failed" + ); +- output ++ Ok(output) + } + + fn protocol_input_remaining(&mut self) -> anyhow::Result { +@@ -1022,26 +1389,21 @@ + if available > 0 { + return Ok(available); + } +- self.pq_buffer_remaining_data ++ let buffered = self ++ .pq_buffer_remaining_data + .call(&mut self.store) +- .context("pq_buffer_remaining_data") ++ .context("pq_buffer_remaining_data")?; ++ ensure!(buffered >= 0, "negative buffered protocol input length"); ++ Ok(buffered) + } + +- fn startup_error(&mut self, error: anyhow::Error) -> Error { ++ fn startup_rejection_error(&self, error: anyhow::Error, protocol: Vec) -> Error { + let error = self.stderr.attach(error); +- let _ = self.pq_flush.call(&mut self.store); +- let protocol = self +- .take_output_js() +- .unwrap_or_else(|_| Uint8Array::new_with_length(0)); +- if protocol.length() == 0 { +- return Error::from(error); +- } +- + let js_error = js_sys::Error::new(&error.to_string()); + let _ = js_sys::Reflect::set( + &js_error, + &wasm_bindgen::JsValue::from_str("protocolResponse"), +- &protocol, ++ &Uint8Array::from(protocol.as_slice()), + ); + let _ = js_sys::Reflect::set( + &js_error, +@@ -1052,6 +1414,88 @@ + } + } + ++fn read_u32_le(bytes: &[u8], offset: usize) -> u32 { ++ u32::from_le_bytes(bytes[offset..offset + 4].try_into().unwrap()) ++} ++ ++fn read_u64_le(bytes: &[u8], offset: usize) -> u64 { ++ u64::from_le_bytes(bytes[offset..offset + 8].try_into().unwrap()) ++} ++ ++fn validate_startup_rejection_protocol(protocol: &[u8]) -> anyhow::Result<()> { ++ let mut offset = 0usize; ++ let mut found_error_response = false; ++ while offset < protocol.len() { ++ ensure!( ++ protocol.len() - offset >= 5, ++ "truncated startup rejection protocol frame" ++ ); ++ let tag = protocol[offset]; ++ let frame_length = ++ u32::from_be_bytes(protocol[offset + 1..offset + 5].try_into().unwrap()) as usize; ++ ensure!( ++ frame_length >= 4, ++ "invalid startup rejection protocol frame length {frame_length}" ++ ); ++ let frame_end = offset ++ .checked_add(1) ++ .and_then(|value| value.checked_add(frame_length)) ++ .context("startup rejection protocol frame bounds overflow")?; ++ ensure!( ++ frame_end <= protocol.len(), ++ "truncated startup rejection protocol frame" ++ ); ++ if tag == b'E' { ++ validate_error_response_fields(&protocol[offset + 5..frame_end])?; ++ found_error_response = true; ++ } ++ offset = frame_end; ++ } ++ ensure!( ++ found_error_response, ++ "startup rejection protocol did not contain an ErrorResponse" ++ ); ++ Ok(()) ++} ++ ++fn validate_error_response_fields(fields: &[u8]) -> anyhow::Result<()> { ++ ensure!( ++ fields.last() == Some(&0), ++ "startup ErrorResponse was not terminated" ++ ); ++ let mut offset = 0usize; ++ let mut found_sqlstate = false; ++ while offset < fields.len() - 1 { ++ ensure!( ++ fields[offset] != 0, ++ "startup ErrorResponse terminated before the end of its frame" ++ ); ++ let field_type = fields[offset]; ++ offset += 1; ++ let terminator = fields[offset..] ++ .iter() ++ .position(|byte| *byte == 0) ++ .context("startup ErrorResponse contained an unterminated field")?; ++ if field_type == b'C' { ++ ensure!( ++ terminator == 5, ++ "startup ErrorResponse carried an invalid SQLSTATE" ++ ); ++ found_sqlstate = true; ++ } ++ offset += terminator + 1; ++ } ++ ensure!( ++ offset == fields.len() - 1, ++ "startup ErrorResponse contained trailing bytes" ++ ); ++ ensure!( ++ found_sqlstate, ++ "startup ErrorResponse did not contain a SQLSTATE" ++ ); ++ Ok(()) ++} ++ + fn typed_export( + store: &mut Store, + instance: &WasmerInstance, diff --git a/src/wasix/browser-host/patches/0027-wasmer-wasix-isolate-thread-local-handle-borrows.patch b/src/wasix/browser-host/patches/0027-wasmer-wasix-isolate-thread-local-handle-borrows.patch new file mode 100644 index 000000000..d4dae9713 --- /dev/null +++ b/src/wasix/browser-host/patches/0027-wasmer-wasix-isolate-thread-local-handle-borrows.patch @@ -0,0 +1,176 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: isolate registry and handle borrows + +Clone each realm-local registered handle under a short registry borrow before +borrowing the handle value. This makes the unsafe extended Ref lifetime derive +from the same Rc retained by the returned guard, while registry removal stays +independent from access to the registered value. + +The JS engine can terminate a guest by trapping across a Rust host callback, +and that boundary is not guaranteed to run every Rust destructor. Use fallible +inner borrows so a stranded handle guard becomes absence, and make both +callback_signal handle acquisitions translate that conflict to the existing +terminal WASI Fault outcome. Registry borrow invariant violations still panic; +this does not suppress unrelated reentrant registry mutation. + +diff --git a/src/state/handles/thread_local.rs b/src/state/handles/thread_local.rs +index fd01509b..a6c4939d 100644 +--- a/src/state/handles/thread_local.rs ++++ b/src/state/handles/thread_local.rs +@@ -17,6 +17,20 @@ + = RefCell::new(HashMap::new()); + } + ++fn clone_registered( ++ registry: &RefCell>>>, ++ id: u64, ++) -> Option>> { ++ registry.borrow().get(&id).cloned() ++} ++ ++fn remove_registered( ++ registry: &RefCell>>>, ++ id: u64, ++) -> Option>> { ++ registry.borrow_mut().remove(&id) ++} ++ + /// This non-sendable guard provides memory safe access + /// to the WasiInstance object but only when it is + /// constructed with certain constraints +@@ -75,48 +89,29 @@ + } + impl WasiInstanceHandlesPointer { + pub fn get(&self) -> Option> { +- self.id +- .iter() +- .filter_map(|id| { +- THREAD_LOCAL_INSTANCE_HANDLES.with(|map| { +- let map = map.borrow(); +- if let Some(inner) = map.get(id) { +- let borrow: Ref = inner.borrow(); +- let borrow: Ref<'static, WasiModuleTreeHandles> = +- unsafe { std::mem::transmute(borrow) }; +- Some(WasiInstanceGuard { +- borrow, +- _pointer: self, +- _inner: inner.clone(), +- }) +- } else { +- None +- } +- }) +- }) +- .next() ++ let id = self.id?; ++ // Clone the registry-owned Rc before extending the inner Ref lifetime. ++ // The returned guard retains that same Rc, and registry cleanup never ++ // needs to borrow the registered value. ++ let inner = THREAD_LOCAL_INSTANCE_HANDLES.with(|map| clone_registered(map, id))?; ++ let borrow: Ref = inner.try_borrow().ok()?; ++ let borrow: Ref<'static, WasiModuleTreeHandles> = unsafe { std::mem::transmute(borrow) }; ++ Some(WasiInstanceGuard { ++ borrow, ++ _pointer: self, ++ _inner: inner.clone(), ++ }) + } + pub fn get_mut(&self) -> Option> { +- self.id +- .into_iter() +- .filter_map(|id| { +- THREAD_LOCAL_INSTANCE_HANDLES.with(|map| { +- let map = map.borrow_mut(); +- if let Some(inner) = map.get(&id) { +- let borrow: RefMut = inner.borrow_mut(); +- let borrow: RefMut<'static, WasiModuleTreeHandles> = +- unsafe { std::mem::transmute(borrow) }; +- Some(WasiInstanceGuardMut { +- borrow, +- _pointer: self, +- _inner: inner.clone(), +- }) +- } else { +- None +- } +- }) +- }) +- .next() ++ let id = self.id?; ++ let inner = THREAD_LOCAL_INSTANCE_HANDLES.with(|map| clone_registered(map, id))?; ++ let borrow: RefMut = inner.try_borrow_mut().ok()?; ++ let borrow: RefMut<'static, WasiModuleTreeHandles> = unsafe { std::mem::transmute(borrow) }; ++ Some(WasiInstanceGuardMut { ++ borrow, ++ _pointer: self, ++ _inner: inner.clone(), ++ }) + } + pub fn set(&mut self, val: WasiModuleTreeHandles) { + self.clear(); +@@ -138,12 +133,33 @@ + } + fn destroy(id: u64) { + THREAD_LOCAL_INSTANCE_HANDLES.with(|map| { +- let mut map = map.borrow_mut(); +- map.remove(&id); ++ remove_registered(map, id); + }) + } + } + ++#[cfg(test)] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn registry_removal_does_not_borrow_the_registered_value() { ++ let registry = RefCell::new(HashMap::new()); ++ registry ++ .borrow_mut() ++ .insert(7, Rc::new(RefCell::new(41_u8))); ++ ++ let inner = clone_registered(®istry, 7).expect("registered value"); ++ let retained = inner.borrow(); ++ assert!(inner.try_borrow_mut().is_err()); ++ ++ let removed = remove_registered(®istry, 7); ++ assert!(removed.is_some()); ++ assert!(registry.borrow().is_empty()); ++ assert_eq!(*retained, 41); ++ } ++} ++ + /// This provides access to the memory inside the instance + pub(crate) struct WasiInstanceGuardMemory<'a> { + // the order is very important as the first value is +diff --git a/src/syscalls/wasix/callback_signal.rs b/src/syscalls/wasix/callback_signal.rs +index 940b34d2..5404ec72 100644 +--- a/src/syscalls/wasix/callback_signal.rs ++++ b/src/syscalls/wasix/callback_signal.rs +@@ -28,7 +28,8 @@ + Span::current().record("name", name.as_str()); + + let funct = env +- .inner() ++ .try_inner() ++ .ok_or_else(|| WasiError::Exit(Errno::Fault.into()))? + .main_module_instance_handles() + .instance + .exports +@@ -37,7 +38,13 @@ + Span::current().record("funct_is_some", funct.is_some()); + + { +- let mut env_inner = ctx.data_mut().inner_mut(); ++ // A JS trap can strand an instance-handle guard. Treat that terminal ++ // lifecycle conflict as a guest fault instead of panicking while ++ // acquiring the RefCell mutably. ++ let mut env_inner = ctx ++ .data_mut() ++ .try_inner_mut() ++ .ok_or_else(|| WasiError::Exit(Errno::Fault.into()))?; + let inner = env_inner.main_module_instance_handles_mut(); + inner.signal = funct; + inner.signal_set = true; diff --git a/src/wasix/browser-host/patches/0033-wasmer-wasix-poll-ready-asyncify-work.patch b/src/wasix/browser-host/patches/0033-wasmer-wasix-poll-ready-asyncify-work.patch index f0dd8d332..3802bcc74 100644 --- a/src/wasix/browser-host/patches/0033-wasmer-wasix-poll-ready-asyncify-work.patch +++ b/src/wasix/browser-host/patches/0033-wasmer-wasix-poll-ready-asyncify-work.patch @@ -1,3 +1,13 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: poll ready work before asyncify suspension + +Complete immediately ready work without unnecessary asyncify setup while +retaining exit checks, signal handling and the pending-work continuation. The +seek path uses this only where its future and shared offset survive suspension. +This general runtime optimization must retain pending/error/signal tests; +ready-only benchmark wins do not justify weakening those paths. + diff --git a/src/syscalls/mod.rs b/src/syscalls/mod.rs index 03e09d75..9decf21b 100644 --- a/src/syscalls/mod.rs diff --git a/src/wasix/browser-host/patches/0034-wasmer-js-preserve-seek-end-errors.patch b/src/wasix/browser-host/patches/0034-wasmer-js-preserve-seek-end-errors.patch index 443904bf4..57a9b6234 100644 --- a/src/wasix/browser-host/patches/0034-wasmer-js-preserve-seek-end-errors.patch +++ b/src/wasix/browser-host/patches/0034-wasmer-js-preserve-seek-end-errors.patch @@ -1,3 +1,13 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-js: preserve seek-from-end provider failures + +Compute seek-from-end using the synchronous filesystem provider's result and +leave the cursor unchanged on provider or arithmetic errors. Do not translate +a failed length lookup into a successful seek. The tests cover positive and +negative offsets, overflow and failed providers; this is a bridge correctness +fix independent of performance claims. + diff --git a/src/fs/sync_bridge.rs b/src/fs/sync_bridge.rs index e37f33e..7e79d76 100644 --- a/src/fs/sync_bridge.rs diff --git a/src/wasix/browser-host/patches/0035-wasmer-wasix-preserve-shared-seek-offset.patch b/src/wasix/browser-host/patches/0035-wasmer-wasix-preserve-shared-seek-offset.patch index 9def9d756..bf288d9f5 100644 --- a/src/wasix/browser-host/patches/0035-wasmer-wasix-preserve-shared-seek-offset.patch +++ b/src/wasix/browser-host/patches/0035-wasmer-wasix-preserve-shared-seek-offset.patch @@ -1,3 +1,12 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] wasmer-wasix: retain the original shared offset across seek + +Keep the original open-file-description offset alive while seek suspends. +A reused numeric descriptor must not redirect completion to a different file's +offset. The regression covers suspension plus descriptor reuse. This is a +general WASIX correctness fix suitable for independent upstream review. + diff --git a/src/syscalls/wasi/fd_seek.rs b/src/syscalls/wasi/fd_seek.rs index 31cfe43e..18cf012e 100644 --- a/src/syscalls/wasi/fd_seek.rs diff --git a/src/wasix/browser-host/protocol-contract.generated.rs b/src/wasix/browser-host/protocol-contract.generated.rs new file mode 100644 index 000000000..ca7fb450c --- /dev/null +++ b/src/wasix/browser-host/protocol-contract.generated.rs @@ -0,0 +1,6 @@ +// Generated by src/wasix/runtime/protocol-contract/generate.mjs. +// Staged as src/protocol_contract.rs in the patched Wasmer JS host. +pub(crate) const PROTOCOL_CALLBACK_CHUNK_BYTES: usize = 65536; +pub(crate) const PROTOCOL_BUFFERED: i32 = 0; +pub(crate) const PROTOCOL_HYBRID: i32 = 2; +pub(crate) const PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT: i32 = 3; diff --git a/src/wasix/browser-host/source.toml b/src/wasix/browser-host/source.toml index abb31cc86..a0001ed0f 100644 --- a/src/wasix/browser-host/source.toml +++ b/src/wasix/browser-host/source.toml @@ -49,12 +49,18 @@ series = [ "0013-wasmer-wasix-fast-single-backend-clock.patch", "0014-wasmer-js-track-directory-mutations.patch", "0015-wasmer-js-add-sync-filesystem-bridge.patch", - "0016-wasmer-wasix-direct-single-backend-clock.patch", + "0016-wasmer-wasix-expose-direct-memory.patch", "0017-wasmer-js-direct-pgwire-memory-bridge.patch", "0018-wasmer-js-bound-direct-stderr.patch", "0019-wasmer-parse-exception-reference-types.patch", "0020-wasmer-wasix-preserve-posix-close-durability.patch", "0021-wasmer-js-stream-direct-pgwire.patch", + # Typed recovery, checked entropy, fail-closed guest phases and independent + # handle ownership fixes. Optional drain/immutable-file experiments stay out. + "0023-wasmer-js-prepare-trusted-embedded-session.patch", + "0024-virtual-fs-propagate-random-errors.patch", + "0025-wasmer-js-fail-closed-direct-guest-phases.patch", + "0027-wasmer-wasix-isolate-thread-local-handle-borrows.patch", # Independently preserve provider failures and cursor state in the # synchronous JavaScript filesystem bridge. "0034-wasmer-js-preserve-seek-end-errors.patch", diff --git a/src/wasix/pgwire-server/src/proxy.rs b/src/wasix/pgwire-server/src/proxy.rs index 484e003e4..4a871a023 100644 --- a/src/wasix/pgwire-server/src/proxy.rs +++ b/src/wasix/pgwire-server/src/proxy.rs @@ -24,6 +24,8 @@ use oliphaunt_wasix::session::{ ProtocolPumpOutcome, ProtocolStream, StartupProtocolResponse, startup_error_response_output, }; +// Per-connection socket batching, not a PostgreSQL frame or response limit. +// Larger batches trade stack space for fewer reads; frames can span reads. const PROXY_READ_BUFFER_BYTES: usize = 64 * 1024; /// Blocking PostgreSQL socket proxy for the embedded Oliphaunt runtime. @@ -321,22 +323,6 @@ impl OliphauntProxy { let response = opened.startup(message)?; let response_accepted = response.accepted && !response_contains_error(&response.output); - if response_accepted - && let Some(user) = startup_parameter(message, "user")? - && user != "postgres" - { - let role_response = opened.set_role(user)?; - if response_contains_error(&role_response) { - let _ = write_frontend( - &mut stream, - &role_response, - "write startup role rejection", - )?; - let _ = opened.close(); - close_after_flush = true; - break; - } - } { if !write_frontend( &mut stream, @@ -630,11 +616,6 @@ impl WireBackend { .send_with_connection_protocol_pump(message, || continuation_prefix.into_vec()) } - fn set_role(&mut self, user: &str) -> Result> { - let sql = format!("SET ROLE \"{}\"", user.replace('"', "\"\"")); - self.send(&simple_query(&sql)?) - } - fn reset_session_state(&mut self) -> Result<()> { for sql in ["ROLLBACK", "DISCARD ALL"] { let response = self.send(&simple_query(sql)?)?; diff --git a/src/wasix/pgwire-server/tests/proxy_smoke.rs b/src/wasix/pgwire-server/tests/proxy_smoke.rs index 0eae820c7..b9d06ee10 100644 --- a/src/wasix/pgwire-server/tests/proxy_smoke.rs +++ b/src/wasix/pgwire-server/tests/proxy_smoke.rs @@ -29,6 +29,44 @@ fn tcp_addr(server: &OliphauntServer) -> Result { .context("PostgreSQL TCP server address") } +#[test] +#[ignore = "requires prepared WASIX runtime"] +fn tcp_proxy_initializes_the_startup_principal_before_admission() -> Result<()> { + let mut server = OliphauntServer::builder().start()?; + let addr = tcp_addr(&server)?; + query_proxy( + addr, + false, + "CREATE ROLE proxy_app LOGIN; CREATE ROLE proxy_no_login NOLOGIN; + ALTER ROLE proxy_app SET work_mem = '9MB'", + )?; + let mut client = TcpStream::connect(addr)?; + client.set_read_timeout(Some(Duration::from_secs(30)))?; + client.write_all(&startup_message_for("proxy_app"))?; + read_until_ready(&mut client)?; + client.write_all(&simple_query_message( + "RESET ROLE; SELECT current_user, session_user, current_setting('work_mem'), system_user IS NULL", + ))?; + assert_eq!( + read_query_values(&mut client)?, + vec!["proxy_app", "proxy_app", "9MB", "t"] + ); + client.write_all(&terminate_message())?; + drop(client); + let mut rejected = TcpStream::connect(addr)?; + rejected.set_read_timeout(Some(Duration::from_secs(30)))?; + rejected.write_all(&startup_message_for("proxy_no_login"))?; + let error = read_until_ready(&mut rejected).expect_err("NOLOGIN must fail at startup"); + assert!( + error.to_string().contains("not permitted to log in"), + "{error}" + ); + drop(rejected); + assert_eq!(query_proxy(addr, false, "SELECT 1")?, vec!["1"]); + server.close()?; + Ok(()) +} + #[test] #[ignore = "requires prepared WASIX runtime"] fn tcp_proxy_handles_psql_style_and_fragmented_connections() -> Result<()> { @@ -408,11 +446,15 @@ fn cancel_request() -> Vec { } fn startup_message() -> Vec { + startup_message_for("postgres") +} + +fn startup_message_for(username: &str) -> Vec { let mut message = Vec::new(); message.extend_from_slice(&0_i32.to_be_bytes()); message.extend_from_slice(&PROTOCOL_3.to_be_bytes()); for (key, value) in [ - ("user", "postgres"), + ("user", username), ("database", "postgres"), ("application_name", "oliphaunt-wasix-test"), ] { diff --git a/src/wasix/postgres-tools/crates/aot/aarch64-apple-darwin/build-support.rs b/src/wasix/postgres-tools/crates/aot/aarch64-apple-darwin/build-support.rs index 68dd1f4d7..38f078a56 100644 --- a/src/wasix/postgres-tools/crates/aot/aarch64-apple-darwin/build-support.rs +++ b/src/wasix/postgres-tools/crates/aot/aarch64-apple-darwin/build-support.rs @@ -231,6 +231,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/postgres-tools/crates/aot/aarch64-unknown-linux-gnu/build-support.rs b/src/wasix/postgres-tools/crates/aot/aarch64-unknown-linux-gnu/build-support.rs index 68dd1f4d7..38f078a56 100644 --- a/src/wasix/postgres-tools/crates/aot/aarch64-unknown-linux-gnu/build-support.rs +++ b/src/wasix/postgres-tools/crates/aot/aarch64-unknown-linux-gnu/build-support.rs @@ -231,6 +231,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/postgres-tools/crates/aot/x86_64-pc-windows-msvc/build-support.rs b/src/wasix/postgres-tools/crates/aot/x86_64-pc-windows-msvc/build-support.rs index 68dd1f4d7..38f078a56 100644 --- a/src/wasix/postgres-tools/crates/aot/x86_64-pc-windows-msvc/build-support.rs +++ b/src/wasix/postgres-tools/crates/aot/x86_64-pc-windows-msvc/build-support.rs @@ -231,6 +231,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/postgres-tools/crates/aot/x86_64-unknown-linux-gnu/build-support.rs b/src/wasix/postgres-tools/crates/aot/x86_64-unknown-linux-gnu/build-support.rs index 68dd1f4d7..38f078a56 100644 --- a/src/wasix/postgres-tools/crates/aot/x86_64-unknown-linux-gnu/build-support.rs +++ b/src/wasix/postgres-tools/crates/aot/x86_64-unknown-linux-gnu/build-support.rs @@ -231,6 +231,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/postmaster/bin/build-sealed-headless-carrier.test.sh b/src/wasix/postmaster/bin/build-sealed-headless-carrier.test.sh index afe922eaa..2d06ff791 100755 --- a/src/wasix/postmaster/bin/build-sealed-headless-carrier.test.sh +++ b/src/wasix/postmaster/bin/build-sealed-headless-carrier.test.sh @@ -160,7 +160,7 @@ target_triple="$(fresh_host_arch | sed 's/linux-amd64/x86_64-unknown-linux-gnu/; printf 'wasmer_napi_commit=%s\n' "$FRESH_WASMER_NAPI_COMMIT" printf 'wasmer_test_files_commit=%s\n' "$FRESH_WASMER_TEST_FILES_COMMIT" printf 'wasmer_spec_commit=%s\n' "$FRESH_WASMER_SPEC_COMMIT" - printf 'wasmer_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$project_root/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch")" + printf 'wasmer_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$project_root/wasmer/patches/wasmer/series")" printf 'wasmer_prepared_signature_sha256=%064d\n' 0 printf 'wasmer_cargo_lock_sha256=%s\n' "$cargo_lock_sha256" printf 'wasmer_binary_sha256=%s\n' "$(fresh_wasmer_bin_hash "$FRESH_UPSTREAM_WASMER_BIN")" @@ -170,7 +170,7 @@ target_triple="$(fresh_host_arch | sed 's/linux-amd64/x86_64-unknown-linux-gnu/; printf 'runtime_abi_id=%s\n' "$runtime_abi_id" printf 'artifact_abi_version=%s\n' "$FRESH_WASMER_ARTIFACT_ABI_VERSION" printf 'wasix_libc_source_commit=%s\n' "$FRESH_WASIX_LIBC_SOURCE_COMMIT" - printf 'wasix_libc_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$project_root/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch")" + printf 'wasix_libc_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$project_root/wasmer/patches/wasix-libc/series")" printf 'wasix_libc_prepared_signature_sha256=%064d\n' 0 printf 'sysroot_carrier_manifest_sha256=%064d\n' 0 printf 'sysroot_variant=%s\n' "$WASIXCC_SYSROOT_VARIANT" diff --git a/src/wasix/postmaster/executor/src/bin/compiler.rs b/src/wasix/postmaster/executor/src/bin/compiler.rs index c323d4db5..0ee2d713f 100644 --- a/src/wasix/postmaster/executor/src/bin/compiler.rs +++ b/src/wasix/postmaster/executor/src/bin/compiler.rs @@ -106,6 +106,16 @@ fn parse() -> Result> { })) } +fn product_compiler() -> LLVM { + let mut compiler = LLVM::new(); + // Retain main's policy; strict shared-memory compilation is a separate change. + compiler + .opt_level(LLVMOptLevel::Aggressive) + .non_volatile_memops(true) + .readonly_funcref_table(true); + compiler +} + fn main() -> Result<()> { let arguments: Vec<_> = env::args_os().skip(1).collect(); if arguments.first().and_then(|value| value.to_str()) == Some("verify-aot") { @@ -120,11 +130,7 @@ fn main() -> Result<()> { verify_module_bytes(&module_bytes) .with_context(|| format!("admit sealed module {}", options.input.display()))?; - let mut compiler = LLVM::new(); - compiler - .opt_level(LLVMOptLevel::Aggressive) - .non_volatile_memops(true) - .readonly_funcref_table(true); + let mut compiler = product_compiler(); if let Some(threads) = options.compiler_threads { compiler.num_threads(threads); } @@ -171,11 +177,7 @@ fn verify_aot(module_path: PathBuf, artifact_path: PathBuf) -> Result<()> { .context("derive product verifier linear-memory profile")?; let mut features = Features::default(); features.threads(true).exceptions(true); - let mut compiler = LLVM::new(); - compiler - .opt_level(LLVMOptLevel::Aggressive) - .non_volatile_memops(true) - .readonly_funcref_table(true); + let compiler = product_compiler(); let mut engine = Engine::new( Box::new(compiler) as Box, target, @@ -209,3 +211,14 @@ fn verify_aot(module_path: PathBuf, artifact_path: PathBuf) -> Result<()> { println!("{}", expected_hash); Ok(()) } + +#[test] +fn product_compiler_uses_main_memory_identity() { + let relaxed_id = Box::new(product_compiler()).compiler().deterministic_id(); + let mut strict = product_compiler(); + strict.non_volatile_memops(false); + let strict_id = Box::new(strict).compiler().deterministic_id(); + assert!(strict_id.contains("-nv0-")); + assert!(relaxed_id.contains("-nv1-")); + assert_ne!(strict_id, relaxed_id); +} diff --git a/src/wasix/postmaster/lib/common.sh b/src/wasix/postmaster/lib/common.sh index 0d988051d..095fbe490 100644 --- a/src/wasix/postmaster/lib/common.sh +++ b/src/wasix/postmaster/lib/common.sh @@ -1009,8 +1009,8 @@ fresh_runtime_abi_id() { local target_triple="$2" local host_platform="$3" local host_abi="$4" - local wasmer_patch="$FRESH_ROOT/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch" - local wasix_libc_patch="$FRESH_ROOT/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch" + local wasmer_patch="$FRESH_ROOT/wasmer/patches/wasmer/series" + local wasix_libc_patch="$FRESH_ROOT/wasmer/patches/wasix-libc/series" fresh_is_sha256 "$cargo_lock_sha256" || { printf 'runtime ABI Cargo.lock identity is not a lowercase SHA-256\n' >&2 @@ -1029,10 +1029,10 @@ fresh_runtime_abi_id() { printf '%s\0%s\0' wasmer-napi-commit "$FRESH_WASMER_NAPI_COMMIT" printf '%s\0%s\0' wasmer-test-files-commit "$FRESH_WASMER_TEST_FILES_COMMIT" printf '%s\0%s\0' wasmer-spec-commit "$FRESH_WASMER_SPEC_COMMIT" - printf '%s\0%s\0' wasmer-patch-sha256 "$(fresh_wasmer_bin_hash "$wasmer_patch")" + printf '%s\0%s\0' wasmer-patch-sha256 "$(fresh_runtime_patch_hash "$wasmer_patch")" printf '%s\0%s\0' wasmer-cargo-lock-sha256 "$cargo_lock_sha256" printf '%s\0%s\0' wasix-libc-source-commit "$FRESH_WASIX_LIBC_SOURCE_COMMIT" - printf '%s\0%s\0' wasix-libc-patch-sha256 "$(fresh_wasmer_bin_hash "$wasix_libc_patch")" + printf '%s\0%s\0' wasix-libc-patch-sha256 "$(fresh_runtime_patch_hash "$wasix_libc_patch")" printf '%s\0%s\0' sysroot-variant "$WASIXCC_SYSROOT_VARIANT" printf '%s\0%s\0' target-triple "$target_triple" printf '%s\0%s\0' host-platform "$host_platform" @@ -1099,14 +1099,14 @@ fresh_require_local_wasmer_build_state() { local wasix_libc_signature="$runtime_root/.prepared/wasix-libc.signature" local carrier_manifest="$WASIXCC_SYSROOT_PREFIX/.oliphaunt-patched-sysroots.manifest" local variant_manifest="$WASIXCC_SYSROOT/.oliphaunt-patched-sysroot.manifest" - local wasmer_patch="$FRESH_ROOT/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch" - local wasix_libc_patch="$FRESH_ROOT/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch" + local wasmer_patch="$FRESH_ROOT/wasmer/patches/wasmer/series" + local wasix_libc_patch="$FRESH_ROOT/wasmer/patches/wasix-libc/series" local wasmer_patch_hash local wasix_libc_patch_hash fresh_require_command git || return - wasmer_patch_hash="$(fresh_wasmer_bin_hash "$wasmer_patch")" - wasix_libc_patch_hash="$(fresh_wasmer_bin_hash "$wasix_libc_patch")" + wasmer_patch_hash="$(fresh_runtime_patch_hash "$wasmer_patch")" + wasix_libc_patch_hash="$(fresh_runtime_patch_hash "$wasix_libc_patch")" fresh_require_prepared_worktree \ Wasmer "$wasmer_root" "$FRESH_WASMER_SOURCE_COMMIT" "$wasmer_patch_hash" \ "$FRESH_WASMER_NAPI_COMMIT:$FRESH_WASMER_TEST_FILES_COMMIT:$FRESH_WASMER_SPEC_COMMIT:$(fresh_executor_source_sha256)" \ @@ -1135,8 +1135,8 @@ fresh_require_local_wasmer_build_state() { fresh_require_patched_wasmer_receipt() { local manifest="${WASMER_BUILD_RECEIPT:-$FRESH_WASMER_BUILD_RECEIPT}" - local wasmer_patch="$FRESH_ROOT/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch" - local wasix_libc_patch="$FRESH_ROOT/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch" + local wasmer_patch="$FRESH_ROOT/wasmer/patches/wasmer/series" + local wasix_libc_patch="$FRESH_ROOT/wasmer/patches/wasix-libc/series" [ -f "$manifest" ] && [ ! -L "$manifest" ] || { printf 'missing regular Wasmer build receipt: %s\n' "$manifest" >&2 @@ -1162,9 +1162,9 @@ fresh_require_patched_wasmer_receipt() { fresh_require_manifest_value \ "$manifest" wasix_libc_source_commit "$FRESH_WASIX_LIBC_SOURCE_COMMIT" || return fresh_require_manifest_value \ - "$manifest" wasmer_patch_sha256 "$(fresh_wasmer_bin_hash "$wasmer_patch")" || return + "$manifest" wasmer_patch_sha256 "$(fresh_runtime_patch_hash "$wasmer_patch")" || return fresh_require_manifest_value \ - "$manifest" wasix_libc_patch_sha256 "$(fresh_wasmer_bin_hash "$wasix_libc_patch")" || return + "$manifest" wasix_libc_patch_sha256 "$(fresh_runtime_patch_hash "$wasix_libc_patch")" || return fresh_require_manifest_value \ "$manifest" wasmer_features "$FRESH_WASMER_COMPILER_FEATURES" || return fresh_require_manifest_value \ @@ -1265,7 +1265,7 @@ fresh_require_patched_postmaster_executor() { "$executor_receipt" wasmer_source_commit "$FRESH_WASMER_SOURCE_COMMIT" || return fresh_require_manifest_value \ "$executor_receipt" wasmer_patch_sha256 \ - "$(fresh_wasmer_bin_hash "$FRESH_ROOT/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch")" || return + "$(fresh_runtime_patch_hash "$FRESH_ROOT/wasmer/patches/wasmer/series")" || return fresh_require_manifest_value \ "$executor_receipt" wasmer_prepared_signature_sha256 \ "$(fresh_manifest_value "$wasmer_receipt" wasmer_prepared_signature_sha256)" || return @@ -1660,6 +1660,58 @@ fresh_git_worktree_state_sha256() { } | fresh_sha256_stream } +# Used by the two runtime patch series; PostgreSQL uses the central helper. Keep the +# ordered manifest authoritative; never apply an incidental directory glob. +fresh_patch_series_files() { + local patches_dir="$1" + local series_file="$2" + local patch_name + local seen=" " + [ -f "$series_file" ] && [ ! -L "$series_file" ] || return 2 + while IFS= read -r patch_name || [ -n "$patch_name" ]; do + case "$patch_name" in + ''|'#'*) continue ;; + .*|*[!a-zA-Z0-9._-]*) printf 'unsafe patch entry: %s\n' "$patch_name" >&2; return 2 ;; + *.patch) ;; + *) printf 'not a patch entry: %s\n' "$patch_name" >&2; return 2 ;; + esac + case "$seen" in + *" $patch_name "*) printf 'duplicate patch entry: %s\n' "$patch_name" >&2; return 2 ;; + esac + seen="$seen$patch_name " + [ -f "$patches_dir/$patch_name" ] && [ ! -L "$patches_dir/$patch_name" ] || return 2 + printf '%s\n' "$patches_dir/$patch_name" + done <"$series_file" + [ "$seen" != " " ] +} + +fresh_apply_patch_series() { + local worktree="$1" + local files + local patch + files="$(fresh_patch_series_files "$2" "$3")" || return + while IFS= read -r patch; do + git -C "$worktree" apply --whitespace=error-all "$patch" || return + done <<<"$files" +} + +# Existing receipt *_patch_sha256 keys now mean this ordered-series digest. +# Include names and the manifest itself as well as every member's contents. +fresh_runtime_patch_hash() { + local series_file="$1" + local files + local patch + local digest + files="$(fresh_patch_series_files "$(dirname "$series_file")" "$series_file")" || return + { + printf 'series\0%s\0' "$(fresh_wasmer_bin_hash "$series_file")" + while IFS= read -r patch; do + digest="$(fresh_wasmer_bin_hash "$patch")" || return + printf 'patch\0%s\0%s\0' "${patch##*/}" "$digest" + done <<<"$files" + } | fresh_sha256_stream +} + fresh_overlay_digest() { local overlay_dir="$FRESH_ROOT/postgres/overlays/wasix-core" local series="$FRESH_ROOT/postgres/series" diff --git a/src/wasix/postmaster/lib/common.test.sh b/src/wasix/postmaster/lib/common.test.sh index 2a71aa5ae..42029235e 100644 --- a/src/wasix/postmaster/lib/common.test.sh +++ b/src/wasix/postmaster/lib/common.test.sh @@ -93,15 +93,15 @@ if env FRESH_PROJECT_SOURCE_ID_PREFIX=src/wasix/postmaster \ fi [ "$(fresh_project_source_identity_path \ - "$project_root/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch")" = \ - "src/wasix/postmaster/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch" ] + "$project_root/wasmer/patches/wasix-libc/series")" = \ + "src/wasix/postmaster/wasmer/patches/wasix-libc/series" ] original_fresh_root="$FRESH_ROOT" frozen_root="$test_root/measurement-tool-closures/example" mkdir -p "$frozen_root/wasmer/patches/wasix-libc" FRESH_ROOT="$frozen_root" [ "$(fresh_project_source_identity_path \ - "$frozen_root/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch")" = \ - "src/wasix/postmaster/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch" ] + "$frozen_root/wasmer/patches/wasix-libc/series")" = \ + "src/wasix/postmaster/wasmer/patches/wasix-libc/series" ] expect_source_identity_failure() { fresh_project_source_identity_path "$1" >/dev/null 2>&1 } @@ -413,7 +413,7 @@ write_receipt() { printf 'wasmer_napi_commit=706383f42391cb4e4e82e5fd5e63a0ebf81ae19d\n' printf 'wasmer_test_files_commit=7f27e84c69af3b772f751d6c4a733d9f448b2c70\n' printf 'wasmer_spec_commit=7e0b83aba9dbbb6e0623c9334b0f73b3bb584b90\n' - printf 'wasmer_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$project_root/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch")" + printf 'wasmer_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$project_root/wasmer/patches/wasmer/series")" printf 'wasmer_prepared_signature_sha256=%064d\n' 0 printf 'wasmer_cargo_lock_sha256=%s\n' "$cargo_lock_sha256" printf 'wasmer_binary_sha256=%s\n' "$(fresh_wasmer_bin_hash "$FRESH_UPSTREAM_WASMER_BIN")" @@ -423,7 +423,7 @@ write_receipt() { printf 'runtime_abi_id=%s\n' "$runtime_abi_id" printf 'artifact_abi_version=%s\n' "$FRESH_WASMER_ARTIFACT_ABI_VERSION" printf 'wasix_libc_source_commit=34178a6272804f90448b5bd08dc7bcf0d85438e3\n' - printf 'wasix_libc_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$project_root/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch")" + printf 'wasix_libc_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$project_root/wasmer/patches/wasix-libc/series")" printf 'wasix_libc_prepared_signature_sha256=%064d\n' 0 printf 'sysroot_carrier_manifest_sha256=%064d\n' 0 printf 'sysroot_variant=%s\n' "$WASIXCC_SYSROOT_VARIANT" @@ -444,7 +444,7 @@ write_postmaster_executor_receipt() { printf 'build_recipe_sha256=%s\n' "$(fresh_runtime_build_recipe_sha256)" printf 'wasmer_build_receipt_sha256=%s\n' "$(fresh_wasmer_bin_hash "$WASMER_BUILD_RECEIPT")" printf 'wasmer_source_commit=%s\n' "$FRESH_WASMER_SOURCE_COMMIT" - printf 'wasmer_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$project_root/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch")" + printf 'wasmer_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$project_root/wasmer/patches/wasmer/series")" printf 'wasmer_prepared_signature_sha256=%s\n' \ "$(fresh_manifest_value "$WASMER_BUILD_RECEIPT" wasmer_prepared_signature_sha256)" printf 'wasmer_cargo_lock_sha256=%s\n' \ @@ -928,4 +928,55 @@ FRESH_UPSTREAM_WASMER_BIN="$test_root/missing" \ PATH="$test_root/bin:$PATH" \ expect_failure fresh_wasmer_bin -printf 'patched Wasmer receipt selection tests passed\n' +series_dir="$test_root/runtime-series" +mkdir -p "$series_dir" +printf 'first payload\n' >"$series_dir/0001-one.patch" +printf 'second payload\n' >"$series_dir/0002-two.patch" +printf '0001-one.patch\n0002-two.patch\n' >"$series_dir/series" +series_hash="$(fresh_runtime_patch_hash "$series_dir/series")" +fresh_is_sha256 "$series_hash" +printf 'changed payload\n' >>"$series_dir/0002-two.patch" +[ "$series_hash" != "$(fresh_runtime_patch_hash "$series_dir/series")" ] +series_hash="$(fresh_runtime_patch_hash "$series_dir/series")" +printf '0002-two.patch\n0001-one.patch\n' >"$series_dir/series" +[ "$series_hash" != "$(fresh_runtime_patch_hash "$series_dir/series")" ] +printf '0001-one.patch\n0001-one.patch\n' >"$series_dir/series" +expect_failure fresh_runtime_patch_hash "$series_dir/series" +printf '../escape.patch\n' >"$series_dir/series" +expect_failure fresh_runtime_patch_hash "$series_dir/series" +printf '0003-link.patch\n' >"$series_dir/series" +ln -s "$series_dir/0001-one.patch" "$series_dir/0003-link.patch" +expect_failure fresh_runtime_patch_hash "$series_dir/series" +printf '# no active patches\n' >"$series_dir/series" +expect_failure fresh_runtime_patch_hash "$series_dir/series" + +# Prepared-tree admission supports dependent patches without trying to reverse +# each patch independently against the final tree. +prepared_fixture="$test_root/prepared-libc" +git init --quiet "$prepared_fixture" +printf 'base\n' >"$prepared_fixture/tracked.txt" +git -C "$prepared_fixture" add tracked.txt +git -C "$prepared_fixture" -c user.name=Fixture -c user.email=fixture@example.invalid \ + commit --quiet -m fixture +prepared_commit="$(git -C "$prepared_fixture" rev-parse HEAD)" +printf 'first\n' >"$prepared_fixture/tracked.txt" +git -C "$prepared_fixture" diff >"$series_dir/0001-one.patch" +git -C "$prepared_fixture" add tracked.txt +printf 'second\n' >"$prepared_fixture/tracked.txt" +git -C "$prepared_fixture" diff >"$series_dir/0002-two.patch" +printf '0001-one.patch\n0002-two.patch\n' >"$series_dir/series" +series_hash="$(fresh_runtime_patch_hash "$series_dir/series")" +prepared_signature="$test_root/prepared.signature" +printf '%s:%s::%s' "$prepared_commit" "$series_hash" \ + "$(fresh_runtime_worktree_state_hash "$prepared_fixture")" >"$prepared_signature" +fresh_require_prepared_worktree fixture "$prepared_fixture" "$prepared_commit" \ + "$series_hash" "" "$prepared_signature" +printf 'tampered\n' >>"$prepared_fixture/tracked.txt" +expect_failure fresh_require_prepared_worktree fixture "$prepared_fixture" \ + "$prepared_commit" "$series_hash" "" "$prepared_signature" +printf 'second\n' >"$prepared_fixture/tracked.txt" +printf '0002-two.patch\n0001-one.patch\n' >"$series_dir/series" +expect_failure fresh_require_prepared_worktree fixture "$prepared_fixture" \ + "$prepared_commit" "$(fresh_runtime_patch_hash "$series_dir/series")" "" "$prepared_signature" + +printf 'patched Wasmer receipt selection and ordered-series tests passed\n' diff --git a/src/wasix/postmaster/moon.yml b/src/wasix/postmaster/moon.yml index 34ee7db55..16e1bf58a 100644 --- a/src/wasix/postmaster/moon.yml +++ b/src/wasix/postmaster/moon.yml @@ -60,6 +60,7 @@ fileGroups: - "!wasmer/capabilities.tsv" - "!wasmer/**/*.md" - "!wasmer/**/*.test.mts" + - "!wasmer/**/*.test.py" tasks: executor-build: script: "cd executor\ncargo build --locked --release --no-default-features --features product-executor --bin oliphaunt-wasix-postmaster-executor --target-dir ../../../target/oliphaunt-wasix-postmaster/runtime/postmaster-executor-target\n" @@ -94,7 +95,7 @@ tasks: - quality - unit - requires-rust - script: "set -e\nbash tools/dev/bun.sh test ./src/wasix/postmaster\ncargo test --locked --manifest-path src/wasix/postmaster/tools/sealed-export-closure/Cargo.toml --target-dir target/oliphaunt-wasix-postmaster/runtime/sealed-export-closure-target\nwhile IFS= read -r test_file; do bash \"$test_file\"; done < <(find src/wasix/postmaster -type f -name '*.test.sh' ! -name 'seal-wasix-linear-memory.test.sh' | LC_ALL=C sort)\n" + script: "set -e\nbash tools/dev/bun.sh test ./src/wasix/postmaster\ncommand -v python3 >/dev/null\nwhile IFS= read -r test_file; do python3 \"$test_file\"; done < <(find src/wasix/postmaster -type f -name '*.test.py' | LC_ALL=C sort)\ncargo test --locked --manifest-path src/wasix/postmaster/tools/sealed-export-closure/Cargo.toml --target-dir target/oliphaunt-wasix-postmaster/runtime/sealed-export-closure-target\nwhile IFS= read -r test_file; do bash \"$test_file\"; done < <(find src/wasix/postmaster -type f -name '*.test.sh' ! -name 'seal-wasix-linear-memory.test.sh' | LC_ALL=C sort)\n" inputs: - "/tools/dev/bun.sh" - "**/*" diff --git a/src/wasix/postmaster/postgres/patches/0001-wasix-use-posix-dsm-not-sysv.patch b/src/wasix/postmaster/postgres/patches/0001-wasix-use-posix-dsm-not-sysv.patch index d32b0129f..a849f829f 100644 --- a/src/wasix/postmaster/postgres/patches/0001-wasix-use-posix-dsm-not-sysv.patch +++ b/src/wasix/postmaster/postgres/patches/0001-wasix-use-posix-dsm-not-sysv.patch @@ -1,3 +1,22 @@ +From 0000000000000000000000000000000000000001 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] postgres: select POSIX dynamic shared memory for WASIX + +Base: PostgreSQL REL_18_4; apply in postgres/patches/series order. + +The WASIX postmaster starts separately instantiated backends that must reopen +named shared-memory objects. WASIX provides POSIX shared memory but not the +SysV IPC headers required by the default non-Windows implementation. + +Exclude unavailable SysV and mmap-file DSM implementations on WASIX while +retaining the existing POSIX DSM implementation and native behavior. This is +a platform-correctness adaptation, not a single-backend optimization. It is +an upstream WASIX-port candidate once that port's runtime contract is accepted. +Validate with ordered application, shared-memory probes and concurrent +backend lifecycle tests; application alone does not establish shared coherence. +--- + diff --git a/src/backend/storage/ipc/dsm_impl.c b/src/backend/storage/ipc/dsm_impl.c index 2f6557adf8..81bd637f3d 100644 --- a/src/backend/storage/ipc/dsm_impl.c diff --git a/src/wasix/postmaster/postgres/patches/0003-wasix-libpq-static-encoding-shim.patch b/src/wasix/postmaster/postgres/patches/0003-wasix-libpq-static-encoding-shim.patch index 552f8fd97..b51333f23 100644 --- a/src/wasix/postmaster/postgres/patches/0003-wasix-libpq-static-encoding-shim.patch +++ b/src/wasix/postmaster/postgres/patches/0003-wasix-libpq-static-encoding-shim.patch @@ -1,3 +1,22 @@ +From 0000000000000000000000000000000000000003 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] postgres: link static libpq encoding wrappers on WASIX + +Base: PostgreSQL REL_18_4; apply after the wasix-core overlay and in series order. + +Static libpgcommon uses private encoding symbol names, whereas libpq clients +need the public wrappers. Link the overlay's wasix_encoding_shim.o only into +the wasix-core static archive instead of adding the full shared common archive +and duplicating unrelated symbols. The wrappers forward to existing encoding +implementations without changing their behavior. + +This is a repository port/linkage seam, not a query optimization. A generic +upstream static-link solution should replace it if PostgreSQL gains this +target. Check frontend links and the libpq regression subset as well as patch +application; the overlay is a required input, not generated upstream code. +--- + diff --git a/src/interfaces/libpq/Makefile b/src/interfaces/libpq/Makefile index 8fab74f..4a90ecf 100644 --- a/src/interfaces/libpq/Makefile diff --git a/src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.patch b/src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.patch index 194bce8e4..88cd54eb8 100644 --- a/src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.patch +++ b/src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.patch @@ -1,3 +1,28 @@ +From 0000000000000000000000000000000000000004 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] postgres: integrate the WASIX EXEC_BACKEND process model + +Base: PostgreSQL REL_18_4; apply after the wasix-core overlay and in series order. + +Use fresh WASIX module instances for postmaster children and initdb servers, +preserving PostgreSQL's explicit EXEC_BACKEND handoff. Use the serialized +header length rather than padded sizeof(BackendParameters), retry interrupted +reads, and avoid parent-mutating exit handlers after failed vfork/exec. + +WASIX has subprocesses and pipes but no ambient shell/locale utility. Launch +owned argument vectors directly and preserve executable-version failures as +errors, not NULL dereferences. Encode the pinned runtime's credential, session, +directory-entry and shared-memory limitations explicitly under WASIX guards; +none of these exceptions grants host privileges or changes native behavior. + +This dependent port integration bundle is repository-specific in its current +form. Before upstream submission, separate portable serialization/error fixes +from process, tooling and credential seams; keep their actual dependencies. +The ordered series, version-probe failure test, initdb, concurrent connection +and immediate-recovery tests cover different contracts and are all relevant. +--- + diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c index 8acbfbb..ee138a5 100644 --- a/src/backend/commands/collationcmds.c @@ -790,7 +815,7 @@ index 8b690a1..e9dac8b 100644 char *line; if (find_my_exec(argv0, retpath) < 0) -@@ -327,10 +338,17 @@ find_other_exec(const char *argv0, const char *target, +@@ -327,10 +338,18 @@ find_other_exec(const char *argv0, const char *target, if (validate_exec(retpath) != 0) return -1; @@ -803,13 +828,16 @@ index 8b690a1..e9dac8b 100644 + } +#else + snprintf(cmd, sizeof(cmd), "\"%s\" -V", retpath); - if ((line = pipe_read_line(cmd)) == NULL) - return -1; +- if ((line = pipe_read_line(cmd)) == NULL) +- return -1; ++ line = pipe_read_line(cmd); +#endif ++ if (line == NULL) ++ return -1; if (strcmp(line, versionstr) != 0) { -@@ -342,6 +360,191 @@ find_other_exec(const char *argv0, const char *target, +@@ -342,6 +361,191 @@ find_other_exec(const char *argv0, const char *target, return 0; } diff --git a/src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.test.py b/src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.test.py new file mode 100644 index 000000000..a6852da8c --- /dev/null +++ b/src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.test.py @@ -0,0 +1,57 @@ +#!/usr/bin/env python3 +"""Exercise the patched executable-version probe's native and WASIX branches.""" + +import os +from pathlib import Path +import shlex +import subprocess +import tempfile + + +patch = Path(__file__).with_suffix("").with_suffix(".patch").read_text() +start = patch.index("@@ -327,10 ") +end = patch.index("\n@@", start + 3) +lines = patch[start:end].splitlines()[1:] +updated = "\n".join(line[1:] for line in lines if line.startswith(("+", " "))) +probe = updated[updated.index("#ifdef __wasi__"):updated.index("\tif (strcmp")] + +source = r''' +#include +#include +#include +static char *result; +static char *wasix_read_first_line_from_argv(char *const argv[]) { return result; } +static char *pipe_read_line(const char *command) { return result; } +static int probe_version(void) { + char *line; + char retpath[] = "/postgres"; + const char *versionstr = "PostgreSQL 18.4\n"; + char cmd[128]; +''' + probe + r''' + return strcmp(line, versionstr) == 0 ? 0 : -2; +} +int main(void) { + /* Spawn/read failure, empty output, and failed child all yield NULL. */ + result = NULL; + assert(probe_version() == -1); + result = "PostgreSQL 18.4\n"; + assert(probe_version() == 0); + result = "PostgreSQL 17.0\n"; + assert(probe_version() == -2); +} +''' + +with tempfile.TemporaryDirectory(prefix="postgres-version-probe-") as temporary: + directory = Path(temporary) + source_path = directory / "probe.c" + source_path.write_text(source) + for mode in ([], ["-D__wasi__"]): + executable = directory / "probe" + subprocess.run( + shlex.split(os.environ.get("CC", "cc")) + + ["-std=c99", *mode, str(source_path), "-o", str(executable)], + check=True, + ) + subprocess.run([str(executable)], check=True) + +print("native and WASIX executable-version failure checks passed") diff --git a/src/wasix/postmaster/postgres/patches/0006-wasix-retry-proc-join-on-eintr.patch b/src/wasix/postmaster/postgres/patches/0006-wasix-retry-proc-join-on-eintr.patch index 50adb5c20..1793e8b0f 100644 --- a/src/wasix/postmaster/postgres/patches/0006-wasix-retry-proc-join-on-eintr.patch +++ b/src/wasix/postmaster/postgres/patches/0006-wasix-retry-proc-join-on-eintr.patch @@ -1,3 +1,22 @@ +From 0000000000000000000000000000000000000006 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] postgres: retry interrupted WASIX child joins + +Base: PostgreSQL REL_18_4 plus earlier entries in postgres/patches/series. +Depends-on: 0004-wasix-core-execbackend-initdb-runtime.patch. + +WASIX proc_join can clear its input/output PID before a signal interrupts the +wait. Restore the requested child on every EINTR retry in both initdb and the +common executable probe. Otherwise a normal child exit can be reported as a +failed command or a subsequent retry can wait for the wrong child. + +Retain the existing handling of actual join errors and exit statuses. This +is a process-correctness repair, not a performance shortcut. It is suitable +for an upstream WASIX process adapter once that prerequisite exists. Validate +signal/child supervision and backend waves, not only uninterrupted initdb. +--- + diff --git a/src/bin/initdb/initdb.c b/src/bin/initdb/initdb.c --- a/src/bin/initdb/initdb.c +++ b/src/bin/initdb/initdb.c diff --git a/src/wasix/postmaster/postgres/patches/0008-wasix-packed-atomic-latch-state.patch b/src/wasix/postmaster/postgres/patches/0008-wasix-packed-atomic-latch-state.patch index f5747f1b2..2388de513 100644 --- a/src/wasix/postmaster/postgres/patches/0008-wasix-packed-atomic-latch-state.patch +++ b/src/wasix/postmaster/postgres/patches/0008-wasix-packed-atomic-latch-state.patch @@ -1,3 +1,29 @@ +From 0000000000000000000000000000000000000008 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] postgres: preserve WASIX cross-instance latch ordering + +Base: PostgreSQL REL_18_4; also applies in postgres/patches/series order. + +The EXEC_BACKEND runtime maps the same shared-memory backing into separate +Wasm instances. Release-O3 ThinLTO can erase fence-only latch ordering before +the final module, leaving a granted lock holder asleep. Pack SET and SLEEPING +into one naturally aligned lock-free atomic word so waiter publication and +the wake decision participate in an explicit sequentially consistent order. + +Preserve latch layout with a reserved word and compile-time size/offset checks. +Reset clears only SET; wait paths retract SLEEPING; SetLatch signals only the +clear-to-set transition that observed a published waiter. Native behavior and +signal delivery primitives remain unchanged. Require PG_WASIX_ATOMIC_LATCH_STATE +only on the port with the matching atomic contract. + +This correctness seam is not an optional speed switch. A general PostgreSQL +upstream proposal needs a portable concurrency argument and tests rather than +simply widening the WASIX guard. Acceptance includes the state-machine check, +clean standalone and series application, final-Wasm atomic verification, +cross-instance latch probes and fresh-postmaster backend-wave qualification. +--- + diff --git a/src/backend/storage/ipc/latch.c b/src/backend/storage/ipc/latch.c index beadeb5..407aa5d 100644 --- a/src/backend/storage/ipc/latch.c diff --git a/src/wasix/postmaster/postgres/series b/src/wasix/postmaster/postgres/series index 80d612b4f..31e0b5d6c 100644 --- a/src/wasix/postmaster/postgres/series +++ b/src/wasix/postmaster/postgres/series @@ -5,10 +5,5 @@ src/wasix/postmaster/postgres/patches/0003-wasix-libpq-static-encoding-shim.patc src/wasix/postmaster/postgres/patches/0004-wasix-core-execbackend-initdb-runtime.patch src/wasix/postmaster/postgres/patches/0006-wasix-retry-proc-join-on-eintr.patch src/wasix/postmaster/postgres/patches/0008-wasix-packed-atomic-latch-state.patch -src/third-party/postgres/patches/wasix/0014-oliphaunt-wasix-speed-up-hash-bytes-unaligned-loads.patch -src/third-party/postgres/patches/wasix/0015-oliphaunt-wasix-add-top-xid-current-transaction-fast-path.patch -src/third-party/postgres/patches/wasix/0016-oliphaunt-wasix-add-btree-int4-compare-fast-path.patch src/third-party/postgres/patches/wasix/0018-oliphaunt-wasix-avoid-pg-dump-executequery-lto-collision.patch -src/third-party/postgres/patches/wasix/0024-oliphaunt-wasix-add-like-literal-substring-fast-path.patch -src/third-party/postgres/patches/wasix/0026-oliphaunt-wasix-add-first-int4-leaf-compare-fast-path.patch -src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-buffer-strong-random.patch +src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-use-checked-getrandom.patch diff --git a/src/wasix/postmaster/wasmer/README.md b/src/wasix/postmaster/wasmer/README.md index 94a5a07ed..e9c310625 100644 --- a/src/wasix/postmaster/wasmer/README.md +++ b/src/wasix/postmaster/wasmer/README.md @@ -7,8 +7,8 @@ compiler-bearing producer plus a compiler-free product executor. Tracked product inputs are: -- `patches/wasmer/0001-postgres-wasix-blockers.patch`; -- `patches/wasix-libc/0001-postgres-wasix-blockers.patch`; +- `patches/wasmer/series`; +- `patches/wasix-libc/series`; - the current contract inventory in `capabilities.tsv`; - focused capability fixtures under `probes/`; - preparation, build, verification, and qualification entrypoints under `bin/`. @@ -25,6 +25,18 @@ identities, Cargo.lock, sysroot manifests, compiler/executor features, host ABI, Rust and LLVM versions, artifact ABI, runtime ABI, CPU policy, and binary hashes. Runtime selection never falls back to a stock or `PATH` Wasmer. +The ordered `series` files, not directory globs, select patches. Their digest +includes the manifest, member names and contents; editing or reordering a patch +invalidates receipts. The product executor lives in `../executor`. +The compiler and verifier use the nonvolatile-memory profile. Compiler policy +changes require rebuilt carriers with matching identities. + +The libc series separates mapping, file, socket, process, exception and resource +contracts. Its `sigsetjmp` fix evaluates the buffer expression once in the live +caller and keeps helper declarations limited to exception-enabled builds. +The host C/C++ probes prove those language contracts, not WASIX signal delivery; +the runtime capability probes remain required for that claim. + The product executor accepts only an independently verified sealed carrier. It does not expose the general Wasmer package, registry, network, or compilation command graph. AOT production uses an explicit generic CPU baseline; native CPU diff --git a/src/wasix/postmaster/wasmer/bin/build-patched-wasix-libc-sysroot.sh b/src/wasix/postmaster/wasmer/bin/build-patched-wasix-libc-sysroot.sh index db1d4429b..2cb04ba2e 100755 --- a/src/wasix/postmaster/wasmer/bin/build-patched-wasix-libc-sysroot.sh +++ b/src/wasix/postmaster/wasmer/bin/build-patched-wasix-libc-sysroot.sh @@ -12,7 +12,7 @@ WASIX_LIBC_ROOT="${WASIX_LIBC_ROOT:-$UPSTREAM_WORK_ROOT/wasix-libc}" OUTPUT_PREFIX="${OUTPUT_PREFIX:-$UPSTREAM_WORK_ROOT/build/patched-wasixcc-sysroot}" BUILD_LOG="${BUILD_LOG:-$UPSTREAM_WORK_ROOT/reports/wasix-libc-build.log}" WASIX_LIBC_VARIANTS="${WASIX_LIBC_VARIANTS:-sysroot-ehpic sysroot-exnref-ehpic}" -WASIX_LIBC_PATCH="$UPSTREAM_SOURCE_ROOT/patches/wasix-libc/0001-postgres-wasix-blockers.patch" +WASIX_LIBC_PATCH="$UPSTREAM_SOURCE_ROOT/patches/wasix-libc/series" VARIANT_MANIFEST_NAME=".oliphaunt-patched-sysroot.manifest" CARRIER_MANIFEST_NAME=".oliphaunt-patched-sysroots.manifest" @@ -313,10 +313,14 @@ if ! command -v sha256sum >/dev/null 2>&1 && ! command -v shasum >/dev/null 2>&1 fi SOURCE_COMMIT="$(git -C "$WASIX_LIBC_ROOT" rev-parse --verify 'HEAD^{commit}')" || fail "wasix-libc checkout is not a Git worktree: $WASIX_LIBC_ROOT" -if ! git -C "$WASIX_LIBC_ROOT" apply --reverse --check "$WASIX_LIBC_PATCH" >/dev/null 2>&1; then - fail "required wasix-libc patch is not applied cleanly: $WASIX_LIBC_PATCH" -fi -SOURCE_PATCH_SHA256="$(sha256_file "$WASIX_LIBC_PATCH")" +SOURCE_PATCH_SHA256="$(fresh_runtime_patch_hash "$WASIX_LIBC_PATCH")" +# Later series members intentionally modify earlier ones. Reverse --check on +# individual patch files cannot validate that composed result; use the existing +# prepared-tree receipt, which binds the pin, ordered series and complete tree. +fresh_require_prepared_worktree wasix-libc "$WASIX_LIBC_ROOT" \ + "$FRESH_WASIX_LIBC_SOURCE_COMMIT" "$SOURCE_PATCH_SHA256" "" \ + "$UPSTREAM_WORK_ROOT/.prepared/wasix-libc.signature" || + fail "wasix-libc does not match its prepared patch series" SOURCE_WORKTREE_SHA256="$(source_worktree_sha256)" if [ "$PORTABLE_INPUTS" -eq 1 ]; then diff --git a/src/wasix/postmaster/wasmer/bin/build-runtime.sh b/src/wasix/postmaster/wasmer/bin/build-runtime.sh index cc85410ee..504a8c481 100755 --- a/src/wasix/postmaster/wasmer/bin/build-runtime.sh +++ b/src/wasix/postmaster/wasmer/bin/build-runtime.sh @@ -19,9 +19,10 @@ esac UPSTREAM_WORK_ROOT="${UPSTREAM_WORK_ROOT:-$FRESH_WORK_ROOT/runtime}" WASMER_ROOT="${WASMER_ROOT:-$UPSTREAM_WORK_ROOT/wasmer}" +WASIX_LIBC_ROOT="${WASIX_LIBC_ROOT:-$UPSTREAM_WORK_ROOT/wasix-libc}" LLVM_MAJOR=22 -WASMER_PATCH="$FRESH_ROOT/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch" -WASIX_LIBC_PATCH="$FRESH_ROOT/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch" +WASMER_PATCH="$FRESH_ROOT/wasmer/patches/wasmer/series" +WASIX_LIBC_PATCH="$FRESH_ROOT/wasmer/patches/wasix-libc/series" WASMER_BUILD_RECEIPT_OUT="${WASMER_BUILD_RECEIPT_OUT:-$FRESH_WASMER_BUILD_RECEIPT}" POSTMASTER_EXECUTOR_BUILD_RECEIPT_OUT="${POSTMASTER_EXECUTOR_BUILD_RECEIPT_OUT:-$FRESH_POSTMASTER_EXECUTOR_BUILD_RECEIPT}" WASMER_TARGET_DIR="$WASMER_ROOT/target" @@ -107,6 +108,7 @@ LLVM_SYS_221_PREFIX="$(find_llvm_prefix)" export LLVM_SYS_221_PREFIX UPSTREAM_WORK_ROOT="$UPSTREAM_WORK_ROOT" \ + WASMER_ROOT="$WASMER_ROOT" WASIX_LIBC_ROOT="$WASIX_LIBC_ROOT" \ "$FRESH_ROOT/wasmer/bin/prepare-upstream-checkouts.sh" [ -f "$WASMER_ROOT/lib/cli/Cargo.toml" ] || { printf 'missing prepared Wasmer checkout: %s\n' "$WASMER_ROOT" >&2 @@ -215,18 +217,22 @@ cargo build \ --features "$FRESH_POSTMASTER_COMPILER_FEATURES" if [ "$PORTABLE_INPUTS" -eq 1 ]; then UPSTREAM_WORK_ROOT="$UPSTREAM_WORK_ROOT" \ + WASIX_LIBC_ROOT="$WASIX_LIBC_ROOT" \ "$FRESH_ROOT/wasmer/bin/build-patched-wasix-libc-sysroot.sh" \ --no-build --portable-inputs elif [ -f "$WASIXCC_SYSROOT_PREFIX/.oliphaunt-patched-sysroots.manifest" ] && \ UPSTREAM_WORK_ROOT="$UPSTREAM_WORK_ROOT" \ + WASIX_LIBC_ROOT="$WASIX_LIBC_ROOT" \ "$FRESH_ROOT/wasmer/bin/build-patched-wasix-libc-sysroot.sh" --no-build; then : else UPSTREAM_WORK_ROOT="$UPSTREAM_WORK_ROOT" \ + WASIX_LIBC_ROOT="$WASIX_LIBC_ROOT" \ "$FRESH_ROOT/wasmer/bin/build-patched-wasix-libc-sysroot.sh" fi UPSTREAM_WORK_ROOT="$UPSTREAM_WORK_ROOT" \ + WASMER_ROOT="$WASMER_ROOT" WASIX_LIBC_ROOT="$WASIX_LIBC_ROOT" \ "$FRESH_ROOT/wasmer/bin/prepare-upstream-checkouts.sh" wasmer_bin="$WASMER_TARGET_DIR/release/wasmer" @@ -270,7 +276,7 @@ trap 'rm -f "$temporary_manifest"' EXIT printf 'wasmer_napi_commit=%s\n' "$(git -C "$WASMER_ROOT/lib/napi" rev-parse HEAD)" printf 'wasmer_test_files_commit=%s\n' "$(git -C "$WASMER_ROOT/wasmer-test-files" rev-parse HEAD)" printf 'wasmer_spec_commit=%s\n' "$(git -C "$WASMER_ROOT/tests/wast/spec" rev-parse HEAD)" - printf 'wasmer_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$WASMER_PATCH")" + printf 'wasmer_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$WASMER_PATCH")" printf 'wasmer_prepared_signature_sha256=%s\n' "$(fresh_wasmer_bin_hash "$prepared_signature")" printf 'wasmer_cargo_lock_sha256=%s\n' "$(fresh_wasmer_bin_hash "$WASMER_ROOT/Cargo.lock")" printf 'wasmer_binary_sha256=%s\n' "$(fresh_wasmer_bin_hash "$wasmer_bin")" @@ -280,7 +286,7 @@ trap 'rm -f "$temporary_manifest"' EXIT printf 'runtime_abi_id=%s\n' "$runtime_abi_id" printf 'artifact_abi_version=%s\n' "$FRESH_WASMER_ARTIFACT_ABI_VERSION" printf 'wasix_libc_source_commit=%s\n' "$(git -C "$UPSTREAM_WORK_ROOT/wasix-libc" rev-parse HEAD)" - printf 'wasix_libc_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$WASIX_LIBC_PATCH")" + printf 'wasix_libc_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$WASIX_LIBC_PATCH")" printf 'wasix_libc_prepared_signature_sha256=%s\n' "$(fresh_wasmer_bin_hash "$libc_prepared_signature")" printf 'sysroot_carrier_manifest_sha256=%s\n' "$(fresh_wasmer_bin_hash "$carrier_manifest")" printf 'sysroot_variant=%s\n' "$WASIXCC_SYSROOT_VARIANT" @@ -308,7 +314,7 @@ trap 'rm -f "$temporary_executor_receipt"' EXIT printf 'build_recipe_sha256=%s\n' "$(fresh_runtime_build_recipe_sha256)" printf 'wasmer_build_receipt_sha256=%s\n' "$(fresh_wasmer_bin_hash "$WASMER_BUILD_RECEIPT_OUT")" printf 'wasmer_source_commit=%s\n' "$(git -C "$WASMER_ROOT" rev-parse HEAD)" - printf 'wasmer_patch_sha256=%s\n' "$(fresh_wasmer_bin_hash "$WASMER_PATCH")" + printf 'wasmer_patch_sha256=%s\n' "$(fresh_runtime_patch_hash "$WASMER_PATCH")" printf 'wasmer_prepared_signature_sha256=%s\n' "$(fresh_wasmer_bin_hash "$prepared_signature")" printf 'wasmer_cargo_lock_sha256=%s\n' "$(fresh_wasmer_bin_hash "$WASMER_ROOT/Cargo.lock")" printf 'runtime_abi_id=%s\n' "$runtime_abi_id" diff --git a/src/wasix/postmaster/wasmer/bin/prepare-upstream-checkouts.sh b/src/wasix/postmaster/wasmer/bin/prepare-upstream-checkouts.sh index 2db23e39b..7e1a56bd6 100755 --- a/src/wasix/postmaster/wasmer/bin/prepare-upstream-checkouts.sh +++ b/src/wasix/postmaster/wasmer/bin/prepare-upstream-checkouts.sh @@ -190,7 +190,7 @@ prepare_patched_worktree() { local extra_signature="${6:-}" local patch_signature="skip" if [ "$SKIP_PATCHES" -ne 1 ]; then - patch_signature="$(sha256_file "$patch")" + patch_signature="$(fresh_runtime_patch_hash "$patch")" fi if [ "$name" = "wasmer" ] && [ "$SKIP_PATCHES" -ne 1 ]; then extra_signature="$extra_signature:$(fresh_executor_source_sha256)" @@ -213,8 +213,7 @@ prepare_patched_worktree() { materialize_gitlink "WebAssembly testsuite" "tests/wast/spec" "$WASMER_SPEC_SOURCE_ROOT" "$WASMER_SPEC_REF" fi if [ "$SKIP_PATCHES" -ne 1 ]; then - git -C "$root" apply --check "$patch" - git -C "$root" apply "$patch" + fresh_apply_patch_series "$root" "$(dirname "$patch")" "$patch" if [ "$name" = "wasmer" ]; then bun "$FRESH_ROOT/executor/prepare-paths.mts" "$root" fi @@ -234,14 +233,14 @@ prepare_patched_worktree \ "$WASMER_SOURCE_ROOT" \ "$WASMER_ROOT" \ "$WASMER_REF" \ - "$UPSTREAM_SOURCE_ROOT/patches/wasmer/0001-postgres-wasix-blockers.patch" \ + "$UPSTREAM_SOURCE_ROOT/patches/wasmer/series" \ "$WASMER_NAPI_REF:$WASMER_TEST_FILES_REF:$WASMER_SPEC_REF" prepare_patched_worktree \ "wasix-libc" \ "$WASIX_LIBC_SOURCE_ROOT" \ "$WASIX_LIBC_ROOT" \ "$WASIX_LIBC_REF" \ - "$UPSTREAM_SOURCE_ROOT/patches/wasix-libc/0001-postgres-wasix-blockers.patch" + "$UPSTREAM_SOURCE_ROOT/patches/wasix-libc/series" printf 'prepared Wasmer worktree: %s\n' "$WASMER_ROOT" printf ' base: %s\n' "$(git -C "$WASMER_ROOT" rev-parse HEAD)" diff --git a/src/wasix/postmaster/wasmer/bin/validate-runtime-capabilities.sh b/src/wasix/postmaster/wasmer/bin/validate-runtime-capabilities.sh index ca5712047..c8187eebd 100755 --- a/src/wasix/postmaster/wasmer/bin/validate-runtime-capabilities.sh +++ b/src/wasix/postmaster/wasmer/bin/validate-runtime-capabilities.sh @@ -302,7 +302,7 @@ validate_exact_sysroot() { local source_commit local source_patch_sha256 local source_worktree_sha256 - local expected_patch="$UPSTREAM_SOURCE_ROOT/patches/wasix-libc/0001-postgres-wasix-blockers.patch" + local expected_patch="$UPSTREAM_SOURCE_ROOT/patches/wasix-libc/series" CARRIER_MANIFEST="$WASIXCC_SYSROOT_PREFIX/$CARRIER_MANIFEST_NAME" PATCHED_SYSROOT_MANIFEST="$WASIXCC_SYSROOT/$VARIANT_MANIFEST_NAME" @@ -405,7 +405,7 @@ validate_exact_sysroot() { source_patch_sha256="$(required_manifest_value "$PATCHED_SYSROOT_MANIFEST" source_patch_sha256)" valid_sha256 "$source_patch_sha256" || fail_sysroot 'source_patch_sha256 is invalid' [ -f "$expected_patch" ] || fail_sysroot "local source patch is missing: $expected_patch" - [ "$source_patch_sha256" = "$(sha256_file "$expected_patch")" ] || + [ "$source_patch_sha256" = "$(fresh_runtime_patch_hash "$expected_patch")" ] || fail_sysroot 'carrier was not built from the current wasix-libc patch' PATCHED_SYSROOT_DOCKER_IMAGE_ID="$(required_manifest_value "$PATCHED_SYSROOT_MANIFEST" docker_image_id)" diff --git a/src/wasix/postmaster/wasmer/capabilities.tsv b/src/wasix/postmaster/wasmer/capabilities.tsv index 2e64be3d2..074089e51 100644 --- a/src/wasix/postmaster/wasmer/capabilities.tsv +++ b/src/wasix/postmaster/wasmer/capabilities.tsv @@ -13,6 +13,7 @@ shared-mapping-lifecycle wasmer generation-safe-owned-registry wasmer:lib/wasix/ cross-instance-latch-order wasmer+postgres packed-atomic-state-and-fences wasmer:lib/compiler-llvm/src/translator/code.rs;project:postgres/patches/0008-wasix-packed-atomic-latch-state.patch final-module:atomic-latch-state;stress:backend-wave Preserve PostgreSQL latch progress across instances sharing one backing supported listener-epoll wasmer+wasix-libc open-file-description-lifecycle wasmer:lib/wasix/src/os/epoll/mod.rs;project:runtime/probes/epoll_ofd_lifecycle_probe.c epoll-ofd-lifecycle Keep listener readiness and duplicated descriptor registration semantics correct supported-linux-gnu child-wait-and-signals wasmer+wasix-libc+postgres retry-and-owned-process-state project:postgres/patches/0006-wasix-retry-proc-join-on-eintr.patch;wasmer:lib/wasix/src/os/task/process.rs exec-wait-signal Preserve child exit, signal, interruption, and cleanup semantics supported +posix-signal-masks wasmer+wasix-libc inherited-success-returning-stub wasix-libc:libc-top-half/musl/src/thread/pthread_sigmask.c;wasix-libc:libc-top-half/musl/src/signal/sigaction.c;wasmer:lib/wasix/src/os/task/thread.rs source-audit:mask-backend-stub Blocking masks, handler sa_mask and saved-mask restoration are not implemented; child-wait support does not imply these semantics unsupported aot-artifact-identity wasmer+oliphaunt receipt-bound-generic-baseline wasmer:lib/compiler-llvm/src/compiler.rs;wasmer:lib/oliphaunt-wasix-postmaster-executor/src/bin/compiler.rs;project:lib/verify-sealed-carrier.mts unit:sealed-manifest-cpu-policy Reject AOT from another target, CPU policy, runtime ABI, artifact ABI, or guest module supported directory-durability wasmer+postgres exact-open-directory-fsync wasmer:lib/virtual-fs/src/host_fs.rs;project:runtime/probes/directory_fsync_probe.c directory-fsync Apply PostgreSQL directory durability barriers to the exact opened directory supported-linux-gnu regression-subset postgres product-lifecycle-harness project:bin/run-wasix-regress-subset.sh pg-regress-subset Run PostgreSQL regression cases through the real postmaster and libpq supported diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0001-mmap-and-allocator-exec-ownership.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0001-mmap-and-allocator-exec-ownership.patch new file mode 100644 index 000000000..bfc92c239 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0001-mmap-and-allocator-exec-ownership.patch @@ -0,0 +1,800 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 1/8] libc: preserve mmap and allocator state across exec + +Keep mapped-file writeback, fixed/shared mappings and allocator/sbrk ownership +changes together. PostgreSQL uses shared mappings across EXEC_BACKEND children; +allocator reset/growth must not overwrite those mappings. The mmap writeback +probe accompanies the implementation. This is preserved integration behavior, +not a claim that allocator or mapping redesign has been newly qualified. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/dlmalloc/src/dlmalloc.c b/dlmalloc/src/dlmalloc.c +index 331536b..e7063c5 100644 +--- a/dlmalloc/src/dlmalloc.c ++++ b/dlmalloc/src/dlmalloc.c +@@ -12,6 +12,16 @@ + // WebAssembly doesn't support shrinking linear memory. + #define MORECORE_CANNOT_TRIM 1 + ++// WASIX runtime facilities such as MAP_SHARED and dynamic linking also grow ++// linear memory. Consecutive MORECORE calls therefore are not contractually ++// contiguous even though each individual sbrk segment is exclusive. ++#define MORECORE_CONTIGUOUS 0 ++ ++// WASIX sbrk returns an exclusive segment of exactly the requested size. ++// dlmalloc must not infer that size from a later sbrk(0): another thread or a ++// direct caller can advance the logical break between those two calls. ++#define MORECORE_RETURNS_EXACT_SIZE 1 ++ + // Disable sanity checks to reduce code size. + #define ABORT __builtin_unreachable() + +diff --git a/dlmalloc/src/malloc.c b/dlmalloc/src/malloc.c +index 96e6cea..e466124 100644 +--- a/dlmalloc/src/malloc.c ++++ b/dlmalloc/src/malloc.c +@@ -4172,6 +4172,15 @@ static void* sys_alloc(mstate m, size_t nb) { + if (HAVE_MORECORE && tbase == CMFAIL) { /* Try noncontiguous MORECORE */ + if (asize < HALF_MAX_SIZE_T) { + char* br = CMFAIL; ++#if defined(MORECORE_RETURNS_EXACT_SIZE) && MORECORE_RETURNS_EXACT_SIZE ++ ACQUIRE_MALLOC_GLOBAL_LOCK(); ++ br = (char*)(CALL_MORECORE(asize)); ++ RELEASE_MALLOC_GLOBAL_LOCK(); ++ if (br != CMFAIL) { ++ tbase = br; ++ tsize = asize; ++ } ++#else + char* end = CMFAIL; + ACQUIRE_MALLOC_GLOBAL_LOCK(); + br = (char*)(CALL_MORECORE(asize)); +@@ -4184,6 +4193,7 @@ static void* sys_alloc(mstate m, size_t nb) { + tsize = ssize; + } + } ++#endif + } + } + +diff --git a/emmalloc/emmalloc.c b/emmalloc/emmalloc.c +index c98e42e..c76a2eb 100644 +--- a/emmalloc/emmalloc.c ++++ b/emmalloc/emmalloc.c +@@ -54,9 +54,6 @@ + #include + #endif + +-// Defind by the linker to have the address of the start of the heap. +-extern unsigned char __heap_base; +- + // Behavior of right shifting a signed integer is compiler implementation defined. + static_assert((((int32_t)0x80000000U) >> 31) == -1, "This malloc implementation requires that right-shifting a signed integer produces a sign-extending (arithmetic) shift!"); + +@@ -110,7 +107,7 @@ typedef struct RootRegion + uint8_t* endPtr; + } RootRegion; + +-#if defined(__EMSCRIPTEN_PTHREADS__) ++#if defined(__EMSCRIPTEN_PTHREADS__) || defined(_REENTRANT) + // In multithreaded builds, use a simple global spinlock strategy to acquire/release access to the memory allocator. + static volatile uint8_t multithreadingLock = 0; + #define MALLOC_ACQUIRE() while(__sync_lock_test_and_set(&multithreadingLock, 1)) { while(multithreadingLock) { /*nop*/ } } +@@ -539,38 +536,24 @@ static bool claim_more_memory(size_t numBytes) + validate_memory_regions(); + #endif + +- uint8_t *startPtr; +- uint8_t *endPtr; +- do { +- // If this is the first time we're called, see if we can use +- // the initial heap memory set up by wasm-ld. +- if (!listOfAllRegions) { +- unsigned char *heap_end = sbrk(0); +- if (numBytes <= (size_t)(heap_end - &__heap_base)) { +- startPtr = &__heap_base; +- endPtr = heap_end; +- break; +- } +- } +- +- // Round numBytes up to the nearest page size. +- numBytes = (numBytes + (PAGE_SIZE-1)) & -PAGE_SIZE; ++ // Runtime facilities can grow linear memory without transferring ownership ++ // to this allocator. Claim only the exact interval returned by positive ++ // sbrk; sbrk(0) is an address-space frontier, not an ownership boundary. ++ numBytes = (numBytes + (PAGE_SIZE-1)) & -PAGE_SIZE; + +- // Claim memory via sbrk +- startPtr = (uint8_t*)sbrk(numBytes); +- if ((intptr_t)startPtr == -1) +- { ++ uint8_t *startPtr = (uint8_t*)sbrk(numBytes); ++ if ((intptr_t)startPtr == -1) ++ { + #ifdef EMMALLOC_VERBOSE +- MAIN_THREAD_ASYNC_EM_ASM(console.error('claim_more_memory: sbrk failed!')); ++ MAIN_THREAD_ASYNC_EM_ASM(console.error('claim_more_memory: sbrk failed!')); + #endif +- return false; +- } ++ return false; ++ } + #ifdef EMMALLOC_VERBOSE +- MAIN_THREAD_ASYNC_EM_ASM(console.log('claim_more_memory: claimed 0x' + ($0>>>0).toString(16) + ' - 0x' + ($1>>>0).toString(16) + ' (' + ($2>>>0) + ' bytes) via sbrk()'), startPtr, startPtr + numBytes, numBytes); ++ MAIN_THREAD_ASYNC_EM_ASM(console.log('claim_more_memory: claimed 0x' + ($0>>>0).toString(16) + ' - 0x' + ($1>>>0).toString(16) + ' (' + ($2>>>0) + ' bytes) via sbrk()'), startPtr, startPtr + numBytes, numBytes); + #endif +- assert(HAS_ALIGNMENT(startPtr, alignof(size_t))); +- endPtr = startPtr + numBytes; +- } while (0); ++ assert(HAS_ALIGNMENT(startPtr, alignof(size_t))); ++ uint8_t *endPtr = startPtr + numBytes; + + // Create a sentinel region at the end of the new heap block + Region *endSentinelRegion = (Region*)(endPtr - sizeof(Region)); +diff --git a/libc-bottom-half/mman/mman.c b/libc-bottom-half/mman/mman.c +index 7e26562..a156655 100644 +--- a/libc-bottom-half/mman/mman.c ++++ b/libc-bottom-half/mman/mman.c +@@ -13,25 +13,188 @@ + #include + #include + #include ++#include ++#include + #include + #include ++#include ++ ++#define RUNTIME_MMAP_ALIGN ((size_t) 65536) ++ ++int32_t __imported_wasix_32v1_mem_mmap(int32_t arg0, int32_t arg1, ++ int32_t arg2, int32_t arg3, ++ int32_t arg4, int64_t arg5, ++ int32_t arg6) __attribute__(( ++ __import_module__("wasix_32v1"), ++ __import_name__("mem_mmap") ++)); ++ ++int32_t __imported_wasix_32v1_mem_munmap(int32_t arg0, int32_t arg1) __attribute__(( ++ __import_module__("wasix_32v1"), ++ __import_name__("mem_munmap") ++)); ++ ++int32_t __imported_wasix_32v1_mem_msync(int32_t arg0, int32_t arg1, ++ int32_t arg2) __attribute__(( ++ __import_module__("wasix_32v1"), ++ __import_name__("mem_msync") ++)); + + struct map { ++ void *addr; + int prot; + int flags; + off_t offset; + size_t length; + int fd; ++ struct map *next; + }; + ++/* Runtime-backed mappings live only in Wasmer's authoritative interval ++ registry. This guest list is solely for malloc-backed legacy emulation. */ ++static struct map *legacy_maps; ++static pthread_mutex_t maps_lock = PTHREAD_MUTEX_INITIALIZER; ++ ++static __wasi_errno_t runtime_mem_mmap(void *addr, size_t length, int prot, ++ int flags, int fd, off_t offset, ++ void **ret_addr) { ++ uintptr_t out = 0; ++ int32_t ret = __imported_wasix_32v1_mem_mmap( ++ (int32_t) (uintptr_t) addr, ++ (int32_t) length, ++ (int32_t) prot, ++ (int32_t) flags, ++ (int32_t) fd, ++ (int64_t) offset, ++ (int32_t) (uintptr_t) &out); ++ *ret_addr = (void *) out; ++ return (__wasi_errno_t) ret; ++} ++ ++static __wasi_errno_t runtime_mem_munmap(void *addr, size_t length) { ++ int32_t ret = __imported_wasix_32v1_mem_munmap( ++ (int32_t) (uintptr_t) addr, ++ (int32_t) length); ++ return (__wasi_errno_t) ret; ++} ++ ++static __wasi_errno_t runtime_mem_msync(void *addr, size_t length, int flags) { ++ int32_t ret = __imported_wasix_32v1_mem_msync( ++ (int32_t) (uintptr_t) addr, ++ (int32_t) length, ++ (int32_t) flags); ++ return (__wasi_errno_t) ret; ++} ++ ++static int range_end(void *addr, size_t length, uintptr_t *end) { ++ return length != 0 && ++ !__builtin_add_overflow((uintptr_t) addr, length, end); ++} ++ ++static struct map *find_overlapping_legacy_map_locked( ++ void *addr, size_t length, struct map ***link_out) { ++ uintptr_t start = (uintptr_t) addr; ++ uintptr_t end; ++ struct map **link = &legacy_maps; ++ ++ if (!range_end(addr, length, &end)) { ++ return NULL; ++ } ++ ++ while (*link != NULL) { ++ uintptr_t map_start = (uintptr_t) (*link)->addr; ++ uintptr_t map_end; ++ ++ if (!range_end((*link)->addr, (*link)->length, &map_end)) { ++ /* Registry entries are validated before insertion. */ ++ abort(); ++ } ++ if (start < map_end && map_start < end) { ++ if (link_out) { ++ *link_out = link; ++ } ++ return *link; ++ } ++ link = &(*link)->next; ++ } ++ ++ return NULL; ++} ++ ++static int map_contains(const struct map *map, void *addr, size_t length) { ++ uintptr_t start = (uintptr_t) addr; ++ uintptr_t end; ++ uintptr_t map_start = (uintptr_t) map->addr; ++ uintptr_t map_end; ++ ++ return range_end(addr, length, &end) && ++ range_end(map->addr, map->length, &map_end) && ++ start >= map_start && end <= map_end; ++} ++ ++static int map_is_exact(const struct map *map, void *addr, size_t length) { ++ return map->addr == addr && map->length == length; ++} ++ ++static void remove_map_locked(struct map **link) { ++ struct map *entry = *link; ++ ++ *link = entry->next; ++} ++ ++static void *runtime_mmap(void *addr, size_t length, int prot, int flags, ++ int fd, off_t offset) { ++ void *target = addr; ++ void *mapped = NULL; ++ int fixed = (flags & MAP_FIXED) != 0; ++ __wasi_errno_t error; ++ ++ if (!fixed) { ++ /* ++ * Address selection belongs to the runtime. It atomically claims new ++ * WebAssembly pages and returns their old end as the selected address. ++ * Allocators claim only exact positive-sbrk return intervals. A ++ * non-NULL POSIX hint is deliberately ignored rather than converted ++ * into an unsafe fixed replacement. ++ */ ++ target = NULL; ++ } else if (((uintptr_t) target & (RUNTIME_MMAP_ALIGN - 1)) != 0) { ++ errno = EINVAL; ++ return MAP_FAILED; ++ } ++ ++ /* The host remap is currently read/write. Reject weaker or broader ++ protection contracts instead of silently violating them. */ ++ if (prot != (PROT_READ | PROT_WRITE)) { ++ errno = EINVAL; ++ return MAP_FAILED; ++ } ++ ++ error = runtime_mem_mmap(target, length, prot, flags, fd, offset, &mapped); ++ if (error != __WASI_ERRNO_SUCCESS) { ++ errno = error; ++ return MAP_FAILED; ++ } ++ if (mapped == NULL || (fixed && mapped != target)) { ++ if (mapped != NULL) { ++ runtime_mem_munmap(mapped, length); ++ } ++ errno = EINVAL; ++ return MAP_FAILED; ++ } ++ ++ return mapped; ++} ++ + void *mmap(void *addr, size_t length, int prot, int flags, + int fd, off_t offset) { ++ int runtime_candidate = ++ (flags & MAP_TYPE) == MAP_SHARED && (flags & MAP_ANON) == 0; ++ + // Check for unsupported flags. +- if ((flags & (MAP_PRIVATE | MAP_SHARED)) == 0 || +- (flags & MAP_FIXED) != 0 || +-#ifdef MAP_SHARED_VALIDATE +- (flags & MAP_SHARED_VALIDATE) == MAP_SHARED_VALIDATE || +-#endif ++ if (((flags & MAP_TYPE) != MAP_PRIVATE && ++ (flags & MAP_TYPE) != MAP_SHARED) || ++ ((flags & MAP_FIXED) != 0 && !runtime_candidate) || + #ifdef MAP_NORESERVE + (flags & MAP_NORESERVE) != 0 || + #endif +@@ -67,6 +230,10 @@ void *mmap(void *addr, size_t length, int prot, int flags, + return MAP_FAILED; + } + ++ if (runtime_candidate) { ++ return runtime_mmap(addr, length, prot, flags, fd, offset); ++ } ++ + // Check for integer overflow. + size_t buf_len = 0; + if (__builtin_add_overflow(length, sizeof(struct map), &buf_len)) { +@@ -97,7 +264,7 @@ void *mmap(void *addr, size_t length, int prot, int flags, + errno = EINVAL; + free(map); + +- return NULL; ++ return MAP_FAILED; + } + + map->fd = new_fd; +@@ -108,6 +275,8 @@ void *mmap(void *addr, size_t length, int prot, int flags, + if (nread < 0) { + if (errno == EINTR) + continue; ++ close(new_fd); ++ free(map); + return MAP_FAILED; + } + if (nread == 0) +@@ -121,28 +290,94 @@ void *mmap(void *addr, size_t length, int prot, int flags, + memset(addr, 0, length); + } + ++ map->addr = addr; ++ pthread_mutex_lock(&maps_lock); ++ if (find_overlapping_legacy_map_locked(addr, map->length, NULL) != NULL) { ++ pthread_mutex_unlock(&maps_lock); ++ if (map->fd >= 0) { ++ close(map->fd); ++ } ++ free(map); ++ errno = EINVAL; ++ return MAP_FAILED; ++ } ++ map->next = legacy_maps; ++ legacy_maps = map; ++ pthread_mutex_unlock(&maps_lock); ++ + return addr; + } + ++static int legacy_msync(struct map *map, void *addr, size_t length) { ++ uintptr_t delta = (uintptr_t) addr - (uintptr_t) map->addr; ++ off_t map_offset; ++ ++ if (__builtin_add_overflow(map->offset, (off_t) delta, &map_offset)) { ++ errno = EOVERFLOW; ++ return -1; ++ } ++ ++ if ((map->flags & MAP_SHARED) != 0 && ++ (map->flags & MAP_ANON) == 0 && ++ (map->prot & PROT_WRITE) != 0) { ++ char *body = (char *) addr; ++ ++ while (length > 0) { ++ const ssize_t nwrite = pwrite(map->fd, body, length, map_offset); ++ ++ if (nwrite > 0) { ++ length -= (size_t) nwrite; ++ map_offset += (size_t) nwrite; ++ body += (size_t) nwrite; ++ } else if (errno == EINTR) { ++ continue; ++ } else { ++ return -1; ++ } ++ } ++ } ++ ++ return 0; ++} ++ + int munmap(void *addr, size_t length) { +- struct map *map = (struct map *)addr - 1; ++ struct map **link = NULL; ++ struct map *map; ++ __wasi_errno_t runtime_error = runtime_mem_munmap(addr, length); + +- // We don't support partial munmapping. +- if (map->length != length) { ++ if (runtime_error == __WASI_ERRNO_SUCCESS) { ++ return 0; ++ } ++ if (runtime_error != __WASI_ERRNO_NOENT) { ++ errno = runtime_error; ++ return -1; ++ } ++ ++ pthread_mutex_lock(&maps_lock); ++ map = find_overlapping_legacy_map_locked(addr, length, &link); ++ if (map == NULL || !map_is_exact(map, addr, length)) { ++ pthread_mutex_unlock(&maps_lock); + errno = EINVAL; + return -1; + } + + // Write the data back to the backing file and close + // the file handle +- if (map->fd > 0) { +- if ((map->prot & PROT_WRITE) != 0) { +- msync(addr, length, MS_SYNC); ++ if (map->fd >= 0) { ++ if ((map->flags & MAP_SHARED) != 0 && ++ (map->prot & PROT_WRITE) != 0) { ++ if (legacy_msync(map, addr, length) < 0) { ++ pthread_mutex_unlock(&maps_lock); ++ return -1; ++ } + } + + close(map->fd); + } + ++ remove_map_locked(link); ++ pthread_mutex_unlock(&maps_lock); ++ + // Release the memory. + free(map); + +@@ -151,39 +386,38 @@ int munmap(void *addr, size_t length) { + } + + int msync (void *addr, size_t length, int flags) { +- struct map *map = (struct map *)addr - 1; +- size_t map_flags = map->flags; +- off_t map_offset = map->offset; +- size_t map_length = map->length; +- int fd = map->fd; ++ struct map *map; ++ __wasi_errno_t runtime_error; + +- if (length > map_length) { ++ if ((flags & ~(MS_ASYNC | MS_INVALIDATE | MS_SYNC)) != 0 || ++ ((flags & MS_ASYNC) != 0 && (flags & MS_SYNC) != 0)) { + errno = EINVAL; + return -1; + } + +- if ((map->prot & PROT_WRITE) != 0) { +- errno = EINVAL; ++ runtime_error = runtime_mem_msync(addr, length, flags); ++ if (runtime_error == __WASI_ERRNO_SUCCESS) { ++ return 0; ++ } ++ if (runtime_error != __WASI_ERRNO_NOENT) { ++ errno = runtime_error; + return -1; + } + +- if ((map_flags & MAP_ANON) == 0) { +- char *body = (char *)addr; +- +- while (length > 0) { +- const ssize_t nwrite = pwrite(fd, body, length, map_offset); +- +- if (nwrite > 0) { +- length -= (size_t)nwrite; +- map_offset += (size_t)nwrite; +- body += (size_t)nwrite; +- } else if (errno == EINTR) { +- continue; +- } else { +- return -1; +- } +- } ++ pthread_mutex_lock(&maps_lock); ++ map = find_overlapping_legacy_map_locked(addr, length, NULL); ++ if (map == NULL) { ++ pthread_mutex_unlock(&maps_lock); ++ errno = ENOMEM; ++ return -1; ++ } ++ if (!map_contains(map, addr, length)) { ++ pthread_mutex_unlock(&maps_lock); ++ errno = EINVAL; ++ return -1; + } + +- return 0; ++ int result = legacy_msync(map, addr, length); ++ pthread_mutex_unlock(&maps_lock); ++ return result; + } +diff --git a/libc-bottom-half/sources/sbrk.c b/libc-bottom-half/sources/sbrk.c +index a26b75e..98cd803 100644 +--- a/libc-bottom-half/sources/sbrk.c ++++ b/libc-bottom-half/sources/sbrk.c +@@ -1,13 +1,18 @@ + #include + #include + #include ++#include + #include <__macro_PAGESIZE.h> + +-// Bare-bones implementation of sbrk. ++/* ++ * Every positive call owns exactly the interval it returns. Runtime facilities ++ * may grow linear memory between calls, so callers must not infer ownership ++ * from a later sbrk(0) or assume that independently acquired intervals are ++ * consecutive. ++ */ + void *sbrk(intptr_t increment) { +- // sbrk(0) returns the current memory size. + if (increment == 0) { +- // The wasm spec doesn't guarantee that memory.grow of 0 always succeeds. ++ // The wasm spec doesn't guarantee that memory.grow of 0 succeeds. + return (void *)(__builtin_wasm_memory_size(0) * PAGESIZE); + } + +@@ -28,5 +33,17 @@ void *sbrk(intptr_t increment) { + return (void *)-1; + } + +- return (void *)(old * PAGESIZE); ++ if (old > UINTPTR_MAX / PAGESIZE) { ++ errno = ENOMEM; ++ return (void *)-1; ++ } ++ old *= PAGESIZE; ++ if ((uintptr_t)increment > UINTPTR_MAX - old) { ++ /* memory.grow has already reserved this terminal interval, but its ++ one-past end is not representable in the guest pointer width. Do ++ not transfer an interval with a wrapping end to the caller. */ ++ errno = ENOMEM; ++ return (void *)-1; ++ } ++ return (void *)old; + } +diff --git a/test/wasix/c_mmap_writeback/CMakeLists.txt b/test/wasix/c_mmap_writeback/CMakeLists.txt +new file mode 100644 +index 0000000..eb29426 +--- /dev/null ++++ b/test/wasix/c_mmap_writeback/CMakeLists.txt +@@ -0,0 +1,4 @@ ++cmake_minimum_required (VERSION 3.5.0) ++project (c_mmap_writeback) ++ ++add_executable(main main.c) +diff --git a/test/wasix/c_mmap_writeback/main.c b/test/wasix/c_mmap_writeback/main.c +new file mode 100644 +index 0000000..3ab7648 +--- /dev/null ++++ b/test/wasix/c_mmap_writeback/main.c +@@ -0,0 +1,194 @@ ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++ ++static int write_full(int fd, const char *buf, size_t len) ++{ ++ while (len > 0) { ++ ssize_t written = write(fd, buf, len); ++ if (written < 0) { ++ if (errno == EINTR) { ++ continue; ++ } ++ return -1; ++ } ++ buf += (size_t)written; ++ len -= (size_t)written; ++ } ++ return 0; ++} ++ ++static int read_full(int fd, char *buf, size_t len) ++{ ++ while (len > 0) { ++ ssize_t got = read(fd, buf, len); ++ if (got < 0) { ++ if (errno == EINTR) { ++ continue; ++ } ++ return -1; ++ } ++ if (got == 0) { ++ errno = EIO; ++ return -1; ++ } ++ buf += (size_t)got; ++ len -= (size_t)got; ++ } ++ return 0; ++} ++ ++static int reset_file(int fd, const char *contents, size_t len) ++{ ++ if (ftruncate(fd, 0) < 0) { ++ return -1; ++ } ++ if (lseek(fd, 0, SEEK_SET) < 0) { ++ return -1; ++ } ++ if (write_full(fd, contents, len) < 0) { ++ return -1; ++ } ++ return lseek(fd, 0, SEEK_SET) < 0 ? -1 : 0; ++} ++ ++static int expect_file(int fd, const char *expected, size_t len) ++{ ++ char buf[32]; ++ ++ if (len > sizeof(buf)) { ++ return -1; ++ } ++ memset(buf, 0, sizeof(buf)); ++ if (lseek(fd, 0, SEEK_SET) < 0) { ++ return -1; ++ } ++ if (read_full(fd, buf, len) < 0) { ++ return -1; ++ } ++ return memcmp(buf, expected, len); ++} ++ ++int main(void) ++{ ++ const char *path = "wasix-libc-mmap-writeback-test.dat"; ++ const char initial[] = "abcdefgh"; ++ int fd; ++ char *map; ++ void *frontier_after_map; ++ unsigned char *morecore_segment; ++ size_t i; ++ ++ unlink(path); ++ fd = open(path, O_RDWR | O_CREAT | O_TRUNC, 0600); ++ if (fd < 0) { ++ perror("open"); ++ return 1; ++ } ++ if (reset_file(fd, initial, sizeof(initial) - 1) < 0) { ++ perror("reset shared"); ++ return 2; ++ } ++ ++ map = mmap(NULL, sizeof(initial) - 1, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0); ++ if (map == MAP_FAILED) { ++ perror("mmap shared"); ++ return 3; ++ } ++ errno = 0; ++ if (munmap(map + 1, 1) == 0 || errno != EINVAL) { ++ fprintf(stderr, "overlapping unaligned munmap escaped runtime ownership\n"); ++ return 18; ++ } ++ errno = 0; ++ if (msync(map + 1, 1, MS_SYNC) == 0 || errno != EINVAL) { ++ fprintf(stderr, "overlapping unaligned msync escaped runtime ownership\n"); ++ return 19; ++ } ++ if (memcmp(map, initial, sizeof(initial) - 1) != 0) { ++ fprintf(stderr, "rejected shared operations mutated the mapping\n"); ++ return 20; ++ } ++ frontier_after_map = sbrk(0); ++ morecore_segment = sbrk(65536); ++ if (morecore_segment == (void *)-1) { ++ perror("sbrk after shared mmap"); ++ return 13; ++ } ++ if (morecore_segment != frontier_after_map || ++ sbrk(0) != morecore_segment + 65536) { ++ fprintf(stderr, "sbrk did not return its exact exclusive interval\n"); ++ return 14; ++ } ++ if ((uintptr_t)morecore_segment < (uintptr_t)map + 65536 && ++ (uintptr_t)map < (uintptr_t)morecore_segment + 65536) { ++ fprintf(stderr, "MORECORE overlapped the runtime-owned mmap range\n"); ++ return 15; ++ } ++ memset(morecore_segment, 0x5a, 65536); ++ memcpy(map, "WXYZ", 4); ++ for (i = 1; i <= 256; ++i) { ++ unsigned char *allocation = malloc(i * 17); ++ if (!allocation) { ++ fprintf(stderr, "malloc failed after nonconsecutive MORECORE\n"); ++ return 16; ++ } ++ memset(allocation, (int)(i & 0xff), i * 17); ++ free(allocation); ++ } ++ if (morecore_segment[0] != 0x5a || morecore_segment[65535] != 0x5a || ++ memcmp(map, "WXYZ", 4) != 0) { ++ fprintf(stderr, "allocator or mmap sentinel was corrupted\n"); ++ return 17; ++ } ++ if (msync(map, sizeof(initial) - 1, MS_SYNC) < 0) { ++ perror("msync shared"); ++ return 4; ++ } ++ if (expect_file(fd, "WXYZefgh", sizeof(initial) - 1) != 0) { ++ fprintf(stderr, "MAP_SHARED changes were not written back\n"); ++ return 5; ++ } ++ if (munmap(map, sizeof(initial) - 1) < 0) { ++ perror("munmap shared"); ++ return 6; ++ } ++ errno = 0; ++ map = mmap(NULL, sizeof(initial) - 1, PROT_READ, MAP_SHARED, fd, 0); ++ if (map != MAP_FAILED || errno != EINVAL) { ++ fprintf(stderr, "runtime mapping accepted an unenforced protection contract\n"); ++ return 21; ++ } ++ ++ if (reset_file(fd, initial, sizeof(initial) - 1) < 0) { ++ perror("reset private"); ++ return 7; ++ } ++ map = mmap(NULL, sizeof(initial) - 1, PROT_READ | PROT_WRITE, MAP_PRIVATE, fd, 0); ++ if (map == MAP_FAILED) { ++ perror("mmap private"); ++ return 8; ++ } ++ memcpy(map, "PRIV", 4); ++ if (msync(map, sizeof(initial) - 1, MS_SYNC) < 0) { ++ perror("msync private"); ++ return 9; ++ } ++ if (munmap(map, sizeof(initial) - 1) < 0) { ++ perror("munmap private"); ++ return 10; ++ } ++ if (expect_file(fd, initial, sizeof(initial) - 1) != 0) { ++ fprintf(stderr, "MAP_PRIVATE changes were written back\n"); ++ return 11; ++ } ++ ++ close(fd); ++ unlink(path); ++ return 0; ++} +diff --git a/test/wasix/c_mmap_writeback/test.sh b/test/wasix/c_mmap_writeback/test.sh +new file mode 100755 +index 0000000..3c120c4 +--- /dev/null ++++ b/test/wasix/c_mmap_writeback/test.sh +@@ -0,0 +1,10 @@ ++#!/bin/bash ++ ++wasmer run --verbose --enable-all ./main ++RESULT=$? ++if [ "$RESULT" != "0" ]; then ++ echo "Test failed: different exit code ($RESULT vs. 0)" > /dev/stderr ++ exit 1 ++fi ++ ++echo "c_mmap_writeback test passed" diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch deleted file mode 100644 index bbf7ad23c..000000000 --- a/src/wasix/postmaster/wasmer/patches/wasix-libc/0001-postgres-wasix-blockers.patch +++ /dev/null @@ -1,1793 +0,0 @@ -diff --git a/dlmalloc/src/dlmalloc.c b/dlmalloc/src/dlmalloc.c -index 331536b..e7063c5 100644 ---- a/dlmalloc/src/dlmalloc.c -+++ b/dlmalloc/src/dlmalloc.c -@@ -12,6 +12,16 @@ - // WebAssembly doesn't support shrinking linear memory. - #define MORECORE_CANNOT_TRIM 1 - -+// WASIX runtime facilities such as MAP_SHARED and dynamic linking also grow -+// linear memory. Consecutive MORECORE calls therefore are not contractually -+// contiguous even though each individual sbrk segment is exclusive. -+#define MORECORE_CONTIGUOUS 0 -+ -+// WASIX sbrk returns an exclusive segment of exactly the requested size. -+// dlmalloc must not infer that size from a later sbrk(0): another thread or a -+// direct caller can advance the logical break between those two calls. -+#define MORECORE_RETURNS_EXACT_SIZE 1 -+ - // Disable sanity checks to reduce code size. - #define ABORT __builtin_unreachable() - -diff --git a/dlmalloc/src/malloc.c b/dlmalloc/src/malloc.c -index 96e6cea..e466124 100644 ---- a/dlmalloc/src/malloc.c -+++ b/dlmalloc/src/malloc.c -@@ -4172,6 +4172,15 @@ static void* sys_alloc(mstate m, size_t nb) { - if (HAVE_MORECORE && tbase == CMFAIL) { /* Try noncontiguous MORECORE */ - if (asize < HALF_MAX_SIZE_T) { - char* br = CMFAIL; -+#if defined(MORECORE_RETURNS_EXACT_SIZE) && MORECORE_RETURNS_EXACT_SIZE -+ ACQUIRE_MALLOC_GLOBAL_LOCK(); -+ br = (char*)(CALL_MORECORE(asize)); -+ RELEASE_MALLOC_GLOBAL_LOCK(); -+ if (br != CMFAIL) { -+ tbase = br; -+ tsize = asize; -+ } -+#else - char* end = CMFAIL; - ACQUIRE_MALLOC_GLOBAL_LOCK(); - br = (char*)(CALL_MORECORE(asize)); -@@ -4184,6 +4193,7 @@ static void* sys_alloc(mstate m, size_t nb) { - tsize = ssize; - } - } -+#endif - } - } - -diff --git a/emmalloc/emmalloc.c b/emmalloc/emmalloc.c -index c98e42e..c76a2eb 100644 ---- a/emmalloc/emmalloc.c -+++ b/emmalloc/emmalloc.c -@@ -54,9 +54,6 @@ - #include - #endif - --// Defind by the linker to have the address of the start of the heap. --extern unsigned char __heap_base; -- - // Behavior of right shifting a signed integer is compiler implementation defined. - static_assert((((int32_t)0x80000000U) >> 31) == -1, "This malloc implementation requires that right-shifting a signed integer produces a sign-extending (arithmetic) shift!"); - -@@ -110,7 +107,7 @@ typedef struct RootRegion - uint8_t* endPtr; - } RootRegion; - --#if defined(__EMSCRIPTEN_PTHREADS__) -+#if defined(__EMSCRIPTEN_PTHREADS__) || defined(_REENTRANT) - // In multithreaded builds, use a simple global spinlock strategy to acquire/release access to the memory allocator. - static volatile uint8_t multithreadingLock = 0; - #define MALLOC_ACQUIRE() while(__sync_lock_test_and_set(&multithreadingLock, 1)) { while(multithreadingLock) { /*nop*/ } } -@@ -539,38 +536,24 @@ static bool claim_more_memory(size_t numBytes) - validate_memory_regions(); - #endif - -- uint8_t *startPtr; -- uint8_t *endPtr; -- do { -- // If this is the first time we're called, see if we can use -- // the initial heap memory set up by wasm-ld. -- if (!listOfAllRegions) { -- unsigned char *heap_end = sbrk(0); -- if (numBytes <= (size_t)(heap_end - &__heap_base)) { -- startPtr = &__heap_base; -- endPtr = heap_end; -- break; -- } -- } -- -- // Round numBytes up to the nearest page size. -- numBytes = (numBytes + (PAGE_SIZE-1)) & -PAGE_SIZE; -+ // Runtime facilities can grow linear memory without transferring ownership -+ // to this allocator. Claim only the exact interval returned by positive -+ // sbrk; sbrk(0) is an address-space frontier, not an ownership boundary. -+ numBytes = (numBytes + (PAGE_SIZE-1)) & -PAGE_SIZE; - -- // Claim memory via sbrk -- startPtr = (uint8_t*)sbrk(numBytes); -- if ((intptr_t)startPtr == -1) -- { -+ uint8_t *startPtr = (uint8_t*)sbrk(numBytes); -+ if ((intptr_t)startPtr == -1) -+ { - #ifdef EMMALLOC_VERBOSE -- MAIN_THREAD_ASYNC_EM_ASM(console.error('claim_more_memory: sbrk failed!')); -+ MAIN_THREAD_ASYNC_EM_ASM(console.error('claim_more_memory: sbrk failed!')); - #endif -- return false; -- } -+ return false; -+ } - #ifdef EMMALLOC_VERBOSE -- MAIN_THREAD_ASYNC_EM_ASM(console.log('claim_more_memory: claimed 0x' + ($0>>>0).toString(16) + ' - 0x' + ($1>>>0).toString(16) + ' (' + ($2>>>0) + ' bytes) via sbrk()'), startPtr, startPtr + numBytes, numBytes); -+ MAIN_THREAD_ASYNC_EM_ASM(console.log('claim_more_memory: claimed 0x' + ($0>>>0).toString(16) + ' - 0x' + ($1>>>0).toString(16) + ' (' + ($2>>>0) + ' bytes) via sbrk()'), startPtr, startPtr + numBytes, numBytes); - #endif -- assert(HAS_ALIGNMENT(startPtr, alignof(size_t))); -- endPtr = startPtr + numBytes; -- } while (0); -+ assert(HAS_ALIGNMENT(startPtr, alignof(size_t))); -+ uint8_t *endPtr = startPtr + numBytes; - - // Create a sentinel region at the end of the new heap block - Region *endSentinelRegion = (Region*)(endPtr - sizeof(Region)); -diff --git a/expected/wasm32-wasi-eh/defined-symbols.txt b/expected/wasm32-wasi-eh/defined-symbols.txt -index 51b29a4..29811fd 100644 ---- a/expected/wasm32-wasi-eh/defined-symbols.txt -+++ b/expected/wasm32-wasi-eh/defined-symbols.txt -@@ -1650,6 +1650,7 @@ swprintf - swscanf - symlink - symlinkat -+sync_file_range - sysconf - syslog - system -diff --git a/expected/wasm32-wasi-eh/undefined-symbols.txt b/expected/wasm32-wasi-eh/undefined-symbols.txt -index 4f926be..b03cc61 100644 ---- a/expected/wasm32-wasi-eh/undefined-symbols.txt -+++ b/expected/wasm32-wasi-eh/undefined-symbols.txt -@@ -10,6 +10,7 @@ __floatsitf - __floatunsitf - __getf2 - __gttf2 -+__imported_oliphaunt_postmaster_v1_fd_sync_range - __imported_wasi_snapshot_preview1_args_get - __imported_wasi_snapshot_preview1_args_sizes_get - __imported_wasi_snapshot_preview1_clock_res_get -diff --git a/expected/wasm32-wasi-ehpic/defined-symbols.txt b/expected/wasm32-wasi-ehpic/defined-symbols.txt -index 3885a66..ddaacd9 100644 ---- a/expected/wasm32-wasi-ehpic/defined-symbols.txt -+++ b/expected/wasm32-wasi-ehpic/defined-symbols.txt -@@ -1658,6 +1658,7 @@ swprintf - swscanf - symlink - symlinkat -+sync_file_range - sysconf - syslog - system -diff --git a/expected/wasm32-wasi-ehpic/undefined-symbols.txt b/expected/wasm32-wasi-ehpic/undefined-symbols.txt -index 7af7e4c..49fbed2 100644 ---- a/expected/wasm32-wasi-ehpic/undefined-symbols.txt -+++ b/expected/wasm32-wasi-ehpic/undefined-symbols.txt -@@ -11,6 +11,7 @@ __floatsitf - __floatunsitf - __getf2 - __gttf2 -+__imported_oliphaunt_postmaster_v1_fd_sync_range - __imported_wasi_snapshot_preview1_args_get - __imported_wasi_snapshot_preview1_args_sizes_get - __imported_wasi_snapshot_preview1_clock_res_get -diff --git a/expected/wasm32-wasi-threads/defined-symbols.txt b/expected/wasm32-wasi-threads/defined-symbols.txt -index c8996ef..ad96b32 100644 ---- a/expected/wasm32-wasi-threads/defined-symbols.txt -+++ b/expected/wasm32-wasi-threads/defined-symbols.txt -@@ -1562,6 +1562,7 @@ swprintf - swscanf - symlink - symlinkat -+sync_file_range - sysconf - syslog - system -diff --git a/expected/wasm32-wasi-threads/undefined-symbols.txt b/expected/wasm32-wasi-threads/undefined-symbols.txt -index e4a2638..9145830 100644 ---- a/expected/wasm32-wasi-threads/undefined-symbols.txt -+++ b/expected/wasm32-wasi-threads/undefined-symbols.txt -@@ -13,6 +13,7 @@ __getf2 - __global_base - __gttf2 - __heap_base -+__imported_oliphaunt_postmaster_v1_fd_sync_range - __imported_wasi_snapshot_preview1_args_get - __imported_wasi_snapshot_preview1_args_sizes_get - __imported_wasi_snapshot_preview1_clock_res_get -diff --git a/expected/wasm32-wasi/defined-symbols.txt b/expected/wasm32-wasi/defined-symbols.txt -index 6d33bff..a7e915b 100644 ---- a/expected/wasm32-wasi/defined-symbols.txt -+++ b/expected/wasm32-wasi/defined-symbols.txt -@@ -1657,6 +1657,7 @@ swprintf - swscanf - symlink - symlinkat -+sync_file_range - sysconf - syslog - system -diff --git a/expected/wasm32-wasi/undefined-symbols.txt b/expected/wasm32-wasi/undefined-symbols.txt -index 4f926be..b03cc61 100644 ---- a/expected/wasm32-wasi/undefined-symbols.txt -+++ b/expected/wasm32-wasi/undefined-symbols.txt -@@ -10,6 +10,7 @@ __floatsitf - __floatunsitf - __getf2 - __gttf2 -+__imported_oliphaunt_postmaster_v1_fd_sync_range - __imported_wasi_snapshot_preview1_args_get - __imported_wasi_snapshot_preview1_args_sizes_get - __imported_wasi_snapshot_preview1_clock_res_get -diff --git a/expected/wasm64-wasi/defined-symbols.txt b/expected/wasm64-wasi/defined-symbols.txt -index cd3d45b..ca54a6d 100644 ---- a/expected/wasm64-wasi/defined-symbols.txt -+++ b/expected/wasm64-wasi/defined-symbols.txt -@@ -1612,6 +1612,7 @@ swprintf - swscanf - symlink - symlinkat -+sync_file_range - sysconf - syslog - system -diff --git a/expected/wasm64-wasi/undefined-symbols.txt b/expected/wasm64-wasi/undefined-symbols.txt -index ae7c7ff..a70377f 100644 ---- a/expected/wasm64-wasi/undefined-symbols.txt -+++ b/expected/wasm64-wasi/undefined-symbols.txt -@@ -14,6 +14,7 @@ __getf2 - __global_base - __gttf2 - __heap_base -+__imported_oliphaunt_postmaster_v1_fd_sync_range - __imported_wasi_snapshot_preview1_args_get - __imported_wasi_snapshot_preview1_args_sizes_get - __imported_wasi_snapshot_preview1_clock_res_get -diff --git a/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c b/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c -index 06b7288..0526fce 100644 ---- a/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c -+++ b/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c -@@ -24,7 +24,8 @@ int fcntl(int fildes, int cmd, ...) { - int flags = va_arg(ap, int); - va_end(ap); - -- __wasi_fdflagsext_t fd_flags = flags | FD_CLOEXEC ? __WASI_FDFLAGSEXT_CLOEXEC : 0; -+ __wasi_fdflagsext_t fd_flags = -+ (flags & FD_CLOEXEC) ? __WASI_FDFLAGSEXT_CLOEXEC : 0; - __wasi_errno_t error = - __wasi_fd_fdflags_set(fildes, fd_flags); - if (error != 0) { -diff --git a/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c b/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c -index 8e469c6..f7d8034 100644 ---- a/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c -+++ b/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c -@@ -52,6 +52,14 @@ int __wasilibc_nocwd_openat_nomode(int fd, const char *path, int oflag) { - return -1; - } - -+ // A directory descriptor is opened read-only, but its contents are -+ // metadata and POSIX permits both fsync() and fdatasync() to establish the -+ // same directory durability barrier. Request both capabilities explicitly; -+ // otherwise fdatasync() is rejected by the WASI rights check before the -+ // runtime can synchronize the retained open directory description. -+ if ((oflag & O_DIRECTORY) != 0) -+ max |= __WASI_RIGHTS_FD_DATASYNC | __WASI_RIGHTS_FD_SYNC; -+ - // Ensure that we can actually obtain the minimal rights needed. - __wasi_fdstat_t fsb_cur; - __wasi_errno_t error = __wasi_fd_fdstat_get(fd, &fsb_cur); -diff --git a/libc-bottom-half/cloudlibc/src/libc/fcntl/sync_file_range.c b/libc-bottom-half/cloudlibc/src/libc/fcntl/sync_file_range.c -new file mode 100644 -index 0000000..47cdad7 ---- /dev/null -+++ b/libc-bottom-half/cloudlibc/src/libc/fcntl/sync_file_range.c -@@ -0,0 +1,33 @@ -+#define _GNU_SOURCE -+ -+#include -+#include -+#include -+ -+int32_t __imported_oliphaunt_postmaster_v1_fd_sync_range( -+ int32_t fd, int64_t offset, int64_t nbytes, int32_t flags) __attribute__(( -+ __import_module__("oliphaunt_postmaster_v1"), -+ __import_name__("fd_sync_range"))); -+ -+int sync_file_range(int fd, off_t offset, off_t nbytes, unsigned flags) -+{ -+ const unsigned valid_flags = SYNC_FILE_RANGE_WAIT_BEFORE | -+ SYNC_FILE_RANGE_WRITE | -+ SYNC_FILE_RANGE_WAIT_AFTER; -+ int32_t error; -+ -+ /* Linux requires a representable exclusive end; nbytes=0 means EOF. */ -+ if (offset < 0 || nbytes < 0 || (flags & ~valid_flags) != 0 || -+ (nbytes > 0 && offset > INT64_MAX - nbytes)) { -+ errno = EINVAL; -+ return -1; -+ } -+ -+ error = __imported_oliphaunt_postmaster_v1_fd_sync_range( -+ (int32_t)fd, (int64_t)offset, (int64_t)nbytes, (int32_t)flags); -+ if (error != 0) { -+ errno = error; -+ return -1; -+ } -+ return 0; -+} -diff --git a/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c b/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c -index 74caf2e..2ab7514 100644 ---- a/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c -+++ b/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c -@@ -4,10 +4,20 @@ - #include - #include - #include -+#include - #include -+#include - - int socket(int domain, int ty, int protocol) { - int fd; -+ int flags = ty & (SOCK_NONBLOCK | SOCK_CLOEXEC); -+ ty &= ~(SOCK_NONBLOCK | SOCK_CLOEXEC); -+ -+ if (ty != SOCK_STREAM && ty != SOCK_DGRAM) { -+ errno = EPROTONOSUPPORT; -+ return -1; -+ } -+ - if(!protocol) { - switch (ty) - { -@@ -25,5 +35,14 @@ int socket(int domain, int ty, int protocol) { - return -1; - } - -+ if ((flags & SOCK_CLOEXEC) != 0 && fcntl(fd, F_SETFD, FD_CLOEXEC) < 0) { -+ close(fd); -+ return -1; -+ } -+ if ((flags & SOCK_NONBLOCK) != 0 && fcntl(fd, F_SETFL, O_NONBLOCK) < 0) { -+ close(fd); -+ return -1; -+ } -+ - return fd; - } -diff --git a/libc-bottom-half/headers/public/wasi/api_wasix.h b/libc-bottom-half/headers/public/wasi/api_wasix.h -index 5b02c4b..3963e83 100644 ---- a/libc-bottom-half/headers/public/wasi/api_wasix.h -+++ b/libc-bottom-half/headers/public/wasi/api_wasix.h -@@ -4399,6 +4399,14 @@ __wasi_errno_t __wasi_proc_spawn2( - __wasi_errno_t __wasi_proc_id( - __wasi_pid_t *retptr0 - ) __attribute__((__warn_unused_result__)); -+/** -+ * Returns the current and maximum resource limit for the current process -+ */ -+__wasi_errno_t __wasi_proc_rlimit_get( -+ uint32_t resource, -+ uint64_t *retptr0, -+ uint64_t *retptr1 -+) __attribute__((__warn_unused_result__)); - /** - * Returns the parent handle of a particular process - */ -diff --git a/libc-bottom-half/mman/mman.c b/libc-bottom-half/mman/mman.c -index 7e26562..a156655 100644 ---- a/libc-bottom-half/mman/mman.c -+++ b/libc-bottom-half/mman/mman.c -@@ -13,25 +13,188 @@ - #include - #include - #include -+#include -+#include - #include - #include -+#include -+ -+#define RUNTIME_MMAP_ALIGN ((size_t) 65536) -+ -+int32_t __imported_wasix_32v1_mem_mmap(int32_t arg0, int32_t arg1, -+ int32_t arg2, int32_t arg3, -+ int32_t arg4, int64_t arg5, -+ int32_t arg6) __attribute__(( -+ __import_module__("wasix_32v1"), -+ __import_name__("mem_mmap") -+)); -+ -+int32_t __imported_wasix_32v1_mem_munmap(int32_t arg0, int32_t arg1) __attribute__(( -+ __import_module__("wasix_32v1"), -+ __import_name__("mem_munmap") -+)); -+ -+int32_t __imported_wasix_32v1_mem_msync(int32_t arg0, int32_t arg1, -+ int32_t arg2) __attribute__(( -+ __import_module__("wasix_32v1"), -+ __import_name__("mem_msync") -+)); - - struct map { -+ void *addr; - int prot; - int flags; - off_t offset; - size_t length; - int fd; -+ struct map *next; - }; - -+/* Runtime-backed mappings live only in Wasmer's authoritative interval -+ registry. This guest list is solely for malloc-backed legacy emulation. */ -+static struct map *legacy_maps; -+static pthread_mutex_t maps_lock = PTHREAD_MUTEX_INITIALIZER; -+ -+static __wasi_errno_t runtime_mem_mmap(void *addr, size_t length, int prot, -+ int flags, int fd, off_t offset, -+ void **ret_addr) { -+ uintptr_t out = 0; -+ int32_t ret = __imported_wasix_32v1_mem_mmap( -+ (int32_t) (uintptr_t) addr, -+ (int32_t) length, -+ (int32_t) prot, -+ (int32_t) flags, -+ (int32_t) fd, -+ (int64_t) offset, -+ (int32_t) (uintptr_t) &out); -+ *ret_addr = (void *) out; -+ return (__wasi_errno_t) ret; -+} -+ -+static __wasi_errno_t runtime_mem_munmap(void *addr, size_t length) { -+ int32_t ret = __imported_wasix_32v1_mem_munmap( -+ (int32_t) (uintptr_t) addr, -+ (int32_t) length); -+ return (__wasi_errno_t) ret; -+} -+ -+static __wasi_errno_t runtime_mem_msync(void *addr, size_t length, int flags) { -+ int32_t ret = __imported_wasix_32v1_mem_msync( -+ (int32_t) (uintptr_t) addr, -+ (int32_t) length, -+ (int32_t) flags); -+ return (__wasi_errno_t) ret; -+} -+ -+static int range_end(void *addr, size_t length, uintptr_t *end) { -+ return length != 0 && -+ !__builtin_add_overflow((uintptr_t) addr, length, end); -+} -+ -+static struct map *find_overlapping_legacy_map_locked( -+ void *addr, size_t length, struct map ***link_out) { -+ uintptr_t start = (uintptr_t) addr; -+ uintptr_t end; -+ struct map **link = &legacy_maps; -+ -+ if (!range_end(addr, length, &end)) { -+ return NULL; -+ } -+ -+ while (*link != NULL) { -+ uintptr_t map_start = (uintptr_t) (*link)->addr; -+ uintptr_t map_end; -+ -+ if (!range_end((*link)->addr, (*link)->length, &map_end)) { -+ /* Registry entries are validated before insertion. */ -+ abort(); -+ } -+ if (start < map_end && map_start < end) { -+ if (link_out) { -+ *link_out = link; -+ } -+ return *link; -+ } -+ link = &(*link)->next; -+ } -+ -+ return NULL; -+} -+ -+static int map_contains(const struct map *map, void *addr, size_t length) { -+ uintptr_t start = (uintptr_t) addr; -+ uintptr_t end; -+ uintptr_t map_start = (uintptr_t) map->addr; -+ uintptr_t map_end; -+ -+ return range_end(addr, length, &end) && -+ range_end(map->addr, map->length, &map_end) && -+ start >= map_start && end <= map_end; -+} -+ -+static int map_is_exact(const struct map *map, void *addr, size_t length) { -+ return map->addr == addr && map->length == length; -+} -+ -+static void remove_map_locked(struct map **link) { -+ struct map *entry = *link; -+ -+ *link = entry->next; -+} -+ -+static void *runtime_mmap(void *addr, size_t length, int prot, int flags, -+ int fd, off_t offset) { -+ void *target = addr; -+ void *mapped = NULL; -+ int fixed = (flags & MAP_FIXED) != 0; -+ __wasi_errno_t error; -+ -+ if (!fixed) { -+ /* -+ * Address selection belongs to the runtime. It atomically claims new -+ * WebAssembly pages and returns their old end as the selected address. -+ * Allocators claim only exact positive-sbrk return intervals. A -+ * non-NULL POSIX hint is deliberately ignored rather than converted -+ * into an unsafe fixed replacement. -+ */ -+ target = NULL; -+ } else if (((uintptr_t) target & (RUNTIME_MMAP_ALIGN - 1)) != 0) { -+ errno = EINVAL; -+ return MAP_FAILED; -+ } -+ -+ /* The host remap is currently read/write. Reject weaker or broader -+ protection contracts instead of silently violating them. */ -+ if (prot != (PROT_READ | PROT_WRITE)) { -+ errno = EINVAL; -+ return MAP_FAILED; -+ } -+ -+ error = runtime_mem_mmap(target, length, prot, flags, fd, offset, &mapped); -+ if (error != __WASI_ERRNO_SUCCESS) { -+ errno = error; -+ return MAP_FAILED; -+ } -+ if (mapped == NULL || (fixed && mapped != target)) { -+ if (mapped != NULL) { -+ runtime_mem_munmap(mapped, length); -+ } -+ errno = EINVAL; -+ return MAP_FAILED; -+ } -+ -+ return mapped; -+} -+ - void *mmap(void *addr, size_t length, int prot, int flags, - int fd, off_t offset) { -+ int runtime_candidate = -+ (flags & MAP_TYPE) == MAP_SHARED && (flags & MAP_ANON) == 0; -+ - // Check for unsupported flags. -- if ((flags & (MAP_PRIVATE | MAP_SHARED)) == 0 || -- (flags & MAP_FIXED) != 0 || --#ifdef MAP_SHARED_VALIDATE -- (flags & MAP_SHARED_VALIDATE) == MAP_SHARED_VALIDATE || --#endif -+ if (((flags & MAP_TYPE) != MAP_PRIVATE && -+ (flags & MAP_TYPE) != MAP_SHARED) || -+ ((flags & MAP_FIXED) != 0 && !runtime_candidate) || - #ifdef MAP_NORESERVE - (flags & MAP_NORESERVE) != 0 || - #endif -@@ -67,6 +230,10 @@ void *mmap(void *addr, size_t length, int prot, int flags, - return MAP_FAILED; - } - -+ if (runtime_candidate) { -+ return runtime_mmap(addr, length, prot, flags, fd, offset); -+ } -+ - // Check for integer overflow. - size_t buf_len = 0; - if (__builtin_add_overflow(length, sizeof(struct map), &buf_len)) { -@@ -97,7 +264,7 @@ void *mmap(void *addr, size_t length, int prot, int flags, - errno = EINVAL; - free(map); - -- return NULL; -+ return MAP_FAILED; - } - - map->fd = new_fd; -@@ -108,6 +275,8 @@ void *mmap(void *addr, size_t length, int prot, int flags, - if (nread < 0) { - if (errno == EINTR) - continue; -+ close(new_fd); -+ free(map); - return MAP_FAILED; - } - if (nread == 0) -@@ -121,28 +290,94 @@ void *mmap(void *addr, size_t length, int prot, int flags, - memset(addr, 0, length); - } - -+ map->addr = addr; -+ pthread_mutex_lock(&maps_lock); -+ if (find_overlapping_legacy_map_locked(addr, map->length, NULL) != NULL) { -+ pthread_mutex_unlock(&maps_lock); -+ if (map->fd >= 0) { -+ close(map->fd); -+ } -+ free(map); -+ errno = EINVAL; -+ return MAP_FAILED; -+ } -+ map->next = legacy_maps; -+ legacy_maps = map; -+ pthread_mutex_unlock(&maps_lock); -+ - return addr; - } - -+static int legacy_msync(struct map *map, void *addr, size_t length) { -+ uintptr_t delta = (uintptr_t) addr - (uintptr_t) map->addr; -+ off_t map_offset; -+ -+ if (__builtin_add_overflow(map->offset, (off_t) delta, &map_offset)) { -+ errno = EOVERFLOW; -+ return -1; -+ } -+ -+ if ((map->flags & MAP_SHARED) != 0 && -+ (map->flags & MAP_ANON) == 0 && -+ (map->prot & PROT_WRITE) != 0) { -+ char *body = (char *) addr; -+ -+ while (length > 0) { -+ const ssize_t nwrite = pwrite(map->fd, body, length, map_offset); -+ -+ if (nwrite > 0) { -+ length -= (size_t) nwrite; -+ map_offset += (size_t) nwrite; -+ body += (size_t) nwrite; -+ } else if (errno == EINTR) { -+ continue; -+ } else { -+ return -1; -+ } -+ } -+ } -+ -+ return 0; -+} -+ - int munmap(void *addr, size_t length) { -- struct map *map = (struct map *)addr - 1; -+ struct map **link = NULL; -+ struct map *map; -+ __wasi_errno_t runtime_error = runtime_mem_munmap(addr, length); - -- // We don't support partial munmapping. -- if (map->length != length) { -+ if (runtime_error == __WASI_ERRNO_SUCCESS) { -+ return 0; -+ } -+ if (runtime_error != __WASI_ERRNO_NOENT) { -+ errno = runtime_error; -+ return -1; -+ } -+ -+ pthread_mutex_lock(&maps_lock); -+ map = find_overlapping_legacy_map_locked(addr, length, &link); -+ if (map == NULL || !map_is_exact(map, addr, length)) { -+ pthread_mutex_unlock(&maps_lock); - errno = EINVAL; - return -1; - } - - // Write the data back to the backing file and close - // the file handle -- if (map->fd > 0) { -- if ((map->prot & PROT_WRITE) != 0) { -- msync(addr, length, MS_SYNC); -+ if (map->fd >= 0) { -+ if ((map->flags & MAP_SHARED) != 0 && -+ (map->prot & PROT_WRITE) != 0) { -+ if (legacy_msync(map, addr, length) < 0) { -+ pthread_mutex_unlock(&maps_lock); -+ return -1; -+ } - } - - close(map->fd); - } - -+ remove_map_locked(link); -+ pthread_mutex_unlock(&maps_lock); -+ - // Release the memory. - free(map); - -@@ -151,39 +386,38 @@ int munmap(void *addr, size_t length) { - } - - int msync (void *addr, size_t length, int flags) { -- struct map *map = (struct map *)addr - 1; -- size_t map_flags = map->flags; -- off_t map_offset = map->offset; -- size_t map_length = map->length; -- int fd = map->fd; -+ struct map *map; -+ __wasi_errno_t runtime_error; - -- if (length > map_length) { -+ if ((flags & ~(MS_ASYNC | MS_INVALIDATE | MS_SYNC)) != 0 || -+ ((flags & MS_ASYNC) != 0 && (flags & MS_SYNC) != 0)) { - errno = EINVAL; - return -1; - } - -- if ((map->prot & PROT_WRITE) != 0) { -- errno = EINVAL; -+ runtime_error = runtime_mem_msync(addr, length, flags); -+ if (runtime_error == __WASI_ERRNO_SUCCESS) { -+ return 0; -+ } -+ if (runtime_error != __WASI_ERRNO_NOENT) { -+ errno = runtime_error; - return -1; - } - -- if ((map_flags & MAP_ANON) == 0) { -- char *body = (char *)addr; -- -- while (length > 0) { -- const ssize_t nwrite = pwrite(fd, body, length, map_offset); -- -- if (nwrite > 0) { -- length -= (size_t)nwrite; -- map_offset += (size_t)nwrite; -- body += (size_t)nwrite; -- } else if (errno == EINTR) { -- continue; -- } else { -- return -1; -- } -- } -+ pthread_mutex_lock(&maps_lock); -+ map = find_overlapping_legacy_map_locked(addr, length, NULL); -+ if (map == NULL) { -+ pthread_mutex_unlock(&maps_lock); -+ errno = ENOMEM; -+ return -1; -+ } -+ if (!map_contains(map, addr, length)) { -+ pthread_mutex_unlock(&maps_lock); -+ errno = EINVAL; -+ return -1; - } - -- return 0; -+ int result = legacy_msync(map, addr, length); -+ pthread_mutex_unlock(&maps_lock); -+ return result; - } -diff --git a/libc-bottom-half/sources/__wasilibc_futex.c b/libc-bottom-half/sources/__wasilibc_futex.c -index 340899b..1a453fa 100644 ---- a/libc-bottom-half/sources/__wasilibc_futex.c -+++ b/libc-bottom-half/sources/__wasilibc_futex.c -@@ -12,7 +12,7 @@ int __wasilibc_futex_wait_wasix(volatile void *addr, int op, int expected, int64 - __wasi_bool_t woken = __WASI_BOOL_FALSE; - - __wasi_option_timestamp_t timeout; -- if (max_wait_ns > 0) { -+ if (max_wait_ns >= 0) { - timeout.tag = __WASI_OPTION_SOME; - timeout.u.some = max_wait_ns; - } else { -@@ -30,8 +30,9 @@ int __wasilibc_futex_wait_wasix(volatile void *addr, int op, int expected, int64 - return -EWOULDBLOCK; - } - -- if (__wasi_futex_wait((uint32_t*)addr, expected, &timeout, &woken) != 0) { -- __builtin_trap(); -+ __wasi_errno_t ret = __wasi_futex_wait((uint32_t*)addr, expected, &timeout, &woken); -+ if (ret != 0) { -+ return -ret; - } - - if (woken == __WASI_BOOL_FALSE && *paddr == expected) { -diff --git a/libc-bottom-half/sources/__wasixlibc_real.c b/libc-bottom-half/sources/__wasixlibc_real.c -index 41a3f0b..56d916e 100644 ---- a/libc-bottom-half/sources/__wasixlibc_real.c -+++ b/libc-bottom-half/sources/__wasixlibc_real.c -@@ -502,6 +502,20 @@ __wasi_errno_t __wasi_proc_id( - return (uint16_t) ret; - } - -+int32_t __imported_wasix_32v1_proc_rlimit_get(int32_t arg0, int32_t arg1, int32_t arg2) __attribute__(( -+ __import_module__("wasix_32v1"), -+ __import_name__("proc_rlimit_get") -+)); -+ -+__wasi_errno_t __wasi_proc_rlimit_get( -+ uint32_t resource, -+ uint64_t *retptr0, -+ uint64_t *retptr1 -+){ -+ int32_t ret = __imported_wasix_32v1_proc_rlimit_get((int32_t) resource, (intptr_t) retptr0, (intptr_t) retptr1); -+ return (uint16_t) ret; -+} -+ - int32_t __imported_wasix_32v1_proc_parent(int32_t arg0, int32_t arg1) __attribute__(( - __import_module__("wasix_32v1"), - __import_name__("proc_parent") -@@ -1278,4 +1292,3 @@ __wasi_errno_t __wasi_context_destroy( - int32_t ret = __imported_wasix_32v1_context_destroy((int64_t) context); - return (uint16_t) ret; - } -- -diff --git a/libc-bottom-half/sources/sbrk.c b/libc-bottom-half/sources/sbrk.c -index a26b75e..98cd803 100644 ---- a/libc-bottom-half/sources/sbrk.c -+++ b/libc-bottom-half/sources/sbrk.c -@@ -1,13 +1,18 @@ - #include - #include - #include -+#include - #include <__macro_PAGESIZE.h> - --// Bare-bones implementation of sbrk. -+/* -+ * Every positive call owns exactly the interval it returns. Runtime facilities -+ * may grow linear memory between calls, so callers must not infer ownership -+ * from a later sbrk(0) or assume that independently acquired intervals are -+ * consecutive. -+ */ - void *sbrk(intptr_t increment) { -- // sbrk(0) returns the current memory size. - if (increment == 0) { -- // The wasm spec doesn't guarantee that memory.grow of 0 always succeeds. -+ // The wasm spec doesn't guarantee that memory.grow of 0 succeeds. - return (void *)(__builtin_wasm_memory_size(0) * PAGESIZE); - } - -@@ -28,5 +33,17 @@ void *sbrk(intptr_t increment) { - return (void *)-1; - } - -- return (void *)(old * PAGESIZE); -+ if (old > UINTPTR_MAX / PAGESIZE) { -+ errno = ENOMEM; -+ return (void *)-1; -+ } -+ old *= PAGESIZE; -+ if ((uintptr_t)increment > UINTPTR_MAX - old) { -+ /* memory.grow has already reserved this terminal interval, but its -+ one-past end is not representable in the guest pointer width. Do -+ not transfer an interval with a wrapping end to the caller. */ -+ errno = ENOMEM; -+ return (void *)-1; -+ } -+ return (void *)old; - } -diff --git a/libc-top-half/musl/include/fcntl.h b/libc-top-half/musl/include/fcntl.h -index 15bef28..1332569 100644 ---- a/libc-top-half/musl/include/fcntl.h -+++ b/libc-top-half/musl/include/fcntl.h -@@ -190,11 +190,9 @@ struct f_owner_ex { - #ifdef __wasilibc_unmodified_upstream /* WASI has no name_to_handle_at */ - #define MAX_HANDLE_SZ 128 - #endif --#ifdef __wasilibc_unmodified_upstream /* WASI has no syc_file_range */ - #define SYNC_FILE_RANGE_WAIT_BEFORE 1 - #define SYNC_FILE_RANGE_WRITE 2 - #define SYNC_FILE_RANGE_WAIT_AFTER 4 --#endif - #ifdef __wasilibc_unmodified_upstream /* WASI has no splice */ - #define SPLICE_F_MOVE 1 - #define SPLICE_F_NONBLOCK 2 -@@ -212,8 +210,8 @@ int open_by_handle_at(int, struct file_handle *, int); - #ifdef __wasilibc_unmodified_upstream /* WASI has no readahead */ - ssize_t readahead(int, off_t, size_t); - #endif --#ifdef __wasilibc_unmodified_upstream /* WASI has no splice, syc_file_range, or tee */ - int sync_file_range(int, off_t, off_t, unsigned); -+#ifdef __wasilibc_unmodified_upstream /* WASI has no splice or tee */ - ssize_t vmsplice(int, const struct iovec *, size_t, unsigned); - ssize_t splice(int, off_t *, int, off_t *, size_t, unsigned); - ssize_t tee(int, int, size_t, unsigned); -diff --git a/libc-top-half/musl/include/setjmp.h b/libc-top-half/musl/include/setjmp.h -index 3961c02..f92f9f0 100644 ---- a/libc-top-half/musl/include/setjmp.h -+++ b/libc-top-half/musl/include/setjmp.h -@@ -48,7 +48,13 @@ _Noreturn void _longjmp (jmp_buf, int); - #if defined(_POSIX_SOURCE) || defined(_POSIX_C_SOURCE) || \ - defined(_XOPEN_SOURCE) || defined(_GNU_SOURCE) || defined(_BSD_SOURCE) - typedef jmp_buf sigjmp_buf; -+#ifndef __WASIX_LIBC_BUILDING_SETJMP -+ int __wasilibc_sigsetjmp_save(sigjmp_buf, int); -+#define sigsetjmp(buf, savesigs) \ -+ (__wasilibc_sigsetjmp_save((buf), (savesigs)), setjmp((buf))) -+#else - int sigsetjmp(sigjmp_buf, int) __setjmp_attr; -+#endif - _Noreturn void siglongjmp(sigjmp_buf, int); - #endif - -diff --git a/libc-top-half/musl/include/unistd.h b/libc-top-half/musl/include/unistd.h -index 95a3ee7..7c30563 100644 ---- a/libc-top-half/musl/include/unistd.h -+++ b/libc-top-half/musl/include/unistd.h -@@ -134,11 +134,9 @@ unsigned sleep(unsigned); - int pause(void); - #endif - --#if defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) - pid_t fork(void); - pid_t _fork_internal(int copy_mem); - pid_t _Fork(int copy_mem); --#endif - int execve(const char *, char *const [], char *const []); - int execv(const char *, char *const []); - int execle(const char *, const char *, ...); -diff --git a/libc-top-half/musl/src/linux/epoll.c b/libc-top-half/musl/src/linux/epoll.c -index 4780ce2..b3ac891 100644 ---- a/libc-top-half/musl/src/linux/epoll.c -+++ b/libc-top-half/musl/src/linux/epoll.c -@@ -2,6 +2,8 @@ - #include - #include - #include -+#include -+#include - - int epoll_create(int size) - { -@@ -17,7 +19,27 @@ int epoll_create(int size) - - int epoll_create1(int flags) - { -- return epoll_create(0); -+ int fd; -+ -+ if (flags & ~EPOLL_CLOEXEC) { -+ errno = EINVAL; -+ return -1; -+ } -+ -+ fd = epoll_create(0); -+ if (fd < 0) { -+ return -1; -+ } -+ -+ if ((flags & EPOLL_CLOEXEC) && fcntl(fd, F_SETFD, FD_CLOEXEC) < 0) { -+ int saved_errno = errno; -+ -+ close(fd); -+ errno = saved_errno; -+ return -1; -+ } -+ -+ return fd; - } - - int epoll_ctl(int fd, int op, int fd2, struct epoll_event *ev) -diff --git a/libc-top-half/musl/src/misc/getrlimit.c b/libc-top-half/musl/src/misc/getrlimit.c -index 67bfc7c..972fb67 100644 ---- a/libc-top-half/musl/src/misc/getrlimit.c -+++ b/libc-top-half/musl/src/misc/getrlimit.c -@@ -2,6 +2,12 @@ - #include - #ifdef __wasilibc_unmodified_upstream - #include "syscall.h" -+#else -+#include -+#endif -+ -+#ifndef SYSCALL_RLIM_INFINITY -+#define SYSCALL_RLIM_INFINITY RLIM_INFINITY - #endif - - #define FIX(x) do{ if ((x)>=SYSCALL_RLIM_INFINITY) (x)=RLIM_INFINITY; }while(0) -@@ -24,8 +30,17 @@ int getrlimit(int resource, struct rlimit *rlim) - FIX(rlim->rlim_cur); - FIX(rlim->rlim_max); - #else -- rlim->rlim_cur = RLIM_INFINITY; -- rlim->rlim_max = RLIM_INFINITY; -+ uint64_t rlim_cur; -+ uint64_t rlim_max; -+ __wasi_errno_t error = __wasi_proc_rlimit_get(resource, &rlim_cur, &rlim_max); -+ if (error != __WASI_ERRNO_SUCCESS) { -+ errno = error; -+ return -1; -+ } -+ rlim->rlim_cur = rlim_cur; -+ rlim->rlim_max = rlim_max; -+ FIX(rlim->rlim_cur); -+ FIX(rlim->rlim_max); - #endif - return 0; - } -diff --git a/libc-top-half/musl/src/process/_Fork.c b/libc-top-half/musl/src/process/_Fork.c -index 76df425..8f0ee8e 100644 ---- a/libc-top-half/musl/src/process/_Fork.c -+++ b/libc-top-half/musl/src/process/_Fork.c -@@ -10,8 +10,6 @@ - #include "pthread_impl.h" - #include "aio_impl.h" - --#if defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) -- - static void dummy(int x) { } - weak_alias(dummy, __aio_atfork); - -@@ -62,5 +60,3 @@ pid_t _Fork(int copy_mem) - __restore_sigs(&set); - return __syscall_ret(ret); - } -- --#endif /* defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) */ -\ No newline at end of file -diff --git a/libc-top-half/musl/src/process/execv.c b/libc-top-half/musl/src/process/execv.c -index bce72e3..5ef4ebe 100644 ---- a/libc-top-half/musl/src/process/execv.c -+++ b/libc-top-half/musl/src/process/execv.c -@@ -7,6 +7,7 @@ extern char **__environ; - #include - #include - #include -+char *__wasilibc_exec_combine_strings(char *const strings[]); - #endif - - int execv(const char *path, char *const argv[]) -@@ -14,22 +15,16 @@ int execv(const char *path, char *const argv[]) - #ifdef __wasilibc_unmodified_upstream - return execve(path, argv, __environ); - #else -- int combined_len = 0; -- for (char **argvp = (char **)argv; *argvp != NULL; argvp++) { -- combined_len += strlen(*argvp) + 1; -+ char *combined_argv = __wasilibc_exec_combine_strings(argv); -+ if (combined_argv == NULL) { -+ return -1; - } - -- char *combined_argv = malloc((combined_len + 1)); -- char *combined_argv_p = combined_argv; -- for (char **argvp = (char **)argv; *argvp != NULL; argvp++) { -- memcpy(combined_argv_p, *argvp, strlen(*argvp)); -- combined_argv_p += strlen(*argvp); -- *combined_argv_p = '\n'; -- combined_argv_p++; -- } -- *combined_argv_p = 0; -- - int e = __wasi_proc_exec3(path, combined_argv, NULL, 0, NULL); -+ /* A successful vfork+exec returns in the shared parent image immediately -+ before the restore longjmp. Release child marshaling storage first so -+ every backend launch returns the parent heap to its prior plateau. */ -+ free(combined_argv); - #ifdef __wasm_exception_handling__ - extern _Noreturn void __vfork_restore(); - if (e == 0) { -@@ -37,7 +32,7 @@ int execv(const char *path, char *const argv[]) - } - #endif - -- // A return from proc_exec automatically means it failed -+ // Any ordinary return from proc_exec means it failed. - errno = e; - return -1; - #endif -diff --git a/libc-top-half/musl/src/process/execvp.c b/libc-top-half/musl/src/process/execvp.c -index b26c275..0c5d1d5 100644 ---- a/libc-top-half/musl/src/process/execvp.c -+++ b/libc-top-half/musl/src/process/execvp.c -@@ -3,6 +3,7 @@ - #include - #include - #include -+#include - - extern char **__wasilibc_environ; - -@@ -64,15 +65,31 @@ int __execvpe(const char *file, char *const argv[], char *const envp[]) - #else - char *__wasilibc_exec_combine_strings(char *const strings[]) - { -- int combined_len = 0; -- for (char **ptr = (char **)strings; *ptr != NULL; ptr++) -+ size_t combined_len = 0; -+ for (char **ptr = (char **)strings; strings != NULL && *ptr != NULL; ptr++) - { -- combined_len += strlen(*ptr) + 1; -+ size_t len = strlen(*ptr); -+ if (len == SIZE_MAX || combined_len > SIZE_MAX - len - 1) -+ { -+ errno = E2BIG; -+ return NULL; -+ } -+ combined_len += len + 1; - } - -- char *combined = malloc((combined_len + 1)); -+ if (combined_len == SIZE_MAX) -+ { -+ errno = E2BIG; -+ return NULL; -+ } -+ char *combined = malloc(combined_len + 1); -+ if (combined == NULL) -+ { -+ errno = ENOMEM; -+ return NULL; -+ } - char *combined_p = combined; -- for (char **ptr = (char **)strings; *ptr != NULL; ptr++) -+ for (char **ptr = (char **)strings; strings != NULL && *ptr != NULL; ptr++) - { - memcpy(combined_p, *ptr, strlen(*ptr)); - combined_p += strlen(*ptr); -@@ -87,11 +104,22 @@ char *__wasilibc_exec_combine_strings(char *const strings[]) - int __execvpe(const char *path, char *const argv[], char *const envp[], uint8_t use_path) - { - char *combined_argv = __wasilibc_exec_combine_strings(argv); -+ if (combined_argv == NULL) -+ return -1; - char *combined_env = __wasilibc_exec_combine_strings(envp); -+ if (combined_env == NULL) -+ { -+ free(combined_argv); -+ return -1; -+ } - - int e = __wasi_proc_exec3( - path, combined_argv, combined_env, - use_path ? __WASI_BOOL_TRUE : __WASI_BOOL_FALSE, getenv("PATH")); -+ /* proc_exec3 has consumed both buffers before returning. On vfork success -+ this is the last point before longjmp restores the shared parent image. */ -+ free(combined_argv); -+ free(combined_env); - #ifdef __wasm_exception_handling__ - extern _Noreturn void __vfork_restore(); - if (e == 0) { -@@ -99,9 +127,6 @@ int __execvpe(const char *path, char *const argv[], char *const envp[], uint8_t - } - #endif - -- free(combined_argv); -- free(combined_env); -- - // A return from proc_exec automatically means it failed - errno = e; - return -1; -diff --git a/libc-top-half/musl/src/process/fork.c b/libc-top-half/musl/src/process/fork.c -index df38cf7..21895b4 100644 ---- a/libc-top-half/musl/src/process/fork.c -+++ b/libc-top-half/musl/src/process/fork.c -@@ -51,8 +51,6 @@ weak_alias(dummy_0, __tl_lock); - weak_alias(dummy_0, __tl_unlock); - #endif - --#if defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) -- - pid_t fork(void) - { - return _fork_internal(1); -@@ -102,5 +100,3 @@ pid_t _fork_internal(int copy_mem) - if (ret<0) errno = errno_save; - return ret; - } -- --#endif /* defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) */ -\ No newline at end of file -diff --git a/libc-top-half/musl/src/process/waitpid.c b/libc-top-half/musl/src/process/waitpid.c -index 456cca9..330aa66 100644 ---- a/libc-top-half/musl/src/process/waitpid.c -+++ b/libc-top-half/musl/src/process/waitpid.c -@@ -35,6 +35,13 @@ pid_t waitpid(pid_t pid, int *status, int options) - errno = ret; - return -1; - } else { -+ if (code.tag == __WASI_JOIN_STATUS_TYPE_NOTHING) { -+ if (status != NULL) { -+ *status = 0; -+ } -+ return 0; -+ } -+ - // Read the PID - if (opid.tag == __WASI_OPTION_SOME) { - pid = opid.u.some; -@@ -44,14 +51,18 @@ pid_t waitpid(pid_t pid, int *status, int options) - } - - // Build the status code depending on what happened -- if (code.tag == __WASI_JOIN_STATUS_TYPE_NOTHING) { -- *status = 0; -- } else if (code.tag == __WASI_JOIN_STATUS_TYPE_EXIT_NORMAL) { -- *status = W_EXITCODE(code.u.exit_normal, 0); -+ if (code.tag == __WASI_JOIN_STATUS_TYPE_EXIT_NORMAL) { -+ if (status != NULL) { -+ *status = W_EXITCODE(code.u.exit_normal, 0); -+ } - } else if (code.tag == __WASI_JOIN_STATUS_TYPE_EXIT_SIGNAL) { -- *status = W_EXITCODE(code.u.exit_signal.exit_code, code.u.exit_signal.signal); -+ if (status != NULL) { -+ *status = W_EXITCODE(code.u.exit_signal.exit_code, code.u.exit_signal.signal); -+ } - } else if (code.tag == __WASI_JOIN_STATUS_TYPE_STOPPED) { -- *status = W_STOPCODE(code.u.stopped); -+ if (status != NULL) { -+ *status = W_STOPCODE(code.u.stopped); -+ } - } else { - errno = EUNKNOWN; - return -1; -diff --git a/libc-top-half/musl/src/setjmp/setjmplongjmp.c b/libc-top-half/musl/src/setjmp/setjmplongjmp.c -index 05ebc4a..a1fab64 100644 ---- a/libc-top-half/musl/src/setjmp/setjmplongjmp.c -+++ b/libc-top-half/musl/src/setjmp/setjmplongjmp.c -@@ -1,4 +1,5 @@ - #ifndef __wasilibc_unmodified_upstream -+#define __WASIX_LIBC_BUILDING_SETJMP - #include - - # ifdef __wasm_exception_handling__ -@@ -93,9 +94,11 @@ int setjmp (jmp_buf buf) { - - # endif - -+# ifndef __wasm_exception_handling__ - // TODO: ignoring signal masking for now - int sigsetjmp(jmp_buf buf, int savesigs) { - return setjmp(buf); - } -+# endif - - #endif -diff --git a/libc-top-half/musl/src/signal/kill.c b/libc-top-half/musl/src/signal/kill.c -index ca43eb8..bb34cce 100644 ---- a/libc-top-half/musl/src/signal/kill.c -+++ b/libc-top-half/musl/src/signal/kill.c -@@ -2,6 +2,7 @@ - #ifdef __wasilibc_unmodified_upstream - #include "syscall.h" - #else -+#include - #include - #endif - -@@ -11,6 +12,10 @@ int kill(pid_t pid, int sig) - return syscall(SYS_kill, pid, sig); - #else - int r = __wasi_proc_signal(pid, (__wasi_signal_t)sig); -- return r; -+ if (r != 0) { -+ errno = r; -+ return -1; -+ } -+ return 0; - #endif --} -\ No newline at end of file -+} -diff --git a/libc-top-half/musl/src/signal/sigaction.c b/libc-top-half/musl/src/signal/sigaction.c -index d2ffaad..332feb2 100644 ---- a/libc-top-half/musl/src/signal/sigaction.c -+++ b/libc-top-half/musl/src/signal/sigaction.c -@@ -133,8 +133,7 @@ static void __default_handler(int sig) { - #endif - case SIGURG: - case SIGWINCH: -- SIG_IGN(sig); -- break; -+ return; - - // Default behavior: "continue". - case SIGCONT: -@@ -170,6 +169,18 @@ static void __default_handler(int sig) { - break; - } - } -+ -+static void __invoke_default_handler(int sig) { -+ /* -+ * The WASIX callback is registered at startup, before a program is -+ * required to call sigaction(). Materialize the built-in handler in the -+ * function table before the first default-disposition signal arrives. -+ */ -+ if (default_handler == NULL) { -+ default_handler = &__default_handler; -+ } -+ default_handler(sig); -+} - #endif - - static sighandler_t handlers[_NSIG]; -@@ -187,12 +198,14 @@ void __wasm_signal(int sig) { - struct k_sigaction ksa = __eintr_handler_callbacks[sig]; - UNLOCK(__eintr_handler_lock); - -- if (ksa.handler != 0) { -+ if (ksa.handler == SIG_IGN) { -+ return; -+ } else if (ksa.handler != 0 && ksa.handler != SIG_DFL) { - ksa.handler(sig); - } else { - unsigned long set[_NSIG/(8*sizeof(long))]; - __block_all_sigs(&set); -- default_handler(sig); -+ __invoke_default_handler(sig); - __restore_sigs(&set); - } - } -diff --git a/libc-top-half/musl/src/signal/siglongjmp.c b/libc-top-half/musl/src/signal/siglongjmp.c -index 577ecb5..4ceea6a 100644 ---- a/libc-top-half/musl/src/signal/siglongjmp.c -+++ b/libc-top-half/musl/src/signal/siglongjmp.c -@@ -8,5 +8,10 @@ - - _Noreturn void siglongjmp(sigjmp_buf buf, int ret) - { -+#if !defined(__wasilibc_unmodified_upstream) && defined(__wasm_exception_handling__) -+ if (buf->__fl) { -+ pthread_sigmask(SIG_SETMASK, (sigset_t *)buf->__ss, 0); -+ } -+#endif - longjmp(buf, ret); --} -\ No newline at end of file -+} -diff --git a/libc-top-half/musl/src/signal/sigsetjmp_tail.c b/libc-top-half/musl/src/signal/sigsetjmp_tail.c -index 57b00c5..f35da65 100644 ---- a/libc-top-half/musl/src/signal/sigsetjmp_tail.c -+++ b/libc-top-half/musl/src/signal/sigsetjmp_tail.c -@@ -1,6 +1,7 @@ - #include - #include - #include -+#include - #ifdef __wasilibc_unmodified_upstream - #include "syscall.h" - #endif -@@ -14,4 +15,18 @@ hidden int __sigsetjmp_tail(sigjmp_buf jb, int ret) - #else - return EINVAL; - #endif --} -\ No newline at end of file -+} -+ -+#if !defined(__wasilibc_unmodified_upstream) && defined(__wasm_exception_handling__) -+int __wasilibc_sigsetjmp_save(sigjmp_buf jb, int savesigs) -+{ -+ jb->__fl = savesigs ? 1 : 0; -+ if (!savesigs) { -+ return 0; -+ } -+ -+ memset(jb->__ss, 0, sizeof jb->__ss); -+ pthread_sigmask(SIG_SETMASK, NULL, (sigset_t *)jb->__ss); -+ return 0; -+} -+#endif -diff --git a/libc-top-half/musl/src/string/memcmp.c b/libc-top-half/musl/src/string/memcmp.c -index bdbce9f..3fbc3c5 100644 ---- a/libc-top-half/musl/src/string/memcmp.c -+++ b/libc-top-half/musl/src/string/memcmp.c -@@ -1,8 +1,43 @@ -+#include -+#include - #include - -+static uint64_t load64(const unsigned char *p) -+{ -+ uint64_t v; -+ __builtin_memcpy(&v, p, sizeof(v)); -+ return v; -+} -+ -+static uint32_t load32(const unsigned char *p) -+{ -+ uint32_t v; -+ __builtin_memcpy(&v, p, sizeof(v)); -+ return v; -+} -+ - int memcmp(const void *vl, const void *vr, size_t n) - { - const unsigned char *l=vl, *r=vr; -+#if defined(__GNUC__) && __BYTE_ORDER == __LITTLE_ENDIAN -+ for (; n >= 8; n-=8, l+=8, r+=8) { -+ uint64_t a = load64(l), b = load64(r); -+ if (a != b) { -+ unsigned shift = __builtin_ctzll(a ^ b) & ~7U; -+ return (int)((a >> shift) & 0xff) - (int)((b >> shift) & 0xff); -+ } -+ } -+ if (n >= 4) { -+ uint32_t a = load32(l), b = load32(r); -+ if (a != b) { -+ unsigned shift = __builtin_ctz(a ^ b) & ~7U; -+ return (int)((a >> shift) & 0xff) - (int)((b >> shift) & 0xff); -+ } -+ n-=4; -+ l+=4; -+ r+=4; -+ } -+#endif - for (; n && *l == *r; n--, l++, r++); - return n ? *l-*r : 0; - } -diff --git a/test/wasix/c_mmap_writeback/CMakeLists.txt b/test/wasix/c_mmap_writeback/CMakeLists.txt -new file mode 100644 -index 0000000..eb29426 ---- /dev/null -+++ b/test/wasix/c_mmap_writeback/CMakeLists.txt -@@ -0,0 +1,4 @@ -+cmake_minimum_required (VERSION 3.5.0) -+project (c_mmap_writeback) -+ -+add_executable(main main.c) -diff --git a/test/wasix/c_mmap_writeback/main.c b/test/wasix/c_mmap_writeback/main.c -new file mode 100644 -index 0000000..3ab7648 ---- /dev/null -+++ b/test/wasix/c_mmap_writeback/main.c -@@ -0,0 +1,194 @@ -+#include -+#include -+#include -+#include -+#include -+#include -+#include -+#include -+ -+static int write_full(int fd, const char *buf, size_t len) -+{ -+ while (len > 0) { -+ ssize_t written = write(fd, buf, len); -+ if (written < 0) { -+ if (errno == EINTR) { -+ continue; -+ } -+ return -1; -+ } -+ buf += (size_t)written; -+ len -= (size_t)written; -+ } -+ return 0; -+} -+ -+static int read_full(int fd, char *buf, size_t len) -+{ -+ while (len > 0) { -+ ssize_t got = read(fd, buf, len); -+ if (got < 0) { -+ if (errno == EINTR) { -+ continue; -+ } -+ return -1; -+ } -+ if (got == 0) { -+ errno = EIO; -+ return -1; -+ } -+ buf += (size_t)got; -+ len -= (size_t)got; -+ } -+ return 0; -+} -+ -+static int reset_file(int fd, const char *contents, size_t len) -+{ -+ if (ftruncate(fd, 0) < 0) { -+ return -1; -+ } -+ if (lseek(fd, 0, SEEK_SET) < 0) { -+ return -1; -+ } -+ if (write_full(fd, contents, len) < 0) { -+ return -1; -+ } -+ return lseek(fd, 0, SEEK_SET) < 0 ? -1 : 0; -+} -+ -+static int expect_file(int fd, const char *expected, size_t len) -+{ -+ char buf[32]; -+ -+ if (len > sizeof(buf)) { -+ return -1; -+ } -+ memset(buf, 0, sizeof(buf)); -+ if (lseek(fd, 0, SEEK_SET) < 0) { -+ return -1; -+ } -+ if (read_full(fd, buf, len) < 0) { -+ return -1; -+ } -+ return memcmp(buf, expected, len); -+} -+ -+int main(void) -+{ -+ const char *path = "wasix-libc-mmap-writeback-test.dat"; -+ const char initial[] = "abcdefgh"; -+ int fd; -+ char *map; -+ void *frontier_after_map; -+ unsigned char *morecore_segment; -+ size_t i; -+ -+ unlink(path); -+ fd = open(path, O_RDWR | O_CREAT | O_TRUNC, 0600); -+ if (fd < 0) { -+ perror("open"); -+ return 1; -+ } -+ if (reset_file(fd, initial, sizeof(initial) - 1) < 0) { -+ perror("reset shared"); -+ return 2; -+ } -+ -+ map = mmap(NULL, sizeof(initial) - 1, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0); -+ if (map == MAP_FAILED) { -+ perror("mmap shared"); -+ return 3; -+ } -+ errno = 0; -+ if (munmap(map + 1, 1) == 0 || errno != EINVAL) { -+ fprintf(stderr, "overlapping unaligned munmap escaped runtime ownership\n"); -+ return 18; -+ } -+ errno = 0; -+ if (msync(map + 1, 1, MS_SYNC) == 0 || errno != EINVAL) { -+ fprintf(stderr, "overlapping unaligned msync escaped runtime ownership\n"); -+ return 19; -+ } -+ if (memcmp(map, initial, sizeof(initial) - 1) != 0) { -+ fprintf(stderr, "rejected shared operations mutated the mapping\n"); -+ return 20; -+ } -+ frontier_after_map = sbrk(0); -+ morecore_segment = sbrk(65536); -+ if (morecore_segment == (void *)-1) { -+ perror("sbrk after shared mmap"); -+ return 13; -+ } -+ if (morecore_segment != frontier_after_map || -+ sbrk(0) != morecore_segment + 65536) { -+ fprintf(stderr, "sbrk did not return its exact exclusive interval\n"); -+ return 14; -+ } -+ if ((uintptr_t)morecore_segment < (uintptr_t)map + 65536 && -+ (uintptr_t)map < (uintptr_t)morecore_segment + 65536) { -+ fprintf(stderr, "MORECORE overlapped the runtime-owned mmap range\n"); -+ return 15; -+ } -+ memset(morecore_segment, 0x5a, 65536); -+ memcpy(map, "WXYZ", 4); -+ for (i = 1; i <= 256; ++i) { -+ unsigned char *allocation = malloc(i * 17); -+ if (!allocation) { -+ fprintf(stderr, "malloc failed after nonconsecutive MORECORE\n"); -+ return 16; -+ } -+ memset(allocation, (int)(i & 0xff), i * 17); -+ free(allocation); -+ } -+ if (morecore_segment[0] != 0x5a || morecore_segment[65535] != 0x5a || -+ memcmp(map, "WXYZ", 4) != 0) { -+ fprintf(stderr, "allocator or mmap sentinel was corrupted\n"); -+ return 17; -+ } -+ if (msync(map, sizeof(initial) - 1, MS_SYNC) < 0) { -+ perror("msync shared"); -+ return 4; -+ } -+ if (expect_file(fd, "WXYZefgh", sizeof(initial) - 1) != 0) { -+ fprintf(stderr, "MAP_SHARED changes were not written back\n"); -+ return 5; -+ } -+ if (munmap(map, sizeof(initial) - 1) < 0) { -+ perror("munmap shared"); -+ return 6; -+ } -+ errno = 0; -+ map = mmap(NULL, sizeof(initial) - 1, PROT_READ, MAP_SHARED, fd, 0); -+ if (map != MAP_FAILED || errno != EINVAL) { -+ fprintf(stderr, "runtime mapping accepted an unenforced protection contract\n"); -+ return 21; -+ } -+ -+ if (reset_file(fd, initial, sizeof(initial) - 1) < 0) { -+ perror("reset private"); -+ return 7; -+ } -+ map = mmap(NULL, sizeof(initial) - 1, PROT_READ | PROT_WRITE, MAP_PRIVATE, fd, 0); -+ if (map == MAP_FAILED) { -+ perror("mmap private"); -+ return 8; -+ } -+ memcpy(map, "PRIV", 4); -+ if (msync(map, sizeof(initial) - 1, MS_SYNC) < 0) { -+ perror("msync private"); -+ return 9; -+ } -+ if (munmap(map, sizeof(initial) - 1) < 0) { -+ perror("munmap private"); -+ return 10; -+ } -+ if (expect_file(fd, initial, sizeof(initial) - 1) != 0) { -+ fprintf(stderr, "MAP_PRIVATE changes were written back\n"); -+ return 11; -+ } -+ -+ close(fd); -+ unlink(path); -+ return 0; -+} -diff --git a/test/wasix/c_mmap_writeback/test.sh b/test/wasix/c_mmap_writeback/test.sh -new file mode 100755 -index 0000000..3c120c4 ---- /dev/null -+++ b/test/wasix/c_mmap_writeback/test.sh -@@ -0,0 +1,10 @@ -+#!/bin/bash -+ -+wasmer run --verbose --enable-all ./main -+RESULT=$? -+if [ "$RESULT" != "0" ]; then -+ echo "Test failed: different exit code ($RESULT vs. 0)" > /dev/stderr -+ exit 1 -+fi -+ -+echo "c_mmap_writeback test passed" -diff --git a/test/wasix/c_sync_file_range/CMakeLists.txt b/test/wasix/c_sync_file_range/CMakeLists.txt -new file mode 100644 -index 0000000..c3e1c79 ---- /dev/null -+++ b/test/wasix/c_sync_file_range/CMakeLists.txt -@@ -0,0 +1,4 @@ -+cmake_minimum_required(VERSION 3.5.0) -+project(c_sync_file_range C) -+ -+add_executable(main main.c) -diff --git a/test/wasix/c_sync_file_range/check-import-signature.sh b/test/wasix/c_sync_file_range/check-import-signature.sh -new file mode 100755 -index 0000000..523b83b ---- /dev/null -+++ b/test/wasix/c_sync_file_range/check-import-signature.sh -@@ -0,0 +1,29 @@ -+#!/bin/bash -+ -+set -euo pipefail -+ -+module="${1:-./main}" -+wasm_dis="${WASM_DIS:-wasm-dis}" -+expected='(import "oliphaunt_postmaster_v1" "fd_sync_range" (func $__imported_oliphaunt_postmaster_v1_fd_sync_range (param i32 i64 i64 i32) (result i32)))' -+ -+command -v "$wasm_dis" >/dev/null 2>&1 || { -+ echo "missing wasm disassembler: $wasm_dis" >&2 -+ exit 2 -+} -+[ -f "$module" ] || { -+ echo "missing Wasm module: $module" >&2 -+ exit 2 -+} -+ -+"$wasm_dis" "$module" -o - | grep -F "$expected" >/dev/null || { -+ echo "fd_sync_range import does not match the versioned ABI" >&2 -+ exit 1 -+} -+ -+if "$wasm_dis" "$module" -o - | -+ grep -F '(import "wasix_32v1" "fd_sync_range"' >/dev/null; then -+ echo "fd_sync_range must not be aliased into wasix_32v1" >&2 -+ exit 1 -+fi -+ -+echo "c_sync_file_range import signature passed" -diff --git a/test/wasix/c_sync_file_range/main.c b/test/wasix/c_sync_file_range/main.c -new file mode 100644 -index 0000000..90fdc92 ---- /dev/null -+++ b/test/wasix/c_sync_file_range/main.c -@@ -0,0 +1,82 @@ -+#define _GNU_SOURCE -+ -+#include -+#include -+#include -+#include -+#include -+ -+_Static_assert(SYNC_FILE_RANGE_WAIT_BEFORE == 1, "WAIT_BEFORE ABI bit changed"); -+_Static_assert(SYNC_FILE_RANGE_WRITE == 2, "WRITE ABI bit changed"); -+_Static_assert(SYNC_FILE_RANGE_WAIT_AFTER == 4, "WAIT_AFTER ABI bit changed"); -+ -+static int expect_einval(int fd, off_t offset, off_t len, unsigned flags, -+ const char *label) -+{ -+ errno = 0; -+ if (sync_file_range(fd, offset, len, flags) != -1 || errno != EINVAL) { -+ fprintf(stderr, "%s: expected EINVAL, got errno=%d\n", label, errno); -+ return -1; -+ } -+ return 0; -+} -+ -+int main(void) -+{ -+ static const char path[] = "wasix-libc-sync-file-range-test.dat"; -+ char page[4096] = {1}; -+ int fd; -+ -+ unlink(path); -+ fd = open(path, O_RDWR | O_CREAT | O_TRUNC, 0600); -+ if (fd < 0) { -+ perror("open"); -+ return 1; -+ } -+ if (write(fd, page, sizeof(page)) != (ssize_t)sizeof(page)) { -+ perror("write"); -+ return 2; -+ } -+ -+ if (sync_file_range(fd, 0, sizeof(page), SYNC_FILE_RANGE_WRITE) < 0) { -+ perror("sync_file_range WRITE"); -+ return 3; -+ } -+ if (sync_file_range(fd, 0, sizeof(page), -+ SYNC_FILE_RANGE_WAIT_BEFORE | -+ SYNC_FILE_RANGE_WRITE | -+ SYNC_FILE_RANGE_WAIT_AFTER) < 0) { -+ perror("sync_file_range full combination"); -+ return 4; -+ } -+ if (sync_file_range(fd, 0, 0, 0) < 0) { -+ perror("sync_file_range through EOF no-op"); -+ return 5; -+ } -+ if (sync_file_range(fd, INT64_MAX - 1, 1, 0) < 0) { -+ perror("sync_file_range maximum finite range"); -+ return 6; -+ } -+ -+ if (expect_einval(-1, 0, 0, 8, "unknown flags before import") < 0 || -+ expect_einval(fd, -1, 0, 0, "negative offset before import") < 0 || -+ expect_einval(fd, 0, -1, 0, "negative length before import") < 0 || -+ expect_einval(fd, INT64_MAX, 1, 0, "overflow before import") < 0 || -+ expect_einval(fd, INT64_MAX - 1, 2, 0, "overflow before import") < 0) { -+ return 7; -+ } -+ -+ /* flags=0 is legal but must still validate the descriptor in the host. */ -+ errno = 0; -+ if (sync_file_range(-1, 0, 0, 0) != -1 || errno != EBADF) { -+ fprintf(stderr, "flags=0 did not delegate: errno=%d\n", errno); -+ return 8; -+ } -+ -+ if (close(fd) < 0) { -+ perror("close"); -+ return 9; -+ } -+ unlink(path); -+ return 0; -+} -diff --git a/test/wasix/c_sync_file_range/test.sh b/test/wasix/c_sync_file_range/test.sh -new file mode 100755 -index 0000000..036c772 ---- /dev/null -+++ b/test/wasix/c_sync_file_range/test.sh -@@ -0,0 +1,10 @@ -+#!/bin/bash -+ -+wasmer run --verbose --enable-all ./main -+RESULT=$? -+if [ "$RESULT" != "0" ]; then -+ echo "Test failed: different exit code ($RESULT vs. 0)" > /dev/stderr -+ exit 1 -+fi -+ -+echo "c_sync_file_range test passed" diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0002-file-flags-and-sync-range.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0002-file-flags-and-sync-range.patch new file mode 100644 index 000000000..e6ce4e8e4 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0002-file-flags-and-sync-range.patch @@ -0,0 +1,259 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 2/8] libc: preserve file flags and expose sync_file_range + +Carry descriptor/open flags and the WASIX sync_file_range import with its +public constants/prototype and executable/import-signature tests. PostgreSQL +needs explicit writeback semantics without silently pretending a missing +syscall succeeded. Pair this with the host fd_sync_range implementation. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c b/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c +index 06b7288..0526fce 100644 +--- a/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c ++++ b/libc-bottom-half/cloudlibc/src/libc/fcntl/fcntl.c +@@ -24,7 +24,8 @@ int fcntl(int fildes, int cmd, ...) { + int flags = va_arg(ap, int); + va_end(ap); + +- __wasi_fdflagsext_t fd_flags = flags | FD_CLOEXEC ? __WASI_FDFLAGSEXT_CLOEXEC : 0; ++ __wasi_fdflagsext_t fd_flags = ++ (flags & FD_CLOEXEC) ? __WASI_FDFLAGSEXT_CLOEXEC : 0; + __wasi_errno_t error = + __wasi_fd_fdflags_set(fildes, fd_flags); + if (error != 0) { +diff --git a/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c b/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c +index 8e469c6..f7d8034 100644 +--- a/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c ++++ b/libc-bottom-half/cloudlibc/src/libc/fcntl/openat.c +@@ -52,6 +52,14 @@ int __wasilibc_nocwd_openat_nomode(int fd, const char *path, int oflag) { + return -1; + } + ++ // A directory descriptor is opened read-only, but its contents are ++ // metadata and POSIX permits both fsync() and fdatasync() to establish the ++ // same directory durability barrier. Request both capabilities explicitly; ++ // otherwise fdatasync() is rejected by the WASI rights check before the ++ // runtime can synchronize the retained open directory description. ++ if ((oflag & O_DIRECTORY) != 0) ++ max |= __WASI_RIGHTS_FD_DATASYNC | __WASI_RIGHTS_FD_SYNC; ++ + // Ensure that we can actually obtain the minimal rights needed. + __wasi_fdstat_t fsb_cur; + __wasi_errno_t error = __wasi_fd_fdstat_get(fd, &fsb_cur); +diff --git a/libc-bottom-half/cloudlibc/src/libc/fcntl/sync_file_range.c b/libc-bottom-half/cloudlibc/src/libc/fcntl/sync_file_range.c +new file mode 100644 +index 0000000..47cdad7 +--- /dev/null ++++ b/libc-bottom-half/cloudlibc/src/libc/fcntl/sync_file_range.c +@@ -0,0 +1,33 @@ ++#define _GNU_SOURCE ++ ++#include ++#include ++#include ++ ++int32_t __imported_oliphaunt_postmaster_v1_fd_sync_range( ++ int32_t fd, int64_t offset, int64_t nbytes, int32_t flags) __attribute__(( ++ __import_module__("oliphaunt_postmaster_v1"), ++ __import_name__("fd_sync_range"))); ++ ++int sync_file_range(int fd, off_t offset, off_t nbytes, unsigned flags) ++{ ++ const unsigned valid_flags = SYNC_FILE_RANGE_WAIT_BEFORE | ++ SYNC_FILE_RANGE_WRITE | ++ SYNC_FILE_RANGE_WAIT_AFTER; ++ int32_t error; ++ ++ /* Linux requires a representable exclusive end; nbytes=0 means EOF. */ ++ if (offset < 0 || nbytes < 0 || (flags & ~valid_flags) != 0 || ++ (nbytes > 0 && offset > INT64_MAX - nbytes)) { ++ errno = EINVAL; ++ return -1; ++ } ++ ++ error = __imported_oliphaunt_postmaster_v1_fd_sync_range( ++ (int32_t)fd, (int64_t)offset, (int64_t)nbytes, (int32_t)flags); ++ if (error != 0) { ++ errno = error; ++ return -1; ++ } ++ return 0; ++} +diff --git a/libc-top-half/musl/include/fcntl.h b/libc-top-half/musl/include/fcntl.h +index 15bef28..1332569 100644 +--- a/libc-top-half/musl/include/fcntl.h ++++ b/libc-top-half/musl/include/fcntl.h +@@ -190,11 +190,9 @@ struct f_owner_ex { + #ifdef __wasilibc_unmodified_upstream /* WASI has no name_to_handle_at */ + #define MAX_HANDLE_SZ 128 + #endif +-#ifdef __wasilibc_unmodified_upstream /* WASI has no syc_file_range */ + #define SYNC_FILE_RANGE_WAIT_BEFORE 1 + #define SYNC_FILE_RANGE_WRITE 2 + #define SYNC_FILE_RANGE_WAIT_AFTER 4 +-#endif + #ifdef __wasilibc_unmodified_upstream /* WASI has no splice */ + #define SPLICE_F_MOVE 1 + #define SPLICE_F_NONBLOCK 2 +@@ -212,8 +210,8 @@ int open_by_handle_at(int, struct file_handle *, int); + #ifdef __wasilibc_unmodified_upstream /* WASI has no readahead */ + ssize_t readahead(int, off_t, size_t); + #endif +-#ifdef __wasilibc_unmodified_upstream /* WASI has no splice, syc_file_range, or tee */ + int sync_file_range(int, off_t, off_t, unsigned); ++#ifdef __wasilibc_unmodified_upstream /* WASI has no splice or tee */ + ssize_t vmsplice(int, const struct iovec *, size_t, unsigned); + ssize_t splice(int, off_t *, int, off_t *, size_t, unsigned); + ssize_t tee(int, int, size_t, unsigned); +diff --git a/test/wasix/c_sync_file_range/CMakeLists.txt b/test/wasix/c_sync_file_range/CMakeLists.txt +new file mode 100644 +index 0000000..c3e1c79 +--- /dev/null ++++ b/test/wasix/c_sync_file_range/CMakeLists.txt +@@ -0,0 +1,4 @@ ++cmake_minimum_required(VERSION 3.5.0) ++project(c_sync_file_range C) ++ ++add_executable(main main.c) +diff --git a/test/wasix/c_sync_file_range/check-import-signature.sh b/test/wasix/c_sync_file_range/check-import-signature.sh +new file mode 100755 +index 0000000..523b83b +--- /dev/null ++++ b/test/wasix/c_sync_file_range/check-import-signature.sh +@@ -0,0 +1,29 @@ ++#!/bin/bash ++ ++set -euo pipefail ++ ++module="${1:-./main}" ++wasm_dis="${WASM_DIS:-wasm-dis}" ++expected='(import "oliphaunt_postmaster_v1" "fd_sync_range" (func $__imported_oliphaunt_postmaster_v1_fd_sync_range (param i32 i64 i64 i32) (result i32)))' ++ ++command -v "$wasm_dis" >/dev/null 2>&1 || { ++ echo "missing wasm disassembler: $wasm_dis" >&2 ++ exit 2 ++} ++[ -f "$module" ] || { ++ echo "missing Wasm module: $module" >&2 ++ exit 2 ++} ++ ++"$wasm_dis" "$module" -o - | grep -F "$expected" >/dev/null || { ++ echo "fd_sync_range import does not match the versioned ABI" >&2 ++ exit 1 ++} ++ ++if "$wasm_dis" "$module" -o - | ++ grep -F '(import "wasix_32v1" "fd_sync_range"' >/dev/null; then ++ echo "fd_sync_range must not be aliased into wasix_32v1" >&2 ++ exit 1 ++fi ++ ++echo "c_sync_file_range import signature passed" +diff --git a/test/wasix/c_sync_file_range/main.c b/test/wasix/c_sync_file_range/main.c +new file mode 100644 +index 0000000..90fdc92 +--- /dev/null ++++ b/test/wasix/c_sync_file_range/main.c +@@ -0,0 +1,82 @@ ++#define _GNU_SOURCE ++ ++#include ++#include ++#include ++#include ++#include ++ ++_Static_assert(SYNC_FILE_RANGE_WAIT_BEFORE == 1, "WAIT_BEFORE ABI bit changed"); ++_Static_assert(SYNC_FILE_RANGE_WRITE == 2, "WRITE ABI bit changed"); ++_Static_assert(SYNC_FILE_RANGE_WAIT_AFTER == 4, "WAIT_AFTER ABI bit changed"); ++ ++static int expect_einval(int fd, off_t offset, off_t len, unsigned flags, ++ const char *label) ++{ ++ errno = 0; ++ if (sync_file_range(fd, offset, len, flags) != -1 || errno != EINVAL) { ++ fprintf(stderr, "%s: expected EINVAL, got errno=%d\n", label, errno); ++ return -1; ++ } ++ return 0; ++} ++ ++int main(void) ++{ ++ static const char path[] = "wasix-libc-sync-file-range-test.dat"; ++ char page[4096] = {1}; ++ int fd; ++ ++ unlink(path); ++ fd = open(path, O_RDWR | O_CREAT | O_TRUNC, 0600); ++ if (fd < 0) { ++ perror("open"); ++ return 1; ++ } ++ if (write(fd, page, sizeof(page)) != (ssize_t)sizeof(page)) { ++ perror("write"); ++ return 2; ++ } ++ ++ if (sync_file_range(fd, 0, sizeof(page), SYNC_FILE_RANGE_WRITE) < 0) { ++ perror("sync_file_range WRITE"); ++ return 3; ++ } ++ if (sync_file_range(fd, 0, sizeof(page), ++ SYNC_FILE_RANGE_WAIT_BEFORE | ++ SYNC_FILE_RANGE_WRITE | ++ SYNC_FILE_RANGE_WAIT_AFTER) < 0) { ++ perror("sync_file_range full combination"); ++ return 4; ++ } ++ if (sync_file_range(fd, 0, 0, 0) < 0) { ++ perror("sync_file_range through EOF no-op"); ++ return 5; ++ } ++ if (sync_file_range(fd, INT64_MAX - 1, 1, 0) < 0) { ++ perror("sync_file_range maximum finite range"); ++ return 6; ++ } ++ ++ if (expect_einval(-1, 0, 0, 8, "unknown flags before import") < 0 || ++ expect_einval(fd, -1, 0, 0, "negative offset before import") < 0 || ++ expect_einval(fd, 0, -1, 0, "negative length before import") < 0 || ++ expect_einval(fd, INT64_MAX, 1, 0, "overflow before import") < 0 || ++ expect_einval(fd, INT64_MAX - 1, 2, 0, "overflow before import") < 0) { ++ return 7; ++ } ++ ++ /* flags=0 is legal but must still validate the descriptor in the host. */ ++ errno = 0; ++ if (sync_file_range(-1, 0, 0, 0) != -1 || errno != EBADF) { ++ fprintf(stderr, "flags=0 did not delegate: errno=%d\n", errno); ++ return 8; ++ } ++ ++ if (close(fd) < 0) { ++ perror("close"); ++ return 9; ++ } ++ unlink(path); ++ return 0; ++} +diff --git a/test/wasix/c_sync_file_range/test.sh b/test/wasix/c_sync_file_range/test.sh +new file mode 100755 +index 0000000..036c772 +--- /dev/null ++++ b/test/wasix/c_sync_file_range/test.sh +@@ -0,0 +1,10 @@ ++#!/bin/bash ++ ++wasmer run --verbose --enable-all ./main ++RESULT=$? ++if [ "$RESULT" != "0" ]; then ++ echo "Test failed: different exit code ($RESULT vs. 0)" > /dev/stderr ++ exit 1 ++fi ++ ++echo "c_sync_file_range test passed" diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0003-socket-and-epoll-flags.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0003-socket-and-epoll-flags.patch new file mode 100644 index 000000000..250f7c849 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0003-socket-and-epoll-flags.patch @@ -0,0 +1,94 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 3/8] libc: preserve socket and epoll descriptor flags + +Preserve nonblocking and close-on-exec behavior when libc constructs sockets +and epoll descriptors. PostgreSQL's listener and child event loop require +those flags to survive the libc-to-WASIX boundary. The matching host readiness +and descriptor-lifetime patches are required. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c b/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c +index 74caf2e..2ab7514 100644 +--- a/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c ++++ b/libc-bottom-half/cloudlibc/src/libc/sys/socket/socket.c +@@ -4,10 +4,20 @@ + #include + #include + #include ++#include + #include ++#include + + int socket(int domain, int ty, int protocol) { + int fd; ++ int flags = ty & (SOCK_NONBLOCK | SOCK_CLOEXEC); ++ ty &= ~(SOCK_NONBLOCK | SOCK_CLOEXEC); ++ ++ if (ty != SOCK_STREAM && ty != SOCK_DGRAM) { ++ errno = EPROTONOSUPPORT; ++ return -1; ++ } ++ + if(!protocol) { + switch (ty) + { +@@ -25,5 +35,14 @@ int socket(int domain, int ty, int protocol) { + return -1; + } + ++ if ((flags & SOCK_CLOEXEC) != 0 && fcntl(fd, F_SETFD, FD_CLOEXEC) < 0) { ++ close(fd); ++ return -1; ++ } ++ if ((flags & SOCK_NONBLOCK) != 0 && fcntl(fd, F_SETFL, O_NONBLOCK) < 0) { ++ close(fd); ++ return -1; ++ } ++ + return fd; + } +diff --git a/libc-top-half/musl/src/linux/epoll.c b/libc-top-half/musl/src/linux/epoll.c +index 4780ce2..b3ac891 100644 +--- a/libc-top-half/musl/src/linux/epoll.c ++++ b/libc-top-half/musl/src/linux/epoll.c +@@ -2,6 +2,8 @@ + #include + #include + #include ++#include ++#include + + int epoll_create(int size) + { +@@ -17,7 +19,27 @@ int epoll_create(int size) + + int epoll_create1(int flags) + { +- return epoll_create(0); ++ int fd; ++ ++ if (flags & ~EPOLL_CLOEXEC) { ++ errno = EINVAL; ++ return -1; ++ } ++ ++ fd = epoll_create(0); ++ if (fd < 0) { ++ return -1; ++ } ++ ++ if ((flags & EPOLL_CLOEXEC) && fcntl(fd, F_SETFD, FD_CLOEXEC) < 0) { ++ int saved_errno = errno; ++ ++ close(fd); ++ errno = saved_errno; ++ return -1; ++ } ++ ++ return fd; + } + + int epoll_ctl(int fd, int op, int fd2, struct epoll_event *ev) diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0004-process-wait-signal-and-futex-semantics.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0004-process-wait-signal-and-futex-semantics.patch new file mode 100644 index 000000000..d153098dd --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0004-process-wait-signal-and-futex-semantics.patch @@ -0,0 +1,345 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 4/8] libc: adapt exec, wait, signals and futex clocks + +Retain process entrypoints for the Wasm exception-handling build, preserve exec +argument/search behavior, and map wait, signal and futex semantics required by +Postmaster's concurrent children. This layer must match host lifecycle and +clock domains; native tests alone do not qualify the Wasm/exec behavior. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/libc-bottom-half/sources/__wasilibc_futex.c b/libc-bottom-half/sources/__wasilibc_futex.c +index 340899b..1a453fa 100644 +--- a/libc-bottom-half/sources/__wasilibc_futex.c ++++ b/libc-bottom-half/sources/__wasilibc_futex.c +@@ -12,7 +12,7 @@ int __wasilibc_futex_wait_wasix(volatile void *addr, int op, int expected, int64 + __wasi_bool_t woken = __WASI_BOOL_FALSE; + + __wasi_option_timestamp_t timeout; +- if (max_wait_ns > 0) { ++ if (max_wait_ns >= 0) { + timeout.tag = __WASI_OPTION_SOME; + timeout.u.some = max_wait_ns; + } else { +@@ -30,8 +30,9 @@ int __wasilibc_futex_wait_wasix(volatile void *addr, int op, int expected, int64 + return -EWOULDBLOCK; + } + +- if (__wasi_futex_wait((uint32_t*)addr, expected, &timeout, &woken) != 0) { +- __builtin_trap(); ++ __wasi_errno_t ret = __wasi_futex_wait((uint32_t*)addr, expected, &timeout, &woken); ++ if (ret != 0) { ++ return -ret; + } + + if (woken == __WASI_BOOL_FALSE && *paddr == expected) { +diff --git a/libc-top-half/musl/include/unistd.h b/libc-top-half/musl/include/unistd.h +index 95a3ee7..7c30563 100644 +--- a/libc-top-half/musl/include/unistd.h ++++ b/libc-top-half/musl/include/unistd.h +@@ -134,11 +134,9 @@ unsigned sleep(unsigned); + int pause(void); + #endif + +-#if defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) + pid_t fork(void); + pid_t _fork_internal(int copy_mem); + pid_t _Fork(int copy_mem); +-#endif + int execve(const char *, char *const [], char *const []); + int execv(const char *, char *const []); + int execle(const char *, const char *, ...); +diff --git a/libc-top-half/musl/src/process/_Fork.c b/libc-top-half/musl/src/process/_Fork.c +index 76df425..8f0ee8e 100644 +--- a/libc-top-half/musl/src/process/_Fork.c ++++ b/libc-top-half/musl/src/process/_Fork.c +@@ -10,8 +10,6 @@ + #include "pthread_impl.h" + #include "aio_impl.h" + +-#if defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) +- + static void dummy(int x) { } + weak_alias(dummy, __aio_atfork); + +@@ -62,5 +60,3 @@ pid_t _Fork(int copy_mem) + __restore_sigs(&set); + return __syscall_ret(ret); + } +- +-#endif /* defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) */ +\ No newline at end of file +diff --git a/libc-top-half/musl/src/process/execv.c b/libc-top-half/musl/src/process/execv.c +index bce72e3..5ef4ebe 100644 +--- a/libc-top-half/musl/src/process/execv.c ++++ b/libc-top-half/musl/src/process/execv.c +@@ -7,6 +7,7 @@ extern char **__environ; + #include + #include + #include ++char *__wasilibc_exec_combine_strings(char *const strings[]); + #endif + + int execv(const char *path, char *const argv[]) +@@ -14,22 +15,16 @@ int execv(const char *path, char *const argv[]) + #ifdef __wasilibc_unmodified_upstream + return execve(path, argv, __environ); + #else +- int combined_len = 0; +- for (char **argvp = (char **)argv; *argvp != NULL; argvp++) { +- combined_len += strlen(*argvp) + 1; ++ char *combined_argv = __wasilibc_exec_combine_strings(argv); ++ if (combined_argv == NULL) { ++ return -1; + } + +- char *combined_argv = malloc((combined_len + 1)); +- char *combined_argv_p = combined_argv; +- for (char **argvp = (char **)argv; *argvp != NULL; argvp++) { +- memcpy(combined_argv_p, *argvp, strlen(*argvp)); +- combined_argv_p += strlen(*argvp); +- *combined_argv_p = '\n'; +- combined_argv_p++; +- } +- *combined_argv_p = 0; +- + int e = __wasi_proc_exec3(path, combined_argv, NULL, 0, NULL); ++ /* A successful vfork+exec returns in the shared parent image immediately ++ before the restore longjmp. Release child marshaling storage first so ++ every backend launch returns the parent heap to its prior plateau. */ ++ free(combined_argv); + #ifdef __wasm_exception_handling__ + extern _Noreturn void __vfork_restore(); + if (e == 0) { +@@ -37,7 +32,7 @@ int execv(const char *path, char *const argv[]) + } + #endif + +- // A return from proc_exec automatically means it failed ++ // Any ordinary return from proc_exec means it failed. + errno = e; + return -1; + #endif +diff --git a/libc-top-half/musl/src/process/execvp.c b/libc-top-half/musl/src/process/execvp.c +index b26c275..0c5d1d5 100644 +--- a/libc-top-half/musl/src/process/execvp.c ++++ b/libc-top-half/musl/src/process/execvp.c +@@ -3,6 +3,7 @@ + #include + #include + #include ++#include + + extern char **__wasilibc_environ; + +@@ -64,15 +65,31 @@ int __execvpe(const char *file, char *const argv[], char *const envp[]) + #else + char *__wasilibc_exec_combine_strings(char *const strings[]) + { +- int combined_len = 0; +- for (char **ptr = (char **)strings; *ptr != NULL; ptr++) ++ size_t combined_len = 0; ++ for (char **ptr = (char **)strings; strings != NULL && *ptr != NULL; ptr++) + { +- combined_len += strlen(*ptr) + 1; ++ size_t len = strlen(*ptr); ++ if (len == SIZE_MAX || combined_len > SIZE_MAX - len - 1) ++ { ++ errno = E2BIG; ++ return NULL; ++ } ++ combined_len += len + 1; + } + +- char *combined = malloc((combined_len + 1)); ++ if (combined_len == SIZE_MAX) ++ { ++ errno = E2BIG; ++ return NULL; ++ } ++ char *combined = malloc(combined_len + 1); ++ if (combined == NULL) ++ { ++ errno = ENOMEM; ++ return NULL; ++ } + char *combined_p = combined; +- for (char **ptr = (char **)strings; *ptr != NULL; ptr++) ++ for (char **ptr = (char **)strings; strings != NULL && *ptr != NULL; ptr++) + { + memcpy(combined_p, *ptr, strlen(*ptr)); + combined_p += strlen(*ptr); +@@ -87,11 +104,22 @@ char *__wasilibc_exec_combine_strings(char *const strings[]) + int __execvpe(const char *path, char *const argv[], char *const envp[], uint8_t use_path) + { + char *combined_argv = __wasilibc_exec_combine_strings(argv); ++ if (combined_argv == NULL) ++ return -1; + char *combined_env = __wasilibc_exec_combine_strings(envp); ++ if (combined_env == NULL) ++ { ++ free(combined_argv); ++ return -1; ++ } + + int e = __wasi_proc_exec3( + path, combined_argv, combined_env, + use_path ? __WASI_BOOL_TRUE : __WASI_BOOL_FALSE, getenv("PATH")); ++ /* proc_exec3 has consumed both buffers before returning. On vfork success ++ this is the last point before longjmp restores the shared parent image. */ ++ free(combined_argv); ++ free(combined_env); + #ifdef __wasm_exception_handling__ + extern _Noreturn void __vfork_restore(); + if (e == 0) { +@@ -99,9 +127,6 @@ int __execvpe(const char *path, char *const argv[], char *const envp[], uint8_t + } + #endif + +- free(combined_argv); +- free(combined_env); +- + // A return from proc_exec automatically means it failed + errno = e; + return -1; +diff --git a/libc-top-half/musl/src/process/fork.c b/libc-top-half/musl/src/process/fork.c +index df38cf7..21895b4 100644 +--- a/libc-top-half/musl/src/process/fork.c ++++ b/libc-top-half/musl/src/process/fork.c +@@ -51,8 +51,6 @@ weak_alias(dummy_0, __tl_lock); + weak_alias(dummy_0, __tl_unlock); + #endif + +-#if defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) +- + pid_t fork(void) + { + return _fork_internal(1); +@@ -102,5 +100,3 @@ pid_t _fork_internal(int copy_mem) + if (ret<0) errno = errno_save; + return ret; + } +- +-#endif /* defined(__wasilibc_unmodified_upstream) || !defined(__wasm_exception_handling__) */ +\ No newline at end of file +diff --git a/libc-top-half/musl/src/process/waitpid.c b/libc-top-half/musl/src/process/waitpid.c +index 456cca9..330aa66 100644 +--- a/libc-top-half/musl/src/process/waitpid.c ++++ b/libc-top-half/musl/src/process/waitpid.c +@@ -35,6 +35,13 @@ pid_t waitpid(pid_t pid, int *status, int options) + errno = ret; + return -1; + } else { ++ if (code.tag == __WASI_JOIN_STATUS_TYPE_NOTHING) { ++ if (status != NULL) { ++ *status = 0; ++ } ++ return 0; ++ } ++ + // Read the PID + if (opid.tag == __WASI_OPTION_SOME) { + pid = opid.u.some; +@@ -44,14 +51,18 @@ pid_t waitpid(pid_t pid, int *status, int options) + } + + // Build the status code depending on what happened +- if (code.tag == __WASI_JOIN_STATUS_TYPE_NOTHING) { +- *status = 0; +- } else if (code.tag == __WASI_JOIN_STATUS_TYPE_EXIT_NORMAL) { +- *status = W_EXITCODE(code.u.exit_normal, 0); ++ if (code.tag == __WASI_JOIN_STATUS_TYPE_EXIT_NORMAL) { ++ if (status != NULL) { ++ *status = W_EXITCODE(code.u.exit_normal, 0); ++ } + } else if (code.tag == __WASI_JOIN_STATUS_TYPE_EXIT_SIGNAL) { +- *status = W_EXITCODE(code.u.exit_signal.exit_code, code.u.exit_signal.signal); ++ if (status != NULL) { ++ *status = W_EXITCODE(code.u.exit_signal.exit_code, code.u.exit_signal.signal); ++ } + } else if (code.tag == __WASI_JOIN_STATUS_TYPE_STOPPED) { +- *status = W_STOPCODE(code.u.stopped); ++ if (status != NULL) { ++ *status = W_STOPCODE(code.u.stopped); ++ } + } else { + errno = EUNKNOWN; + return -1; +diff --git a/libc-top-half/musl/src/signal/kill.c b/libc-top-half/musl/src/signal/kill.c +index ca43eb8..bb34cce 100644 +--- a/libc-top-half/musl/src/signal/kill.c ++++ b/libc-top-half/musl/src/signal/kill.c +@@ -2,6 +2,7 @@ + #ifdef __wasilibc_unmodified_upstream + #include "syscall.h" + #else ++#include + #include + #endif + +@@ -11,6 +12,10 @@ int kill(pid_t pid, int sig) + return syscall(SYS_kill, pid, sig); + #else + int r = __wasi_proc_signal(pid, (__wasi_signal_t)sig); +- return r; ++ if (r != 0) { ++ errno = r; ++ return -1; ++ } ++ return 0; + #endif +-} +\ No newline at end of file ++} +diff --git a/libc-top-half/musl/src/signal/sigaction.c b/libc-top-half/musl/src/signal/sigaction.c +index d2ffaad..332feb2 100644 +--- a/libc-top-half/musl/src/signal/sigaction.c ++++ b/libc-top-half/musl/src/signal/sigaction.c +@@ -133,8 +133,7 @@ static void __default_handler(int sig) { + #endif + case SIGURG: + case SIGWINCH: +- SIG_IGN(sig); +- break; ++ return; + + // Default behavior: "continue". + case SIGCONT: +@@ -170,6 +169,18 @@ static void __default_handler(int sig) { + break; + } + } ++ ++static void __invoke_default_handler(int sig) { ++ /* ++ * The WASIX callback is registered at startup, before a program is ++ * required to call sigaction(). Materialize the built-in handler in the ++ * function table before the first default-disposition signal arrives. ++ */ ++ if (default_handler == NULL) { ++ default_handler = &__default_handler; ++ } ++ default_handler(sig); ++} + #endif + + static sighandler_t handlers[_NSIG]; +@@ -187,12 +198,14 @@ void __wasm_signal(int sig) { + struct k_sigaction ksa = __eintr_handler_callbacks[sig]; + UNLOCK(__eintr_handler_lock); + +- if (ksa.handler != 0) { ++ if (ksa.handler == SIG_IGN) { ++ return; ++ } else if (ksa.handler != 0 && ksa.handler != SIG_DFL) { + ksa.handler(sig); + } else { + unsigned long set[_NSIG/(8*sizeof(long))]; + __block_all_sigs(&set); +- default_handler(sig); ++ __invoke_default_handler(sig); + __restore_sigs(&set); + } + } diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0005-wasm-eh-signal-context.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0005-wasm-eh-signal-context.patch new file mode 100644 index 000000000..d43418874 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0005-wasm-eh-signal-context.patch @@ -0,0 +1,108 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 5/8] libc: establish EH signal jump context in the caller + +Capture signal jump state in the calling frame for the Wasm EH configuration; +install mask preparation/restoration hooks around longjmp. The pinned WASIX +pthread_sigmask backend is a success-returning stub: these hooks do not provide +actual guest signal masking. Header and implementation guards match, +including unmodified-upstream and libc-internal builds. The sigsetjmp macro +still evaluates its buffer more than once at this point in the series; patch +0008 fixes that without introducing a wrapper that captures the wrong frame. + +Extracted from the inherited integration bundle, with the missing EH build +guard corrected so non-EH builds do not reference an unavailable helper. +The complete ordered series is the build and qualification unit. + +diff --git a/libc-top-half/musl/include/setjmp.h b/libc-top-half/musl/include/setjmp.h +index 3961c02..f92f9f0 100644 +--- a/libc-top-half/musl/include/setjmp.h ++++ b/libc-top-half/musl/include/setjmp.h +@@ -48,7 +48,15 @@ _Noreturn void _longjmp (jmp_buf, int); + #if defined(_POSIX_SOURCE) || defined(_POSIX_C_SOURCE) || \ + defined(_XOPEN_SOURCE) || defined(_GNU_SOURCE) || defined(_BSD_SOURCE) + typedef jmp_buf sigjmp_buf; ++#if !defined(__wasilibc_unmodified_upstream) && \ ++ defined(__wasm_exception_handling__) && \ ++ !defined(__WASIX_LIBC_BUILDING_SETJMP) ++ int __wasilibc_sigsetjmp_save(sigjmp_buf, int); ++#define sigsetjmp(buf, savesigs) \ ++ (__wasilibc_sigsetjmp_save((buf), (savesigs)), setjmp((buf))) ++#else + int sigsetjmp(sigjmp_buf, int) __setjmp_attr; ++#endif + _Noreturn void siglongjmp(sigjmp_buf, int); + #endif + +diff --git a/libc-top-half/musl/src/setjmp/setjmplongjmp.c b/libc-top-half/musl/src/setjmp/setjmplongjmp.c +index 05ebc4a..a1fab64 100644 +--- a/libc-top-half/musl/src/setjmp/setjmplongjmp.c ++++ b/libc-top-half/musl/src/setjmp/setjmplongjmp.c +@@ -1,4 +1,5 @@ + #ifndef __wasilibc_unmodified_upstream ++#define __WASIX_LIBC_BUILDING_SETJMP + #include + + # ifdef __wasm_exception_handling__ +@@ -93,9 +94,11 @@ int setjmp (jmp_buf buf) { + + # endif + ++# ifndef __wasm_exception_handling__ + // TODO: ignoring signal masking for now + int sigsetjmp(jmp_buf buf, int savesigs) { + return setjmp(buf); + } ++# endif + + #endif +diff --git a/libc-top-half/musl/src/signal/siglongjmp.c b/libc-top-half/musl/src/signal/siglongjmp.c +index 577ecb5..4ceea6a 100644 +--- a/libc-top-half/musl/src/signal/siglongjmp.c ++++ b/libc-top-half/musl/src/signal/siglongjmp.c +@@ -8,5 +8,10 @@ + + _Noreturn void siglongjmp(sigjmp_buf buf, int ret) + { ++#if !defined(__wasilibc_unmodified_upstream) && defined(__wasm_exception_handling__) ++ if (buf->__fl) { ++ pthread_sigmask(SIG_SETMASK, (sigset_t *)buf->__ss, 0); ++ } ++#endif + longjmp(buf, ret); +-} +\ No newline at end of file ++} +diff --git a/libc-top-half/musl/src/signal/sigsetjmp_tail.c b/libc-top-half/musl/src/signal/sigsetjmp_tail.c +index 57b00c5..f35da65 100644 +--- a/libc-top-half/musl/src/signal/sigsetjmp_tail.c ++++ b/libc-top-half/musl/src/signal/sigsetjmp_tail.c +@@ -1,6 +1,7 @@ + #include + #include + #include ++#include + #ifdef __wasilibc_unmodified_upstream + #include "syscall.h" + #endif +@@ -14,4 +15,18 @@ hidden int __sigsetjmp_tail(sigjmp_buf jb, int ret) + #else + return EINVAL; + #endif +-} +\ No newline at end of file ++} ++ ++#if !defined(__wasilibc_unmodified_upstream) && defined(__wasm_exception_handling__) ++int __wasilibc_sigsetjmp_save(sigjmp_buf jb, int savesigs) ++{ ++ jb->__fl = savesigs ? 1 : 0; ++ if (!savesigs) { ++ return 0; ++ } ++ ++ memset(jb->__ss, 0, sizeof jb->__ss); ++ pthread_sigmask(SIG_SETMASK, NULL, (sigset_t *)jb->__ss); ++ return 0; ++} ++#endif diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0005-wasm-eh-signal-context.test.py b/src/wasix/postmaster/wasmer/patches/wasix-libc/0005-wasm-eh-signal-context.test.py new file mode 100644 index 000000000..45fec7513 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0005-wasm-eh-signal-context.test.py @@ -0,0 +1,36 @@ +#!/usr/bin/env python3 +"""Keep sigsetjmp's helper macro within its implementation's EH build scope.""" + +import os +from pathlib import Path +import shlex +import subprocess + + +patch = Path(__file__).with_suffix("").with_suffix(".patch").read_text() +section = patch.split("+++ b/libc-top-half/musl/include/setjmp.h\n", 1)[1] +section = section.split("\ndiff --git ", 1)[0] +lines = section.splitlines()[1:] +header = "\n".join(line[1:] for line in lines if line.startswith(("+", " "))) +source = header + "\nsigsetjmp(probe_buffer, 1);\n" + +for defines, helper_expected in ( + ([], False), + (["__wasm_exception_handling__"], True), + (["__wasm_exception_handling__", "__WASIX_LIBC_BUILDING_SETJMP"], False), + (["__wasm_exception_handling__", "__wasilibc_unmodified_upstream"], False), +): + preprocessed = subprocess.run( + shlex.split(os.environ.get("CC", "cc")) + + ["-E", "-P", "-x", "c", "-D_POSIX_SOURCE"] + + [f"-D{define}" for define in defines] + + ["-"], + input=source, + text=True, + capture_output=True, + check=True, + ).stdout + assert ("__wasilibc_sigsetjmp_save" in preprocessed) == helper_expected, defines + assert ("sigsetjmp(probe_buffer, 1)" not in preprocessed) == helper_expected, defines + +print("sigsetjmp helper matches EH, non-EH, libc-build and upstream header scopes") diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0006-process-resource-limit-import.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0006-process-resource-limit-import.patch new file mode 100644 index 000000000..c7e847444 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0006-process-resource-limit-import.patch @@ -0,0 +1,218 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 6/8] libc: query process resource limits through WASIX + +Expose proc_rlimit_get in the public ABI and import implementation so libc can +report the host-selected stack limit rather than a fabricated constant. Update +expected symbol inventories across supported libc variants with the same ABI. +Requires the host resource-limit syscall in the Wasmer process patch. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/expected/wasm32-wasi-eh/defined-symbols.txt b/expected/wasm32-wasi-eh/defined-symbols.txt +index 51b29a4..29811fd 100644 +--- a/expected/wasm32-wasi-eh/defined-symbols.txt ++++ b/expected/wasm32-wasi-eh/defined-symbols.txt +@@ -1650,6 +1650,7 @@ swprintf + swscanf + symlink + symlinkat ++sync_file_range + sysconf + syslog + system +diff --git a/expected/wasm32-wasi-eh/undefined-symbols.txt b/expected/wasm32-wasi-eh/undefined-symbols.txt +index 4f926be..b03cc61 100644 +--- a/expected/wasm32-wasi-eh/undefined-symbols.txt ++++ b/expected/wasm32-wasi-eh/undefined-symbols.txt +@@ -10,6 +10,7 @@ __floatsitf + __floatunsitf + __getf2 + __gttf2 ++__imported_oliphaunt_postmaster_v1_fd_sync_range + __imported_wasi_snapshot_preview1_args_get + __imported_wasi_snapshot_preview1_args_sizes_get + __imported_wasi_snapshot_preview1_clock_res_get +diff --git a/expected/wasm32-wasi-ehpic/defined-symbols.txt b/expected/wasm32-wasi-ehpic/defined-symbols.txt +index 3885a66..ddaacd9 100644 +--- a/expected/wasm32-wasi-ehpic/defined-symbols.txt ++++ b/expected/wasm32-wasi-ehpic/defined-symbols.txt +@@ -1658,6 +1658,7 @@ swprintf + swscanf + symlink + symlinkat ++sync_file_range + sysconf + syslog + system +diff --git a/expected/wasm32-wasi-ehpic/undefined-symbols.txt b/expected/wasm32-wasi-ehpic/undefined-symbols.txt +index 7af7e4c..49fbed2 100644 +--- a/expected/wasm32-wasi-ehpic/undefined-symbols.txt ++++ b/expected/wasm32-wasi-ehpic/undefined-symbols.txt +@@ -11,6 +11,7 @@ __floatsitf + __floatunsitf + __getf2 + __gttf2 ++__imported_oliphaunt_postmaster_v1_fd_sync_range + __imported_wasi_snapshot_preview1_args_get + __imported_wasi_snapshot_preview1_args_sizes_get + __imported_wasi_snapshot_preview1_clock_res_get +diff --git a/expected/wasm32-wasi-threads/defined-symbols.txt b/expected/wasm32-wasi-threads/defined-symbols.txt +index c8996ef..ad96b32 100644 +--- a/expected/wasm32-wasi-threads/defined-symbols.txt ++++ b/expected/wasm32-wasi-threads/defined-symbols.txt +@@ -1562,6 +1562,7 @@ swprintf + swscanf + symlink + symlinkat ++sync_file_range + sysconf + syslog + system +diff --git a/expected/wasm32-wasi-threads/undefined-symbols.txt b/expected/wasm32-wasi-threads/undefined-symbols.txt +index e4a2638..9145830 100644 +--- a/expected/wasm32-wasi-threads/undefined-symbols.txt ++++ b/expected/wasm32-wasi-threads/undefined-symbols.txt +@@ -13,6 +13,7 @@ __getf2 + __global_base + __gttf2 + __heap_base ++__imported_oliphaunt_postmaster_v1_fd_sync_range + __imported_wasi_snapshot_preview1_args_get + __imported_wasi_snapshot_preview1_args_sizes_get + __imported_wasi_snapshot_preview1_clock_res_get +diff --git a/expected/wasm32-wasi/defined-symbols.txt b/expected/wasm32-wasi/defined-symbols.txt +index 6d33bff..a7e915b 100644 +--- a/expected/wasm32-wasi/defined-symbols.txt ++++ b/expected/wasm32-wasi/defined-symbols.txt +@@ -1657,6 +1657,7 @@ swprintf + swscanf + symlink + symlinkat ++sync_file_range + sysconf + syslog + system +diff --git a/expected/wasm32-wasi/undefined-symbols.txt b/expected/wasm32-wasi/undefined-symbols.txt +index 4f926be..b03cc61 100644 +--- a/expected/wasm32-wasi/undefined-symbols.txt ++++ b/expected/wasm32-wasi/undefined-symbols.txt +@@ -10,6 +10,7 @@ __floatsitf + __floatunsitf + __getf2 + __gttf2 ++__imported_oliphaunt_postmaster_v1_fd_sync_range + __imported_wasi_snapshot_preview1_args_get + __imported_wasi_snapshot_preview1_args_sizes_get + __imported_wasi_snapshot_preview1_clock_res_get +diff --git a/expected/wasm64-wasi/defined-symbols.txt b/expected/wasm64-wasi/defined-symbols.txt +index cd3d45b..ca54a6d 100644 +--- a/expected/wasm64-wasi/defined-symbols.txt ++++ b/expected/wasm64-wasi/defined-symbols.txt +@@ -1612,6 +1612,7 @@ swprintf + swscanf + symlink + symlinkat ++sync_file_range + sysconf + syslog + system +diff --git a/expected/wasm64-wasi/undefined-symbols.txt b/expected/wasm64-wasi/undefined-symbols.txt +index ae7c7ff..a70377f 100644 +--- a/expected/wasm64-wasi/undefined-symbols.txt ++++ b/expected/wasm64-wasi/undefined-symbols.txt +@@ -14,6 +14,7 @@ __getf2 + __global_base + __gttf2 + __heap_base ++__imported_oliphaunt_postmaster_v1_fd_sync_range + __imported_wasi_snapshot_preview1_args_get + __imported_wasi_snapshot_preview1_args_sizes_get + __imported_wasi_snapshot_preview1_clock_res_get +diff --git a/libc-bottom-half/headers/public/wasi/api_wasix.h b/libc-bottom-half/headers/public/wasi/api_wasix.h +index 5b02c4b..3963e83 100644 +--- a/libc-bottom-half/headers/public/wasi/api_wasix.h ++++ b/libc-bottom-half/headers/public/wasi/api_wasix.h +@@ -4399,6 +4399,14 @@ __wasi_errno_t __wasi_proc_spawn2( + __wasi_errno_t __wasi_proc_id( + __wasi_pid_t *retptr0 + ) __attribute__((__warn_unused_result__)); ++/** ++ * Returns the current and maximum resource limit for the current process ++ */ ++__wasi_errno_t __wasi_proc_rlimit_get( ++ uint32_t resource, ++ uint64_t *retptr0, ++ uint64_t *retptr1 ++) __attribute__((__warn_unused_result__)); + /** + * Returns the parent handle of a particular process + */ +diff --git a/libc-bottom-half/sources/__wasixlibc_real.c b/libc-bottom-half/sources/__wasixlibc_real.c +index 41a3f0b..56d916e 100644 +--- a/libc-bottom-half/sources/__wasixlibc_real.c ++++ b/libc-bottom-half/sources/__wasixlibc_real.c +@@ -502,6 +502,20 @@ __wasi_errno_t __wasi_proc_id( + return (uint16_t) ret; + } + ++int32_t __imported_wasix_32v1_proc_rlimit_get(int32_t arg0, int32_t arg1, int32_t arg2) __attribute__(( ++ __import_module__("wasix_32v1"), ++ __import_name__("proc_rlimit_get") ++)); ++ ++__wasi_errno_t __wasi_proc_rlimit_get( ++ uint32_t resource, ++ uint64_t *retptr0, ++ uint64_t *retptr1 ++){ ++ int32_t ret = __imported_wasix_32v1_proc_rlimit_get((int32_t) resource, (intptr_t) retptr0, (intptr_t) retptr1); ++ return (uint16_t) ret; ++} ++ + int32_t __imported_wasix_32v1_proc_parent(int32_t arg0, int32_t arg1) __attribute__(( + __import_module__("wasix_32v1"), + __import_name__("proc_parent") +@@ -1278,4 +1292,3 @@ __wasi_errno_t __wasi_context_destroy( + int32_t ret = __imported_wasix_32v1_context_destroy((int64_t) context); + return (uint16_t) ret; + } +- +diff --git a/libc-top-half/musl/src/misc/getrlimit.c b/libc-top-half/musl/src/misc/getrlimit.c +index 67bfc7c..972fb67 100644 +--- a/libc-top-half/musl/src/misc/getrlimit.c ++++ b/libc-top-half/musl/src/misc/getrlimit.c +@@ -2,6 +2,12 @@ + #include + #ifdef __wasilibc_unmodified_upstream + #include "syscall.h" ++#else ++#include ++#endif ++ ++#ifndef SYSCALL_RLIM_INFINITY ++#define SYSCALL_RLIM_INFINITY RLIM_INFINITY + #endif + + #define FIX(x) do{ if ((x)>=SYSCALL_RLIM_INFINITY) (x)=RLIM_INFINITY; }while(0) +@@ -24,8 +30,17 @@ int getrlimit(int resource, struct rlimit *rlim) + FIX(rlim->rlim_cur); + FIX(rlim->rlim_max); + #else +- rlim->rlim_cur = RLIM_INFINITY; +- rlim->rlim_max = RLIM_INFINITY; ++ uint64_t rlim_cur; ++ uint64_t rlim_max; ++ __wasi_errno_t error = __wasi_proc_rlimit_get(resource, &rlim_cur, &rlim_max); ++ if (error != __WASI_ERRNO_SUCCESS) { ++ errno = error; ++ return -1; ++ } ++ rlim->rlim_cur = rlim_cur; ++ rlim->rlim_max = rlim_max; ++ FIX(rlim->rlim_cur); ++ FIX(rlim->rlim_max); + #endif + return 0; + } diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0007-word-at-a-time-memcmp.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0007-word-at-a-time-memcmp.patch new file mode 100644 index 000000000..aeaaad081 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0007-word-at-a-time-memcmp.patch @@ -0,0 +1,62 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 7/8] libc: retain bounded word-at-a-time memcmp + +Isolate the inherited little-endian word-at-a-time comparison optimization +from the runtime correctness patches. Loads remain bounded by the requested +length and use memcpy for alignment/aliasing safety. This decomposition does +not establish a new performance win or cross-compiler portability: retain its +existing aggregate semantics here and require factor-isolated evidence before +promoting it as an independent upstream optimization. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/libc-top-half/musl/src/string/memcmp.c b/libc-top-half/musl/src/string/memcmp.c +index bdbce9f..3fbc3c5 100644 +--- a/libc-top-half/musl/src/string/memcmp.c ++++ b/libc-top-half/musl/src/string/memcmp.c +@@ -1,8 +1,43 @@ ++#include ++#include + #include + ++static uint64_t load64(const unsigned char *p) ++{ ++ uint64_t v; ++ __builtin_memcpy(&v, p, sizeof(v)); ++ return v; ++} ++ ++static uint32_t load32(const unsigned char *p) ++{ ++ uint32_t v; ++ __builtin_memcpy(&v, p, sizeof(v)); ++ return v; ++} ++ + int memcmp(const void *vl, const void *vr, size_t n) + { + const unsigned char *l=vl, *r=vr; ++#if defined(__GNUC__) && __BYTE_ORDER == __LITTLE_ENDIAN ++ for (; n >= 8; n-=8, l+=8, r+=8) { ++ uint64_t a = load64(l), b = load64(r); ++ if (a != b) { ++ unsigned shift = __builtin_ctzll(a ^ b) & ~7U; ++ return (int)((a >> shift) & 0xff) - (int)((b >> shift) & 0xff); ++ } ++ } ++ if (n >= 4) { ++ uint32_t a = load32(l), b = load32(r); ++ if (a != b) { ++ unsigned shift = __builtin_ctz(a ^ b) & ~7U; ++ return (int)((a >> shift) & 0xff) - (int)((b >> shift) & 0xff); ++ } ++ n-=4; ++ l+=4; ++ r+=4; ++ } ++#endif + for (; n && *l == *r; n--, l++, r++); + return n ? *l-*r : 0; + } diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0007-word-at-a-time-memcmp.test.py b/src/wasix/postmaster/wasmer/patches/wasix-libc/0007-word-at-a-time-memcmp.test.py new file mode 100644 index 000000000..c6a7f243c --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0007-word-at-a-time-memcmp.test.py @@ -0,0 +1,68 @@ +#!/usr/bin/env python3 +"""Check inherited word loads against byte semantics, including protected-page ends.""" + +import os +from pathlib import Path +import shlex +import subprocess +import tempfile + + +patch = Path(__file__).name.replace(".test.py", ".patch") +section = Path(__file__).with_name(patch).read_text().split("@@ -1,8 +1,43 @@\n", 1)[1] +implementation = "\n".join(line[1:] for line in section.splitlines() + if line.startswith(("+", " "))) +source = "#define memcmp candidate_memcmp\n" + implementation + r""" +#undef memcmp +#include +#include +#include + +int main(void) +{ + unsigned char left[96], right[96]; + for (size_t a = 0; a < 8; a++) + for (size_t b = 0; b < 8; b++) + for (size_t n = 0; n <= 80; n++) { + memset(left, 0x80, sizeof left); + memset(right, 0x80, sizeof right); + assert(candidate_memcmp(left + a, right + b, n) == 0); + for (size_t mismatch = 0; mismatch < n; mismatch++) { + right[b + mismatch] = 0xff; + assert(candidate_memcmp(left + a, right + b, n) == -127); + assert(candidate_memcmp(right + b, left + a, n) == 127); + right[b + mismatch] = 0; + assert(candidate_memcmp(left + a, right + b, n) == 128); + right[b + mismatch] = 0x80; + } + } + long page = sysconf(_SC_PAGESIZE); + assert(page >= 4096); + unsigned char *mapping = mmap(NULL, (size_t)page * 4, + PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + assert(mapping != MAP_FAILED); + assert(mprotect(mapping + page, (size_t)page, PROT_NONE) == 0); + assert(mprotect(mapping + 3 * page, (size_t)page, PROT_NONE) == 0); + for (size_t n = 0; n <= 80; n++) { + unsigned char *l = mapping + page - n; + unsigned char *r = mapping + 3 * page - n; + memset(l, 0xff, n); + memset(r, 0xff, n); + assert(candidate_memcmp(l, r, n) == 0); + if (n) { + r[n - 1] = 0; + assert(candidate_memcmp(l, r, n) == 255); + } + } + assert(munmap(mapping, (size_t)page * 4) == 0); + return 0; +} +""" +with tempfile.TemporaryDirectory(prefix="libc-memcmp-") as temporary: + executable = str(Path(temporary) / "probe") + for optimization in ("-O0", "-O2"): + subprocess.run(shlex.split(os.environ.get("CC", "cc")) + ["-x", "c", + "-D_GNU_SOURCE", optimization, "-Wall", "-Wextra", "-Werror", + "-", "-o", executable], input=source, text=True, check=True) + subprocess.run([executable], check=True) +print("memcmp O0/O2: all byte positions, independent alignments, unsigned order and protected-page bounds") diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0008-single-evaluation-signal-jump-buffer.patch b/src/wasix/postmaster/wasmer/patches/wasix-libc/0008-single-evaluation-signal-jump-buffer.patch new file mode 100644 index 000000000..90449762c --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0008-single-evaluation-signal-jump-buffer.patch @@ -0,0 +1,83 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] libc: evaluate the signal jump buffer exactly once + +Return the original buffer from the signal-mask preparation helper and pass +that result directly to setjmp. Previously a side-effectful buffer expression +could save the mask into one buffer and establish the jump context in another. +Both arguments now evaluate once while setjmp remains in the caller's frame; +no returned helper frame or compiler-specific statement expression is used. + +This is a separate correctness delta after the source-equivalent 0001-0007 +split. The private helper return ABI changes, so rebuild libc and its callers; +the ordered series identity invalidates previous sysroot/runtime receipts. +PostgreSQL's current callers use stable buffers, but the public libc macro must +also satisfy its argument-evaluation contract. This fix is upstream-suitable +for the WASIX EH adaptation; it does not change non-EH or upstream musl paths. + +diff --git a/libc-top-half/musl/include/setjmp.h b/libc-top-half/musl/include/setjmp.h +index 47bb2e0..775aa1f 100644 +--- a/libc-top-half/musl/include/setjmp.h ++++ b/libc-top-half/musl/include/setjmp.h +@@ -42,27 +42,27 @@ _Noreturn void _longjmp (jmp_buf, int); + #if __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 1) + #define __setjmp_attr __attribute__((__returns_twice__)) + #else + #define __setjmp_attr + #endif + + #if defined(_POSIX_SOURCE) || defined(_POSIX_C_SOURCE) || \ + defined(_XOPEN_SOURCE) || defined(_GNU_SOURCE) || defined(_BSD_SOURCE) + typedef jmp_buf sigjmp_buf; + #if !defined(__wasilibc_unmodified_upstream) && \ + defined(__wasm_exception_handling__) && \ + !defined(__WASIX_LIBC_BUILDING_SETJMP) +- int __wasilibc_sigsetjmp_save(sigjmp_buf, int); ++ struct __jmp_buf_tag *__wasilibc_sigsetjmp_save(sigjmp_buf, int); + #define sigsetjmp(buf, savesigs) \ +- (__wasilibc_sigsetjmp_save((buf), (savesigs)), setjmp((buf))) ++ setjmp(__wasilibc_sigsetjmp_save((buf), (savesigs))) + #else + int sigsetjmp(sigjmp_buf, int) __setjmp_attr; + #endif + _Noreturn void siglongjmp(sigjmp_buf, int); + #endif + + #if defined(_XOPEN_SOURCE) || defined(_GNU_SOURCE) || defined(_BSD_SOURCE) + int _setjmp(jmp_buf) __setjmp_attr; + _Noreturn void _longjmp(jmp_buf, int); + #endif + + #undef __setjmp_attr +diff --git a/libc-top-half/musl/src/signal/sigsetjmp_tail.c b/libc-top-half/musl/src/signal/sigsetjmp_tail.c +index f35da65..2eeb9cc 100644 +--- a/libc-top-half/musl/src/signal/sigsetjmp_tail.c ++++ b/libc-top-half/musl/src/signal/sigsetjmp_tail.c +@@ -9,24 +9,24 @@ + hidden int __sigsetjmp_tail(sigjmp_buf jb, int ret) + { + #ifdef __wasilibc_unmodified_upstream + void *p = jb->__ss; + __syscall(SYS_rt_sigprocmask, SIG_SETMASK, ret?p:0, ret?0:p, _NSIG/8); + return ret; + #else + return EINVAL; + #endif + } + + #if !defined(__wasilibc_unmodified_upstream) && defined(__wasm_exception_handling__) +-int __wasilibc_sigsetjmp_save(sigjmp_buf jb, int savesigs) ++struct __jmp_buf_tag *__wasilibc_sigsetjmp_save(sigjmp_buf jb, int savesigs) + { + jb->__fl = savesigs ? 1 : 0; + if (!savesigs) { +- return 0; ++ return jb; + } + + memset(jb->__ss, 0, sizeof jb->__ss); + pthread_sigmask(SIG_SETMASK, NULL, (sigset_t *)jb->__ss); +- return 0; ++ return jb; + } + #endif diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/0008-single-evaluation-signal-jump-buffer.test.py b/src/wasix/postmaster/wasmer/patches/wasix-libc/0008-single-evaluation-signal-jump-buffer.test.py new file mode 100644 index 000000000..7ecaf9d54 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/0008-single-evaluation-signal-jump-buffer.test.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Execute the patched EH helpers with host frame/mask primitives, not a WASIX mask claim.""" + +import os +from pathlib import Path +import shlex +import subprocess +import tempfile + + +directory = Path(__file__).parent + + +def after_image(patch, path): + section = (directory / patch).read_text().split(f"+++ b/{path}\n", 1)[1] + section = section.split("\ndiff --git ", 1)[0] + return "\n".join(line[1:] for line in section.splitlines() + if line.startswith(("+", " "))) + + +new_patch = Path(__file__).name.replace(".test.py", ".patch") +old_patch = "0005-wasm-eh-signal-context.patch" +header = after_image(new_patch, "libc-top-half/musl/include/setjmp.h") +declaration = header[header.index("\tstruct __jmp_buf_tag *"):header.index("\n#else", header.index("#define sigsetjmp"))] +helper = after_image(new_patch, "libc-top-half/musl/src/signal/sigsetjmp_tail.c") +helper = helper[helper.index("struct __jmp_buf_tag *__wasilibc_sigsetjmp_save"):helper.rindex("\n#endif")] +restore = after_image(old_patch, "libc-top-half/musl/src/signal/siglongjmp.c") +restore = restore[restore.index("_Noreturn void siglongjmp"):] +probe = (directory.parent.parent / "probes/wasm_eh_sjlj_probe.c").read_text() + +# Only frame representation is adapted. Both helper bodies and the public macro +# come from the actual patches. _setjmp captures the live caller, never a wrapper. +adapter = r""" +#include +#include +#include +#include +struct probe_jmp_buf { + jmp_buf native; + unsigned long __fl; + unsigned long __ss[128 / sizeof(long)]; +}; +typedef struct probe_jmp_buf probe_sigjmp_buf[1]; +#ifdef __cplusplus +#define _Noreturn [[noreturn]] +#endif +#define __jmp_buf_tag probe_jmp_buf +#define sigjmp_buf probe_sigjmp_buf +#define jmp_buf probe_sigjmp_buf +#undef setjmp +#define setjmp(buf) _setjmp((buf)->native) +#define longjmp(buf, value) (longjmp)((buf)->native, (value)) +#define siglongjmp probe_siglongjmp +#undef sigsetjmp +#define __wasm_exception_handling__ 1 +""" + +with tempfile.TemporaryDirectory(prefix="libc-sigsetjmp-") as temporary: + for language, compiler in (("c", os.environ.get("CC", "cc")), + ("c++", os.environ.get("CXX", "c++"))): + for optimization in ("-O0", "-O2"): + for fixed in (False, True): + macro = declaration if fixed else declaration.replace( + "setjmp(__wasilibc_sigsetjmp_save((buf), (savesigs)))", + "(__wasilibc_sigsetjmp_save((buf), (savesigs)), setjmp((buf)))") + source = "\n".join((adapter, macro, helper, restore, probe)) + executable = str(Path(temporary) / "probe") + subprocess.run(shlex.split(compiler) + ["-x", language, + "-D_POSIX_C_SOURCE=200809L", optimization, "-Wall", "-Wextra", + "-Werror", "-", "-o", executable], input=source, text=True, check=True) + result = subprocess.run([executable], capture_output=True, text=True) + assert result.returncode == (0 if fixed else 21), (language, optimization, fixed, result) + if fixed: + assert "mask-preserved" in result.stdout + +print("C/C++ O0/O2: caller-owned jumps, arguments once, zero/nonzero returns, host mask hooks; old macro rejected") diff --git a/src/wasix/postmaster/wasmer/patches/wasix-libc/series b/src/wasix/postmaster/wasmer/patches/wasix-libc/series new file mode 100644 index 000000000..3dc910f01 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasix-libc/series @@ -0,0 +1,8 @@ +0001-mmap-and-allocator-exec-ownership.patch +0002-file-flags-and-sync-range.patch +0003-socket-and-epoll-flags.patch +0004-process-wait-signal-and-futex-semantics.patch +0005-wasm-eh-signal-context.patch +0006-process-resource-limit-import.patch +0007-word-at-a-time-memcmp.patch +0008-single-evaluation-signal-jump-buffer.patch diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0001-engine-memory-and-exception-lifetimes.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0001-engine-memory-and-exception-lifetimes.patch new file mode 100644 index 000000000..eefba7295 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0001-engine-memory-and-exception-lifetimes.patch @@ -0,0 +1,7688 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 1/9] engine: retain memory, code and exception lifetimes + +Keep the engine/VM/compiler/API changes together: instantiated modules share +owned memories, tables, code mappings and exception metadata across exec and +sealed loading. They are the engine foundations used by the later WASIX +lifecycle patches. Generic lifetime/exception fixes need smaller upstream +submissions; executable-memory sealing is downstream policy. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/lib/api/src/backend/sys/entities/engine.rs b/lib/api/src/backend/sys/entities/engine.rs +index 40696a9..1fd35ea 100644 +--- a/lib/api/src/backend/sys/entities/engine.rs ++++ b/lib/api/src/backend/sys/entities/engine.rs +@@ -3,11 +3,112 @@ + use std::{path::Path, sync::Arc}; + + use shared_buffer::OwnedBuffer; +-pub use wasmer_compiler::{Artifact, BaseTunables, Engine, EngineBuilder, Tunables}; +-use wasmer_types::{CompilationProgressCallback, DeserializeError, Features, target::Target}; ++use wasmer_compiler::{ArtifactCreate, PendingArtifact}; ++pub use wasmer_compiler::{ ++ Artifact, BaseTunables, CodeMemoryPolicy, Engine, EngineBuilder, Tunables, ++}; ++use wasmer_types::{ ++ CompilationProgressCallback, DeserializeError, Features, ModuleHash, target::Target, ++}; ++use wasmer_vm::MemoryStyle; + + use crate::{BackendEngine, BackendModule}; + ++/// A detached module activation that still owns rollback of its exact ++/// engine-side executable-code allocation. ++/// ++/// Sealed loaders can inspect the activated module and complete fallible audit ++/// work before calling [`Self::commit`]. Dropping this value rolls the ++/// allocation back, including frame and unwind registrations. ++#[doc(hidden)] ++pub struct PendingModuleActivation { ++ artifact: PendingArtifact, ++} ++ ++/// The complete linear-memory allocation plan embedded in a serialized AOT ++/// artifact. ++/// ++/// A sealed loader can compare this plan with its admitted module contract ++/// before executable-code ownership escapes the activation transaction. ++#[derive(Clone, Copy, Debug, PartialEq, Eq)] ++pub struct SerializedLinearMemoryPlan { ++ /// The module's initial memory size in WebAssembly pages. ++ pub minimum_pages: u32, ++ /// The module's declared maximum memory size in WebAssembly pages. ++ pub maximum_pages: Option, ++ /// Whether the module declares shared linear memory. ++ pub shared: bool, ++ /// The allocation strategy compiled into the artifact. ++ pub style: SerializedLinearMemoryStyle, ++} ++ ++/// A stable, inspection-only description of an AOT artifact's memory style. ++#[derive(Clone, Copy, Debug, PartialEq, Eq)] ++pub enum SerializedLinearMemoryStyle { ++ /// A nonmoving reservation with a fixed bound and offset guard. ++ Static { ++ /// Reserved linear-memory capacity in WebAssembly pages. ++ bound_pages: u32, ++ /// Bytes reserved after the static bound for unchecked offsets. ++ offset_guard_bytes: u64, ++ }, ++ /// A growable allocation whose host base may need to move. ++ Dynamic { ++ /// Bytes reserved after the current allocation for unchecked offsets. ++ offset_guard_bytes: u64, ++ }, ++} ++ ++impl PendingModuleActivation { ++ fn new(artifact: PendingArtifact) -> Self { ++ Self { artifact } ++ } ++ ++ /// Return the activated module hash without committing code memory. ++ pub fn module_hash(&self) -> Option { ++ self.artifact.module_hash() ++ } ++ ++ /// Return every linear-memory type and allocation style embedded in this ++ /// artifact without committing its executable-code allocation. ++ pub fn linear_memory_plans(&self) -> Vec { ++ let artifact = self.artifact.artifact(); ++ artifact ++ .module_info() ++ .memories ++ .values() ++ .zip(artifact.memory_styles().values()) ++ .map(|(memory, style)| SerializedLinearMemoryPlan { ++ minimum_pages: memory.minimum.0, ++ maximum_pages: memory.maximum.map(|pages| pages.0), ++ shared: memory.shared, ++ style: match style { ++ MemoryStyle::Static { ++ bound, ++ offset_guard_size, ++ } => SerializedLinearMemoryStyle::Static { ++ bound_pages: bound.0, ++ offset_guard_bytes: *offset_guard_size, ++ }, ++ MemoryStyle::Dynamic { offset_guard_size } => { ++ SerializedLinearMemoryStyle::Dynamic { ++ offset_guard_bytes: *offset_guard_size, ++ } ++ } ++ }, ++ }) ++ .collect() ++ } ++ ++ /// Commit executable-code ownership and construct the public module. ++ pub fn commit(self) -> crate::Module { ++ let artifact = Arc::new(self.artifact.commit()); ++ crate::Module(BackendModule::Sys(super::module::Module::from_artifact( ++ artifact, ++ ))) ++ } ++} ++ + /// Get the default config for the sys Engine + #[allow(unreachable_code)] + #[cfg(feature = "compiler")] +@@ -83,6 +184,9 @@ pub trait NativeEngineExt { + /// Get a reference to attached Tunable of this engine + fn tunables(&self) -> &dyn Tunables; + ++ /// Select executable-memory ownership before the first artifact allocation. ++ fn set_code_memory_policy(&mut self, policy: CodeMemoryPolicy) -> Result<(), String>; ++ + /// Compile a module from bytes with a progress callback. + /// + /// The callback is invoked with progress updates during the compilation process. +@@ -120,6 +224,55 @@ pub trait NativeEngineExt { + &self, + file_ref: &Path, + ) -> Result; ++ ++ /// Load and validate a serialized WebAssembly module from an existing ++ /// memory mapping. ++ /// ++ /// This is the descriptor-bound counterpart to ++ /// [`Self::deserialize_from_mmapped_file`]: callers can verify the exact ++ /// mapping before transferring it into the checked artifact deserializer. ++ /// ++ /// # Safety ++ /// See [`Artifact::deserialize`]. ++ unsafe fn deserialize_from_mmapped_buffer( ++ &self, ++ buffer: OwnedBuffer, ++ ) -> Result; ++ ++ /// Validate a serialized universal artifact and return its embedded module ++ /// hash without allocating or publishing executable code. ++ /// ++ /// This checks the universal magic, metadata ABI, complete archived value, ++ /// and native CPU-feature requirements. The buffer is borrowed so a caller ++ /// can retain the exact verified snapshot for later lazy activation. ++ fn inspect_serialized_artifact( ++ &self, ++ buffer: &OwnedBuffer, ++ ) -> Result; ++ ++ /// Load and validate a descriptor-bound artifact, materialize only the ++ /// metadata needed for later instantiations, and release the archive. ++ /// ++ /// The resulting module cannot be re-serialized. This is intended for ++ /// sealed precompiled-only executors that value reclaimable steady-state ++ /// memory over artifact round-tripping. ++ /// ++ /// # Safety ++ /// See [`Artifact::deserialize_detached`]. ++ unsafe fn deserialize_from_mmapped_buffer_detached( ++ &self, ++ buffer: OwnedBuffer, ++ ) -> Result; ++ ++ /// Load and validate a detached descriptor-bound artifact while retaining ++ /// rollback ownership until the caller explicitly commits activation. ++ /// ++ /// # Safety ++ /// See [`Artifact::deserialize_detached_pending`]. ++ unsafe fn deserialize_from_mmapped_buffer_detached_pending( ++ &self, ++ buffer: OwnedBuffer, ++ ) -> Result; + } + + impl NativeEngineExt for crate::engine::Engine { +@@ -163,6 +316,13 @@ impl NativeEngineExt for crate::engine::Engine { + } + } + ++ fn set_code_memory_policy(&mut self, policy: CodeMemoryPolicy) -> Result<(), String> { ++ match self.be { ++ BackendEngine::Sys(ref mut engine) => engine.set_code_memory_policy(policy), ++ _ => Err("code-memory policies require a sys engine".to_string()), ++ } ++ } ++ + fn new_module_with_progress( + &self, + bytes: &[u8], +@@ -196,6 +356,39 @@ impl NativeEngineExt for crate::engine::Engine { + super::module::Module::from_artifact(artifact), + ))) + } ++ ++ unsafe fn deserialize_from_mmapped_buffer( ++ &self, ++ buffer: OwnedBuffer, ++ ) -> Result { ++ let artifact = unsafe { Arc::new(Artifact::deserialize(self.as_sys(), buffer)?) }; ++ Ok(crate::Module(BackendModule::Sys( ++ super::module::Module::from_artifact(artifact), ++ ))) ++ } ++ ++ fn inspect_serialized_artifact( ++ &self, ++ buffer: &OwnedBuffer, ++ ) -> Result { ++ Artifact::inspect_serialized(self.as_sys(), buffer.as_ref()) ++ } ++ ++ unsafe fn deserialize_from_mmapped_buffer_detached( ++ &self, ++ buffer: OwnedBuffer, ++ ) -> Result { ++ unsafe { self.deserialize_from_mmapped_buffer_detached_pending(buffer) } ++ .map(PendingModuleActivation::commit) ++ } ++ ++ unsafe fn deserialize_from_mmapped_buffer_detached_pending( ++ &self, ++ buffer: OwnedBuffer, ++ ) -> Result { ++ let artifact = unsafe { Artifact::deserialize_detached_pending(self.as_sys(), buffer)? }; ++ Ok(PendingModuleActivation::new(artifact)) ++ } + } + + impl crate::Engine { +diff --git a/lib/api/src/backend/sys/entities/instance.rs b/lib/api/src/backend/sys/entities/instance.rs +index 57c4a07..63d84c9 100644 +--- a/lib/api/src/backend/sys/entities/instance.rs ++++ b/lib/api/src/backend/sys/entities/instance.rs +@@ -44,6 +44,33 @@ impl Instance { + Ok((instance, exports)) + } + ++ /// Instantiates an instance while materializing only the named exports. ++ /// ++ /// The VM's complete `ModuleInfo::exports` remains authoritative and can ++ /// be queried later through [`Self::lookup_export`]. This is an internal ++ /// opt-in path for runtimes whose modules export very large symbol tables; ++ /// ordinary Wasmer instances stay eager. ++ #[allow(clippy::result_large_err)] ++ pub(crate) fn new_with_export_names( ++ store: &mut impl AsStoreMut, ++ module: &Module, ++ imports: &Imports, ++ export_names: &[&str], ++ ) -> Result<(Self, Exports), InstantiationError> { ++ let externs = imports ++ .imports_for_module(module) ++ .map_err(InstantiationError::Link)?; ++ let mut handle = module.as_sys().instantiate(store, &externs)?; ++ handle.unwrap_sys_mut().enable_deferred_export_cache(); ++ let exports = Self::get_named_exports(store, handle.unwrap_sys_mut(), export_names); ++ ++ let instance = Self { ++ _handle: StoreHandle::new(store.objects_mut().as_sys_mut(), handle.unwrap_sys()), ++ }; ++ ++ Ok((instance, exports)) ++ } ++ + #[allow(clippy::result_large_err)] + pub(crate) fn new_by_index( + store: &mut impl AsStoreMut, +@@ -78,6 +105,34 @@ impl Instance { + }) + .collect::() + } ++ ++ fn get_named_exports( ++ store: &mut impl AsStoreMut, ++ handle: &mut VMInstance, ++ export_names: &[&str], ++ ) -> Exports { ++ export_names ++ .iter() ++ .filter_map(|name| { ++ let export = handle.lookup(name)?; ++ let extern_ = Extern::from_vm_extern(store, crate::vm::VMExtern::Sys(export)); ++ Some(((*name).to_string(), extern_)) ++ }) ++ .collect::() ++ } ++ ++ /// Looks up one export from the VM's complete module metadata, materializing ++ /// it on demand for deferred instances. ++ pub(crate) fn lookup_export(&self, store: &mut impl AsStoreMut, name: &str) -> Option { ++ let export = self ++ ._handle ++ .get_mut(store.objects_mut().as_sys_mut()) ++ .lookup(name)?; ++ Some(Extern::from_vm_extern( ++ store, ++ crate::vm::VMExtern::Sys(export), ++ )) ++ } + } + + impl crate::BackendInstance { +diff --git a/lib/api/src/backend/sys/entities/memory/mod.rs b/lib/api/src/backend/sys/entities/memory/mod.rs +index e8ad5f5..39a84de 100644 +--- a/lib/api/src/backend/sys/entities/memory/mod.rs ++++ b/lib/api/src/backend/sys/entities/memory/mod.rs +@@ -89,6 +89,87 @@ impl Memory { + Ok(()) + } + ++ pub(crate) fn supports_persistent_shared_fixed_remap(&self, store: &impl AsStoreRef) -> bool { ++ self.handle ++ .get(store.as_store_ref().objects().as_sys()) ++ .supports_persistent_shared_fixed_remap() ++ } ++ ++ /// Replace a page-aligned range inside this memory with a shared, ++ /// read-write file mapping at the same host address. ++ /// ++ /// # Safety ++ /// No concurrent linear-memory access or live Rust reference may overlap the range during ++ /// replacement. The backing inode must not shrink below its size validated at mapping time; ++ /// only the validated final partial page may extend beyond EOF. ++ pub unsafe fn remap_shared_file_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.handle ++ .get_mut(store.as_store_mut().objects_mut().as_sys_mut()) ++ .remap_shared_file_fixed(start, len, file, file_offset) ++ } ++ } ++ ++ /// Replace a page-aligned range inside this memory with a private, ++ /// copy-on-write file mapping at the same host address. ++ /// ++ /// # Safety ++ /// No concurrent linear-memory access or live Rust reference may overlap the range during ++ /// replacement, and the backing inode must not be truncated or replaced below ++ /// `file_offset + len` while accessible. ++ pub unsafe fn remap_private_file_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.handle ++ .get_mut(store.as_store_mut().objects_mut().as_sys_mut()) ++ .remap_private_file_fixed(start, len, file, file_offset) ++ } ++ } ++ ++ /// Replace a page-aligned range inside this memory with private, ++ /// zero-filled memory at the same host address. ++ /// ++ /// # Safety ++ /// No access or Rust reference may overlap the range during replacement. ++ pub unsafe fn remap_private_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.handle ++ .get_mut(store.as_store_mut().objects_mut().as_sys_mut()) ++ .remap_private_fixed(start, len) ++ } ++ } ++ ++ /// Synchronize a page-aligned memory range with its backing file. ++ pub fn msync( ++ &self, ++ store: &impl AsStoreRef, ++ start: usize, ++ len: usize, ++ flags: i32, ++ ) -> Result<(), MemoryError> { ++ self.handle ++ .get(store.as_store_ref().objects().as_sys()) ++ .msync(start, len, flags) ++ } ++ + pub(crate) fn from_vm_extern(store: &impl AsStoreRef, vm_extern: VMExternMemory) -> Self { + Self { + handle: unsafe { +diff --git a/lib/api/src/backend/sys/entities/module.rs b/lib/api/src/backend/sys/entities/module.rs +index 4e6ae27..de6887f 100644 +--- a/lib/api/src/backend/sys/entities/module.rs ++++ b/lib/api/src/backend/sys/entities/module.rs +@@ -49,6 +49,7 @@ impl Module { + unsafe { Self::from_binary_unchecked(engine, binary) } + } + ++ #[cfg(feature = "compiler")] + pub(crate) fn from_binary_with_progress( + engine: &impl AsEngineRef, + binary: &[u8], +@@ -64,6 +65,15 @@ impl Module { + Ok(Self::from_artifact(artifact)) + } + ++ #[cfg(not(feature = "compiler"))] ++ pub(crate) fn from_binary_with_progress( ++ engine: &impl AsEngineRef, ++ binary: &[u8], ++ _callback: CompilationProgressCallback, ++ ) -> Result { ++ Self::from_binary(engine, binary) ++ } ++ + pub(crate) unsafe fn from_binary_unchecked( + engine: &impl AsEngineRef, + binary: &[u8], +diff --git a/lib/api/src/backend/sys/mod.rs b/lib/api/src/backend/sys/mod.rs +index 4e8ea87..fa6c1c5 100644 +--- a/lib/api/src/backend/sys/mod.rs ++++ b/lib/api/src/backend/sys/mod.rs +@@ -7,7 +7,10 @@ pub(crate) mod error; + pub(crate) mod tunables; + pub mod vm; + +-pub use engine::NativeEngineExt; ++pub use engine::{ ++ NativeEngineExt, PendingModuleActivation, SerializedLinearMemoryPlan, ++ SerializedLinearMemoryStyle, ++}; + pub use entities::*; + pub use tunables::*; + +@@ -16,7 +19,8 @@ pub use wasmer_compiler::{ + CompilerConfig, FunctionMiddleware, MiddlewareReaderState, ModuleMiddleware, wasmparser, + }; + +-pub use wasmer_compiler::{Artifact, EngineBuilder, Features, Tunables}; ++pub use shared_buffer::OwnedBuffer; ++pub use wasmer_compiler::{Artifact, CodeMemoryPolicy, EngineBuilder, Features, Tunables}; + + pub use wasmer_types::MiddlewareError; + pub use wasmer_types::target::{Architecture, CpuFeature, OperatingSystem, Target, Triple}; +diff --git a/lib/api/src/entities/exports.rs b/lib/api/src/entities/exports.rs +index 5bda728..9dc3d9b 100644 +--- a/lib/api/src/entities/exports.rs ++++ b/lib/api/src/entities/exports.rs +@@ -3,6 +3,7 @@ use crate::{Extern, Function, Global, Memory, Table, TypedFunction, WasmTypeList + use indexmap::IndexMap; + use std::fmt; + use std::iter::{ExactSizeIterator, FromIterator}; ++use std::sync::Arc; + use thiserror::Error; + + /// The `ExportError` can happen when trying to get a specific +@@ -60,7 +61,7 @@ pub enum ExportError { + #[derive(Clone, Default, PartialEq, Eq)] + #[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] + pub struct Exports { +- map: IndexMap, ++ map: Arc>, + } + + impl Exports { +@@ -72,7 +73,7 @@ impl Exports { + /// Creates a new `Exports` with capacity `n`. + pub fn with_capacity(n: usize) -> Self { + Self { +- map: IndexMap::with_capacity(n), ++ map: Arc::new(IndexMap::with_capacity(n)), + } + } + +@@ -92,7 +93,7 @@ impl Exports { + S: Into, + E: Into, + { +- self.map.insert(name.into(), value.into()); ++ Arc::make_mut(&mut self.map).insert(name.into(), value.into()); + } + + /// Get an export given a `name`. +@@ -256,7 +257,7 @@ where + impl FromIterator<(String, Extern)> for Exports { + fn from_iter>(iter: I) -> Self { + Self { +- map: IndexMap::from_iter(iter), ++ map: Arc::new(IndexMap::from_iter(iter)), + } + } + } +@@ -266,7 +267,9 @@ impl IntoIterator for Exports { + type Item = (String, Extern); + + fn into_iter(self) -> Self::IntoIter { +- self.map.into_iter() ++ Arc::try_unwrap(self.map) ++ .unwrap_or_else(|map| (*map).clone()) ++ .into_iter() + } + } + +@@ -279,6 +282,55 @@ impl<'a> IntoIterator for &'a Exports { + } + } + ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use crate::{Store, Value}; ++ ++ #[test] ++ fn clone_shares_map_until_mutated() { ++ let mut store = Store::default(); ++ let mut exports = Exports::new(); ++ exports.insert("one", Global::new(&mut store, Value::I32(1))); ++ ++ let mut cloned = exports.clone(); ++ assert!(Arc::ptr_eq(&exports.map, &cloned.map)); ++ assert_eq!(exports, cloned); ++ ++ cloned.insert("two", Global::new(&mut store, Value::I32(2))); ++ ++ assert!(!Arc::ptr_eq(&exports.map, &cloned.map)); ++ assert_eq!(exports.len(), 1); ++ assert_eq!(cloned.len(), 2); ++ assert!(exports.get_extern("two").is_none()); ++ assert!(cloned.get_extern("two").is_some()); ++ assert_ne!(exports, cloned); ++ } ++ ++ #[test] ++ fn consuming_iteration_owns_unique_and_shared_maps() { ++ let mut store = Store::default(); ++ let mut unique = Exports::new(); ++ unique.insert("one", Global::new(&mut store, Value::I32(1))); ++ let unique_names = unique.into_iter().map(|(name, _)| name).collect::>(); ++ assert_eq!(unique_names, ["one"]); ++ ++ let mut original = Exports::new(); ++ original.insert("one", Global::new(&mut store, Value::I32(1))); ++ original.insert("two", Global::new(&mut store, Value::I32(2))); ++ let shared = original.clone(); ++ assert_eq!(Arc::strong_count(&original.map), 2); ++ ++ let shared_names = shared.into_iter().map(|(name, _)| name).collect::>(); ++ ++ assert_eq!(shared_names, ["one", "two"]); ++ assert_eq!(Arc::strong_count(&original.map), 1); ++ assert_eq!(original.len(), 2); ++ assert!(original.get_extern("one").is_some()); ++ assert!(original.get_extern("two").is_some()); ++ } ++} ++ + /// This trait is used to mark types as gettable from an [`Instance`]. + /// + /// [`Instance`]: crate::Instance +diff --git a/lib/api/src/entities/instance.rs b/lib/api/src/entities/instance.rs +index edea32f..72944fe 100644 +--- a/lib/api/src/entities/instance.rs ++++ b/lib/api/src/entities/instance.rs +@@ -81,6 +81,72 @@ impl Instance { + }) + } + ++ /// Instantiates an instance with only a selected initial export set. ++ /// ++ /// This is an internal runtime optimization for modules with very large ++ /// dynamic-link symbol tables. The native backend retains the complete ++ /// module export metadata for store-aware lookup through ++ /// [`Self::lookup_export`]. Other backends remain eager to preserve their ++ /// existing semantics. ++ #[doc(hidden)] ++ #[allow(clippy::result_large_err)] ++ pub fn new_with_export_names( ++ store: &mut impl AsStoreMut, ++ module: &Module, ++ imports: &Imports, ++ export_names: &[&str], ++ ) -> Result { ++ let (_inner, exports) = match &store.as_store_mut().inner.store { ++ #[cfg(feature = "sys")] ++ crate::BackendStore::Sys(_) => { ++ let (i, e) = crate::backend::sys::instance::Instance::new_with_export_names( ++ store, ++ module, ++ imports, ++ export_names, ++ )?; ++ (crate::BackendInstance::Sys(i), e) ++ } ++ #[cfg(feature = "v8")] ++ crate::BackendStore::V8(_) => { ++ let (i, e) = crate::backend::v8::instance::Instance::new(store, module, imports)?; ++ (crate::BackendInstance::V8(i), e) ++ } ++ #[cfg(feature = "js")] ++ crate::BackendStore::Js(_) => { ++ let (i, e) = crate::backend::js::instance::Instance::new(store, module, imports)?; ++ (crate::BackendInstance::Js(i), e) ++ } ++ }; ++ ++ Ok(Self { ++ _inner, ++ module: module.clone(), ++ exports, ++ }) ++ } ++ ++ /// Resolves one export using the complete module export table. ++ /// ++ /// For selectively materialized native instances this may allocate the ++ /// corresponding runtime handle on first use. Ordinary eager instances and ++ /// non-native backends return the already materialized export. ++ #[doc(hidden)] ++ pub fn lookup_export(&self, store: &mut impl AsStoreMut, name: &str) -> Option { ++ if let Some(export) = self.exports.get_extern(name) { ++ return Some(export.clone()); ++ } ++ ++ match &self._inner { ++ #[cfg(feature = "sys")] ++ crate::BackendInstance::Sys(instance) => instance.lookup_export(store, name), ++ #[cfg(feature = "v8")] ++ crate::BackendInstance::V8(_) => self.exports.get_extern(name).cloned(), ++ #[cfg(feature = "js")] ++ crate::BackendInstance::Js(_) => self.exports.get_extern(name).cloned(), ++ } ++ } ++ + /// Creates a new `Instance` from a WebAssembly [`Module`] and a + /// vector of imports. + /// +diff --git a/lib/api/src/entities/memory/inner.rs b/lib/api/src/entities/memory/inner.rs +index 5adfc92..6526e21 100644 +--- a/lib/api/src/entities/memory/inner.rs ++++ b/lib/api/src/entities/memory/inner.rs +@@ -170,6 +170,18 @@ impl BackendMemory { + }) + } + ++ #[inline] ++ pub fn supports_persistent_shared_fixed_remap(&self, store: &impl AsStoreRef) -> bool { ++ match self { ++ #[cfg(feature = "sys")] ++ Self::Sys(s) => s.supports_persistent_shared_fixed_remap(store), ++ #[cfg(feature = "v8")] ++ Self::V8(_) => false, ++ #[cfg(feature = "js")] ++ Self::Js(_) => false, ++ } ++ } ++ + /// Attempts to duplicate this memory (if its clonable) in a new store + /// (copied memory) + #[inline] +@@ -205,6 +217,117 @@ impl BackendMemory { + } + } + ++ /// # Safety ++ /// No concurrent linear-memory access or live Rust reference may overlap the replaced range. ++ /// The backing inode must not shrink below its size validated at mapping time; only the ++ /// validated final partial page may extend beyond EOF. ++ #[inline] ++ pub unsafe fn remap_shared_file_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ match self { ++ #[cfg(feature = "sys")] ++ Self::Sys(s) => unsafe { ++ s.remap_shared_file_fixed(store, start, len, file, file_offset) ++ }, ++ #[cfg(feature = "v8")] ++ Self::V8(_) => Err(MemoryError::UnsupportedOperation { ++ message: ++ "fixed file-backed shared memory remapping is only supported by the sys backend" ++ .to_string(), ++ }), ++ #[cfg(feature = "js")] ++ Self::Js(_) => Err(MemoryError::UnsupportedOperation { ++ message: ++ "fixed file-backed shared memory remapping is only supported by the sys backend" ++ .to_string(), ++ }), ++ } ++ } ++ ++ /// # Safety ++ /// No concurrent linear-memory access or live Rust reference may overlap the replaced range, ++ /// and the backing inode must not be truncated or replaced below `file_offset + len` while ++ /// accessible. ++ #[inline] ++ pub unsafe fn remap_private_file_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ match self { ++ #[cfg(feature = "sys")] ++ Self::Sys(s) => unsafe { ++ s.remap_private_file_fixed(store, start, len, file, file_offset) ++ }, ++ #[cfg(feature = "v8")] ++ Self::V8(_) => Err(MemoryError::UnsupportedOperation { ++ message: "fixed private file remapping is only supported by the sys backend" ++ .to_string(), ++ }), ++ #[cfg(feature = "js")] ++ Self::Js(_) => Err(MemoryError::UnsupportedOperation { ++ message: "fixed private file remapping is only supported by the sys backend" ++ .to_string(), ++ }), ++ } ++ } ++ ++ /// # Safety ++ /// No access or Rust reference may overlap the replaced range. ++ #[inline] ++ pub unsafe fn remap_private_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ ) -> Result<(), MemoryError> { ++ match self { ++ #[cfg(feature = "sys")] ++ Self::Sys(s) => unsafe { s.remap_private_fixed(store, start, len) }, ++ #[cfg(feature = "v8")] ++ Self::V8(_) => Err(MemoryError::UnsupportedOperation { ++ message: "fixed private memory remapping is only supported by the sys backend" ++ .to_string(), ++ }), ++ #[cfg(feature = "js")] ++ Self::Js(_) => Err(MemoryError::UnsupportedOperation { ++ message: "fixed private memory remapping is only supported by the sys backend" ++ .to_string(), ++ }), ++ } ++ } ++ ++ #[inline] ++ pub fn msync( ++ &self, ++ store: &impl AsStoreRef, ++ start: usize, ++ len: usize, ++ flags: i32, ++ ) -> Result<(), MemoryError> { ++ match self { ++ #[cfg(feature = "sys")] ++ Self::Sys(s) => s.msync(store, start, len, flags), ++ #[cfg(feature = "v8")] ++ Self::V8(_) => Err(MemoryError::UnsupportedOperation { ++ message: "memory synchronization is only supported by the sys backend".to_string(), ++ }), ++ #[cfg(feature = "js")] ++ Self::Js(_) => Err(MemoryError::UnsupportedOperation { ++ message: "memory synchronization is only supported by the sys backend".to_string(), ++ }), ++ } ++ } ++ + #[inline] + pub(crate) fn from_vm_extern(store: &mut impl AsStoreMut, vm_extern: VMExternMemory) -> Self { + match &store.as_store_mut().inner.store { +diff --git a/lib/api/src/entities/memory/mod.rs b/lib/api/src/entities/memory/mod.rs +index e869b2f..efdfaaf 100644 +--- a/lib/api/src/entities/memory/mod.rs ++++ b/lib/api/src/entities/memory/mod.rs +@@ -151,6 +151,12 @@ impl Memory { + self.0.reset(store) + } + ++ /// Returns whether shared fixed file mappings can remain valid across all ++ /// future growth of this memory. ++ pub fn supports_persistent_shared_fixed_remap(&self, store: &impl AsStoreRef) -> bool { ++ self.0.supports_persistent_shared_fixed_remap(store) ++ } ++ + /// Attempts to duplicate this memory (if its clonable) in a new store + /// (copied memory) + pub fn copy_to_store( +@@ -161,6 +167,81 @@ impl Memory { + self.0.copy_to_store(store, new_store).map(Self) + } + ++ /// Replace an existing page-aligned range in this memory with a shared, ++ /// file-backed host mapping. ++ /// ++ /// # Safety ++ /// ++ /// No guest or host thread may concurrently access the linear memory, and no live Rust ++ /// reference may point into the replaced range while the mapping is changed. The backing inode ++ /// must not shrink below the size validated by this call while the mapping is accessible; only ++ /// the validated final partial page may extend beyond EOF. ++ pub unsafe fn remap_shared_file_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.0 ++ .remap_shared_file_fixed(store, start, len, file, file_offset) ++ } ++ } ++ ++ /// Replace an existing page-aligned range in this memory with a private, ++ /// copy-on-write file-backed host mapping. ++ /// ++ /// Clean pages may be shared by independent memories, while writes remain ++ /// private to the memory that performed them. ++ /// ++ /// # Safety ++ /// ++ /// No guest or host thread may concurrently access the linear memory, and no live Rust ++ /// reference may point into the replaced range while the mapping is changed. The backing inode ++ /// must not be truncated or replaced below `file_offset + len` while the mapping is accessible. ++ pub unsafe fn remap_private_file_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.0 ++ .remap_private_file_fixed(store, start, len, file, file_offset) ++ } ++ } ++ ++ /// Replace an existing page-aligned range in this memory with private, ++ /// zero-filled host memory. ++ /// ++ /// # Safety ++ /// ++ /// No guest or host thread may access the replaced range while the mapping is changed, and ++ /// no live Rust reference may point into it. ++ pub unsafe fn remap_private_fixed( ++ &self, ++ store: &mut impl AsStoreMut, ++ start: usize, ++ len: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { self.0.remap_private_fixed(store, start, len) } ++ } ++ ++ /// Synchronize a page-aligned memory range with its backing file. ++ pub fn msync( ++ &self, ++ store: &impl AsStoreRef, ++ start: usize, ++ len: usize, ++ flags: i32, ++ ) -> Result<(), MemoryError> { ++ self.0.msync(store, start, len, flags) ++ } ++ + pub(crate) fn from_vm_extern(store: &mut impl AsStoreMut, vm_extern: VMExternMemory) -> Self { + Self(BackendMemory::from_vm_extern(store, vm_extern)) + } +diff --git a/lib/api/tests/instance.rs b/lib/api/tests/instance.rs +index 7a57c37..bd3a664 100644 +--- a/lib/api/tests/instance.rs ++++ b/lib/api/tests/instance.rs +@@ -48,6 +48,197 @@ fn exports_work_after_multiple_instances_have_been_freed() -> Result<(), String> + Ok(()) + } + ++#[cfg(feature = "sys")] ++#[test] ++fn selectively_materialized_exports_remain_available_by_module_identity() { ++ let mut store = Store::default(); ++ let module = Module::new( ++ &store, ++ r#" ++(module ++ (func $initialize ++ i32.const 7 ++ global.set $state) ++ (start $initialize) ++ (func $answer (result i32) ++ i32.const 42) ++ (global $state (mut i32) (i32.const 0)) ++ (memory 1) ++ (export "first" (func $answer)) ++ (export "second" (func $answer)) ++ (export "state" (global $state)) ++ (export "memory" (memory 0))) ++"#, ++ ) ++ .unwrap(); ++ ++ // The ordinary API remains eager and keeps its established surface. ++ let eager = Instance::new(&mut store, &module, &Imports::new()).unwrap(); ++ assert_eq!(eager.exports.len(), 4); ++ ++ let deferred = ++ Instance::new_with_export_names(&mut store, &module, &Imports::new(), &["first"]).unwrap(); ++ assert_eq!(deferred.exports.len(), 1); ++ assert!(deferred.exports.get_extern("second").is_none()); ++ ++ let first = deferred.exports.get_function("first").unwrap().clone(); ++ let Extern::Function(second) = deferred.lookup_export(&mut store, "second").unwrap() else { ++ panic!("second must be a function") ++ }; ++ assert_eq!(first, second, "function aliases must share VM identity"); ++ assert_eq!(second.call(&mut store, &[]).unwrap()[0], Value::I32(42)); ++ ++ // A clone reaches exports that were never part of the initial materialized ++ // set, and Wasm start execution was not bypassed by selective export setup. ++ let cloned = deferred.clone(); ++ let Extern::Global(state) = cloned.lookup_export(&mut store, "state").unwrap() else { ++ panic!("state must be a global") ++ }; ++ assert_eq!(state.get(&mut store), Value::I32(7)); ++ assert!(cloned.lookup_export(&mut store, "memory").is_some()); ++ assert!(cloned.lookup_export(&mut store, "does-not-exist").is_none()); ++ ++ // Deferred lookups stay in the compact identity cache rather than growing ++ // the public string-keyed export map on every Instance clone. ++ assert_eq!(deferred.exports.len(), 1); ++} ++ ++#[engine_test] ++fn passive_data_drop_is_local_to_each_instance() -> Result<(), String> { ++ let mut store = Store::default(); ++ let module = Module::new( ++ &store, ++ r#" ++(module ++ (memory (export "memory") 1) ++ (data $payload "shared") ++ (func (export "init") (param $dst i32) (param $src i32) (param $len i32) ++ local.get $dst ++ local.get $src ++ local.get $len ++ memory.init $payload) ++ (func (export "drop") ++ data.drop $payload)) ++"#, ++ ) ++ .unwrap(); ++ ++ let first = Instance::new(&mut store, &module, &Imports::new()).unwrap(); ++ let second = Instance::new(&mut store, &module, &Imports::new()).unwrap(); ++ let third = Instance::new(&mut store, &module, &Imports::new()).unwrap(); ++ let first_init: TypedFunction<(i32, i32, i32), ()> = ++ first.exports.get_typed_function(&store, "init").unwrap(); ++ let first_drop: TypedFunction<(), ()> = ++ first.exports.get_typed_function(&store, "drop").unwrap(); ++ let second_init: TypedFunction<(i32, i32, i32), ()> = ++ second.exports.get_typed_function(&store, "init").unwrap(); ++ let second_memory = second.exports.get_memory("memory").unwrap().clone(); ++ let third_init: TypedFunction<(i32, i32, i32), ()> = ++ third.exports.get_typed_function(&store, "init").unwrap(); ++ let third_memory = third.exports.get_memory("memory").unwrap().clone(); ++ ++ first_init.call(&mut store, 0, 0, 6).unwrap(); ++ first_drop.call(&mut store).unwrap(); ++ first_drop.call(&mut store).unwrap(); ++ ++ // A dropped segment has length zero. Its empty range remains valid while ++ // a non-empty read traps, including after repeated `data.drop`s. ++ first_init.call(&mut store, 65_536, 0, 0).unwrap(); ++ first_init ++ .call(&mut store, 16, 0, 1) ++ .expect_err("memory.init after data.drop must trap"); ++ ++ // Dropping one instance must not mutate the bytes shared by its module or ++ // the independent dropped-segment state of another instance. ++ second_init.call(&mut store, 16, 0, 6).unwrap(); ++ let mut second_bytes = [0_u8; 6]; ++ second_memory ++ .view(&store) ++ .read(16, &mut second_bytes) ++ .unwrap(); ++ assert_eq!(&second_bytes, b"shared"); ++ ++ // Surviving instances own the shared module bytes. They remain usable after ++ // both the original `Module` handle and the instance that dropped its ++ // segment have gone away. ++ drop(first_init); ++ drop(first_drop); ++ drop(first); ++ drop(module); ++ third_init.call(&mut store, 24, 0, 6).unwrap(); ++ let mut third_bytes = [0_u8; 6]; ++ third_memory ++ .view(&store) ++ .read(24, &mut third_bytes) ++ .unwrap(); ++ assert_eq!(&third_bytes, b"shared"); ++ ++ Ok(()) ++} ++ ++#[engine_test] ++fn passive_data_memory_init_preserves_contents_and_bounds() -> Result<(), String> { ++ let mut store = Store::default(); ++ let module = Module::new( ++ &store, ++ r#" ++(module ++ (memory (export "memory") 1) ++ (data $payload "shared") ++ (func (export "init") (param $dst i32) (param $src i32) (param $len i32) ++ local.get $dst ++ local.get $src ++ local.get $len ++ memory.init $payload) ++ (func (export "drop") ++ data.drop $payload)) ++"#, ++ ) ++ .unwrap(); ++ ++ let instance = Instance::new(&mut store, &module, &Imports::new()).unwrap(); ++ let init: TypedFunction<(i32, i32, i32), ()> = ++ instance.exports.get_typed_function(&store, "init").unwrap(); ++ let drop_data: TypedFunction<(), ()> = ++ instance.exports.get_typed_function(&store, "drop").unwrap(); ++ let memory = instance.exports.get_memory("memory").unwrap().clone(); ++ ++ init.call(&mut store, 8, 0, 6).unwrap(); ++ init.call(&mut store, 24, 1, 4).unwrap(); ++ let mut complete = [0_u8; 6]; ++ let mut partial = [0_u8; 4]; ++ memory.view(&store).read(8, &mut complete).unwrap(); ++ memory.view(&store).read(24, &mut partial).unwrap(); ++ assert_eq!(&complete, b"shared"); ++ assert_eq!(&partial, b"hare"); ++ ++ // Empty ranges at the exact source/destination ends are valid. ++ init.call(&mut store, 65_536, 6, 0).unwrap(); ++ init.call(&mut store, 0, 6, 0).unwrap(); ++ ++ // Source and destination ranges retain the WebAssembly bounds behavior. ++ init.call(&mut store, 0, 7, 0) ++ .expect_err("an empty source range beyond the segment must trap"); ++ init.call(&mut store, 0, 5, 2) ++ .expect_err("a non-empty source range beyond the segment must trap"); ++ init.call(&mut store, 65_537, 0, 0) ++ .expect_err("an empty destination range beyond memory must trap"); ++ init.call(&mut store, 65_535, 0, 2) ++ .expect_err("a non-empty destination range beyond memory must trap"); ++ ++ drop_data.call(&mut store).unwrap(); ++ ++ // A dropped segment behaves as an empty slice: its only valid source range ++ // is (0, 0), while destination bounds are still checked normally. ++ init.call(&mut store, 65_536, 0, 0).unwrap(); ++ init.call(&mut store, 0, 1, 0) ++ .expect_err("an offset beyond a dropped segment must trap"); ++ init.call(&mut store, 0, 0, 1) ++ .expect_err("a non-empty read from a dropped segment must trap"); ++ ++ Ok(()) ++} ++ + #[engine_test] + fn unit_native_function_env() -> Result<(), String> { + let mut store = Store::default(); +diff --git a/lib/api/tests/memory.rs b/lib/api/tests/memory.rs +index 52c3c9b..1b671e4 100644 +--- a/lib/api/tests/memory.rs ++++ b/lib/api/tests/memory.rs +@@ -2,7 +2,7 @@ use std::sync::{ + Arc, + atomic::{AtomicBool, Ordering}, + }; +-use wasmer::{Instance, Memory, MemoryLocation, MemoryType, Module, Store, imports}; ++use wasmer::{Instance, Memory, MemoryLocation, MemoryType, Module, Pages, Store, imports}; + + #[test] + #[allow(unused_attributes)] +@@ -115,6 +115,137 @@ fn test_wasm_slice_issue_5444() { + )) + } + ++#[cfg(all(feature = "sys", unix))] ++#[test] ++fn private_file_remap_preserves_memory_base_growth_and_mapping_lifetime() -> anyhow::Result<()> { ++ use std::io::{Read, Seek, Write}; ++ ++ const WASM_PAGE_SIZE: usize = 64 * 1024; ++ let mut store = Store::default(); ++ let memory = Memory::new(&mut store, MemoryType::new(Pages(1), Some(Pages(3)), false))?; ++ let base = memory.view(&store).data_ptr(); ++ ++ let mut backing = tempfile::tempfile()?; ++ backing.write_all(&vec![0x7a; WASM_PAGE_SIZE])?; ++ backing.sync_all()?; ++ backing.rewind()?; ++ // SAFETY: the test has exclusive access to the store and memory, and the ++ // backing remains alive and unmodified until after remapping completes. ++ unsafe { ++ memory.remap_private_file_fixed(&mut store, 0, WASM_PAGE_SIZE, &backing, 0)?; ++ } ++ ++ assert_eq!(memory.view(&store).data_ptr(), base); ++ memory.view(&store).write(0, &[0x42])?; ++ let mut backing_byte = [0_u8; 1]; ++ backing.read_exact(&mut backing_byte)?; ++ assert_eq!(backing_byte, [0x7a]); ++ ++ drop(backing); ++ let mut mapped_byte = [0_u8; 1]; ++ memory.view(&store).read(0, &mut mapped_byte)?; ++ assert_eq!(mapped_byte, [0x42]); ++ ++ memory.grow(&mut store, Pages(1))?; ++ assert_eq!(memory.view(&store).data_ptr(), base); ++ assert_eq!(memory.view(&store).size(), Pages(2)); ++ memory.view(&store).write(WASM_PAGE_SIZE as u64, &[0x55])?; ++ let mut grown_byte = [0_u8; 1]; ++ memory ++ .view(&store) ++ .read(WASM_PAGE_SIZE as u64, &mut grown_byte)?; ++ assert_eq!(grown_byte, [0x55]); ++ Ok(()) ++} ++ ++#[cfg(feature = "llvm")] ++#[test] ++fn test_llvm_imported_dynamic_memory_bulk_ops() -> anyhow::Result<()> { ++ const PAGE_SIZE: u64 = 65536; ++ ++ fn read_memory( ++ memory: &Memory, ++ store: &Store, ++ offset: u64, ++ len: usize, ++ ) -> anyhow::Result> { ++ let mut bytes = vec![0; len]; ++ memory.view(store).read(offset, &mut bytes)?; ++ Ok(bytes) ++ } ++ ++ let engine: wasmer::Engine = wasmer::sys::EngineBuilder::new(wasmer::sys::LLVM::default()) ++ .engine() ++ .into(); ++ let mut store = Store::new(engine); ++ let wat = r#" ++ (module ++ (import "env" "memory" (memory 1 1)) ++ (func (export "copy") (param i32 i32 i32) ++ local.get 0 ++ local.get 1 ++ local.get 2 ++ memory.copy) ++ (func (export "fill") (param i32 i32 i32) ++ local.get 0 ++ local.get 1 ++ local.get 2 ++ memory.fill) ++ ) ++ "#; ++ let module = Module::new(&store, wat)?; ++ let memory = Memory::new(&mut store, MemoryType::new(Pages(1), Some(Pages(1)), false))?; ++ let imports = imports! { ++ "env" => { ++ "memory" => memory.clone(), ++ } ++ }; ++ let instance = Instance::new(&mut store, &module, &imports)?; ++ let copy: wasmer::TypedFunction<(i32, i32, i32), ()> = ++ instance.exports.get_typed_function(&store, "copy")?; ++ let fill: wasmer::TypedFunction<(i32, i32, i32), ()> = ++ instance.exports.get_typed_function(&store, "fill")?; ++ ++ memory.view(&store).write(8, b"abcdef")?; ++ copy.call(&mut store, 16, 8, 6)?; ++ assert_eq!(read_memory(&memory, &store, 16, 6)?, b"abcdef"); ++ ++ memory.view(&store).write(32, b"0123456789")?; ++ copy.call(&mut store, 34, 32, 8)?; ++ assert_eq!(read_memory(&memory, &store, 32, 10)?, b"0101234567"); ++ ++ fill.call(&mut store, 48, 0x7a, 4)?; ++ assert_eq!(read_memory(&memory, &store, 48, 4)?, b"zzzz"); ++ ++ memory.view(&store).write(PAGE_SIZE - 4, &[1, 2, 3, 4])?; ++ let result = fill ++ .call(&mut store, (PAGE_SIZE - 2) as i32, 0x41, 4) ++ .unwrap_err(); ++ assert_eq!( ++ result.to_trap(), ++ Some(wasmer_types::TrapCode::HeapAccessOutOfBounds) ++ ); ++ assert_eq!( ++ read_memory(&memory, &store, PAGE_SIZE - 4, 4)?, ++ &[1, 2, 3, 4] ++ ); ++ ++ memory.view(&store).write(PAGE_SIZE - 4, &[5, 6, 7, 8])?; ++ let result = copy ++ .call(&mut store, (PAGE_SIZE - 2) as i32, 8, 4) ++ .unwrap_err(); ++ assert_eq!( ++ result.to_trap(), ++ Some(wasmer_types::TrapCode::HeapAccessOutOfBounds) ++ ); ++ assert_eq!( ++ read_memory(&memory, &store, PAGE_SIZE - 4, 4)?, ++ &[5, 6, 7, 8] ++ ); ++ ++ Ok(()) ++} ++ + #[test] + fn test_wasm_memory_size() { + let mut store = Store::default(); +diff --git a/lib/api/tests/module.rs b/lib/api/tests/module.rs +index df286eb..6ad59ec 100644 +--- a/lib/api/tests/module.rs ++++ b/lib/api/tests/module.rs +@@ -4,6 +4,27 @@ use wasm_bindgen_test::*; + + use wasmer::*; + ++#[cfg(all(feature = "sys", feature = "compiler"))] ++use std::io::{Seek, Write}; ++#[cfg(all( ++ feature = "sys", ++ feature = "compiler", ++ target_os = "linux", ++ target_arch = "x86_64" ++))] ++use std::os::unix::fs::PermissionsExt; ++#[cfg(all( ++ feature = "sys", ++ feature = "compiler", ++ target_os = "linux", ++ target_arch = "x86_64" ++))] ++use wasmer::sys::CodeMemoryPolicy; ++#[cfg(all(feature = "sys", feature = "compiler"))] ++use wasmer::sys::{NativeEngineExt, OwnedBuffer}; ++#[cfg(all(feature = "sys", feature = "compiler"))] ++use wasmer_types::MetadataHeader; ++ + #[cfg(unix)] + use std::ffi::OsStr; + #[cfg(unix)] +@@ -32,6 +53,367 @@ fn module_set_name() -> Result<(), String> { + Ok(()) + } + ++#[cfg(all(feature = "sys", feature = "compiler"))] ++fn mapped_artifact(bytes: &[u8]) -> Result { ++ let mut artifact_file = tempfile::tempfile().map_err(|error| error.to_string())?; ++ artifact_file ++ .write_all(bytes) ++ .map_err(|error| error.to_string())?; ++ artifact_file.rewind().map_err(|error| error.to_string())?; ++ OwnedBuffer::from_file(&artifact_file).map_err(|error| error.to_string()) ++} ++ ++#[test] ++#[cfg(all(feature = "sys", feature = "compiler"))] ++fn serialized_artifact_inspector_returns_embedded_hash() -> Result<(), String> { ++ let compiler_store = Store::default(); ++ let compiled = Module::new( ++ &compiler_store, ++ r#"(module (func (export "answer") (result i32) i32.const 42))"#, ++ ) ++ .map_err(|error| error.to_string())?; ++ let expected_hash = compiled ++ .info() ++ .hash ++ .ok_or_else(|| "compiled module is missing its hash".to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ let mapping = mapped_artifact(&serialized)?; ++ ++ let engine = Engine::headless(); ++ let inspected_hash = engine ++ .inspect_serialized_artifact(&mapping) ++ .map_err(|error| error.to_string())?; ++ ++ assert_eq!(inspected_hash, expected_hash); ++ Ok(()) ++} ++ ++#[test] ++#[cfg(all(feature = "sys", feature = "compiler"))] ++fn serialized_artifact_inspector_checks_magic_abi_and_archive() -> Result<(), String> { ++ let compiler_store = Store::default(); ++ let compiled = Module::new(&compiler_store, "(module)").map_err(|error| error.to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ let engine = Engine::headless(); ++ ++ let mut wrong_magic = serialized.to_vec(); ++ wrong_magic[0] ^= 0xff; ++ let error = engine ++ .inspect_serialized_artifact(&mapped_artifact(&wrong_magic)?) ++ .expect_err("an invalid universal magic must be rejected"); ++ assert!(matches!(error, DeserializeError::Incompatible(_))); ++ ++ let mut wrong_abi = serialized.to_vec(); ++ const UNIVERSAL_MAGIC_LEN: usize = 16; ++ const METADATA_VERSION_OFFSET: usize = UNIVERSAL_MAGIC_LEN + 8; ++ wrong_abi[METADATA_VERSION_OFFSET..METADATA_VERSION_OFFSET + std::mem::size_of::()] ++ .copy_from_slice(&(MetadataHeader::CURRENT_VERSION + 1).to_ne_bytes()); ++ let error = engine ++ .inspect_serialized_artifact(&mapped_artifact(&wrong_abi)?) ++ .expect_err("an incompatible metadata ABI must be rejected"); ++ assert!(matches!(error, DeserializeError::Incompatible(_))); ++ ++ let mut invalid_archive = b"wasmer-universal".to_vec(); ++ invalid_archive.extend(MetadataHeader::new(1).into_bytes()); ++ invalid_archive.push(0); ++ let error = engine ++ .inspect_serialized_artifact(&mapped_artifact(&invalid_archive)?) ++ .expect_err("a malformed rkyv archive must be rejected"); ++ assert!(matches!(error, DeserializeError::CorruptedBinary(_))); ++ ++ Ok(()) ++} ++ ++#[test] ++#[cfg(all(feature = "sys", feature = "compiler"))] ++fn detached_artifact_passive_data_has_instance_local_drop_state() -> Result<(), String> { ++ let compiler_store = Store::default(); ++ let compiled = Module::new( ++ &compiler_store, ++ r#"(module ++ (memory (export "memory") 1) ++ (data $payload "artifact") ++ (func (export "init") (param $dst i32) (param $src i32) (param $len i32) ++ local.get $dst ++ local.get $src ++ local.get $len ++ memory.init $payload) ++ (func (export "drop") ++ data.drop $payload) ++ )"#, ++ ) ++ .map_err(|error| error.to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ ++ let engine = Engine::headless(); ++ let detached = ++ unsafe { engine.deserialize_from_mmapped_buffer_detached(mapped_artifact(&serialized)?) } ++ .map_err(|error| error.to_string())?; ++ ++ let mut store = Store::new(engine); ++ let first = ++ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; ++ let second = ++ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; ++ let first_init = first ++ .exports ++ .get_typed_function::<(i32, i32, i32), ()>(&store, "init") ++ .map_err(|error| error.to_string())?; ++ let first_drop = first ++ .exports ++ .get_typed_function::<(), ()>(&store, "drop") ++ .map_err(|error| error.to_string())?; ++ let second_init = second ++ .exports ++ .get_typed_function::<(i32, i32, i32), ()>(&store, "init") ++ .map_err(|error| error.to_string())?; ++ let second_memory = second ++ .exports ++ .get_memory("memory") ++ .map_err(|error| error.to_string())? ++ .clone(); ++ ++ first_drop ++ .call(&mut store) ++ .map_err(|error| error.to_string())?; ++ first_init ++ .call(&mut store, 0, 0, 1) ++ .expect_err("the first detached instance must observe its data.drop"); ++ ++ // Instance module handles keep the detached metadata, including passive ++ // bytes, alive independently of the original detached `Module` handle. ++ drop(detached); ++ second_init ++ .call(&mut store, 32, 0, 8) ++ .map_err(|error| error.to_string())?; ++ let mut copied = [0_u8; 8]; ++ second_memory ++ .view(&store) ++ .read(32, &mut copied) ++ .map_err(|error| error.to_string())?; ++ assert_eq!(&copied, b"artifact"); ++ ++ Ok(()) ++} ++ ++#[test] ++#[cfg(all( ++ feature = "sys", ++ feature = "compiler", ++ feature = "cranelift", ++ any(target_arch = "x86_64", target_arch = "aarch64") ++))] ++fn serialized_artifact_inspector_checks_native_cpu_features() -> Result<(), String> { ++ use wasmer::sys::{CpuFeature, Cranelift, Features, Target, Triple}; ++ ++ let compiler_store = Store::default(); ++ let compiled = Module::new(&compiler_store, "(module)").map_err(|error| error.to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ let mapping = mapped_artifact(&serialized)?; ++ ++ let inspection_engine = ::new( ++ Box::new(Cranelift::default()), ++ Target::new(Triple::host(), CpuFeature::set()), ++ Features::default(), ++ ); ++ let error = inspection_engine ++ .inspect_serialized_artifact(&mapping) ++ .expect_err("an artifact requiring unavailable CPU features must be rejected"); ++ ++ assert!(matches!(error, DeserializeError::Incompatible(_))); ++ Ok(()) ++} ++ ++#[test] ++#[cfg(all(feature = "sys", feature = "compiler"))] ++fn detached_mmapped_module_executes_without_retaining_serializable_state() -> Result<(), String> { ++ let compiler_store = Store::default(); ++ let compiled = Module::new( ++ &compiler_store, ++ r#"(module ++ (memory 1) ++ (data (i32.const 1024) "sealed") ++ (func (export "answer") (result i32) i32.const 42) ++ )"#, ++ ) ++ .map_err(|error| error.to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ ++ let mapping = mapped_artifact(&serialized)?; ++ ++ let engine = Engine::headless(); ++ let detached = unsafe { engine.deserialize_from_mmapped_buffer_detached(mapping) } ++ .map_err(|error| error.to_string())?; ++ assert!( ++ detached.serialize().is_err(), ++ "a detached runtime module must not retain state only needed for re-serialization" ++ ); ++ ++ let mut store = Store::new(engine); ++ let instance = ++ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; ++ let answer = instance ++ .exports ++ .get_typed_function::<(), i32>(&store, "answer") ++ .map_err(|error| error.to_string())?; ++ assert_eq!( ++ answer.call(&mut store).map_err(|error| error.to_string())?, ++ 42 ++ ); ++ Ok(()) ++} ++ ++#[test] ++#[cfg(all( ++ feature = "sys", ++ feature = "compiler", ++ target_os = "linux", ++ target_arch = "x86_64" ++))] ++fn abandoned_pending_detached_activation_releases_strict_code_memory() -> Result<(), String> { ++ let compiler_store = Store::default(); ++ let compiled = Module::new( ++ &compiler_store, ++ r#"(module (func (export "answer") (result i32) i32.const 42))"#, ++ ) ++ .map_err(|error| error.to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ ++ let directory = tempfile::Builder::new() ++ .prefix("wasmer-pending-code-memory-module-test-") ++ .tempdir_in("/var/tmp") ++ .map_err(|error| error.to_string())?; ++ std::fs::set_permissions(directory.path(), std::fs::Permissions::from_mode(0o700)) ++ .map_err(|error| error.to_string())?; ++ let policy = CodeMemoryPolicy::strict_linux_x86_64_file_backed(directory.path())?; ++ let mut engine = Engine::headless(); ++ engine.set_code_memory_policy(policy)?; ++ ++ let baseline = engine.as_sys().code_memory_allocation_count(); ++ let pending = unsafe { ++ engine.deserialize_from_mmapped_buffer_detached_pending(mapped_artifact(&serialized)?) ++ } ++ .map_err(|error| error.to_string())?; ++ assert_eq!(engine.as_sys().code_memory_allocation_count(), baseline + 1); ++ assert!(pending.module_hash().is_some()); ++ ++ drop(pending); ++ assert_eq!(engine.as_sys().code_memory_allocation_count(), baseline); ++ Ok(()) ++} ++ ++#[test] ++#[cfg(all( ++ feature = "sys", ++ feature = "compiler", ++ target_os = "linux", ++ target_arch = "x86_64" ++))] ++fn detached_module_executes_from_strict_relocated_regular_file_code_memory() -> Result<(), String> { ++ let compiler_store = Store::default(); ++ let compiled = Module::new( ++ &compiler_store, ++ r#"(module $strict_detached_frames ++ (func $inner (unreachable)) ++ (func (export "answer") (result i32) i32.const 42) ++ (func (export "run") (call $inner)) ++ )"#, ++ ) ++ .map_err(|error| error.to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ ++ let directory = tempfile::Builder::new() ++ .prefix("wasmer-strict-code-memory-module-test-") ++ .tempdir_in("/var/tmp") ++ .map_err(|error| error.to_string())?; ++ std::fs::set_permissions(directory.path(), std::fs::Permissions::from_mode(0o700)) ++ .map_err(|error| error.to_string())?; ++ let policy = CodeMemoryPolicy::strict_linux_x86_64_file_backed(directory.path())?; ++ let mut engine = Engine::headless(); ++ engine.set_code_memory_policy(policy)?; ++ let detached = ++ unsafe { engine.deserialize_from_mmapped_buffer_detached(mapped_artifact(&serialized)?) } ++ .map_err(|error| error.to_string())?; ++ ++ let mut store = Store::new(engine); ++ let instance = ++ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; ++ let answer = instance ++ .exports ++ .get_typed_function::<(), i32>(&store, "answer") ++ .map_err(|error| error.to_string())?; ++ assert_eq!( ++ answer.call(&mut store).map_err(|error| error.to_string())?, ++ 42 ++ ); ++ ++ let run = instance ++ .exports ++ .get_typed_function::<(), ()>(&store, "run") ++ .map_err(|error| error.to_string())?; ++ let error = run ++ .call(&mut store) ++ .expect_err("strict detached code must trap"); ++ let trace = error.trace(); ++ assert_eq!(trace.len(), 2, "strict code must retain both trap frames"); ++ assert_eq!(trace[0].module_name(), "strict_detached_frames"); ++ assert_eq!(trace[0].func_index(), 0); ++ assert_eq!(trace[0].function_name(), Some("inner")); ++ assert_eq!(trace[1].module_name(), "strict_detached_frames"); ++ assert_eq!(trace[1].func_index(), 2); ++ assert_eq!(trace[1].function_name(), None); ++ assert!(error.message().contains("unreachable")); ++ Ok(()) ++} ++ ++#[test] ++#[cfg(all(feature = "sys", feature = "compiler"))] ++#[cfg_attr(target_env = "musl", ignore)] ++fn detached_mmapped_module_retains_trap_frame_metadata() -> Result<(), String> { ++ let compiler_store = Store::default(); ++ let compiled = Module::new( ++ &compiler_store, ++ r#"(module $detached_frames ++ (func $inner (unreachable)) ++ (func (export "run") (call $inner)) ++ )"#, ++ ) ++ .map_err(|error| error.to_string())?; ++ let serialized = compiled.serialize().map_err(|error| error.to_string())?; ++ ++ let engine = Engine::headless(); ++ let detached = ++ unsafe { engine.deserialize_from_mmapped_buffer_detached(mapped_artifact(&serialized)?) } ++ .map_err(|error| error.to_string())?; ++ ++ let mut store = Store::new(engine); ++ let instance = ++ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; ++ let run = instance ++ .exports ++ .get_typed_function::<(), ()>(&store, "run") ++ .map_err(|error| error.to_string())?; ++ let error = run ++ .call(&mut store) ++ .expect_err("the detached module must trap"); ++ let trace = error.trace(); ++ ++ assert_eq!( ++ trace.len(), ++ 2, ++ "detached trap metadata must retain both frames" ++ ); ++ assert_eq!(trace[0].module_name(), "detached_frames"); ++ assert_eq!(trace[0].func_index(), 0); ++ assert_eq!(trace[0].function_name(), Some("inner")); ++ assert_eq!(trace[1].module_name(), "detached_frames"); ++ assert_eq!(trace[1].func_index(), 1); ++ assert_eq!(trace[1].function_name(), None); ++ assert!(error.message().contains("unreachable")); ++ ++ Ok(()) ++} ++ + #[engine_test] + fn imports() -> Result<(), String> { + let store = Store::default(); +diff --git a/lib/compiler-llvm/src/compiler.rs b/lib/compiler-llvm/src/compiler.rs +index b04e437..3950012 100644 +--- a/lib/compiler-llvm/src/compiler.rs ++++ b/lib/compiler-llvm/src/compiler.rs +@@ -178,13 +178,17 @@ impl Compiler for LLVMCompiler { + + fn deterministic_id(&self) -> String { + format!( +- "llvm-{}", ++ "llvm-v3-{}-nan{}-nv{}-rotable{}-pic{}", + match self.config.opt_level { + inkwell::OptimizationLevel::None => "opt0", + inkwell::OptimizationLevel::Less => "optl", + inkwell::OptimizationLevel::Default => "optd", + inkwell::OptimizationLevel::Aggressive => "opta", +- } ++ }, ++ u8::from(self.config.enable_nan_canonicalization), ++ u8::from(self.config.enable_non_volatile_memops), ++ u8::from(self.config.enable_readonly_funcref_table), ++ u8::from(self.config.is_pic), + ) + } + +diff --git a/lib/compiler-llvm/src/config.rs b/lib/compiler-llvm/src/config.rs +index d732134..58918f0 100644 +--- a/lib/compiler-llvm/src/config.rs ++++ b/lib/compiler-llvm/src/config.rs +@@ -109,7 +109,7 @@ pub struct LLVM { + pub(crate) enable_verifier: bool, + pub(crate) enable_perfmap: bool, + pub(crate) opt_level: LLVMOptLevel, +- is_pic: bool, ++ pub(crate) is_pic: bool, + pub(crate) callbacks: Option, + /// The middleware chain. + pub(crate) middlewares: Vec>, +diff --git a/lib/compiler-llvm/src/object_file.rs b/lib/compiler-llvm/src/object_file.rs +index 7265bca..816be2e 100644 +--- a/lib/compiler-llvm/src/object_file.rs ++++ b/lib/compiler-llvm/src/object_file.rs +@@ -46,6 +46,10 @@ static LIBCALLS_ELF: phf::Map<&'static str, LibCall> = phf::phf_map! { + "truncf" => LibCall::TruncF32, + "trunc" => LibCall::TruncF64, + "__chkstk" => LibCall::Probestack, ++ "bzero" => LibCall::HostBzero, ++ "memset" => LibCall::HostMemset, ++ "memcpy" => LibCall::HostMemcpy, ++ "memmove" => LibCall::HostMemmove, + "wasmer_vm_f32_ceil" => LibCall::CeilF32, + "wasmer_vm_f64_ceil" => LibCall::CeilF64, + "wasmer_vm_f32_floor" => LibCall::FloorF32, +@@ -101,6 +105,10 @@ static LIBCALLS_MACHO: phf::Map<&'static str, LibCall> = phf::phf_map! { + "_nearbyint" => LibCall::NearestF64, + "_truncf" => LibCall::TruncF32, + "_trunc" => LibCall::TruncF64, ++ "_bzero" => LibCall::HostBzero, ++ "_memset" => LibCall::HostMemset, ++ "_memcpy" => LibCall::HostMemcpy, ++ "_memmove" => LibCall::HostMemmove, + "_wasmer_vm_f32_ceil" => LibCall::CeilF32, + "_wasmer_vm_f64_ceil" => LibCall::CeilF64, + "_wasmer_vm_f32_floor" => LibCall::FloorF32, +diff --git a/lib/compiler-llvm/src/translator/code.rs b/lib/compiler-llvm/src/translator/code.rs +index fdf0509..9fc1f51 100644 +--- a/lib/compiler-llvm/src/translator/code.rs ++++ b/lib/compiler-llvm/src/translator/code.rs +@@ -361,9 +361,8 @@ impl FuncTranslator { + ); + + while fcg.state.has_control_frames() { +- let pos = reader.current_position() as u32; + let op = reader.read_operator()?; +- fcg.translate_operator(op, pos)?; ++ fcg.translate_operator(op)?; + } + + fcg.finalize(wasm_fn_type)?; +@@ -445,16 +444,25 @@ impl FuncTranslator { + target: &Triple, + ) -> Result { + let func_index = wasm_module.func_index(*local_func_index); ++ let function_body_len = function_body.data.len() as u64; + let opt_style = if Some(func_index) == self.wasm_apply_data_relocs_fn_index { + // `__wasm_apply_data_relocs` can become a very large function made up + // mostly of loads and stores, and even `-O1` can spend significant + // time optimizing it. + OptimizationStyle::Disabled +- } else if function_body.data.len() as u64 > WASM_LARGE_FUNCTION_THRESHOLD { ++ } else if function_body_len > WASM_LARGE_FUNCTION_THRESHOLD { + OptimizationStyle::ForSize + } else { + OptimizationStyle::ForSpeed + }; ++ if function_body_len > WASM_LARGE_FUNCTION_THRESHOLD { ++ tracing::debug!( ++ function = %wasm_module.get_function_name(func_index), ++ body_len = function_body_len, ++ ?opt_style, ++ "selected LLVM optimization style for large function" ++ ); ++ } + let module = self.translate_to_module( + wasm_module, + module_translation, +@@ -1247,7 +1255,7 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + // If this memory access must trap when out of bounds (i.e. it is a memory + // access written in the user program as opposed to one used by our VM) + // then mark that it can't be deleted. +- if let MemoryCache::Static { base_ptr: _ } = self.ctx.memory( ++ if let MemoryCache::Static { .. } = self.ctx.memory( + memory_index, + self.intrinsics, + self.module, +@@ -1739,7 +1747,7 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + ); + ptr_to_base + } +- MemoryCache::Static { base_ptr } => base_ptr, ++ MemoryCache::Static { base_ptr, .. } => base_ptr, + } + }; + let value_ptr = unsafe { +@@ -1752,6 +1760,160 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + ) + } + ++ fn memory_bulk_parts( ++ &mut self, ++ memory_index: MemoryIndex, ++ label: &str, ++ ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { ++ match self.ctx.memory( ++ memory_index, ++ self.intrinsics, ++ self.module, ++ self.memory_styles, ++ )? { ++ MemoryCache::Dynamic { ++ ptr_to_base_ptr, ++ ptr_to_current_length, ++ } => { ++ let base = self.build_dynamic_memory_base(ptr_to_base_ptr, memory_index, label)?; ++ Ok((base, ptr_to_current_length)) ++ } ++ MemoryCache::Static { ++ base_ptr, ++ ptr_to_current_length, ++ } => Ok((base_ptr, ptr_to_current_length)), ++ } ++ } ++ ++ fn int_value_as_u64( ++ &self, ++ value: IntValue<'ctx>, ++ label: &str, ++ ) -> Result, CompileError> { ++ match value.get_type().get_bit_width() { ++ 64 => Ok(value), ++ width if width < 64 => Ok(err!(self.builder.build_int_z_extend( ++ value, ++ self.intrinsics.i64_ty, ++ &format!("{label}_zext") ++ ))), ++ width => Err(CompileError::Codegen(format!( ++ "unsupported memory index width {width}" ++ ))), ++ } ++ } ++ ++ fn build_memory_range_check( ++ &self, ++ offset: IntValue<'ctx>, ++ len: IntValue<'ctx>, ++ ptr_to_current_length: PointerValue<'ctx>, ++ label: &str, ++ ) -> Result, CompileError> { ++ let offset = self.int_value_as_u64(offset, &format!("{label}_offset"))?; ++ let len = self.int_value_as_u64(len, &format!("{label}_len"))?; ++ let current_length = err!(self.builder.build_load( ++ self.intrinsics.i32_ty, ++ ptr_to_current_length, ++ &format!("{label}_current_length") ++ )) ++ .into_int_value(); ++ tbaa_label( ++ self.module, ++ self.intrinsics, ++ format!("{label} memory length"), ++ current_length.as_instruction_value().unwrap(), ++ ); ++ let current_length = err!(self.builder.build_int_z_extend( ++ current_length, ++ self.intrinsics.i64_ty, ++ &format!("{label}_current_length_zext") ++ )); ++ let offset_in_bounds = err!(self.builder.build_int_compare( ++ IntPredicate::ULE, ++ offset, ++ current_length, ++ &format!("{label}_offset_in_bounds") ++ )); ++ let remaining = err!(self.builder.build_int_sub( ++ current_length, ++ offset, ++ &format!("{label}_remaining") ++ )); ++ let len_in_bounds = err!(self.builder.build_int_compare( ++ IntPredicate::ULE, ++ len, ++ remaining, ++ &format!("{label}_len_in_bounds") ++ )); ++ Ok(err!(self.builder.build_and( ++ offset_in_bounds, ++ len_in_bounds, ++ &format!("{label}_in_bounds") ++ ))) ++ } ++ ++ fn build_memory_oob_trap( ++ &self, ++ in_bounds: IntValue<'ctx>, ++ label: &str, ++ ) -> Result<(), CompileError> { ++ let in_bounds = err!(self.build_call_with_param_attributes( ++ self.intrinsics.expect_i1, ++ &[ ++ in_bounds.into(), ++ self.intrinsics.i1_ty.const_int(1, true).into(), ++ ], ++ &format!("{label}_in_bounds_expect"), ++ )) ++ .try_as_basic_value() ++ .unwrap_basic() ++ .into_int_value(); ++ ++ let continue_block = self ++ .context ++ .append_basic_block(self.function, &format!("{label}_in_bounds_continue")); ++ let trap_block = self ++ .context ++ .append_basic_block(self.function, &format!("{label}_oob_trap")); ++ err!( ++ self.builder ++ .build_conditional_branch(in_bounds, continue_block, trap_block) ++ ); ++ ++ self.builder.position_at_end(trap_block); ++ err!(self.build_call_with_param_attributes( ++ self.intrinsics.throw_trap, ++ &[self.intrinsics.trap_memory_oob.into()], ++ "throw", ++ )); ++ err!(self.builder.build_unreachable()); ++ ++ self.builder.position_at_end(continue_block); ++ Ok(()) ++ } ++ ++ fn build_dynamic_memory_base( ++ &self, ++ ptr_to_base_ptr: PointerValue<'ctx>, ++ memory_index: MemoryIndex, ++ label: &str, ++ ) -> Result, CompileError> { ++ let base = err!(self.builder.build_load( ++ self.intrinsics.ptr_ty, ++ ptr_to_base_ptr, ++ &format!("{label}_base") ++ )) ++ .into_pointer_value(); ++ tbaa_label( ++ self.module, ++ self.intrinsics, ++ format!("memory base_ptr {}", memory_index.as_u32()), ++ base.as_instruction_value().unwrap(), ++ ); ++ Ok(base) ++ } ++ + fn trap_if_misaligned( + &self, + _memarg: &MemArg, +@@ -1952,6 +2114,303 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + }) + } + ++ fn emit_indirect_call_from_state( ++ &mut self, ++ sigindex: SignatureIndex, ++ table_index: TableIndex, ++ is_return_call: bool, ++ ) -> Result>, CompileError> { ++ let func_type = &self.wasm_module.signatures[sigindex]; ++ let table = self.wasm_module.tables.get(table_index).unwrap(); ++ let local_fixed_funcref_table = self ++ .wasm_module ++ .local_table_index(table_index) ++ .filter(|_| table.is_fixed_funcref_table()); ++ let expected_signature_hash = self ++ .intrinsics ++ .i32_ty ++ .const_int(u64::from(self.signature_hashes[sigindex].as_u32()), false); ++ ++ let func_index = self.state.pop1()?.into_int_value(); ++ let generic_table = if local_fixed_funcref_table.is_none() { ++ Some( ++ self.ctx ++ .table(table_index, self.intrinsics, self.module, &self.builder)?, ++ ) ++ } else { ++ None ++ }; ++ ++ let table_bound = if local_fixed_funcref_table.is_some() { ++ self.intrinsics ++ .i32_ty ++ .const_int(table.minimum.into(), false) ++ } else { ++ let (_, table_bound) = *generic_table.as_ref().unwrap(); ++ err!(self.builder.build_int_truncate( ++ table_bound, ++ self.intrinsics.i32_ty, ++ "truncated_table_bounds", ++ )) ++ }; ++ ++ let index_in_bounds = err!(self.builder.build_int_compare( ++ IntPredicate::ULT, ++ func_index, ++ table_bound, ++ "index_in_bounds", ++ )); ++ ++ let index_in_bounds = self ++ .build_call_with_param_attributes( ++ self.intrinsics.expect_i1, ++ &[ ++ index_in_bounds.into(), ++ self.intrinsics.i1_ty.const_int(1, false).into(), ++ ], ++ "index_in_bounds_expect", ++ )? ++ .try_as_basic_value() ++ .unwrap_basic() ++ .into_int_value(); ++ ++ let in_bounds_continue_block = self ++ .context ++ .append_basic_block(self.function, "in_bounds_continue_block"); ++ let not_in_bounds_block = self ++ .context ++ .append_basic_block(self.function, "not_in_bounds_block"); ++ err!(self.builder.build_conditional_branch( ++ index_in_bounds, ++ in_bounds_continue_block, ++ not_in_bounds_block, ++ )); ++ self.builder.position_at_end(not_in_bounds_block); ++ self.build_call_with_param_attributes( ++ self.intrinsics.throw_trap, ++ &[self.intrinsics.trap_table_access_oob.into()], ++ "throw", ++ )?; ++ err!(self.builder.build_unreachable()); ++ self.builder.position_at_end(in_bounds_continue_block); ++ ++ let anyfunc_struct_ptr = if let Some(local_table_index) = local_fixed_funcref_table { ++ let anyfuncs = self.ctx.fixed_funcref_table_anyfuncs( ++ local_table_index, ++ self.intrinsics, ++ &self.builder, ++ )?; ++ unsafe { ++ err!(self.builder.build_in_bounds_gep( ++ self.intrinsics.anyfunc_ty, ++ anyfuncs, ++ &[func_index], ++ "anyfunc_struct_ptr", ++ )) ++ } ++ } else if table.ty == Type::FuncRef { ++ let anyfuncs = self.ctx.table_anyfuncs( ++ table_index, ++ self.intrinsics, ++ self.module, ++ &self.builder, ++ )?; ++ unsafe { ++ err!(self.builder.build_in_bounds_gep( ++ self.intrinsics.anyfunc_ty, ++ anyfuncs, ++ &[func_index], ++ "anyfunc_struct_ptr", ++ )) ++ } ++ } else { ++ let (table_base, _) = *generic_table.as_ref().unwrap(); ++ ++ let casted_table_base = err!(self.builder.build_pointer_cast( ++ table_base, ++ self.context.ptr_type(AddressSpace::default()), ++ "casted_table_base", ++ )); ++ ++ let funcref_ptr = unsafe { ++ err!(self.builder.build_in_bounds_gep( ++ self.intrinsics.ptr_ty, ++ casted_table_base, ++ &[func_index], ++ "funcref_ptr", ++ )) ++ }; ++ ++ let anyfunc_struct_ptr = err!(self.builder.build_load( ++ self.intrinsics.ptr_ty, ++ funcref_ptr, ++ "anyfunc_struct_ptr", ++ )) ++ .into_pointer_value(); ++ ++ if !table.readonly { ++ let funcref_not_null = err!( ++ self.builder ++ .build_is_not_null(anyfunc_struct_ptr, "null_funcref_check") ++ ); ++ ++ let funcref_continue_deref_block = self ++ .context ++ .append_basic_block(self.function, "funcref_continue_deref_block"); ++ ++ let funcref_is_null_block = self ++ .context ++ .append_basic_block(self.function, "funcref_is_null_block"); ++ err!(self.builder.build_conditional_branch( ++ funcref_not_null, ++ funcref_continue_deref_block, ++ funcref_is_null_block, ++ )); ++ self.builder.position_at_end(funcref_is_null_block); ++ self.build_call_with_param_attributes( ++ self.intrinsics.throw_trap, ++ &[self.intrinsics.trap_call_indirect_null.into()], ++ "throw", ++ )?; ++ err!(self.builder.build_unreachable()); ++ self.builder.position_at_end(funcref_continue_deref_block); ++ } ++ ++ anyfunc_struct_ptr ++ }; ++ ++ let sig_hash_ptr = self ++ .builder ++ .build_struct_gep( ++ self.intrinsics.anyfunc_ty, ++ anyfunc_struct_ptr, ++ 1, ++ "sig_hash_ptr", ++ ) ++ .unwrap(); ++ let func_ptr_ptr = self ++ .builder ++ .build_struct_gep( ++ self.intrinsics.anyfunc_ty, ++ anyfunc_struct_ptr, ++ 0, ++ "func_ptr_ptr", ++ ) ++ .unwrap(); ++ let (func_ptr, found_signature_hash) = ( ++ err!( ++ self.builder ++ .build_load(self.intrinsics.ptr_ty, func_ptr_ptr, "func_ptr") ++ ) ++ .into_pointer_value(), ++ err!( ++ self.builder ++ .build_load(self.intrinsics.i32_ty, sig_hash_ptr, "sig_hash") ++ ) ++ .into_int_value(), ++ ); ++ ++ let elem_initialized = err!(self.builder.build_is_not_null(func_ptr, "")); ++ let sig_hashes_equal = err!(self.builder.build_int_compare( ++ IntPredicate::EQ, ++ expected_signature_hash, ++ found_signature_hash, ++ "sig_hashes_equal", ++ )); ++ ++ let initialized_and_sig_hashes_match = err!(self.builder.build_and( ++ elem_initialized, ++ sig_hashes_equal, ++ "" ++ )); ++ ++ let initialized_and_sig_hashes_match = self ++ .build_call_with_param_attributes( ++ self.intrinsics.expect_i1, ++ &[ ++ initialized_and_sig_hashes_match.into(), ++ self.intrinsics.i1_ty.const_int(1, false).into(), ++ ], ++ "initialized_and_sig_hashes_match_expect", ++ )? ++ .try_as_basic_value() ++ .unwrap_basic() ++ .into_int_value(); ++ ++ let continue_block = self ++ .context ++ .append_basic_block(self.function, "continue_block"); ++ let sighashes_notequal_block = self ++ .context ++ .append_basic_block(self.function, "sighashes_notequal_block"); ++ err!(self.builder.build_conditional_branch( ++ initialized_and_sig_hashes_match, ++ continue_block, ++ sighashes_notequal_block, ++ )); ++ ++ self.builder.position_at_end(sighashes_notequal_block); ++ let trap_code = err!(self.builder.build_select( ++ elem_initialized, ++ self.intrinsics.trap_call_indirect_sig, ++ self.intrinsics.trap_call_indirect_null, ++ "", ++ )); ++ self.build_call_with_param_attributes( ++ self.intrinsics.throw_trap, ++ &[trap_code.into()], ++ "throw", ++ )?; ++ err!(self.builder.build_unreachable()); ++ self.builder.position_at_end(continue_block); ++ ++ let callee_vmctx = if table.readonly { ++ self.ctx.basic().into_pointer_value() ++ } else { ++ let ctx_ptr_ptr = self ++ .builder ++ .build_struct_gep( ++ self.intrinsics.anyfunc_ty, ++ anyfunc_struct_ptr, ++ 2, ++ "ctx_ptr_ptr", ++ ) ++ .unwrap(); ++ err!( ++ self.builder ++ .build_load(self.intrinsics.ptr_ty, ctx_ptr_ptr, "ctx_ptr") ++ ) ++ .into_pointer_value() ++ }; ++ ++ let rets = if self.m0_param.is_some() { ++ self.build_m0_indirect_call( ++ table_index.as_u32(), ++ callee_vmctx, ++ func_type, ++ func_ptr, ++ func_index, ++ is_return_call, ++ )? ++ } else { ++ let (call_site, llvm_func_type) = ++ self.build_indirect_call(callee_vmctx, func_type, func_ptr, None, is_return_call)?; ++ ++ if is_return_call { ++ self.emit_return_call(call_site, llvm_func_type)?; ++ Vec::new() ++ } else { ++ self.abi ++ .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? ++ .iter() ++ .copied() ++ .collect() ++ } ++ }; ++ ++ Ok(rets) ++ } ++ + fn build_m0_indirect_call( + &mut self, + table_index: u32, +@@ -1960,14 +2419,40 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + func_ptr: PointerValue<'ctx>, + func_index: IntValue<'ctx>, + is_return_call: bool, +- ) -> Result<(), CompileError> { +- let Some(m0) = self.m0_param else { ++ ) -> Result>, CompileError> { ++ if self.m0_param.is_none() { + return Err(CompileError::Codegen( + "Call to build_m0_indirect_call without m0 parameter!".to_string(), + )); +- }; ++ } + + let params = self.state.popn_save_extra(func_type.params().len())?; ++ self.build_m0_indirect_call_with_params( ++ table_index, ++ ctx_ptr, ++ func_type, ++ func_ptr, ++ func_index, ++ is_return_call, ++ ¶ms, ++ ) ++ } ++ ++ fn build_m0_indirect_call_with_params( ++ &mut self, ++ table_index: u32, ++ ctx_ptr: PointerValue<'ctx>, ++ func_type: &FunctionType, ++ func_ptr: PointerValue<'ctx>, ++ func_index: IntValue<'ctx>, ++ is_return_call: bool, ++ params: &[(BasicValueEnum<'ctx>, ExtraInfo)], ++ ) -> Result>, CompileError> { ++ let Some(m0) = self.m0_param else { ++ return Err(CompileError::Codegen( ++ "Call to build_m0_indirect_call_with_params without m0 parameter!".to_string(), ++ )); ++ }; + + let mut local_func_indices = vec![]; + let mut foreign_func_indices = vec![]; +@@ -2082,20 +2567,22 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + }; + + if is_return_call { +- return Ok(()); ++ return Ok(Vec::new()); + } + + self.builder + .position_at_end(cont.expect("non-return call requires cont")); + ++ let mut rets = Vec::with_capacity(foreign_rets.len()); + for i in 0..foreign_rets.len() { + let f_i = foreign_rets[i]; + let l_i = local_rets[i]; + let ty = f_i.get_type(); + let v = err!(self.builder.build_phi(ty, "")); + v.add_incoming(&[(&f_i, foreign_idx_block), (&l_i, local_idx_block)]); +- self.state.push1(v.as_basic_value()); ++ rets.push(v.as_basic_value()); + } ++ Ok(rets) + } else if foreign_func_indices.is_empty() { + let (call_site, llvm_func_type) = self.build_indirect_call_with_params( + ctx_ptr, +@@ -2108,11 +2595,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + + if is_return_call { + self.emit_return_call(call_site, llvm_func_type)?; ++ Ok(Vec::new()) + } else { +- self.abi ++ Ok(self ++ .abi + .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? + .iter() +- .for_each(|ret| self.state.push1(*ret)); ++ .copied() ++ .collect()) + } + } else { + let (call_site, llvm_func_type) = self.build_indirect_call_with_params( +@@ -2125,15 +2615,16 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + )?; + if is_return_call { + self.emit_return_call(call_site, llvm_func_type)?; ++ Ok(Vec::new()) + } else { +- self.abi ++ Ok(self ++ .abi + .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? + .iter() +- .for_each(|ret| self.state.push1(*ret)); ++ .copied() ++ .collect()) + } + } +- +- Ok(()) + } + + fn build_indirect_call( +@@ -2965,8 +3456,6 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + // Basic instructions. + // https://github.com/sunfishcode/wasm-reference-manual/blob/master/WebAssembly.md#basic-instructions + fn translate_basic_operator(&mut self, op: Operator) -> Result<(), CompileError> { +- let vmctx = &self.ctx.basic().into_pointer_value(); +- + match op { + Operator::Nop => { + // Do nothing. +@@ -3476,300 +3965,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + let is_return_call = matches!(op, Operator::ReturnCallIndirect { .. }); + let sigindex = SignatureIndex::from_u32(type_index); + let table_index = TableIndex::from_u32(table_index); +- let func_type = &self.wasm_module.signatures[sigindex]; +- let table = self.wasm_module.tables.get(table_index).unwrap(); +- let local_fixed_funcref_table = self +- .wasm_module +- .local_table_index(table_index) +- .filter(|_| table.is_fixed_funcref_table()); +- let expected_signature_hash = self +- .intrinsics +- .i32_ty +- .const_int(u64::from(self.signature_hashes[sigindex].as_u32()), false); +- +- let func_index = self.state.pop1()?.into_int_value(); +- let generic_table = if local_fixed_funcref_table.is_none() { +- Some(self.ctx.table( +- table_index, +- self.intrinsics, +- self.module, +- &self.builder, +- )?) +- } else { +- None +- }; +- +- let table_bound = if local_fixed_funcref_table.is_some() { +- self.intrinsics +- .i32_ty +- .const_int(table.minimum.into(), false) +- } else { +- let (_, table_bound) = *generic_table.as_ref().unwrap(); +- err!(self.builder.build_int_truncate( +- table_bound, +- self.intrinsics.i32_ty, +- "truncated_table_bounds", +- )) +- }; +- +- // First, check if the index is outside of the table bounds. +- let index_in_bounds = err!(self.builder.build_int_compare( +- IntPredicate::ULT, +- func_index, +- table_bound, +- "index_in_bounds", +- )); +- +- let index_in_bounds = self +- .build_call_with_param_attributes( +- self.intrinsics.expect_i1, +- &[ +- index_in_bounds.into(), +- self.intrinsics.i1_ty.const_int(1, false).into(), +- ], +- "index_in_bounds_expect", +- )? +- .try_as_basic_value() +- .unwrap_basic() +- .into_int_value(); +- +- let in_bounds_continue_block = self +- .context +- .append_basic_block(self.function, "in_bounds_continue_block"); +- let not_in_bounds_block = self +- .context +- .append_basic_block(self.function, "not_in_bounds_block"); +- err!(self.builder.build_conditional_branch( +- index_in_bounds, +- in_bounds_continue_block, +- not_in_bounds_block, +- )); +- self.builder.position_at_end(not_in_bounds_block); +- self.build_call_with_param_attributes( +- self.intrinsics.throw_trap, +- &[self.intrinsics.trap_table_access_oob.into()], +- "throw", +- )?; +- err!(self.builder.build_unreachable()); +- self.builder.position_at_end(in_bounds_continue_block); +- +- let anyfunc_struct_ptr = if let Some(local_table_index) = local_fixed_funcref_table +- { +- let anyfuncs = self.ctx.fixed_funcref_table_anyfuncs( +- local_table_index, +- self.intrinsics, +- &self.builder, +- )?; +- unsafe { +- err!(self.builder.build_in_bounds_gep( +- self.intrinsics.anyfunc_ty, +- anyfuncs, +- &[func_index], +- "anyfunc_struct_ptr", +- )) +- } +- } else { +- let (table_base, _) = *generic_table.as_ref().unwrap(); +- +- // We assume the table has the `funcref` (pointer to `anyfunc`) +- // element type. +- let casted_table_base = err!(self.builder.build_pointer_cast( +- table_base, +- self.context.ptr_type(AddressSpace::default()), +- "casted_table_base", +- )); +- +- let funcref_ptr = unsafe { +- err!(self.builder.build_in_bounds_gep( +- self.intrinsics.ptr_ty, +- casted_table_base, +- &[func_index], +- "funcref_ptr", +- )) +- }; +- +- // a funcref (pointer to `anyfunc`) +- let anyfunc_struct_ptr = err!(self.builder.build_load( +- self.intrinsics.ptr_ty, +- funcref_ptr, +- "anyfunc_struct_ptr", +- )) +- .into_pointer_value(); +- +- if !table.readonly { +- // trap if we're trying to call a null funcref +- let funcref_not_null = err!( +- self.builder +- .build_is_not_null(anyfunc_struct_ptr, "null_funcref_check") +- ); +- +- let funcref_continue_deref_block = self +- .context +- .append_basic_block(self.function, "funcref_continue_deref_block"); +- +- let funcref_is_null_block = self +- .context +- .append_basic_block(self.function, "funcref_is_null_block"); +- err!(self.builder.build_conditional_branch( +- funcref_not_null, +- funcref_continue_deref_block, +- funcref_is_null_block, +- )); +- self.builder.position_at_end(funcref_is_null_block); +- self.build_call_with_param_attributes( +- self.intrinsics.throw_trap, +- &[self.intrinsics.trap_call_indirect_null.into()], +- "throw", +- )?; +- err!(self.builder.build_unreachable()); +- self.builder.position_at_end(funcref_continue_deref_block); +- } +- +- anyfunc_struct_ptr +- }; +- +- // Load things from the anyfunc data structure. +- let sig_hash_ptr = self +- .builder +- .build_struct_gep( +- self.intrinsics.anyfunc_ty, +- anyfunc_struct_ptr, +- 1, +- "sig_hash_ptr", +- ) +- .unwrap(); +- let func_ptr_ptr = self +- .builder +- .build_struct_gep( +- self.intrinsics.anyfunc_ty, +- anyfunc_struct_ptr, +- 0, +- "func_ptr_ptr", +- ) +- .unwrap(); +- let (func_ptr, found_signature_hash) = ( +- err!( +- self.builder +- .build_load(self.intrinsics.ptr_ty, func_ptr_ptr, "func_ptr") +- ) +- .into_pointer_value(), +- err!( +- self.builder +- .build_load(self.intrinsics.i32_ty, sig_hash_ptr, "sig_hash") +- ) +- .into_int_value(), +- ); +- +- // Next, check if the table element is initialized. +- +- // TODO: we may not need this check anymore +- let elem_initialized = err!(self.builder.build_is_not_null(func_ptr, "")); +- +- // Next, check if the signature id is correct. +- +- let sig_hashes_equal = err!(self.builder.build_int_compare( +- IntPredicate::EQ, +- expected_signature_hash, +- found_signature_hash, +- "sig_hashes_equal", +- )); +- +- let initialized_and_sig_hashes_match = err!(self.builder.build_and( +- elem_initialized, +- sig_hashes_equal, +- "" +- )); +- +- // Tell llvm that the expected and found signature hashes should match. +- let initialized_and_sig_hashes_match = self +- .build_call_with_param_attributes( +- self.intrinsics.expect_i1, +- &[ +- initialized_and_sig_hashes_match.into(), +- self.intrinsics.i1_ty.const_int(1, false).into(), +- ], +- "initialized_and_sig_hashes_match_expect", +- )? +- .try_as_basic_value() +- .unwrap_basic() +- .into_int_value(); +- +- let continue_block = self +- .context +- .append_basic_block(self.function, "continue_block"); +- let sighashes_notequal_block = self +- .context +- .append_basic_block(self.function, "sighashes_notequal_block"); +- err!(self.builder.build_conditional_branch( +- initialized_and_sig_hashes_match, +- continue_block, +- sighashes_notequal_block, +- )); +- +- self.builder.position_at_end(sighashes_notequal_block); +- let trap_code = err!(self.builder.build_select( +- elem_initialized, +- self.intrinsics.trap_call_indirect_sig, +- self.intrinsics.trap_call_indirect_null, +- "", +- )); +- self.build_call_with_param_attributes( +- self.intrinsics.throw_trap, +- &[trap_code.into()], +- "throw", +- )?; +- err!(self.builder.build_unreachable()); +- self.builder.position_at_end(continue_block); +- +- let callee_vmctx = if table.readonly { +- *vmctx +- } else { +- let ctx_ptr_ptr = self +- .builder +- .build_struct_gep( +- self.intrinsics.anyfunc_ty, +- anyfunc_struct_ptr, +- 2, +- "ctx_ptr_ptr", +- ) +- .unwrap(); +- err!( +- self.builder +- .build_load(self.intrinsics.ptr_ty, ctx_ptr_ptr, "ctx_ptr") +- ) +- .into_pointer_value() +- }; +- +- if self.m0_param.is_some() { +- self.build_m0_indirect_call( +- table_index.as_u32(), +- callee_vmctx, +- func_type, +- func_ptr, +- func_index, +- is_return_call, +- )?; +- } else { +- let (call_site, llvm_func_type) = self.build_indirect_call( +- callee_vmctx, +- func_type, +- func_ptr, +- None, +- is_return_call, +- )?; +- +- if is_return_call { +- self.emit_return_call(call_site, llvm_func_type)?; +- } else { +- self.abi +- .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? +- .iter() +- .for_each(|ret| self.state.push1(*ret)); +- } +- } +- ++ let rets = ++ self.emit_indirect_call_from_state(sigindex, table_index, is_return_call)?; + if is_return_call { + self.state.reachable = false; ++ } else { ++ for ret in rets { ++ self.state.push1(ret); ++ } + } + } + _ => unreachable!(), +@@ -5289,13 +5492,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); + let mask = self.intrinsics.i32_ty.const_int(31u64, false); + let v2 = err!(self.builder.build_and(v2, mask, "")); +- let lhs = err!(self.builder.build_left_shift(v1, v2, "")); +- let rhs = { +- let negv2 = err!(self.builder.build_int_neg(v2, "")); +- let rhs = err!(self.builder.build_and(negv2, mask, "")); +- err!(self.builder.build_right_shift(v1, rhs, false, "")) +- }; +- let res = err!(self.builder.build_or(lhs, rhs, "")); ++ let res = self ++ .build_call_with_param_attributes( ++ self.intrinsics.fshl_i32, ++ &[v1.into(), v1.into(), v2.into()], ++ "", ++ )? ++ .try_as_basic_value() ++ .unwrap_basic(); + self.state.push1(res); + } + Operator::I64Rotl => { +@@ -5305,13 +5509,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); + let mask = self.intrinsics.i64_ty.const_int(63u64, false); + let v2 = err!(self.builder.build_and(v2, mask, "")); +- let lhs = err!(self.builder.build_left_shift(v1, v2, "")); +- let rhs = { +- let negv2 = err!(self.builder.build_int_neg(v2, "")); +- let rhs = err!(self.builder.build_and(negv2, mask, "")); +- err!(self.builder.build_right_shift(v1, rhs, false, "")) +- }; +- let res = err!(self.builder.build_or(lhs, rhs, "")); ++ let res = self ++ .build_call_with_param_attributes( ++ self.intrinsics.fshl_i64, ++ &[v1.into(), v1.into(), v2.into()], ++ "", ++ )? ++ .try_as_basic_value() ++ .unwrap_basic(); + self.state.push1(res); + } + Operator::I32Rotr => { +@@ -5321,13 +5526,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); + let mask = self.intrinsics.i32_ty.const_int(31u64, false); + let v2 = err!(self.builder.build_and(v2, mask, "")); +- let lhs = err!(self.builder.build_right_shift(v1, v2, false, "")); +- let rhs = { +- let negv2 = err!(self.builder.build_int_neg(v2, "")); +- let rhs = err!(self.builder.build_and(negv2, mask, "")); +- err!(self.builder.build_left_shift(v1, rhs, "")) +- }; +- let res = err!(self.builder.build_or(lhs, rhs, "")); ++ let res = self ++ .build_call_with_param_attributes( ++ self.intrinsics.fshr_i32, ++ &[v1.into(), v1.into(), v2.into()], ++ "", ++ )? ++ .try_as_basic_value() ++ .unwrap_basic(); + self.state.push1(res); + } + Operator::I64Rotr => { +@@ -5337,13 +5543,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); + let mask = self.intrinsics.i64_ty.const_int(63u64, false); + let v2 = err!(self.builder.build_and(v2, mask, "")); +- let lhs = err!(self.builder.build_right_shift(v1, v2, false, "")); +- let rhs = { +- let negv2 = err!(self.builder.build_int_neg(v2, "")); +- let rhs = err!(self.builder.build_and(negv2, mask, "")); +- err!(self.builder.build_left_shift(v1, rhs, "")) +- }; +- let res = err!(self.builder.build_or(lhs, rhs, "")); ++ let res = self ++ .build_call_with_param_attributes( ++ self.intrinsics.fshr_i64, ++ &[v1.into(), v1.into(), v2.into()], ++ "", ++ )? ++ .try_as_basic_value() ++ .unwrap_basic(); + self.state.push1(res); + } + Operator::I32Clz => { +@@ -9650,7 +9857,11 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + Operator::V128Load { ref memarg } => { + let offset = self.state.pop1()?.into_int_value(); + let result = +- self.build_annotated_load(self.intrinsics.i128_ty, offset, memarg, 1)?; ++ self.build_annotated_load(self.intrinsics.i64x2_ty, offset, memarg, 1)?; ++ let result = err!( ++ self.builder ++ .build_bit_cast(result, self.intrinsics.i128_ty, "") ++ ); + self.state.push1(result); + } + Operator::V128Load8Lane { ref memarg, lane } => { +@@ -9735,8 +9946,9 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + Operator::V128Store { ref memarg } => { + let (v, i) = self.state.pop1_extra()?; + let v = self.apply_pending_canonicalization(v, i)?; ++ let v = err!(self.builder.build_bit_cast(v, self.intrinsics.i64x2_ty, "")); + let offset = self.state.pop1()?.into_int_value(); +- self.build_annotated_store(self.intrinsics.i128_ty, offset, v, memarg, 1)?; ++ self.build_annotated_store(self.intrinsics.i64x2_ty, offset, v, memarg, 1)?; + } + Operator::V128Store8Lane { ref memarg, lane } => { + let (v, i) = self.state.pop1_extra()?; +@@ -10570,54 +10782,107 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + )?; + } + Operator::MemoryCopy { dst_mem, src_mem } => { +- // ignored until we support multiple memories +- let _dst = dst_mem; +- let (memory_copy, src) = if let Some(local_memory_index) = self +- .wasm_module +- .local_memory_index(MemoryIndex::from_u32(src_mem)) +- { +- (self.intrinsics.memory_copy, local_memory_index.as_u32()) +- } else { +- (self.intrinsics.imported_memory_copy, src_mem) +- }; +- + let (dest_pos, src_pos, len) = self.state.pop3()?; +- let src_index = self.intrinsics.i32_ty.const_int(src.into(), false); +- self.build_call_with_param_attributes( +- memory_copy, +- &[ +- vmctx.as_basic_value_enum().into(), +- src_index.into(), +- dest_pos.into(), +- src_pos.into(), +- len.into(), +- ], +- "", ++ let dst_memory_index = MemoryIndex::from_u32(dst_mem); ++ let src_memory_index = MemoryIndex::from_u32(src_mem); ++ ++ let (dst_base, dst_current_length) = ++ self.memory_bulk_parts(dst_memory_index, "memory_copy_dst")?; ++ let (src_base, src_current_length) = ++ self.memory_bulk_parts(src_memory_index, "memory_copy_src")?; ++ ++ let dest_pos = dest_pos.into_int_value(); ++ let src_pos = src_pos.into_int_value(); ++ let len = len.into_int_value(); ++ ++ let dst_in_bounds = self.build_memory_range_check( ++ dest_pos, ++ len, ++ dst_current_length, ++ "memory_copy_dst", ++ )?; ++ let src_in_bounds = self.build_memory_range_check( ++ src_pos, ++ len, ++ src_current_length, ++ "memory_copy_src", + )?; ++ let in_bounds = err!(self.builder.build_and( ++ dst_in_bounds, ++ src_in_bounds, ++ "memory_copy_in_bounds" ++ )); ++ self.build_memory_oob_trap(in_bounds, "memory_copy")?; ++ ++ let dest_offset = err!(self.builder.build_int_z_extend( ++ dest_pos, ++ self.intrinsics.i64_ty, ++ "memory_copy_dst_offset" ++ )); ++ let src_offset = err!(self.builder.build_int_z_extend( ++ src_pos, ++ self.intrinsics.i64_ty, ++ "memory_copy_src_offset" ++ )); ++ let dst = unsafe { ++ err!(self.builder.build_gep( ++ self.intrinsics.i8_ty, ++ dst_base, ++ &[dest_offset], ++ "memory_copy_dst_ptr" ++ )) ++ }; ++ let src = unsafe { ++ err!(self.builder.build_gep( ++ self.intrinsics.i8_ty, ++ src_base, ++ &[src_offset], ++ "memory_copy_src_ptr" ++ )) ++ }; ++ let len = err!(self.builder.build_int_z_extend( ++ len, ++ self.intrinsics.i64_ty, ++ "memory_copy_len" ++ )); ++ err!(self.builder.build_memmove(dst, 1, src, 1, len)); + } + Operator::MemoryFill { mem } => { +- let (memory_fill, mem) = if let Some(local_memory_index) = self +- .wasm_module +- .local_memory_index(MemoryIndex::from_u32(mem)) +- { +- (self.intrinsics.memory_fill, local_memory_index.as_u32()) +- } else { +- (self.intrinsics.imported_memory_fill, mem) +- }; +- + let (dst, val, len) = self.state.pop3()?; +- let mem_index = self.intrinsics.i32_ty.const_int(mem.into(), false); +- self.build_call_with_param_attributes( +- memory_fill, +- &[ +- vmctx.as_basic_value_enum().into(), +- mem_index.into(), +- dst.into(), +- val.into(), +- len.into(), +- ], +- "", +- )?; ++ let memory_index = MemoryIndex::from_u32(mem); ++ ++ let (base, current_length) = self.memory_bulk_parts(memory_index, "memory_fill")?; ++ let dst = dst.into_int_value(); ++ let val = val.into_int_value(); ++ let len = len.into_int_value(); ++ let in_bounds = ++ self.build_memory_range_check(dst, len, current_length, "memory_fill")?; ++ self.build_memory_oob_trap(in_bounds, "memory_fill")?; ++ ++ let dst_offset = err!(self.builder.build_int_z_extend( ++ dst, ++ self.intrinsics.i64_ty, ++ "memory_fill_offset" ++ )); ++ let dst = unsafe { ++ err!(self.builder.build_gep( ++ self.intrinsics.i8_ty, ++ base, ++ &[dst_offset], ++ "memory_fill_ptr" ++ )) ++ }; ++ let val = err!(self.builder.build_int_truncate( ++ val, ++ self.intrinsics.i8_ty, ++ "memory_fill_val" ++ )); ++ let len = err!(self.builder.build_int_z_extend( ++ len, ++ self.intrinsics.i64_ty, ++ "memory_fill_len" ++ )); ++ err!(self.builder.build_memset(dst, 1, val, len)); + } + _ => unreachable!(), + } +@@ -10630,13 +10895,15 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + + match op { + Operator::AtomicFence => { +- // Fence is a nop. +- // +- // Fence was added to preserve information about fences from +- // source languages. If in the future Wasm extends the memory +- // model, and if we hadn't recorded what fences used to be there, +- // it would lead to data races that weren't present in the +- // original source language. ++ // WASIX can remap the same file-backed PostgreSQL region into ++ // multiple Wasm instances. Preserve source-language fences for ++ // that cross-instance memory, which is stronger than the ++ // ordinary single-shared-memory Wasm execution model. ++ err!(self.builder.build_fence( ++ AtomicOrdering::SequentiallyConsistent, ++ false, ++ "atomic_fence" ++ )); + } + Operator::I32AtomicLoad { ref memarg } => { + let offset = self.state.pop1()?.into_int_value(); +@@ -11984,7 +12251,7 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { + Ok(()) + } + +- fn translate_operator(&mut self, op: Operator, _source_loc: u32) -> Result<(), CompileError> { ++ fn translate_operator(&mut self, op: Operator) -> Result<(), CompileError> { + //let opcode_offset: Option = None; + + if !self.state.reachable { +diff --git a/lib/compiler-llvm/src/translator/intrinsics.rs b/lib/compiler-llvm/src/translator/intrinsics.rs +index 1863283..6fdb84d 100644 +--- a/lib/compiler-llvm/src/translator/intrinsics.rs ++++ b/lib/compiler-llvm/src/translator/intrinsics.rs +@@ -79,6 +79,11 @@ pub struct Intrinsics<'ctx> { + pub ctpop_i64: FunctionValue<'ctx>, + pub ctpop_i8x16: FunctionValue<'ctx>, + ++ pub fshl_i32: FunctionValue<'ctx>, ++ pub fshl_i64: FunctionValue<'ctx>, ++ pub fshr_i32: FunctionValue<'ctx>, ++ pub fshr_i64: FunctionValue<'ctx>, ++ + pub fp_rounding_md: BasicMetadataValueEnum<'ctx>, + pub fp_exception_md: BasicMetadataValueEnum<'ctx>, + pub fp_ogt_md: BasicMetadataValueEnum<'ctx>, +@@ -421,6 +426,10 @@ impl<'ctx> Intrinsics<'ctx> { + + let ret_i32_take_i32 = i32_ty.fn_type(&[i32_ty_basic_md], false); + let ret_i64_take_i64 = i64_ty.fn_type(&[i64_ty_basic_md], false); ++ let ret_i32_take_i32_i32_i32 = ++ i32_ty.fn_type(&[i32_ty_basic_md, i32_ty_basic_md, i32_ty_basic_md], false); ++ let ret_i64_take_i64_i64_i64 = ++ i64_ty.fn_type(&[i64_ty_basic_md, i64_ty_basic_md, i64_ty_basic_md], false); + + let ret_f32_take_f32 = f32_ty.fn_type(&[f32_ty_basic_md], false); + let ret_f64_take_f64 = f64_ty.fn_type(&[f64_ty_basic_md], false); +@@ -581,6 +590,11 @@ impl<'ctx> Intrinsics<'ctx> { + ctpop_i64: add_function_with_attrs("llvm.ctpop.i64", ret_i64_take_i64, None), + ctpop_i8x16: add_function_with_attrs("llvm.ctpop.v16i8", ret_i8x16_take_i8x16, None), + ++ fshl_i32: add_function_with_attrs("llvm.fshl.i32", ret_i32_take_i32_i32_i32, None), ++ fshl_i64: add_function_with_attrs("llvm.fshl.i64", ret_i64_take_i64_i64_i64, None), ++ fshr_i32: add_function_with_attrs("llvm.fshr.i32", ret_i32_take_i32_i32_i32, None), ++ fshr_i64: add_function_with_attrs("llvm.fshr.i64", ret_i64_take_i64_i64_i64, None), ++ + fp_rounding_md: context.metadata_string("round.tonearest").into(), + fp_exception_md: context.metadata_string("fpexcept.strict").into(), + +@@ -1417,13 +1431,17 @@ pub enum MemoryCache<'ctx> { + ptr_to_current_length: PointerValue<'ctx>, + }, + /// The memory is always in the same place. +- Static { base_ptr: PointerValue<'ctx> }, ++ Static { ++ base_ptr: PointerValue<'ctx>, ++ ptr_to_current_length: PointerValue<'ctx>, ++ }, + } + + #[derive(Clone)] + struct TableCache<'ctx> { + ptr_to_base_ptr: PointerValue<'ctx>, + ptr_to_bounds: PointerValue<'ctx>, ++ ptr_to_anyfuncs_ptr: PointerValue<'ctx>, + } + + #[derive(Clone, Copy)] +@@ -1567,13 +1585,13 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + intrinsics.vmmemory_definition_base_element, + "", + )); ++ let current_length_ptr = err!(cache_builder.build_struct_gep( ++ intrinsics.vmmemory_definition_ty, ++ memory_definition_ptr, ++ intrinsics.vmmemory_definition_current_length_element, ++ "", ++ )); + let value = if let MemoryStyle::Dynamic { .. } = memory_style { +- let current_length_ptr = err!(cache_builder.build_struct_gep( +- intrinsics.vmmemory_definition_ty, +- memory_definition_ptr, +- intrinsics.vmmemory_definition_current_length_element, +- "", +- )); + MemoryCache::Dynamic { + ptr_to_base_ptr: base_ptr, + ptr_to_current_length: current_length_ptr, +@@ -1587,7 +1605,10 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + format!("memory base_ptr {}", index.as_u32()), + base_ptr.as_instruction_value().unwrap(), + ); +- MemoryCache::Static { base_ptr } ++ MemoryCache::Static { ++ base_ptr, ++ ptr_to_current_length: current_length_ptr, ++ } + }; + + self.cached_memories.insert(index, value); +@@ -1604,7 +1625,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + ctx_ptr_value: PointerValue<'ctx>, + offsets: &VMOffsets, + builder: &Builder<'ctx>, +- ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { ++ ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { + if let Some(local_table_index) = wasm_module.local_table_index(table_index) { + let offset = intrinsics.i64_ty.const_int( + offsets +@@ -1627,7 +1648,18 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + unsafe { err!(builder.build_gep(intrinsics.i8_ty, ctx_ptr_value, &[offset], "")) }; + let ptr_to_bounds = err!(builder.build_bit_cast(ptr_to_bounds, intrinsics.ptr_ty, "")) + .into_pointer_value(); +- Ok((ptr_to_base_ptr, ptr_to_bounds)) ++ let offset = intrinsics.i64_ty.const_int( ++ offsets ++ .vmctx_vmtable_definition_anyfuncs(local_table_index) ++ .into(), ++ false, ++ ); ++ let ptr_to_anyfuncs_ptr = ++ unsafe { err!(builder.build_gep(intrinsics.i8_ty, ctx_ptr_value, &[offset], "")) }; ++ let ptr_to_anyfuncs_ptr = ++ err!(builder.build_bit_cast(ptr_to_anyfuncs_ptr, intrinsics.ptr_ty, "")) ++ .into_pointer_value(); ++ Ok((ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr)) + } else { + let offset = intrinsics.i64_ty.const_int( + offsets.vmctx_vmtable_import_definition(table_index).into(), +@@ -1663,7 +1695,15 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + unsafe { err!(builder.build_gep(intrinsics.i8_ty, definition_ptr, &[offset], "")) }; + let ptr_to_bounds = err!(builder.build_bit_cast(ptr_to_bounds, intrinsics.ptr_ty, "")) + .into_pointer_value(); +- Ok((ptr_to_base_ptr, ptr_to_bounds)) ++ let offset = intrinsics ++ .i64_ty ++ .const_int(offsets.vmtable_definition_anyfuncs().into(), false); ++ let ptr_to_anyfuncs_ptr = ++ unsafe { err!(builder.build_gep(intrinsics.i8_ty, definition_ptr, &[offset], "")) }; ++ let ptr_to_anyfuncs_ptr = ++ err!(builder.build_bit_cast(ptr_to_anyfuncs_ptr, intrinsics.ptr_ty, "")) ++ .into_pointer_value(); ++ Ok((ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr)) + } + } + +@@ -1673,7 +1713,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + intrinsics: &Intrinsics<'ctx>, + module: &Module<'ctx>, + body_builder: &Builder<'ctx>, +- ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { ++ ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { + let (cached_tables, wasm_module, ctx_ptr_value, cache_builder, offsets) = ( + &mut self.cached_tables, + self.wasm_module, +@@ -1689,8 +1729,8 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + )) + })?; + +- // If the table is growable, it may change, so we can't cache the pointers; they need to +- // go directly in the function body at the point where they're needed ++ // If the table is growable, it may change, so keep preparing the field ++ // pointers in the function body at the point where the table is used. + if is_growable { + Self::build_table_prepare( + table_index, +@@ -1705,22 +1745,25 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + let TableCache { + ptr_to_base_ptr, + ptr_to_bounds, ++ ptr_to_anyfuncs_ptr, + } = match cached_tables.entry(table_index) { + Entry::Occupied(entry) => entry.get().clone(), + Entry::Vacant(entry) => { +- let (ptr_to_base_ptr, ptr_to_bounds) = Self::build_table_prepare( +- table_index, +- intrinsics, +- module, +- wasm_module, +- ctx_ptr_value, +- offsets, +- cache_builder, +- )?; ++ let (ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr) = ++ Self::build_table_prepare( ++ table_index, ++ intrinsics, ++ module, ++ wasm_module, ++ ctx_ptr_value, ++ offsets, ++ cache_builder, ++ )?; + + let v = TableCache { + ptr_to_base_ptr, + ptr_to_bounds, ++ ptr_to_anyfuncs_ptr, + }; + + entry.insert(v.clone()); +@@ -1729,7 +1772,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + } + }; + +- Ok((ptr_to_base_ptr, ptr_to_bounds)) ++ Ok((ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr)) + } + } + +@@ -1740,7 +1783,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + module: &Module<'ctx>, + body_builder: &Builder<'ctx>, + ) -> Result<(PointerValue<'ctx>, IntValue<'ctx>), CompileError> { +- let (ptr_to_base_ptr, ptr_to_bounds) = ++ let (ptr_to_base_ptr, ptr_to_bounds, _) = + self.table_prepare(index, intrinsics, module, body_builder)?; + + // Safe to unwrap since an out-of-bounds index will be caught be table_prepare +@@ -1769,6 +1812,34 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { + Ok((base_ptr, bounds)) + } + ++ pub fn table_anyfuncs( ++ &mut self, ++ index: TableIndex, ++ intrinsics: &Intrinsics<'ctx>, ++ module: &Module<'ctx>, ++ body_builder: &Builder<'ctx>, ++ ) -> Result, CompileError> { ++ let (_, _, ptr_to_anyfuncs_ptr) = ++ self.table_prepare(index, intrinsics, module, body_builder)?; ++ ++ let builder = if is_table_growable(self.wasm_module, index).unwrap() { ++ &body_builder ++ } else { ++ &self.cache_builder ++ }; ++ ++ let anyfuncs = ++ err!(builder.build_load(intrinsics.ptr_ty, ptr_to_anyfuncs_ptr, "table_anyfuncs")) ++ .into_pointer_value(); ++ tbaa_label( ++ module, ++ intrinsics, ++ format!("table_anyfuncs {}", index.index()), ++ anyfuncs.as_instruction_value().unwrap(), ++ ); ++ Ok(anyfuncs) ++ } ++ + // Return a pointer to the beginning of a local funcref Table (a pointer related to vmctx). + pub fn fixed_funcref_table_anyfuncs( + &self, +diff --git a/lib/compiler-llvm/src/translator/trampoline.rs b/lib/compiler-llvm/src/translator/trampoline.rs +index e3a9eae..5075136 100644 +--- a/lib/compiler-llvm/src/translator/trampoline.rs ++++ b/lib/compiler-llvm/src/translator/trampoline.rs +@@ -358,6 +358,7 @@ impl FuncTrampoline { + compact_unwind_section_indices, + gcc_except_table_section_indices, + data_dw_ref_personality_section_indices, ++ .. + } = load_object_file( + mem_buf_slice, + &self.func_section, +diff --git a/lib/compiler/src/artifact_builders/artifact_builder.rs b/lib/compiler/src/artifact_builders/artifact_builder.rs +index 14fc0f4..8a7ff25 100644 +--- a/lib/compiler/src/artifact_builders/artifact_builder.rs ++++ b/lib/compiler/src/artifact_builders/artifact_builder.rs +@@ -106,7 +106,6 @@ impl ArtifactBuild { + translation.function_body_inputs, + progress_callback, + )?; +- + let data_initializers = translation + .data_initializers + .iter() +@@ -262,6 +261,126 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuild { + } + } + ++/// Runtime metadata materialized from a serialized artifact after its code, ++/// relocations, unwind sections, and frame registrations have been installed. ++/// ++/// Unlike [`ArtifactBuildFromArchive`], this representation does not retain the ++/// complete serialized archive. It is intended for sealed, precompiled-only ++/// runtimes where re-serialization is neither required nor desirable. Keeping ++/// only the metadata used by later instantiations lets immutable AOT pages be ++/// reclaimed once checked deserialization has completed. ++#[derive(Debug)] ++#[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] ++pub struct DetachedArtifactBuild { ++ compile_info: CompileModuleInfo, ++ data_initializers: Box<[OwnedDataInitializer]>, ++ cpu_features: u64, ++ // This is staging storage only. Registration transfers the map into the ++ // global trap registry, which then becomes its sole long-lived owner. ++ function_frame_info: Option>, ++} ++ ++impl DetachedArtifactBuild { ++ /// Materialize only metadata that remains live after allocation and linking. ++ pub fn from_archive(archive: &ArtifactBuildFromArchive) -> Result { ++ let archived = archive.cell.borrow_dependent(); ++ let data_initializers = rkyv::deserialize::<_, RkyvError>(archived.data_initializers) ++ .map_err(|e| DeserializeError::CorruptedBinary(format!("{e:?}")))?; ++ let function_frame_info = rkyv::deserialize::<_, RkyvError>( ++ &archived.compilation.function_frame_info, ++ ) ++ .map_err(|e| DeserializeError::CorruptedBinary(format!("{e:?}")))?; ++ ++ Ok(Self { ++ compile_info: archive.compile_info.clone(), ++ data_initializers, ++ cpu_features: archived.cpu_features, ++ function_frame_info: Some(function_frame_info), ++ }) ++ } ++ ++ /// Transfer frame metadata to the global trap registration. ++ /// ++ /// A detached build stages an owned map only until code registration. The ++ /// returned map must be moved into [`crate::FrameInfosVariant::Owned`] so ++ /// the detached artifact does not retain a duplicate of all trap and ++ /// address-map allocations. ++ pub(crate) fn take_frame_info_for_registration( ++ &mut self, ++ ) -> Result, DeserializeError> { ++ self.function_frame_info.take().ok_or_else(|| { ++ DeserializeError::Generic( ++ "detached artifact frame metadata was already transferred before registration" ++ .to_string(), ++ ) ++ }) ++ } ++ ++ #[cfg(test)] ++ /// Construct a detached build with frame metadata for engine ownership tests. ++ pub(crate) fn for_frame_info_test( ++ function_frame_info: PrimaryMap, ++ ) -> Self { ++ Self { ++ compile_info: CompileModuleInfo { ++ features: Features::default(), ++ module: Arc::new(ModuleInfo::default()), ++ memory_styles: PrimaryMap::new(), ++ table_styles: PrimaryMap::new(), ++ }, ++ data_initializers: Vec::new().into_boxed_slice(), ++ cpu_features: 0, ++ function_frame_info: Some(function_frame_info), ++ } ++ } ++} ++ ++impl<'a> ArtifactCreate<'a> for DetachedArtifactBuild { ++ type OwnedDataInitializer = &'a OwnedDataInitializer; ++ type OwnedDataInitializerIterator = core::slice::Iter<'a, OwnedDataInitializer>; ++ ++ fn create_module_info(&self) -> Arc { ++ self.compile_info.module.clone() ++ } ++ ++ fn set_module_info_name(&mut self, name: String) -> bool { ++ Arc::get_mut(&mut self.compile_info.module).is_some_and(|module_info| { ++ module_info.name = Some(name); ++ true ++ }) ++ } ++ ++ fn module_info(&self) -> &ModuleInfo { ++ &self.compile_info.module ++ } ++ ++ fn features(&self) -> &Features { ++ &self.compile_info.features ++ } ++ ++ fn cpu_features(&self) -> EnumSet { ++ EnumSet::from_u64(self.cpu_features) ++ } ++ ++ fn data_initializers(&'a self) -> Self::OwnedDataInitializerIterator { ++ self.data_initializers.iter() ++ } ++ ++ fn memory_styles(&self) -> &PrimaryMap { ++ &self.compile_info.memory_styles ++ } ++ ++ fn table_styles(&self) -> &PrimaryMap { ++ &self.compile_info.table_styles ++ } ++ ++ fn serialize(&self) -> Result, SerializeError> { ++ Err(SerializeError::Generic( ++ "detached runtime artifacts cannot be re-serialized".to_string(), ++ )) ++ } ++} ++ + /// Module loaded from an archive. Since `CompileModuleInfo` is part of the public + /// interface of this crate and has to be mutable, it has to be deserialized completely. + #[derive(Debug)] +diff --git a/lib/compiler/src/artifact_builders/mod.rs b/lib/compiler/src/artifact_builders/mod.rs +index acadda9..aff8e89 100644 +--- a/lib/compiler/src/artifact_builders/mod.rs ++++ b/lib/compiler/src/artifact_builders/mod.rs +@@ -3,7 +3,9 @@ + mod artifact_builder; + mod trampoline; + +-pub use self::artifact_builder::{ArtifactBuild, ArtifactBuildFromArchive, ModuleFromArchive}; ++pub use self::artifact_builder::{ ++ ArtifactBuild, ArtifactBuildFromArchive, DetachedArtifactBuild, ModuleFromArchive, ++}; + pub use self::trampoline::get_libcall_trampoline; + #[cfg(feature = "compiler")] + pub use self::trampoline::*; +diff --git a/lib/compiler/src/engine/artifact.rs b/lib/compiler/src/engine/artifact.rs +index d807fdf..7847369 100644 +--- a/lib/compiler/src/engine/artifact.rs ++++ b/lib/compiler/src/engine/artifact.rs +@@ -9,16 +9,20 @@ use std::sync::{ + #[cfg(feature = "compiler")] + use crate::ModuleEnvironment; + use crate::{ +- ArtifactBuild, ArtifactBuildFromArchive, ArtifactCreate, Engine, EngineInner, Features, +- FrameInfosVariant, FunctionExtent, GlobalFrameInfoRegistration, InstantiationError, Tunables, ++ ArtifactBuild, ArtifactBuildFromArchive, ArtifactCreate, CodeMemoryId, DetachedArtifactBuild, ++ Engine, EngineInner, Features, FrameInfosVariant, FunctionExtent, GlobalFrameInfoRegistration, ++ InstantiationError, Tunables, + engine::{link::link_module, resolver::resolve_tags}, + lib::std::vec::IntoIter, + register_frame_info, resolve_imports, +- serialize::{MetadataHeader, SerializableModule}, +- types::relocation::{RelocationLike, RelocationTarget}, ++ serialize::{ArchivedSerializableModule, MetadataHeader, SerializableModule}, ++ types::{ ++ module::CompileModuleInfo, ++ relocation::{RelocationLike, RelocationTarget}, ++ }, + }; + #[cfg(feature = "static-artifact-create")] +-use crate::{Compiler, FunctionBodyData, ModuleTranslationState, types::module::CompileModuleInfo}; ++use crate::{Compiler, FunctionBodyData, ModuleTranslationState}; + #[cfg(any(feature = "static-artifact-create", feature = "static-artifact-load"))] + use crate::{serialize::SerializableCompilation, types::symbols::ModuleMetadata}; + +@@ -33,11 +37,13 @@ use crate::object::{ + Object, ObjectMetadataBuilder, emit_compilation, emit_data, get_object_for_target, + }; + ++#[cfg(feature = "compiler")] ++use wasmer_types::CompilationProgressCallback; + use wasmer_types::{ +- ArchivedDataInitializerLocation, ArchivedOwnedDataInitializer, CompilationProgressCallback, +- CompileError, DataInitializer, DataInitializerLike, DataInitializerLocation, +- DataInitializerLocationLike, DeserializeError, FunctionIndex, LocalFunctionIndex, MemoryIndex, +- ModuleInfo, OwnedDataInitializer, SerializeError, SignatureIndex, TableIndex, ++ ArchivedDataInitializerLocation, ArchivedOwnedDataInitializer, CompileError, DataInitializer, ++ DataInitializerLike, DataInitializerLocation, DataInitializerLocationLike, DeserializeError, ++ FunctionIndex, LocalFunctionIndex, MemoryIndex, ModuleHash, ModuleInfo, OwnedDataInitializer, ++ SerializeError, SignatureIndex, TableIndex, VMOffsets, + entity::{BoxedSlice, PrimaryMap}, + target::{CpuFeature, Target}, + }; +@@ -58,13 +64,24 @@ pub struct AllocatedArtifact { + // using 'Artifact::take_frame_info_registration' method + // so the GloabelFrameInfo and MMap stays in sync and get dropped at the same time + frame_info_registration: Option, +- finished_functions: BoxedSlice, ++ // These immutable pointer tables are shared by every VM instance of this ++ // artifact. The executable allocation itself remains owned by EngineInner. ++ finished_functions: Arc>, + + #[cfg_attr(feature = "artifact-size", loupe(skip))] +- finished_function_call_trampolines: BoxedSlice, ++ finished_function_call_trampolines: Arc>, + finished_dynamic_function_trampolines: BoxedSlice, + signatures: BoxedSlice, + finished_function_lengths: BoxedSlice, ++ ++ /// Host-layout offsets derived from this artifact's immutable module. ++ /// ++ /// Instance creation clones this compact value instead of rescanning the ++ /// module and recomputing every VMContext offset. The module name is the ++ /// only artifact metadata that can change after allocation, and it is not ++ /// an input to `VMOffsets`. ++ #[cfg_attr(feature = "artifact-size", loupe(skip))] ++ vm_offsets: VMOffsets, + } + + #[derive(Debug, PartialEq, Eq, PartialOrd, Ord)] +@@ -101,12 +118,72 @@ impl Default for ArtifactId { + #[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] + pub struct Artifact { + id: ArtifactId, ++ code_memory_id: Option, + artifact: ArtifactBuildVariant, + // The artifact will only be allocated in memory in case we can execute it + // (that means, if the target != host then this will be None). + allocated: Option, + } + ++/// A deserialized artifact whose published executable memory has not escaped ++/// its activation transaction. ++/// ++/// Dropping this value releases the artifact and then removes its exact ++/// engine-owned `CodeMemory` allocation, which deregisters frame and unwind ++/// metadata before unmapping code. [`PendingArtifact::commit`] transfers the ++/// artifact to its caller and permanently commits that allocation. ++#[doc(hidden)] ++pub struct PendingArtifact { ++ artifact: Option, ++ engine: Engine, ++ code_memory_id: Option, ++} ++ ++impl PendingArtifact { ++ fn new(engine: &Engine, artifact: Artifact) -> Self { ++ let code_memory_id = artifact.code_memory_id; ++ Self { ++ artifact: Some(artifact), ++ engine: engine.clone(), ++ code_memory_id, ++ } ++ } ++ ++ /// Borrow the validated artifact while rollback ownership remains pending. ++ pub fn artifact(&self) -> &Artifact { ++ self.artifact ++ .as_ref() ++ .expect("pending artifact must exist before commit") ++ } ++ ++ /// Return the module hash while rollback ownership remains pending. ++ pub fn module_hash(&self) -> Option { ++ self.artifact().module_info().hash ++ } ++ ++ /// Commit executable-memory ownership and return the activated artifact. ++ pub fn commit(mut self) -> Artifact { ++ self.code_memory_id = None; ++ self.artifact ++ .take() ++ .expect("pending artifact may only be committed once") ++ } ++} ++ ++impl Drop for PendingArtifact { ++ fn drop(&mut self) { ++ let artifact = self.artifact.take(); ++ drop(artifact); ++ if let Some(code_memory_id) = self.code_memory_id.take() { ++ let removed = self.engine.rollback_code_memory_activation(code_memory_id); ++ debug_assert!( ++ removed, ++ "pending artifact lost rollback ownership of its code memory" ++ ); ++ } ++ } ++} ++ + /// Artifacts may be created as the result of the compilation of a wasm + /// module, corresponding to `ArtifactBuildVariant::Plain`, or loaded + /// from an archive, corresponding to `ArtifactBuildVariant::Archived`. +@@ -115,9 +192,64 @@ pub struct Artifact { + pub enum ArtifactBuildVariant { + Plain(ArtifactBuild), + Archived(ArtifactBuildFromArchive), ++ Detached(DetachedArtifactBuild), + } + + impl Artifact { ++ fn checked_serialized_module( ++ bytes: &[u8], ++ ) -> Result<&ArchivedSerializableModule, DeserializeError> { ++ if !ArtifactBuild::is_deserializable(bytes) { ++ return Err(DeserializeError::Incompatible( ++ "The provided bytes are not a Wasmer universal artifact".to_string(), ++ )); ++ } ++ ++ let bytes = Self::get_byte_slice(bytes, ArtifactBuild::MAGIC_HEADER.len(), bytes.len())?; ++ let metadata_len = MetadataHeader::parse(bytes)?; ++ let metadata_slice = Self::get_byte_slice(bytes, MetadataHeader::LEN, bytes.len())?; ++ let metadata_slice = Self::get_byte_slice(metadata_slice, 0, metadata_len)?; ++ SerializableModule::archive_from_slice_checked(metadata_slice) ++ } ++ ++ fn validate_cpu_features( ++ target: &Target, ++ cpu_features: EnumSet, ++ ) -> Result<(), DeserializeError> { ++ if target.is_native() && !target.cpu_features().is_superset(cpu_features) { ++ return Err(DeserializeError::Incompatible(format!( ++ "Some CPU Features needed for the artifact are missing: {:?}", ++ cpu_features.difference(*target.cpu_features()) ++ ))); ++ } ++ Ok(()) ++ } ++ ++ /// Inspect a serialized universal artifact without allocating or publishing ++ /// executable code. ++ /// ++ /// This performs the same magic, metadata ABI, checked archive, and native ++ /// CPU-feature validation as checked deserialization, then returns the ++ /// module hash embedded by the compiler. Static-object artifacts are not ++ /// accepted because inspecting them requires the static artifact loader. ++ pub fn inspect_serialized( ++ engine: &Engine, ++ bytes: &[u8], ++ ) -> Result { ++ let archived = Self::checked_serialized_module(bytes)?; ++ let compile_info: CompileModuleInfo = ++ rkyv::deserialize::<_, rkyv::rancor::Error>(&archived.compile_info) ++ .map_err(|error| DeserializeError::CorruptedBinary(error.to_string()))?; ++ let cpu_features = EnumSet::from_u64(archived.cpu_features.to_native()); ++ Self::validate_cpu_features(engine.target(), cpu_features)?; ++ ++ compile_info.module.hash.ok_or_else(|| { ++ DeserializeError::CorruptedBinary( ++ "serialized artifact does not contain an embedded module hash".to_string(), ++ ) ++ }) ++ } ++ + /// Compile a data buffer into a `ArtifactBuild`, which may then be instantiated. + #[cfg(feature = "compiler")] + pub fn new( +@@ -222,14 +354,7 @@ impl Artifact { + } + + let artifact = ArtifactBuildFromArchive::try_new(bytes, |bytes| { +- let bytes = +- Self::get_byte_slice(bytes, ArtifactBuild::MAGIC_HEADER.len(), bytes.len())?; +- +- let metadata_len = MetadataHeader::parse(bytes)?; +- let metadata_slice = Self::get_byte_slice(bytes, MetadataHeader::LEN, bytes.len())?; +- let metadata_slice = Self::get_byte_slice(metadata_slice, 0, metadata_len)?; +- +- SerializableModule::archive_from_slice_checked(metadata_slice) ++ Self::checked_serialized_module(bytes.as_ref()) + })?; + + let mut inner_engine = engine.inner_mut(); +@@ -241,6 +366,56 @@ impl Artifact { + } + } + ++ /// Deserialize and validate a serialized artifact, then detach the ++ /// serialized archive once all runtime metadata has been materialized. ++ /// ++ /// This is intended for sealed, precompiled-only executors. The resulting ++ /// artifact can be instantiated normally but cannot be re-serialized. ++ /// Releasing the archive avoids pinning code and relocation bytes that have ++ /// already been copied, linked, and published into executable memory. ++ /// ++ /// # Safety ++ /// See [`Self::deserialize`]. ++ pub unsafe fn deserialize_detached( ++ engine: &Engine, ++ bytes: OwnedBuffer, ++ ) -> Result { ++ unsafe { Self::deserialize_detached_pending(engine, bytes) }.map(PendingArtifact::commit) ++ } ++ ++ /// Deserialize a detached artifact while retaining rollback ownership of ++ /// its published executable memory. ++ /// ++ /// Callers that perform fallible admission or evidence work after ++ /// deserialization should use this entry point and call ++ /// [`PendingArtifact::commit`] only after those steps succeed. ++ /// ++ /// # Safety ++ /// See [`Self::deserialize`]. ++ pub unsafe fn deserialize_detached_pending( ++ engine: &Engine, ++ bytes: OwnedBuffer, ++ ) -> Result { ++ unsafe { ++ let artifact = if !ArtifactBuild::is_deserializable(bytes.as_ref()) { ++ Self::deserialize(engine, bytes)? ++ } else { ++ let artifact = ArtifactBuildFromArchive::try_new(bytes, |bytes| { ++ Self::checked_serialized_module(bytes.as_ref()) ++ })?; ++ ++ let mut inner_engine = engine.inner_mut(); ++ Self::from_parts_with_archive_policy( ++ &mut inner_engine, ++ ArtifactBuildVariant::Archived(artifact), ++ engine.target(), ++ true, ++ )? ++ }; ++ Ok(PendingArtifact::new(engine, artifact)) ++ } ++ } ++ + /// Deserialize a serialized artifact. + /// + /// NOTE: You should prefer [`Self::deserialize`]. +@@ -293,23 +468,34 @@ impl Artifact { + engine_inner: &mut EngineInner, + artifact: ArtifactBuildVariant, + target: &Target, ++ ) -> Result { ++ Self::from_parts_with_archive_policy(engine_inner, artifact, target, false) ++ } ++ ++ fn from_parts_with_archive_policy( ++ engine_inner: &mut EngineInner, ++ artifact: ArtifactBuildVariant, ++ target: &Target, ++ detach_archive: bool, + ) -> Result { + if !target.is_native() { + return Ok(Self { + id: Default::default(), ++ code_memory_id: None, + artifact, + allocated: None, + }); + } else { +- // check if cpu features are compatible before anything else +- let cpu_features = artifact.cpu_features(); +- if !target.cpu_features().is_superset(cpu_features) { +- return Err(DeserializeError::Incompatible(format!( +- "Some CPU Features needed for the artifact are missing: {:?}", +- cpu_features.difference(*target.cpu_features()) +- ))); +- } ++ // Check CPU compatibility before allocating executable memory. ++ Self::validate_cpu_features(target, artifact.cpu_features())?; + } ++ ++ // Allocation, relocation, publication, unwind/frame registration, and ++ // detached metadata materialization are one engine-owned transaction. ++ // Any error or panic before the final commit removes every allocation ++ // made by this artifact build. ++ let mut code_memory_transaction = engine_inner.begin_code_memory_transaction(); ++ let engine_inner = code_memory_transaction.inner_mut(); + let module_info = artifact.module_info(); + let ( + finished_functions, +@@ -331,7 +517,15 @@ impl Artifact { + a.get_dynamic_function_trampolines_ref().values(), + a.get_custom_sections_ref().values(), + )?, ++ ArtifactBuildVariant::Detached(_) => { ++ unreachable!("detached artifacts have already been allocated and linked") ++ } + }; ++ let code_memory_id = engine_inner.last_code_memory_id().ok_or_else(|| { ++ DeserializeError::Generic( ++ "artifact allocation completed without engine-owned code memory".to_string(), ++ ) ++ })?; + + let get_got_address: Box Option> = match &artifact { + ArtifactBuildVariant::Plain(p) => { +@@ -369,6 +563,9 @@ impl Artifact { + Box::new(|_: RelocationTarget| None) + } + } ++ ArtifactBuildVariant::Detached(_) => { ++ unreachable!("detached artifacts have already been allocated and linked") ++ } + }; + + match &artifact { +@@ -402,6 +599,9 @@ impl Artifact { + a.get_libcall_trampoline_len(), + &get_got_address, + ), ++ ArtifactBuildVariant::Detached(_) => { ++ unreachable!("detached artifacts have already been allocated and linked") ++ } + }; + + // Compute indices into the shared signature table. +@@ -429,6 +629,9 @@ impl Artifact { + a.get_custom_sections_ref()[v].bytes.len(), + ) + }), ++ ArtifactBuildVariant::Detached(_) => { ++ unreachable!("detached artifacts have already been allocated and linked") ++ } + }; + #[allow(unused_variables)] + let compact_unwind = match &artifact { +@@ -446,6 +649,9 @@ impl Artifact { + ) + }) + } ++ ArtifactBuildVariant::Detached(_) => { ++ unreachable!("detached artifacts have already been allocated and linked") ++ } + }; + + #[cfg(all(not(target_arch = "wasm32"), feature = "compiler"))] +@@ -454,7 +660,7 @@ impl Artifact { + } + + // Make all code compiled thus far executable. +- engine_inner.publish_compiled_code(); ++ engine_inner.publish_compiled_code()?; + + #[cfg(all(target_os = "macos", target_arch = "aarch64"))] + if let Some(compact_unwind) = compact_unwind { +@@ -476,19 +682,34 @@ impl Artifact { + .map(|extent| extent.length) + .collect::>() + .into_boxed_slice(); +- let finished_functions = finished_functions +- .values() +- .map(|extent| extent.ptr) +- .collect::>() +- .into_boxed_slice(); ++ let finished_functions = Arc::new( ++ finished_functions ++ .values() ++ .map(|extent| extent.ptr) ++ .collect::>() ++ .into_boxed_slice(), ++ ); + let finished_function_call_trampolines = +- finished_function_call_trampolines.into_boxed_slice(); ++ Arc::new(finished_function_call_trampolines.into_boxed_slice()); + let finished_dynamic_function_trampolines = + finished_dynamic_function_trampolines.into_boxed_slice(); + let signatures = signatures.into_boxed_slice(); + ++ let artifact = if detach_archive { ++ match artifact { ++ ArtifactBuildVariant::Archived(archive) => { ++ ArtifactBuildVariant::Detached(DetachedArtifactBuild::from_archive(&archive)?) ++ } ++ artifact => artifact, ++ } ++ } else { ++ artifact ++ }; ++ let vm_offsets = VMOffsets::new(std::mem::size_of::() as u8, artifact.module_info()); ++ + let mut artifact = Self { + id: Default::default(), ++ code_memory_id: Some(code_memory_id), + artifact, + allocated: Some(AllocatedArtifact { + frame_info_registered: false, +@@ -498,6 +719,7 @@ impl Artifact { + finished_dynamic_function_trampolines, + signatures, + finished_function_lengths, ++ vm_offsets, + }), + }; + +@@ -508,6 +730,7 @@ impl Artifact { + engine_inner.register_frame_info(frame_info); + } + ++ code_memory_transaction.commit(); + Ok(artifact) + } + +@@ -583,6 +806,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.create_module_info(), + Self::Archived(artifact) => artifact.create_module_info(), ++ Self::Detached(artifact) => artifact.create_module_info(), + } + } + +@@ -590,6 +814,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.set_module_info_name(name), + Self::Archived(artifact) => artifact.set_module_info_name(name), ++ Self::Detached(artifact) => artifact.set_module_info_name(name), + } + } + +@@ -597,6 +822,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.module_info(), + Self::Archived(artifact) => artifact.module_info(), ++ Self::Detached(artifact) => artifact.module_info(), + } + } + +@@ -604,6 +830,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.features(), + Self::Archived(artifact) => artifact.features(), ++ Self::Detached(artifact) => artifact.features(), + } + } + +@@ -611,6 +838,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.cpu_features(), + Self::Archived(artifact) => artifact.cpu_features(), ++ Self::Detached(artifact) => artifact.cpu_features(), + } + } + +@@ -618,6 +846,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.memory_styles(), + Self::Archived(artifact) => artifact.memory_styles(), ++ Self::Detached(artifact) => artifact.memory_styles(), + } + } + +@@ -625,6 +854,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.table_styles(), + Self::Archived(artifact) => artifact.table_styles(), ++ Self::Detached(artifact) => artifact.table_styles(), + } + } + +@@ -640,6 +870,11 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + .map(OwnedDataInitializerVariant::Archived) + .collect::>() + .into_iter(), ++ Self::Detached(artifact) => artifact ++ .data_initializers() ++ .map(OwnedDataInitializerVariant::Plain) ++ .collect::>() ++ .into_iter(), + } + } + +@@ -647,6 +882,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { + match self { + Self::Plain(artifact) => artifact.serialize(), + Self::Archived(artifact) => artifact.serialize(), ++ Self::Detached(artifact) => artifact.serialize(), + } + } + } +@@ -741,27 +977,22 @@ impl Artifact { + .collect::>() + .into_boxed_slice(); + +- let frame_info_registration = &mut self +- .allocated +- .as_mut() +- .expect("It must be allocated") +- .frame_info_registration; ++ let module_info = self.artifact.create_module_info(); ++ let frame_infos = match &mut self.artifact { ++ ArtifactBuildVariant::Plain(p) => { ++ FrameInfosVariant::Owned(p.get_frame_info_ref().clone()) ++ } ++ ArtifactBuildVariant::Archived(a) => FrameInfosVariant::Archived(a.clone()), ++ ArtifactBuildVariant::Detached(a) => { ++ FrameInfosVariant::Owned(a.take_frame_info_for_registration()?) ++ } ++ }; ++ let frame_info_registration = ++ register_frame_info(module_info, &finished_function_extents, frame_infos); + +- *frame_info_registration = register_frame_info( +- self.artifact.create_module_info(), +- &finished_function_extents, +- match &self.artifact { +- ArtifactBuildVariant::Plain(p) => { +- FrameInfosVariant::Owned(p.get_frame_info_ref().clone()) +- } +- ArtifactBuildVariant::Archived(a) => FrameInfosVariant::Archived(a.clone()), +- }, +- ); +- +- self.allocated +- .as_mut() +- .expect("It must be allocated") +- .frame_info_registered = true; ++ let allocated = self.allocated.as_mut().expect("It must be allocated"); ++ allocated.frame_info_registration = frame_info_registration; ++ allocated.frame_info_registered = true; + + Ok(()) + } +@@ -786,6 +1017,16 @@ impl Artifact { + .finished_functions + } + ++ fn shared_finished_functions(&self) -> Arc> { ++ Arc::clone( ++ &self ++ .allocated ++ .as_ref() ++ .expect("It must be allocated") ++ .finished_functions, ++ ) ++ } ++ + /// Returns the function call trampolines allocated in memory of this + /// `Artifact`, ready to be run. + pub fn finished_function_call_trampolines(&self) -> &BoxedSlice { +@@ -796,6 +1037,18 @@ impl Artifact { + .finished_function_call_trampolines + } + ++ fn shared_finished_function_call_trampolines( ++ &self, ++ ) -> Arc> { ++ Arc::clone( ++ &self ++ .allocated ++ .as_ref() ++ .expect("It must be allocated") ++ .finished_function_call_trampolines, ++ ) ++ } ++ + /// Returns the dynamic function trampolines allocated in memory + /// of this `Artifact`, ready to be run. + pub fn finished_dynamic_function_trampolines( +@@ -870,7 +1123,14 @@ impl Artifact { + memory_definition_locations, + table_definition_locations, + global_definition_locations, +- ) = InstanceAllocator::new(&module); ++ ) = InstanceAllocator::new_with_offsets( ++ self.allocated ++ .as_ref() ++ .expect("Artifact::instantiate called on a non-host artifact") ++ .vm_offsets ++ .clone(), ++ &module, ++ ); + let finished_memories = tunables + .create_memories( + context, +@@ -898,8 +1158,8 @@ impl Artifact { + allocator, + module, + context, +- self.finished_functions().clone(), +- self.finished_function_call_trampolines().clone(), ++ self.shared_finished_functions(), ++ self.shared_finished_function_call_trampolines(), + finished_memories, + finished_tables, + finished_globals, +@@ -1284,21 +1544,114 @@ impl Artifact { + .collect::>() + .into_boxed_slice(); + ++ // Build the variant first so its module remains available while ++ // deriving the cached host-layout offsets. ++ let artifact = ArtifactBuildVariant::Plain(artifact); ++ let vm_offsets = ++ VMOffsets::new(std::mem::size_of::() as u8, artifact.module_info()); ++ + Ok(Self { + id: Default::default(), +- artifact: ArtifactBuildVariant::Plain(artifact), ++ code_memory_id: None, ++ artifact, + allocated: Some(AllocatedArtifact { + frame_info_registered: false, + frame_info_registration: None, +- finished_functions: finished_functions.into_boxed_slice(), +- finished_function_call_trampolines: finished_function_call_trampolines +- .into_boxed_slice(), ++ finished_functions: Arc::new(finished_functions.into_boxed_slice()), ++ finished_function_call_trampolines: Arc::new( ++ finished_function_call_trampolines.into_boxed_slice(), ++ ), + finished_dynamic_function_trampolines: finished_dynamic_function_trampolines + .into_boxed_slice(), + signatures: signatures.into_boxed_slice(), + finished_function_lengths, ++ vm_offsets, + }), + }) + } + } + } ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use crate::types::function::CompiledFunctionFrameInfo; ++ use wasmer_types::{TrapCode, TrapInformation}; ++ use wasmer_vm::VMFunctionBody; ++ ++ fn empty_boxed_slice() -> BoxedSlice ++ where ++ K: wasmer_types::entity::EntityRef, ++ { ++ PrimaryMap::::new().into_boxed_slice() ++ } ++ ++ fn detached_artifact_with_one_frame(code: *const VMFunctionBody) -> Artifact { ++ let mut frame = CompiledFunctionFrameInfo::default(); ++ frame.traps.push(TrapInformation { ++ code_offset: 0, ++ trap_code: TrapCode::UnreachableCodeReached, ++ }); ++ let mut frame_infos = PrimaryMap::new(); ++ frame_infos.push(frame); ++ ++ let mut finished_functions = PrimaryMap::new(); ++ finished_functions.push(FunctionBodyPtr(code)); ++ let mut finished_function_lengths = PrimaryMap::new(); ++ finished_function_lengths.push(1); ++ ++ let artifact = ++ ArtifactBuildVariant::Detached(DetachedArtifactBuild::for_frame_info_test(frame_infos)); ++ let vm_offsets = VMOffsets::new(std::mem::size_of::() as u8, artifact.module_info()); ++ ++ Artifact { ++ id: ArtifactId::default(), ++ code_memory_id: None, ++ artifact, ++ allocated: Some(AllocatedArtifact { ++ frame_info_registered: false, ++ frame_info_registration: None, ++ finished_functions: Arc::new(finished_functions.into_boxed_slice()), ++ finished_function_call_trampolines: Arc::new(empty_boxed_slice()), ++ finished_dynamic_function_trampolines: empty_boxed_slice(), ++ signatures: empty_boxed_slice(), ++ finished_function_lengths: finished_function_lengths.into_boxed_slice(), ++ vm_offsets, ++ }), ++ } ++ } ++ ++ #[test] ++ fn detached_frame_info_registration_moves_once_and_is_idempotent() { ++ // Registration treats code addresses as opaque range keys. Back this ++ // test range with a live, uniquely allocated byte so it cannot overlap ++ // another concurrent registration. ++ let code = Box::new(0_u8); ++ let code_ptr = (&*code as *const u8).cast::(); ++ let mut artifact = detached_artifact_with_one_frame(code_ptr); ++ ++ artifact.internal_register_frame_info().unwrap(); ++ let registration = artifact ++ .internal_take_frame_info_registration() ++ .expect("one function must produce a global frame registration"); ++ ++ let detached = match &mut artifact.artifact { ++ ArtifactBuildVariant::Detached(detached) => detached, ++ _ => unreachable!("test artifact must remain detached"), ++ }; ++ let error = detached ++ .take_frame_info_for_registration() ++ .expect_err("registration must consume detached frame metadata"); ++ assert!( ++ error.to_string().contains("already transferred"), ++ "unexpected ownership error: {error}" ++ ); ++ ++ artifact ++ .internal_register_frame_info() ++ .expect("repeated registration must be a harmless no-op"); ++ assert!(artifact.internal_take_frame_info_registration().is_none()); ++ ++ drop(registration); ++ } ++} +diff --git a/lib/compiler/src/engine/code_memory.rs b/lib/compiler/src/engine/code_memory.rs +index 19166c6..c4b3739 100644 +--- a/lib/compiler/src/engine/code_memory.rs ++++ b/lib/compiler/src/engine/code_memory.rs +@@ -11,8 +11,23 @@ use crate::{ + unwind::{CompiledFunctionUnwindInfoLike, CompiledFunctionUnwindInfoReference}, + }, + }; ++use std::path::Path; ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++use std::path::PathBuf; ++use std::sync::atomic::{AtomicUsize, Ordering}; + use wasmer_vm::{Mmap, VMFunctionBody}; + ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++use std::{ ++ ffi::CStr, ++ fs::{File, OpenOptions}, ++ os::{ ++ fd::{AsRawFd, FromRawFd}, ++ unix::{fs::MetadataExt, fs::OpenOptionsExt}, ++ }, ++ sync::Arc, ++}; ++ + /// The optimal alignment for functions. + /// + /// On x86-64, this is 16 since it's what the optimizations assume. +@@ -24,26 +39,671 @@ const ARCH_FUNCTION_ALIGNMENT: usize = 16; + /// + const DATA_SECTION_ALIGNMENT: usize = 64; + ++/// Stable identity of the strict Linux x86-64 file-backed code-memory policy. ++pub const STRICT_LINUX_X86_64_CODE_MEMORY_POLICY_ID: &str = ++ "wasmer.code-memory.relocated-regular-file.linux-x86_64.v1"; ++ ++/// Process-unique identity of one engine-owned executable-code allocation. ++/// ++/// This identity exists so a sealed activation can retain rollback ownership ++/// after executable publication without relying on vector position. It is not ++/// serialized and has no cross-process meaning. ++#[doc(hidden)] ++#[derive(Clone, Copy, Debug, PartialEq, Eq)] ++#[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] ++pub(crate) struct CodeMemoryId(usize); ++ ++static NEXT_CODE_MEMORY_ID: AtomicUsize = AtomicUsize::new(0); ++ ++fn next_code_memory_id() -> CodeMemoryId { ++ let id = NEXT_CODE_MEMORY_ID ++ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |id| id.checked_add(1)) ++ .expect("process exhausted unique code-memory allocation identities"); ++ CodeMemoryId(id) ++} ++ ++/// Per-engine ownership policy for relocated executable memory. ++/// ++/// Generic Wasmer engines use anonymous memory. The strict file-backed mode is ++/// deliberately opt-in and is available only on Linux x86-64. Its constructor ++/// opens and pins the selected directory; allocation never searches for a ++/// different directory or silently falls back to anonymous memory. ++#[derive(Clone, Debug)] ++pub struct CodeMemoryPolicy { ++ kind: CodeMemoryPolicyKind, ++} ++ ++#[derive(Clone, Debug)] ++enum CodeMemoryPolicyKind { ++ Anonymous, ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ StrictLinuxX86_64FileBacked(StrictLinuxX86_64FileBackedPolicy), ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++#[derive(Clone, Debug)] ++struct StrictLinuxX86_64FileBackedPolicy { ++ directory: Arc, ++ canonical_directory: Arc, ++ directory_device: u64, ++ directory_inode: u64, ++} ++ ++impl Default for CodeMemoryPolicy { ++ fn default() -> Self { ++ Self::anonymous() ++ } ++} ++ ++impl CodeMemoryPolicy { ++ /// Select ordinary anonymous executable memory. ++ pub fn anonymous() -> Self { ++ Self { ++ kind: CodeMemoryPolicyKind::Anonymous, ++ } ++ } ++ ++ /// Select strict regular-file-backed relocated code memory on Linux x86-64. ++ /// ++ /// `directory` is opened once and retained by descriptor. Every later code ++ /// allocation uses `openat(O_TMPFILE)` against that descriptor, so path ++ /// replacement after engine configuration cannot redirect allocations. ++ pub fn strict_linux_x86_64_file_backed(directory: impl AsRef) -> Result { ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ { ++ let requested_metadata = ++ std::fs::symlink_metadata(directory.as_ref()).map_err(|error| { ++ format!( ++ "inspect strict code-memory directory {}: {error}", ++ directory.as_ref().display() ++ ) ++ })?; ++ if requested_metadata.file_type().is_symlink() || !requested_metadata.is_dir() { ++ return Err(format!( ++ "strict code-memory path must be a non-symlink directory: {}", ++ directory.as_ref().display() ++ )); ++ } ++ let directory_file = OpenOptions::new() ++ .read(true) ++ .custom_flags(libc::O_CLOEXEC | libc::O_DIRECTORY | libc::O_NOFOLLOW) ++ .open(directory.as_ref()) ++ .map_err(|error| { ++ format!( ++ "open strict code-memory directory {}: {error}", ++ directory.as_ref().display() ++ ) ++ })?; ++ let metadata = directory_file.metadata().map_err(|error| { ++ format!( ++ "inspect strict code-memory directory {}: {error}", ++ directory.as_ref().display() ++ ) ++ })?; ++ if requested_metadata.dev() != metadata.dev() ++ || requested_metadata.ino() != metadata.ino() ++ { ++ return Err(format!( ++ "strict code-memory directory changed identity while being pinned: {}", ++ directory.as_ref().display() ++ )); ++ } ++ if !metadata.is_dir() { ++ return Err(format!( ++ "strict code-memory path is not a directory: {}", ++ directory.as_ref().display() ++ )); ++ } ++ if metadata.uid() != unsafe { libc::geteuid() } { ++ return Err(format!( ++ "strict code-memory directory is not owned by the effective user: {}", ++ directory.as_ref().display() ++ )); ++ } ++ if metadata.mode() & 0o7777 != 0o700 { ++ return Err(format!( ++ "strict code-memory directory must have exact mode 0700: {}", ++ directory.as_ref().display() ++ )); ++ } ++ let canonical_directory = ++ std::fs::canonicalize(directory.as_ref()).map_err(|error| { ++ format!( ++ "canonicalize pinned strict code-memory directory {}: {error}", ++ directory.as_ref().display() ++ ) ++ })?; ++ let canonical_metadata = std::fs::metadata(&canonical_directory).map_err(|error| { ++ format!( ++ "inspect canonical strict code-memory directory {}: {error}", ++ canonical_directory.display() ++ ) ++ })?; ++ if canonical_metadata.dev() != metadata.dev() ++ || canonical_metadata.ino() != metadata.ino() ++ { ++ return Err(format!( ++ "strict code-memory directory changed identity while canonicalizing: {}", ++ directory.as_ref().display() ++ )); ++ } ++ reject_memory_backed_filesystem(directory_file.as_raw_fd()).map_err(|error| { ++ format!( ++ "reject strict code-memory directory {}: {error}", ++ canonical_directory.display() ++ ) ++ })?; ++ Ok(Self { ++ kind: CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked( ++ StrictLinuxX86_64FileBackedPolicy { ++ directory: Arc::new(directory_file), ++ canonical_directory: Arc::new(canonical_directory), ++ directory_device: metadata.dev(), ++ directory_inode: metadata.ino(), ++ }, ++ ), ++ }) ++ } ++ ++ #[cfg(not(all(target_os = "linux", target_arch = "x86_64")))] ++ { ++ let _ = directory; ++ Err("strict file-backed code memory is supported only on Linux x86-64".to_string()) ++ } ++ } ++ ++ /// Return the stable policy identity. ++ pub fn id(&self) -> &'static str { ++ match self.kind { ++ CodeMemoryPolicyKind::Anonymous => "wasmer.code-memory.anonymous.v1", ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(_) => { ++ STRICT_LINUX_X86_64_CODE_MEMORY_POLICY_ID ++ } ++ } ++ } ++ ++ /// Return the pinned directory selected by a strict file-backed policy. ++ pub fn pinned_directory(&self) -> Option<&Path> { ++ match &self.kind { ++ CodeMemoryPolicyKind::Anonymous => None, ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { ++ Some(policy.canonical_directory.as_path()) ++ } ++ } ++ } ++ ++ /// Return the device containing the pinned strict-policy directory. ++ pub fn pinned_directory_device(&self) -> Option { ++ match &self.kind { ++ CodeMemoryPolicyKind::Anonymous => None, ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { ++ Some(policy.directory_device) ++ } ++ } ++ } ++ ++ /// Return the inode of the pinned strict-policy directory. ++ pub fn pinned_directory_inode(&self) -> Option { ++ match &self.kind { ++ CodeMemoryPolicyKind::Anonymous => None, ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { ++ Some(policy.directory_inode) ++ } ++ } ++ } ++} ++ ++enum CodeMemoryBacking { ++ Anonymous(Mmap), ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ StrictLinuxX86_64FileBacked(StrictLinuxX86_64CodeMapping), ++} ++ ++impl CodeMemoryBacking { ++ fn empty() -> Self { ++ Self::Anonymous(Mmap::new()) ++ } ++ ++ fn allocate(policy: &CodeMemoryPolicy, len: usize) -> Result { ++ match &policy.kind { ++ CodeMemoryPolicyKind::Anonymous => Mmap::with_at_least(len).map(Self::Anonymous), ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { ++ StrictLinuxX86_64CodeMapping::allocate(policy, len) ++ .map(Self::StrictLinuxX86_64FileBacked) ++ } ++ } ++ } ++ ++ fn as_mut_slice(&mut self) -> &mut [u8] { ++ match self { ++ Self::Anonymous(mapping) => mapping.as_mut_slice(), ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ Self::StrictLinuxX86_64FileBacked(mapping) => mapping.as_mut_slice(), ++ } ++ } ++ ++ fn len(&self) -> usize { ++ match self { ++ Self::Anonymous(mapping) => mapping.len(), ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ Self::StrictLinuxX86_64FileBacked(mapping) => mapping.len, ++ } ++ } ++ ++ fn is_empty(&self) -> bool { ++ self.len() == 0 ++ } ++ ++ fn publish(&mut self, executable_prefix_len: usize) -> Result<(), String> { ++ match self { ++ Self::Anonymous(mapping) => { ++ if executable_prefix_len == 0 { ++ return Ok(()); ++ } ++ unsafe { ++ region::protect( ++ mapping.as_mut_ptr(), ++ executable_prefix_len, ++ region::Protection::READ_EXECUTE, ++ ) ++ } ++ .map_err(|error| format!("make anonymous code memory executable: {error}")) ++ } ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ Self::StrictLinuxX86_64FileBacked(mapping) => mapping.publish(executable_prefix_len), ++ } ++ } ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++struct StrictLinuxX86_64CodeMapping { ++ ptr: usize, ++ len: usize, ++ writable_file: Option, ++ read_only_file: Option, ++ device: u64, ++ inode: u64, ++ published: bool, ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++impl StrictLinuxX86_64CodeMapping { ++ fn allocate(policy: &StrictLinuxX86_64FileBackedPolicy, len: usize) -> Result { ++ if len == 0 { ++ return Ok(Self { ++ ptr: Vec::::new().as_ptr() as usize, ++ len: 0, ++ writable_file: None, ++ read_only_file: None, ++ device: 0, ++ inode: 0, ++ published: false, ++ }); ++ } ++ let len = len.next_multiple_of(region::page::size()); ++ ++ verify_pinned_directory(policy)?; ++ let dot = c"."; ++ let descriptor = unsafe { ++ libc::openat( ++ policy.directory.as_raw_fd(), ++ dot.as_ptr(), ++ libc::O_CLOEXEC | libc::O_RDWR | libc::O_TMPFILE | libc::O_EXCL, ++ 0o600, ++ ) ++ }; ++ if descriptor < 0 { ++ return Err(format!( ++ "create O_TMPFILE in pinned code-memory directory {}: {}", ++ policy.canonical_directory.display(), ++ std::io::Error::last_os_error() ++ )); ++ } ++ let writable_file = unsafe { File::from_raw_fd(descriptor) }; ++ let metadata = validate_anonymous_regular_file(&writable_file)?; ++ if metadata.dev() != policy.directory_device { ++ return Err("strict code-memory O_TMPFILE changed filesystem device".to_string()); ++ } ++ if metadata.uid() != unsafe { libc::geteuid() } { ++ return Err( ++ "strict code-memory O_TMPFILE is not owned by the effective user".to_string(), ++ ); ++ } ++ if metadata.mode() & 0o7777 != 0o600 { ++ return Err("strict code-memory O_TMPFILE must have exact mode 0600".to_string()); ++ } ++ reject_memory_backed_filesystem(writable_file.as_raw_fd())?; ++ ++ let length: libc::off_t = len ++ .try_into() ++ .map_err(|_| "code-memory allocation does not fit off_t".to_string())?; ++ if unsafe { libc::ftruncate(writable_file.as_raw_fd(), length) } != 0 { ++ return Err(format!( ++ "ftruncate strict code-memory file to {len} bytes: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ let allocation_result = ++ unsafe { libc::posix_fallocate(writable_file.as_raw_fd(), 0, length) }; ++ if allocation_result != 0 { ++ return Err(format!( ++ "posix_fallocate strict code-memory file to {len} bytes: {}", ++ std::io::Error::from_raw_os_error(allocation_result) ++ )); ++ } ++ ++ // Probe and retain the exact read-only inode descriptor before any ++ // relocation writes begin. A missing /proc descriptor bridge is a hard ++ // admission error, never a reason to fall back after linking starts. ++ let read_only_file = reopen_read_only(&writable_file)?; ++ validate_same_anonymous_regular_file(&writable_file, &read_only_file)?; ++ ++ let ptr = unsafe { ++ libc::mmap( ++ std::ptr::null_mut(), ++ len, ++ libc::PROT_READ | libc::PROT_WRITE, ++ libc::MAP_SHARED, ++ writable_file.as_raw_fd(), ++ 0, ++ ) ++ }; ++ if ptr == libc::MAP_FAILED { ++ return Err(format!( ++ "map strict code-memory file shared read-write: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ ++ Ok(Self { ++ ptr: ptr as usize, ++ len, ++ writable_file: Some(writable_file), ++ read_only_file: Some(read_only_file), ++ device: metadata.dev(), ++ inode: metadata.ino(), ++ published: false, ++ }) ++ } ++ ++ fn as_mut_slice(&mut self) -> &mut [u8] { ++ unsafe { std::slice::from_raw_parts_mut(self.ptr as *mut u8, self.len) } ++ } ++ ++ fn publish(&mut self, executable_prefix_len: usize) -> Result<(), String> { ++ if self.len == 0 { ++ self.published = true; ++ return Ok(()); ++ } ++ if self.published { ++ return Err("strict code-memory mapping was already published".to_string()); ++ } ++ if executable_prefix_len > self.len { ++ return Err("executable code-memory prefix exceeds its mapping".to_string()); ++ } ++ let page_size = region::page::size(); ++ let executable_pages = if executable_prefix_len == 0 { ++ 0 ++ } else { ++ executable_prefix_len.next_multiple_of(page_size) ++ }; ++ if executable_pages > self.len { ++ return Err("rounded executable code-memory prefix exceeds its mapping".to_string()); ++ } ++ ++ let writable_file = self ++ .writable_file ++ .as_ref() ++ .ok_or_else(|| "strict code-memory writable descriptor is missing".to_string())?; ++ let read_only_file = self ++ .read_only_file ++ .as_ref() ++ .ok_or_else(|| "strict code-memory read-only descriptor is missing".to_string())?; ++ validate_same_anonymous_regular_file(writable_file, read_only_file)?; ++ let current_metadata = writable_file ++ .metadata() ++ .map_err(|error| format!("inspect strict code-memory inode before publish: {error}"))?; ++ if current_metadata.dev() != self.device || current_metadata.ino() != self.inode { ++ return Err("strict code-memory inode identity changed before publish".to_string()); ++ } ++ ++ if unsafe { libc::fchmod(writable_file.as_raw_fd(), 0o400) } != 0 { ++ return Err(format!( ++ "make strict code-memory inode read-only: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ if unsafe { libc::msync(self.ptr as *mut libc::c_void, self.len, libc::MS_SYNC) } != 0 { ++ return Err(format!( ++ "synchronize relocated strict code memory: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ ++ // Remove the writable mapping before dropping the only writable file ++ // description. From this point onward there is no writable alias. ++ if unsafe { libc::mprotect(self.ptr as *mut libc::c_void, self.len, libc::PROT_READ) } != 0 ++ { ++ return Err(format!( ++ "remove write permission from relocated strict code memory: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ drop(self.writable_file.take()); ++ ++ let remapped = unsafe { ++ libc::mmap( ++ self.ptr as *mut libc::c_void, ++ self.len, ++ libc::PROT_READ, ++ libc::MAP_PRIVATE | libc::MAP_FIXED, ++ read_only_file.as_raw_fd(), ++ 0, ++ ) ++ }; ++ if remapped == libc::MAP_FAILED { ++ return Err(format!( ++ "remap relocated strict code memory private read-only: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ if remapped as usize != self.ptr { ++ return Err("fixed strict code-memory remap changed its base address".to_string()); ++ } ++ if executable_pages != 0 ++ && unsafe { ++ libc::mprotect( ++ self.ptr as *mut libc::c_void, ++ executable_pages, ++ libc::PROT_READ | libc::PROT_EXEC, ++ ) ++ } != 0 ++ { ++ return Err(format!( ++ "make relocated strict code-memory prefix executable: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ ++ // The MAP_PRIVATE view is clean and reconstructible from its unlinked ++ // regular inode. Discard residency now; demanded code pages fault back ++ // as file cache instead of permanently accounting as anonymous RSS. ++ if unsafe { libc::madvise(self.ptr as *mut libc::c_void, self.len, libc::MADV_DONTNEED) } ++ != 0 ++ { ++ return Err(format!( ++ "discard clean strict code-memory pages: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ let advice_result = unsafe { ++ libc::posix_fadvise(read_only_file.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) ++ }; ++ if advice_result != 0 { ++ return Err(format!( ++ "discard strict code-memory file cache: {}", ++ std::io::Error::from_raw_os_error(advice_result) ++ )); ++ } ++ drop(self.read_only_file.take()); ++ self.published = true; ++ Ok(()) ++ } ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++impl Drop for StrictLinuxX86_64CodeMapping { ++ fn drop(&mut self) { ++ if self.len != 0 { ++ unsafe { ++ libc::munmap(self.ptr as *mut libc::c_void, self.len); ++ } ++ } ++ } ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++fn verify_pinned_directory(policy: &StrictLinuxX86_64FileBackedPolicy) -> Result<(), String> { ++ let metadata = policy ++ .directory ++ .metadata() ++ .map_err(|error| format!("inspect pinned code-memory directory: {error}"))?; ++ if !metadata.is_dir() ++ || metadata.dev() != policy.directory_device ++ || metadata.ino() != policy.directory_inode ++ { ++ return Err("pinned code-memory directory identity changed".to_string()); ++ } ++ if metadata.uid() != unsafe { libc::geteuid() } { ++ return Err("pinned code-memory directory is not owned by the effective user".to_string()); ++ } ++ if metadata.mode() & 0o7777 != 0o700 { ++ return Err("pinned code-memory directory must retain exact mode 0700".to_string()); ++ } ++ reject_memory_backed_filesystem(policy.directory.as_raw_fd()) ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++fn validate_anonymous_regular_file(file: &File) -> Result { ++ let metadata = file ++ .metadata() ++ .map_err(|error| format!("inspect anonymous code-memory file: {error}"))?; ++ if !metadata.file_type().is_file() { ++ return Err("strict code-memory O_TMPFILE inode is not a regular file".to_string()); ++ } ++ if metadata.nlink() != 0 { ++ return Err("strict code-memory O_TMPFILE inode unexpectedly has a name".to_string()); ++ } ++ Ok(metadata) ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++fn validate_same_anonymous_regular_file(writable: &File, read_only: &File) -> Result<(), String> { ++ let writable_metadata = validate_anonymous_regular_file(writable)?; ++ let read_only_metadata = validate_anonymous_regular_file(read_only)?; ++ if writable_metadata.dev() != read_only_metadata.dev() ++ || writable_metadata.ino() != read_only_metadata.ino() ++ { ++ return Err("read-only code-memory descriptor changed inode identity".to_string()); ++ } ++ let writable_flags = unsafe { libc::fcntl(writable.as_raw_fd(), libc::F_GETFL) }; ++ if writable_flags < 0 { ++ return Err(format!( ++ "inspect writable code-memory descriptor: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ if writable_flags & libc::O_ACCMODE != libc::O_RDWR { ++ return Err("code-memory relocation descriptor is not read-write".to_string()); ++ } ++ let flags = unsafe { libc::fcntl(read_only.as_raw_fd(), libc::F_GETFL) }; ++ if flags < 0 { ++ return Err(format!( ++ "inspect read-only code-memory descriptor: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ if flags & libc::O_ACCMODE != libc::O_RDONLY { ++ return Err("code-memory descriptor bridge did not remove write access".to_string()); ++ } ++ Ok(()) ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++fn reopen_read_only(file: &File) -> Result { ++ let descriptor_path = format!("/proc/self/fd/{}\0", file.as_raw_fd()); ++ let descriptor_path = CStr::from_bytes_with_nul(descriptor_path.as_bytes()) ++ .map_err(|error| format!("construct code-memory descriptor path: {error}"))?; ++ let descriptor = ++ unsafe { libc::open(descriptor_path.as_ptr(), libc::O_RDONLY | libc::O_CLOEXEC) }; ++ if descriptor < 0 { ++ return Err(format!( ++ "reopen anonymous code-memory inode read-only: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ Ok(unsafe { File::from_raw_fd(descriptor) }) ++} ++ ++#[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++fn reject_memory_backed_filesystem(descriptor: libc::c_int) -> Result<(), String> { ++ let mut status = std::mem::MaybeUninit::::uninit(); ++ if unsafe { libc::fstatfs(descriptor, status.as_mut_ptr()) } != 0 { ++ return Err(format!( ++ "identify code-memory filesystem: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ let filesystem_type = unsafe { status.assume_init() }.f_type as u64; ++ const TMPFS_MAGIC: u64 = 0x0102_1994; ++ const RAMFS_MAGIC: u64 = 0x8584_58f6; ++ const HUGETLBFS_MAGIC: u64 = 0x9584_58f6; ++ if matches!(filesystem_type, TMPFS_MAGIC | RAMFS_MAGIC | HUGETLBFS_MAGIC) { ++ return Err(format!( ++ "code-memory filesystem is memory-backed (type 0x{filesystem_type:x})" ++ )); ++ } ++ Ok(()) ++} ++ + /// Memory manager for executable code. + pub struct CodeMemory { ++ id: CodeMemoryId, + // frame info is placed first, to ensure it's dropped before the mmap + frame_info_registration: Option, + unwind_registry: UnwindRegistry, +- mmap: Mmap, ++ backing: CodeMemoryBacking, ++ policy: CodeMemoryPolicy, + start_of_nonexecutable_pages: usize, + } + + impl CodeMemory { + /// Create a new `CodeMemory` instance. + pub fn new() -> Self { ++ Self::with_policy(CodeMemoryPolicy::anonymous()) ++ } ++ ++ pub(crate) fn with_policy(policy: CodeMemoryPolicy) -> Self { + Self { ++ id: next_code_memory_id(), + unwind_registry: UnwindRegistry::new(), +- mmap: Mmap::new(), ++ backing: CodeMemoryBacking::empty(), ++ policy, + start_of_nonexecutable_pages: 0, + frame_info_registration: None, + } + } + ++ /// Return this process-local allocation identity. ++ pub(crate) fn id(&self) -> CodeMemoryId { ++ self.id ++ } ++ + /// Mutably get the UnwindRegistry. + pub fn unwind_registry_mut(&mut self) -> &mut UnwindRegistry { + &mut self.unwind_registry +@@ -100,13 +760,13 @@ impl CodeMemory { + + // 2. Allocate the pages. Mark them all read-write. + +- self.mmap = Mmap::with_at_least(total_len)?; ++ self.backing = CodeMemoryBacking::allocate(&self.policy, total_len)?; + + // 3. Determine where the pointers to each function, executable section + // or data section are. Copy the functions. Collect the addresses of each and return them. + + let mut bytes = 0; +- let mut buf = self.mmap.as_mut_slice(); ++ let mut buf = self.backing.as_mut_slice(); + for func in functions { + let len = round_up( + Self::function_allocation_size(*func), +@@ -158,19 +818,14 @@ impl CodeMemory { + } + + /// Apply the page permissions. +- pub fn publish(&mut self) { +- if self.mmap.is_empty() || self.start_of_nonexecutable_pages == 0 { +- return; +- } +- assert!(self.mmap.len() >= self.start_of_nonexecutable_pages); +- unsafe { +- region::protect( +- self.mmap.as_mut_ptr(), +- self.start_of_nonexecutable_pages, +- region::Protection::READ_EXECUTE, +- ) ++ pub fn publish(&mut self) -> Result<(), String> { ++ if self.backing.is_empty() { ++ return Ok(()); + } +- .expect("unable to make memory readonly and executable"); ++ if self.backing.len() < self.start_of_nonexecutable_pages { ++ return Err("executable code-memory prefix exceeds allocation".to_string()); ++ } ++ self.backing.publish(self.start_of_nonexecutable_pages) + } + + /// Calculates the allocation size of the given compiled function. +@@ -243,9 +898,261 @@ fn round_up(size: usize, multiple: usize) -> usize { + + #[cfg(test)] + mod tests { +- use super::CodeMemory; ++ use super::{CodeMemory, CodeMemoryPolicy}; ++ + fn _assert() { + fn _assert_send_sync() {} + _assert_send_sync::(); + } ++ ++ #[test] ++ fn generic_code_memory_policy_remains_anonymous() { ++ let policy = CodeMemoryPolicy::default(); ++ assert_eq!(policy.id(), "wasmer.code-memory.anonymous.v1"); ++ assert!(policy.pinned_directory().is_none()); ++ assert!(policy.pinned_directory_device().is_none()); ++ assert!(policy.pinned_directory_inode().is_none()); ++ } ++ ++ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] ++ mod strict_linux_x86_64 { ++ use super::super::{ ++ CodeMemoryPolicy, CodeMemoryPolicyKind, StrictLinuxX86_64CodeMapping, ++ StrictLinuxX86_64FileBackedPolicy, ++ }; ++ use std::{ ++ ffi::CString, ++ fs, ++ os::{ ++ fd::AsRawFd, ++ unix::{ ++ ffi::OsStrExt, ++ fs::{PermissionsExt, symlink}, ++ }, ++ }, ++ path::Path, ++ }; ++ ++ fn disk_directory() -> tempfile::TempDir { ++ let directory = tempfile::Builder::new() ++ .prefix("wasmer-code-memory-test-") ++ .tempdir_in("/var/tmp") ++ .expect("create strict code-memory test directory on /var/tmp"); ++ fs::set_permissions(directory.path(), fs::Permissions::from_mode(0o700)) ++ .expect("make strict code-memory test directory private"); ++ directory ++ } ++ ++ fn strict_policy(directory: &Path) -> StrictLinuxX86_64FileBackedPolicy { ++ match CodeMemoryPolicy::strict_linux_x86_64_file_backed(directory) ++ .expect("admit strict code-memory test directory") ++ .kind ++ { ++ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => policy, ++ CodeMemoryPolicyKind::Anonymous => panic!("strict policy became anonymous"), ++ } ++ } ++ ++ fn mapping_line(address: usize) -> String { ++ fs::read_to_string("/proc/self/maps") ++ .expect("read process mappings") ++ .lines() ++ .find(|line| { ++ let range = line.split_whitespace().next().expect("mapping range"); ++ let (start, end) = range.split_once('-').expect("mapping range separator"); ++ let start = usize::from_str_radix(start, 16).expect("mapping start"); ++ let end = usize::from_str_radix(end, 16).expect("mapping end"); ++ start <= address && address < end ++ }) ++ .expect("address is present in process mappings") ++ .to_string() ++ } ++ ++ fn smaps_kib(address: usize, field: &str) -> usize { ++ let smaps = fs::read_to_string("/proc/self/smaps").expect("read process smaps"); ++ let mut selected = false; ++ for line in smaps.lines() { ++ let range = line ++ .split_whitespace() ++ .next() ++ .and_then(|value| value.split_once('-')) ++ .and_then(|(start, end)| { ++ Some(( ++ usize::from_str_radix(start, 16).ok()?, ++ usize::from_str_radix(end, 16).ok()?, ++ )) ++ }); ++ if let Some((start, end)) = range { ++ if selected { ++ break; ++ } ++ selected = start <= address && address < end; ++ continue; ++ } ++ if selected && line.starts_with(field) { ++ return line ++ .split_whitespace() ++ .nth(1) ++ .expect("smaps value") ++ .parse() ++ .expect("numeric smaps value"); ++ } ++ } ++ panic!("smaps field {field} is present for address {address:#x}"); ++ } ++ ++ #[test] ++ fn relocated_regular_file_preserves_base_bytes_permissions_and_execution() { ++ let directory = disk_directory(); ++ let policy = strict_policy(directory.path()); ++ let page_size = region::page::size(); ++ let mut mapping = StrictLinuxX86_64CodeMapping::allocate(&policy, page_size * 2) ++ .expect("allocate strict code memory"); ++ let base = mapping.ptr; ++ let inode = mapping.inode; ++ ++ // O_EXCL has a distinct meaning with O_TMPFILE: it permanently ++ // forbids turning the anonymous inode into a named file. Prove ++ // that property before publication while the writer descriptor is ++ // still available. ++ let writable_fd = mapping ++ .writable_file ++ .as_ref() ++ .expect("strict allocation retains its writer before publication") ++ .as_raw_fd(); ++ let source = CString::new(format!("/proc/self/fd/{writable_fd}")) ++ .expect("descriptor path contains no NUL"); ++ let linked_path = directory.path().join("must-remain-unnameable"); ++ let destination = CString::new(linked_path.as_os_str().as_bytes()) ++ .expect("test destination contains no NUL"); ++ let link_result = unsafe { ++ libc::linkat( ++ libc::AT_FDCWD, ++ source.as_ptr(), ++ libc::AT_FDCWD, ++ destination.as_ptr(), ++ libc::AT_SYMLINK_FOLLOW, ++ ) ++ }; ++ assert_eq!( ++ link_result, -1, ++ "O_TMPFILE|O_EXCL code image must never become nameable" ++ ); ++ assert!(!linked_path.exists()); ++ ++ // ENDBR64; mov eax, 42; ret. ENDBR64 keeps this indirect call valid ++ // on hosts enforcing Intel CET while remaining a NOP elsewhere. ++ let function = [0xf3, 0x0f, 0x1e, 0xfa, 0xb8, 42, 0, 0, 0, 0xc3]; ++ mapping.as_mut_slice()[..function.len()].copy_from_slice(&function); ++ let data = b"relocated-code-memory-byte-stability"; ++ mapping.as_mut_slice()[page_size..page_size + data.len()].copy_from_slice(data); ++ ++ mapping ++ .publish(page_size) ++ .expect("publish strict code memory"); ++ assert_eq!(mapping.ptr, base, "fixed remap must preserve the link base"); ++ assert!(mapping.writable_file.is_none()); ++ assert!(mapping.read_only_file.is_none()); ++ ++ let executable = mapping_line(base); ++ let read_only = mapping_line(base + page_size); ++ assert_eq!( ++ executable.split_whitespace().nth(1), ++ Some("r-xp"), ++ "executable prefix must be private RX: {executable}" ++ ); ++ assert_eq!( ++ read_only.split_whitespace().nth(1), ++ Some("r--p"), ++ "data suffix must be private RO: {read_only}" ++ ); ++ assert!(executable.contains("(deleted)")); ++ assert!(read_only.contains("(deleted)")); ++ assert_eq!( ++ smaps_kib(base, "Rss:"), ++ 0, ++ "MADV_DONTNEED must remove the relocated executable PTE" ++ ); ++ assert_eq!( ++ smaps_kib(base + page_size, "Rss:"), ++ 0, ++ "MADV_DONTNEED must remove the relocated data PTE" ++ ); ++ let inode = inode.to_string(); ++ for line in fs::read_to_string("/proc/self/maps") ++ .expect("read process mappings") ++ .lines() ++ { ++ let fields = line.split_whitespace().collect::>(); ++ if fields.get(4) == Some(&inode.as_str()) { ++ assert!( ++ !fields[1].contains('w'), ++ "published inode retained a writable mapping: {line}" ++ ); ++ } ++ } ++ ++ assert_eq!( ++ unsafe { std::slice::from_raw_parts(base as *const u8, function.len()) }, ++ function ++ ); ++ assert_eq!( ++ unsafe { std::slice::from_raw_parts((base + page_size) as *const u8, data.len()) }, ++ data ++ ); ++ ++ let call: unsafe extern "C" fn() -> u32 = unsafe { std::mem::transmute(base) }; ++ assert_eq!(unsafe { call() }, 42); ++ for address in [base, base + page_size] { ++ assert_eq!(smaps_kib(address, "Anonymous:"), 0); ++ assert_eq!(smaps_kib(address, "Private_Dirty:"), 0); ++ assert_eq!(smaps_kib(address, "Shared_Dirty:"), 0); ++ } ++ } ++ ++ #[test] ++ fn strict_policy_rejects_memory_backed_symlink_and_non_private_directories() { ++ if Path::new("/dev/shm").is_dir() { ++ let memory_backed = tempfile::Builder::new() ++ .prefix("wasmer-code-memory-test-") ++ .tempdir_in("/dev/shm") ++ .expect("create tmpfs rejection fixture"); ++ let error = CodeMemoryPolicy::strict_linux_x86_64_file_backed(memory_backed.path()) ++ .expect_err("tmpfs must be rejected"); ++ assert!(error.contains("memory-backed"), "unexpected error: {error}"); ++ } ++ ++ let owner = disk_directory(); ++ let target = owner.path().join("target"); ++ fs::create_dir(&target).expect("create symlink target"); ++ fs::set_permissions(&target, fs::Permissions::from_mode(0o700)) ++ .expect("set target permissions"); ++ let link = owner.path().join("link"); ++ symlink(&target, &link).expect("create directory symlink"); ++ let error = CodeMemoryPolicy::strict_linux_x86_64_file_backed(&link) ++ .expect_err("symlink must be rejected"); ++ assert!(error.contains("non-symlink"), "unexpected error: {error}"); ++ ++ fs::set_permissions(&target, fs::Permissions::from_mode(0o750)) ++ .expect("make target non-private"); ++ let error = CodeMemoryPolicy::strict_linux_x86_64_file_backed(&target) ++ .expect_err("non-0700 directory must be rejected"); ++ assert!( ++ error.contains("exact mode 0700"), ++ "unexpected error: {error}" ++ ); ++ ++ let pinned = disk_directory(); ++ let policy = strict_policy(pinned.path()); ++ fs::set_permissions(pinned.path(), fs::Permissions::from_mode(0o750)) ++ .expect("mutate pinned directory permissions"); ++ let error = StrictLinuxX86_64CodeMapping::allocate(&policy, region::page::size()) ++ .err() ++ .expect("allocation must revalidate its pinned directory"); ++ assert!( ++ error.contains("retain exact mode 0700"), ++ "unexpected error: {error}" ++ ); ++ } ++ } + } +diff --git a/lib/compiler/src/engine/inner.rs b/lib/compiler/src/engine/inner.rs +index a1e6661..a62c16d 100644 +--- a/lib/compiler/src/engine/inner.rs ++++ b/lib/compiler/src/engine/inner.rs +@@ -25,7 +25,8 @@ use wasmer_types::{ + + #[cfg(not(target_arch = "wasm32"))] + use crate::{ +- Artifact, BaseTunables, CodeMemory, FunctionExtent, GlobalFrameInfoRegistration, Tunables, ++ Artifact, BaseTunables, CodeMemory, CodeMemoryId, CodeMemoryPolicy, FunctionExtent, ++ GlobalFrameInfoRegistration, Tunables, + types::{ + function::FunctionBodyLike, + section::{CustomSectionLike, CustomSectionProtection, SectionIndex}, +@@ -69,6 +70,8 @@ impl Engine { + #[cfg(not(target_arch = "wasm32"))] + code_memory: vec![], + #[cfg(not(target_arch = "wasm32"))] ++ code_memory_policy: CodeMemoryPolicy::anonymous(), ++ #[cfg(not(target_arch = "wasm32"))] + signatures: SignatureRegistry::new(), + })), + target: Arc::new(target), +@@ -128,6 +131,8 @@ impl Engine { + #[cfg(not(target_arch = "wasm32"))] + code_memory: vec![], + #[cfg(not(target_arch = "wasm32"))] ++ code_memory_policy: CodeMemoryPolicy::anonymous(), ++ #[cfg(not(target_arch = "wasm32"))] + signatures: SignatureRegistry::new(), + })), + target: Arc::new(target), +@@ -300,6 +305,46 @@ impl Engine { + self.tunables = Arc::new(tunables); + } + ++ /// Select executable-memory ownership before this engine allocates code. ++ /// ++ /// Generic engines retain the anonymous default. A policy change after any ++ /// artifact allocation is rejected so one engine cannot mix ownership ++ /// contracts or silently downgrade a strict product policy. ++ #[cfg(not(target_arch = "wasm32"))] ++ pub fn set_code_memory_policy(&mut self, policy: CodeMemoryPolicy) -> Result<(), String> { ++ let mut inner = self.inner_mut(); ++ if !inner.code_memory.is_empty() { ++ return Err( ++ "code-memory policy must be configured before the first artifact allocation" ++ .to_string(), ++ ); ++ } ++ inner.code_memory_policy = policy; ++ Ok(()) ++ } ++ ++ /// Return the executable-memory ownership policy selected for this engine. ++ #[cfg(not(target_arch = "wasm32"))] ++ pub fn code_memory_policy(&self) -> CodeMemoryPolicy { ++ self.inner().code_memory_policy.clone() ++ } ++ ++ /// Return the number of executable-code allocations currently owned by ++ /// this engine. ++ /// ++ /// Sealed runtimes use this read-only count to prove that an abandoned ++ /// pending activation released its published allocation. ++ #[doc(hidden)] ++ #[cfg(not(target_arch = "wasm32"))] ++ pub fn code_memory_allocation_count(&self) -> usize { ++ self.inner().code_memory.len() ++ } ++ ++ #[cfg(not(target_arch = "wasm32"))] ++ pub(crate) fn rollback_code_memory_activation(&self, id: CodeMemoryId) -> bool { ++ self.inner_mut().remove_code_memory(id) ++ } ++ + /// Get a reference to attached Tunable of this engine + #[cfg(not(target_arch = "wasm32"))] + pub fn tunables(&self) -> &dyn Tunables { +@@ -347,12 +392,49 @@ pub struct EngineInner { + /// functions to memory. + #[cfg(not(target_arch = "wasm32"))] + code_memory: Vec, ++ /// Policy copied into every code-memory allocation owned by this engine. ++ #[cfg(not(target_arch = "wasm32"))] ++ code_memory_policy: CodeMemoryPolicy, + /// The signature registry is used mainly to operate with trampolines + /// performantly. + #[cfg(not(target_arch = "wasm32"))] + signatures: SignatureRegistry, + } + ++/// Rollback owner for all code-memory allocations made by one artifact build. ++/// ++/// `Artifact::from_parts` holds the engine mutex for the complete transaction, ++/// so no unrelated allocation can be appended between this checkpoint and ++/// commit. Dropping this value after any error or unwind deregisters metadata ++/// and unmaps every uncommitted allocation through `CodeMemory`'s normal drop ++/// order. ++#[cfg(not(target_arch = "wasm32"))] ++pub(crate) struct CodeMemoryTransaction<'a> { ++ engine_inner: &'a mut EngineInner, ++ checkpoint: usize, ++ committed: bool, ++} ++ ++#[cfg(not(target_arch = "wasm32"))] ++impl CodeMemoryTransaction<'_> { ++ pub(crate) fn inner_mut(&mut self) -> &mut EngineInner { ++ self.engine_inner ++ } ++ ++ pub(crate) fn commit(mut self) { ++ self.committed = true; ++ } ++} ++ ++#[cfg(not(target_arch = "wasm32"))] ++impl Drop for CodeMemoryTransaction<'_> { ++ fn drop(&mut self) { ++ if !self.committed { ++ self.engine_inner.code_memory.truncate(self.checkpoint); ++ } ++ } ++} ++ + impl std::fmt::Debug for EngineInner { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + let mut formatter = f.debug_struct("EngineInner"); +@@ -372,6 +454,30 @@ impl std::fmt::Debug for EngineInner { + } + + impl EngineInner { ++ #[cfg(not(target_arch = "wasm32"))] ++ pub(crate) fn begin_code_memory_transaction(&mut self) -> CodeMemoryTransaction<'_> { ++ let checkpoint = self.code_memory.len(); ++ CodeMemoryTransaction { ++ engine_inner: self, ++ checkpoint, ++ committed: false, ++ } ++ } ++ ++ #[cfg(not(target_arch = "wasm32"))] ++ pub(crate) fn last_code_memory_id(&self) -> Option { ++ self.code_memory.last().map(CodeMemory::id) ++ } ++ ++ #[cfg(not(target_arch = "wasm32"))] ++ fn remove_code_memory(&mut self, id: CodeMemoryId) -> bool { ++ let Some(index) = self.code_memory.iter().position(|memory| memory.id() == id) else { ++ return false; ++ }; ++ self.code_memory.remove(index); ++ true ++ } ++ + /// Gets the compiler associated to this engine. + #[cfg(feature = "compiler")] + pub fn compiler(&self) -> Result<&dyn Compiler, CompileError> { +@@ -429,7 +535,8 @@ impl EngineInner { + let (executable_sections, data_sections): (Vec<_>, _) = custom_sections + .clone() + .partition(|section| section.protection() == CustomSectionProtection::ReadExecute); +- self.code_memory.push(CodeMemory::new()); ++ self.code_memory ++ .push(CodeMemory::with_policy(self.code_memory_policy.clone())); + + let (mut allocated_functions, allocated_executable_sections, allocated_data_sections) = + self.code_memory +@@ -495,8 +602,22 @@ impl EngineInner { + + #[cfg(not(target_arch = "wasm32"))] + /// Make memory containing compiled code executable. +- pub(crate) fn publish_compiled_code(&mut self) { +- self.code_memory.last_mut().unwrap().publish(); ++ pub(crate) fn publish_compiled_code(&mut self) -> Result<(), CompileError> { ++ let result = self ++ .code_memory ++ .last_mut() ++ .expect("code memory must be allocated before publication") ++ .publish(); ++ if let Err(message) = result { ++ // No Artifact escapes on a publication failure. Remove the failed ++ // allocation immediately so a later deserialization cannot retain ++ // an unpublished shared mapping or writable file descriptor. ++ self.code_memory.pop(); ++ return Err(CompileError::Resource(format!( ++ "failed to publish executable code memory: {message}" ++ ))); ++ } ++ Ok(()) + } + + #[cfg(not(target_arch = "wasm32"))] +@@ -570,16 +691,31 @@ impl EngineInner { + })?; + let mut file = std::io::BufWriter::new(file); + ++ let module_label = module_info ++ .name ++ .as_deref() ++ .map(str::to_owned) ++ .or_else(|| { ++ module_info ++ .hash() ++ .map(|hash| format!("module_{}", hash.short_hash())) ++ }) ++ .unwrap_or_else(|| "module".to_string()); ++ let module_label = module_label.replace(char::is_whitespace, "_"); ++ + for (func_index, code) in finished_functions.iter() { + let func_index = module_info.func_index(func_index); +- if let Some(func_name) = module_info.function_names.get(&func_index) { +- let sanitized_name = func_name.replace(['\n', '\r'], "_"); +- let line = format!( +- "{:p} {:x} {sanitized_name}\n", +- code.ptr.0 as *const _, code.length +- ); +- write!(file, "{line}").map_err(|e| CompileError::Codegen(e.to_string()))?; +- } ++ let func_name = module_info ++ .function_names ++ .get(&func_index) ++ .cloned() ++ .unwrap_or_else(|| format!("{module_label}::function_{}", func_index.as_u32())); ++ let sanitized_name = func_name.replace(char::is_whitespace, "_"); ++ let line = format!( ++ "{:p} {:x} {sanitized_name}\n", ++ code.ptr.0 as *const _, code.length ++ ); ++ write!(file, "{line}").map_err(|e| CompileError::Codegen(e.to_string()))?; + } + + file.flush() +@@ -637,3 +773,65 @@ impl Default for EngineId { + } + } + } ++ ++#[cfg(all(test, not(target_arch = "wasm32")))] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn code_memory_policy_is_explicitly_selectable_only_before_allocation() { ++ let mut engine = Engine::headless(); ++ engine ++ .set_code_memory_policy(CodeMemoryPolicy::anonymous()) ++ .unwrap(); ++ engine.inner_mut().code_memory.push(CodeMemory::new()); ++ ++ let error = engine ++ .set_code_memory_policy(CodeMemoryPolicy::anonymous()) ++ .unwrap_err(); ++ assert!(error.contains("before the first artifact allocation")); ++ } ++ ++ #[test] ++ fn code_memory_transaction_rolls_back_uncommitted_allocations_and_keeps_committed_ones() { ++ let engine = Engine::headless(); ++ let mut inner = engine.inner_mut(); ++ ++ { ++ let mut transaction = inner.begin_code_memory_transaction(); ++ transaction.inner_mut().code_memory.push(CodeMemory::new()); ++ } ++ assert!(inner.code_memory.is_empty()); ++ ++ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ let mut transaction = inner.begin_code_memory_transaction(); ++ transaction.inner_mut().code_memory.push(CodeMemory::new()); ++ panic!("injected artifact-build panic"); ++ })); ++ assert!(panic.is_err()); ++ assert!(inner.code_memory.is_empty()); ++ ++ { ++ let mut transaction = inner.begin_code_memory_transaction(); ++ transaction.inner_mut().code_memory.push(CodeMemory::new()); ++ transaction.commit(); ++ } ++ assert_eq!(inner.code_memory.len(), 1); ++ } ++ ++ #[test] ++ fn pending_activation_rollback_removes_the_exact_allocation_out_of_order() { ++ let engine = Engine::headless(); ++ let (first, second) = { ++ let mut inner = engine.inner_mut(); ++ inner.code_memory.push(CodeMemory::new()); ++ inner.code_memory.push(CodeMemory::new()); ++ (inner.code_memory[0].id(), inner.code_memory[1].id()) ++ }; ++ ++ assert!(engine.rollback_code_memory_activation(first)); ++ let inner = engine.inner(); ++ assert_eq!(inner.code_memory.len(), 1); ++ assert_eq!(inner.code_memory[0].id(), second); ++ } ++} +diff --git a/lib/compiler/src/engine/mod.rs b/lib/compiler/src/engine/mod.rs +index ab19cc7..511d032 100644 +--- a/lib/compiler/src/engine/mod.rs ++++ b/lib/compiler/src/engine/mod.rs +@@ -28,10 +28,14 @@ pub use self::trap::*; + pub use self::tunables::{BaseTunables, Tunables}; + + #[cfg(not(target_arch = "wasm32"))] +-pub use self::artifact::Artifact; ++pub use self::artifact::{Artifact, PendingArtifact}; + pub use self::builder::EngineBuilder; + #[cfg(not(target_arch = "wasm32"))] +-pub use self::code_memory::CodeMemory; ++pub(crate) use self::code_memory::CodeMemoryId; ++#[cfg(not(target_arch = "wasm32"))] ++pub use self::code_memory::{ ++ CodeMemory, CodeMemoryPolicy, STRICT_LINUX_X86_64_CODE_MEMORY_POLICY_ID, ++}; + pub use self::inner::{Engine, EngineInner}; + #[cfg(not(target_arch = "wasm32"))] + pub use self::link::link_module; +diff --git a/lib/compiler/src/engine/trap/frame_info.rs b/lib/compiler/src/engine/trap/frame_info.rs +index abee089..9d51660 100644 +--- a/lib/compiler/src/engine/trap/frame_info.rs ++++ b/lib/compiler/src/engine/trap/frame_info.rs +@@ -21,7 +21,7 @@ use crate::types::function::{ArchivedCompiledFunctionFrameInfo, CompiledFunction + use rkyv::vec::ArchivedVec; + use std::collections::BTreeMap; + use std::sync::{Arc, LazyLock, RwLock}; +-use wasmer_types::lib::std::{cmp, ops::Deref}; ++use wasmer_types::lib::std::ops::Deref; + use wasmer_types::{ + FrameInfo, LocalFunctionIndex, ModuleInfo, SourceLoc, TrapInformation, + entity::{BoxedSlice, EntityRef, PrimaryMap}, +@@ -61,7 +61,13 @@ pub struct GlobalFrameInfoRegistration { + #[derive(Debug)] + struct ModuleInfoFrameInfo { + start: usize, +- functions: BTreeMap, ++ /// Function address ranges sorted by their start address. ++ /// ++ /// This is a finished immutable index: a binary search is sufficient for ++ /// trap-time lookup, and contiguous storage avoids one heap node per ++ /// compiled function. Registration validates the ranges before publishing ++ /// the module to the global registry. ++ functions: Box<[FunctionInfo]>, + module: Arc, + frame_infos: FrameInfosVariant, + } +@@ -76,18 +82,16 @@ impl ModuleInfoFrameInfo { + + /// Gets a function given a pc + fn function_info(&self, pc: usize) -> Option<&FunctionInfo> { +- let (end, func) = self.functions.range(pc..).next()?; +- if func.start <= pc && pc <= *end { +- Some(func) +- } else { +- None +- } ++ let insertion = self.functions.partition_point(|func| func.start <= pc); ++ let func = self.functions.get(insertion.checked_sub(1)?)?; ++ (pc <= func.end).then_some(func) + } + } + + #[derive(Debug)] + struct FunctionInfo { + start: usize, ++ end: usize, + local_index: LocalFunctionIndex, + } + +@@ -363,9 +367,7 @@ pub fn register( + finished_functions: &BoxedSlice, + frame_infos: FrameInfosVariant, + ) -> Option { +- let mut min = usize::MAX; +- let mut max = 0; +- let mut functions = BTreeMap::new(); ++ let mut functions = Vec::with_capacity(finished_functions.len()); + for ( + i, + FunctionExtent { +@@ -376,19 +378,37 @@ pub fn register( + { + let start = **start as usize; + // end is "last byte" of the function code +- let end = start + len - 1; +- min = cmp::min(min, start); +- max = cmp::max(max, end); +- let func = FunctionInfo { ++ let end = start ++ .checked_add( ++ len.checked_sub(1) ++ .expect("compiled function extent must not be empty"), ++ ) ++ .expect("compiled function extent address overflow"); ++ functions.push(FunctionInfo { + start, ++ end, + local_index: i, +- }; +- assert!(functions.insert(end, func).is_none()); ++ }); + } + if functions.is_empty() { + return None; + } + ++ functions.sort_unstable_by_key(|func| func.start); ++ for adjacent in functions.windows(2) { ++ assert!( ++ adjacent[0].end < adjacent[1].start, ++ "compiled function extents must be disjoint" ++ ); ++ } ++ let min = functions[0].start; ++ let max = functions ++ .iter() ++ .map(|func| func.end) ++ .max() ++ .expect("nonempty function range index"); ++ let functions = functions.into_boxed_slice(); ++ + let mut info = FRAME_INFO.write().unwrap(); + // First up assert that our chunk of jit functions doesn't collide with + // any other known chunks of jit functions... +@@ -412,3 +432,83 @@ pub fn register( + assert!(prev.is_none()); + Some(GlobalFrameInfoRegistration { key: max }) + } ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use wasmer_vm::VMFunctionBody; ++ ++ fn module_with_ranges(ranges: &[(usize, usize)]) -> ModuleInfoFrameInfo { ++ ModuleInfoFrameInfo { ++ start: ranges.first().map_or(0, |range| range.0), ++ functions: ranges ++ .iter() ++ .enumerate() ++ .map(|(index, &(start, end))| FunctionInfo { ++ start, ++ end, ++ local_index: LocalFunctionIndex::new(index), ++ }) ++ .collect::>() ++ .into_boxed_slice(), ++ module: Arc::new(ModuleInfo::default()), ++ frame_infos: FrameInfosVariant::Owned(PrimaryMap::new()), ++ } ++ } ++ ++ #[test] ++ fn packed_function_ranges_preserve_boundaries_and_gaps() { ++ let module = module_with_ranges(&[(100, 109), (200, 219)]); ++ ++ assert!(module.function_info(99).is_none()); ++ assert_eq!( ++ module.function_info(100).unwrap().local_index, ++ LocalFunctionIndex::new(0) ++ ); ++ assert_eq!( ++ module.function_info(109).unwrap().local_index, ++ LocalFunctionIndex::new(0) ++ ); ++ assert!(module.function_info(110).is_none()); ++ assert!(module.function_info(199).is_none()); ++ assert_eq!( ++ module.function_info(200).unwrap().local_index, ++ LocalFunctionIndex::new(1) ++ ); ++ assert_eq!( ++ module.function_info(219).unwrap().local_index, ++ LocalFunctionIndex::new(1) ++ ); ++ assert!(module.function_info(220).is_none()); ++ } ++ ++ fn extent(start: usize, length: usize) -> FunctionExtent { ++ FunctionExtent { ++ ptr: FunctionBodyPtr(start as *const VMFunctionBody), ++ length, ++ } ++ } ++ ++ #[test] ++ #[should_panic(expected = "compiled function extent must not be empty")] ++ fn registration_rejects_empty_function_extents() { ++ let functions = PrimaryMap::from_iter([extent(0x1000, 0)]).into_boxed_slice(); ++ let _ = register( ++ Arc::new(ModuleInfo::default()), ++ &functions, ++ FrameInfosVariant::Owned(PrimaryMap::new()), ++ ); ++ } ++ ++ #[test] ++ #[should_panic(expected = "compiled function extents must be disjoint")] ++ fn registration_rejects_overlapping_function_extents() { ++ let functions = ++ PrimaryMap::from_iter([extent(0x1000, 32), extent(0x1010, 32)]).into_boxed_slice(); ++ let _ = register( ++ Arc::new(ModuleInfo::default()), ++ &functions, ++ FrameInfosVariant::Owned(PrimaryMap::new()), ++ ); ++ } ++} +diff --git a/lib/types/src/libcalls.rs b/lib/types/src/libcalls.rs +index 7aaada0..f38de45 100644 +--- a/lib/types/src/libcalls.rs ++++ b/lib/types/src/libcalls.rs +@@ -109,6 +109,18 @@ pub enum LibCall { + /// when the `enable_probestack` setting is true. + Probestack, + ++ /// Host compiler-emitted bzero. ++ HostBzero, ++ ++ /// Host compiler-emitted memset. ++ HostMemset, ++ ++ /// Host compiler-emitted memcpy. ++ HostMemcpy, ++ ++ /// Host compiler-emitted memmove. ++ HostMemmove, ++ + /// memory.atomic.wait32 for local memories + Memory32AtomicWait32, + +@@ -188,6 +200,10 @@ impl LibCall { + Self::Probestack => "_wasmer_vm_probestack", + #[cfg(not(target_vendor = "apple"))] + Self::Probestack => "wasmer_vm_probestack", ++ Self::HostBzero => "wasmer_vm_host_bzero", ++ Self::HostMemset => "wasmer_vm_host_memset", ++ Self::HostMemcpy => "wasmer_vm_host_memcpy", ++ Self::HostMemmove => "wasmer_vm_host_memmove", + Self::Memory32AtomicWait32 => "wasmer_vm_memory32_atomic_wait32", + Self::ImportedMemory32AtomicWait32 => "wasmer_vm_imported_memory32_atomic_wait32", + Self::Memory32AtomicWait64 => "wasmer_vm_memory32_atomic_wait64", +diff --git a/lib/types/src/serialize.rs b/lib/types/src/serialize.rs +index ab6c8c9..e5124e3 100644 +--- a/lib/types/src/serialize.rs ++++ b/lib/types/src/serialize.rs +@@ -13,7 +13,7 @@ pub struct MetadataHeader { + impl MetadataHeader { + /// Current ABI version. Increment this any time breaking changes are made + /// to the format of the serialized data. +- pub const CURRENT_VERSION: u32 = 20; ++ pub const CURRENT_VERSION: u32 = 21; + + /// Magic number to identify wasmer metadata. + const MAGIC: [u8; 8] = *b"WASMER\0\0"; +diff --git a/lib/types/src/vmoffsets.rs b/lib/types/src/vmoffsets.rs +index 297118e..9c370ac 100644 +--- a/lib/types/src/vmoffsets.rs ++++ b/lib/types/src/vmoffsets.rs +@@ -544,9 +544,14 @@ impl VMOffsets { + 4 + } + ++ /// The offset of the `anyfuncs` field. ++ pub const fn vmtable_definition_anyfuncs(&self) -> u8 { ++ 2 * self.pointer_size ++ } ++ + /// Return the size of `VMTableDefinition`. + pub const fn size_of_vmtable_definition(&self) -> u8 { +- 2 * self.pointer_size ++ 3 * self.pointer_size + } + } + +@@ -814,6 +819,12 @@ impl VMOffsets { + self.vmctx_vmtable_definition(index) + u32::from(self.vmtable_definition_current_elements()) + } + ++ /// Return the offset to the `anyfuncs` field in `VMTableDefinition` index `index`. ++ /// Remember updating precompute upon changes ++ pub fn vmctx_vmtable_definition_anyfuncs(&self, index: LocalTableIndex) -> u32 { ++ self.vmctx_vmtable_definition(index) + u32::from(self.vmtable_definition_anyfuncs()) ++ } ++ + /// Return the offset to the inline `VMCallerCheckedAnyfunc` array for a local fixed + /// `funcref` table. + pub fn vmctx_fixed_funcref_table_anyfuncs(&self, index: LocalTableIndex) -> Option { +diff --git a/lib/vm/src/export.rs b/lib/vm/src/export.rs +index 5d47e42..6cf092b 100644 +--- a/lib/vm/src/export.rs ++++ b/lib/vm/src/export.rs +@@ -11,6 +11,7 @@ use std::any::Any; + use wasmer_types::{FunctionType, TagKind}; + + /// The value of an export passed from one instance to another. ++#[derive(Clone, Copy, PartialEq, Eq)] + #[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] + pub enum VMExtern { + /// A function export value. +diff --git a/lib/vm/src/instance/allocator.rs b/lib/vm/src/instance/allocator.rs +index c81bfd1..f040a4c 100644 +--- a/lib/vm/src/instance/allocator.rs ++++ b/lib/vm/src/instance/allocator.rs +@@ -78,6 +78,42 @@ impl InstanceAllocator { + Vec>, + ) { + let offsets = VMOffsets::new(mem::size_of::() as u8, module); ++ // SAFETY: the offsets were derived immediately above from this exact ++ // module and the host pointer width. ++ unsafe { Self::new_with_offsets(offsets, module) } ++ } ++ ++ /// Allocates instance data using precomputed host-layout offsets. ++ /// ++ /// This is the cached-offset counterpart of [`InstanceAllocator::new`]. ++ /// It avoids rescanning an immutable module when many instances are ++ /// created from the same compiled artifact. ++ /// ++ /// # Safety ++ /// ++ /// `offsets` must equal ++ /// `VMOffsets::new(size_of::() as u8, module)` for this exact ++ /// `module`. Supplying offsets from another module or pointer width can ++ /// under-allocate the dynamically sized VMContext and lead to invalid ++ /// pointer writes while the instance is initialized. ++ #[allow(clippy::type_complexity)] ++ pub unsafe fn new_with_offsets( ++ offsets: VMOffsets, ++ module: &ModuleInfo, ++ ) -> ( ++ Self, ++ Vec>, ++ Vec>, ++ Vec>, ++ ) { ++ debug_assert_eq!( ++ format!("{offsets:?}"), ++ format!( ++ "{:?}", ++ VMOffsets::new(mem::size_of::() as u8, module) ++ ), ++ "precomputed VMOffsets do not match the supplied module" ++ ); + let instance_layout = Self::instance_layout(&offsets); + + #[allow(clippy::cast_ptr_alignment)] +@@ -257,3 +293,34 @@ impl InstanceAllocator { + &self.offsets + } + } ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn vmoffsets_are_deterministic_for_a_module() { ++ let module = ModuleInfo::default(); ++ let pointer_size = mem::size_of::() as u8; ++ ++ assert_eq!( ++ format!("{:?}", VMOffsets::new(pointer_size, &module)), ++ format!("{:?}", VMOffsets::new(pointer_size, &module)) ++ ); ++ } ++ ++ #[test] ++ fn cached_offsets_produce_the_same_allocator_layout() { ++ let module = ModuleInfo::default(); ++ let (computed, _, _, _) = InstanceAllocator::new(&module); ++ let offsets = VMOffsets::new(mem::size_of::() as u8, &module); ++ // SAFETY: `offsets` was computed from this exact module and host. ++ let (cached, _, _, _) = unsafe { InstanceAllocator::new_with_offsets(offsets, &module) }; ++ ++ assert_eq!(computed.instance_layout, cached.instance_layout); ++ assert_eq!( ++ computed.offsets.size_of_vmctx(), ++ cached.offsets.size_of_vmctx() ++ ); ++ } ++} +diff --git a/lib/vm/src/instance/mod.rs b/lib/vm/src/instance/mod.rs +index 3741bbb..7a16b05 100644 +--- a/lib/vm/src/instance/mod.rs ++++ b/lib/vm/src/instance/mod.rs +@@ -74,18 +74,32 @@ pub(crate) struct Instance { + tags: BoxedSlice>, + + /// Pointers to functions in executable memory. +- functions: BoxedSlice, ++ /// ++ /// These pointers are immutable after the artifact is linked, so every ++ /// instance of the same module shares the typed pointer table. ++ functions: Arc>, + + /// Pointers to function call trampolines in executable memory. +- function_call_trampolines: BoxedSlice, ++ function_call_trampolines: Arc>, + + /// Passive elements in this instantiation. As `elem.drop`s happen, these + /// entries get removed. + passive_elements: RefCell]>>>, + +- /// Passive data segments from our module. As `data.drop`s happen, entries +- /// get removed. A missing entry is considered equivalent to an empty slice. +- passive_data: RefCell>>, ++ /// Passive data segments dropped by this instance. ++ /// ++ /// The immutable segment bytes remain owned by the shared `ModuleInfo`; ++ /// instances only need to record the segments whose logical length has ++ /// become zero after `data.drop`. ++ dropped_passive_data: RefCell>, ++ ++ /// Canonical extern handles for deferred export materialization. ++ /// ++ /// This is deliberately disabled for ordinary eager instances: populating ++ /// both their public export map and a second full cache would only increase ++ /// memory use. Deferred instances opt in, and the cache is keyed by the ++ /// underlying declaration so aliases share one local `VMFunction` handle. ++ deferred_export_cache: Option>, + + /// Mapping of function indices to their func ref backing data. `VMFuncRef`s + /// will point to elements here for functions defined by this instance. +@@ -240,7 +254,7 @@ impl Instance { + return; + }; + unsafe { +- *base.as_ptr().add(index as usize) = anyfunc_from_funcref(funcref); ++ *base.as_ptr().add(index as usize) = VMCallerCheckedAnyfunc::from_funcref(funcref); + } + } + +@@ -254,7 +268,7 @@ impl Instance { + unreachable!("fixed funcref tables cannot contain externrefs"); + }; + unsafe { +- *base.as_ptr().add(index as usize) = anyfunc_from_funcref(funcref); ++ *base.as_ptr().add(index as usize) = VMCallerCheckedAnyfunc::from_funcref(funcref); + } + } + } +@@ -859,8 +873,19 @@ impl Instance { + // https://webassembly.github.io/bulk-memory-operations/core/exec/instructions.html#exec-memory-init + + let memory = self.get_vmmemory(memory_index); +- let passive_data = self.passive_data.borrow(); +- let data = passive_data.get(&data_index).map_or(&[][..], |d| &**d); ++ let was_dropped = self ++ .dropped_passive_data ++ .borrow() ++ .binary_search(&data_index) ++ .is_ok(); ++ let data = if was_dropped { ++ &[][..] ++ } else { ++ self.module ++ .passive_data ++ .get(&data_index) ++ .map_or(&[][..], |data| &**data) ++ }; + + let current_length = unsafe { memory.vmmemory().as_ref().current_length }; + if src.checked_add(len).is_none_or(|n| n as usize > data.len()) +@@ -876,8 +901,12 @@ impl Instance { + + /// Drop the given data segment, truncating its length to zero. + pub(crate) fn data_drop(&self, data_index: DataIndex) { +- let mut passive_data = self.passive_data.borrow_mut(); +- passive_data.remove(&data_index); ++ if self.module.passive_data.contains_key(&data_index) { ++ let mut dropped = self.dropped_passive_data.borrow_mut(); ++ if let Err(position) = dropped.binary_search(&data_index) { ++ dropped.insert(position, data_index); ++ } ++ } + } + + /// Get a table by index regardless of whether it is locally-defined or an +@@ -1140,8 +1169,8 @@ impl VMInstance { + allocator: InstanceAllocator, + module: Arc, + context: &mut StoreObjects, +- finished_functions: BoxedSlice, +- finished_function_call_trampolines: BoxedSlice, ++ finished_functions: Arc>, ++ finished_function_call_trampolines: Arc>, + finished_memories: BoxedSlice>, + finished_tables: BoxedSlice>, + finished_globals: BoxedSlice>, +@@ -1155,15 +1184,6 @@ impl VMInstance { + .map(|m: &InternalStoreHandle| VMSharedTagIndex::new(m.index() as u32)) + .collect::>() + .into_boxed_slice(); +- let passive_data = RefCell::new( +- module +- .passive_data +- .clone() +- .into_iter() +- .map(|(idx, bytes)| (idx, Arc::from(bytes))) +- .collect::>(), +- ); +- + let handle = { + let offsets = allocator.offsets().clone(); + // use dummy value to create an instance so we can get the vmctx pointer +@@ -1181,7 +1201,8 @@ impl VMInstance { + functions: finished_functions, + function_call_trampolines: finished_function_call_trampolines, + passive_elements: Default::default(), +- passive_data, ++ dropped_passive_data: Default::default(), ++ deferred_export_cache: None, + funcrefs, + imported_funcrefs, + vmctx: VMContext {}, +@@ -1243,7 +1264,6 @@ impl VMInstance { + instance.builtin_functions_ptr(), + VMBuiltinFunctionsArray::initialized(), + ); +- + // Perform infallible initialization in this constructor, while fallible + // initialization is deferred to the `initialize` method. + initialize_passive_elements(instance); +@@ -1322,9 +1342,19 @@ impl VMInstance { + + /// Lookup an export with the given export declaration. + pub fn lookup_by_declaration(&mut self, export: ExportIndex) -> VMExtern { ++ if let Some(cached) = self ++ .instance() ++ .deferred_export_cache ++ .as_ref() ++ .and_then(|cache| cache.get(&export)) ++ .copied() ++ { ++ return cached; ++ } ++ + let instance = self.instance(); + +- match export { ++ let extern_ = match export { + ExportIndex::Function(index) => { + let sig_index = &instance.module.functions[index]; + let handle = if let Some(def_index) = instance.module.local_func_index(index) { +@@ -1383,6 +1413,22 @@ impl VMInstance { + let handle = instance.tags[index]; + VMExtern::Tag(handle) + } ++ }; ++ ++ if let Some(cache) = self.instance_mut().deferred_export_cache.as_mut() { ++ cache.insert(export, extern_); ++ } ++ ++ extern_ ++ } ++ ++ /// Enables identity-preserving deferred export lookup for this instance. ++ /// ++ /// Ordinary eager instances intentionally leave this disabled so they do ++ /// not retain a second copy of their complete export set. ++ pub fn enable_deferred_export_cache(&mut self) { ++ if self.instance().deferred_export_cache.is_none() { ++ self.instance_mut().deferred_export_cache = Some(HashMap::new()); + } + } + +@@ -1714,13 +1760,6 @@ fn initialize_globals(instance: &Instance) { + } + } + +-fn anyfunc_from_funcref(funcref: Option) -> VMCallerCheckedAnyfunc { +- match funcref { +- Some(funcref) => unsafe { *funcref.0.as_ptr() }, +- None => VMCallerCheckedAnyfunc::null(), +- } +-} +- + /// Eagerly builds all the `VMFuncRef`s for imported and local functions so that all + /// future funcref operations are just looking up this data. + fn build_funcrefs( +@@ -1763,3 +1802,161 @@ fn build_funcrefs( + imported_func_refs.into_boxed_slice(), + ) + } ++ ++#[cfg(test)] ++mod tests { ++ use super::{InstanceAllocator, VMInstance}; ++ use crate::{ ++ FunctionBodyPtr, Imports, StoreObjects, VMContext, VMFunctionBody, VMSignatureHash, ++ VMTrampoline, ++ }; ++ use std::{ptr::NonNull, sync::Arc}; ++ use wasmer_types::{ ++ FunctionType, LocalFunctionIndex, ModuleInfo, RawValue, SignatureIndex, TagIndex, ++ entity::{BoxedSlice, EntityRef, PrimaryMap}, ++ }; ++ ++ unsafe extern "C" fn test_trampoline( ++ _vmctx: *mut VMContext, ++ _callee: *const VMFunctionBody, ++ _values: *mut RawValue, ++ ) { ++ } ++ ++ fn empty_boxed_slice() -> BoxedSlice ++ where ++ K: wasmer_types::entity::EntityRef, ++ { ++ PrimaryMap::::new().into_boxed_slice() ++ } ++ ++ unsafe fn new_test_instance( ++ module: Arc, ++ context: &mut StoreObjects, ++ functions: Arc>, ++ trampolines: Arc>, ++ ) -> VMInstance { ++ let (allocator, memories, tables, globals) = InstanceAllocator::new(&module); ++ assert!(memories.is_empty()); ++ assert!(tables.is_empty()); ++ assert!(globals.is_empty()); ++ ++ unsafe { ++ VMInstance::new( ++ allocator, ++ module, ++ context, ++ functions, ++ trampolines, ++ empty_boxed_slice(), ++ empty_boxed_slice(), ++ empty_boxed_slice(), ++ empty_boxed_slice::(), ++ Imports::none(), ++ PrimaryMap::from_iter([VMSignatureHash::new(7)]).into_boxed_slice(), ++ ) ++ .unwrap() ++ } ++ } ++ ++ fn test_module_and_function_tables() -> ( ++ Arc, ++ FunctionBodyPtr, ++ Arc>, ++ Arc>, ++ ) { ++ let mut module = ModuleInfo::new(); ++ let signature = module.signatures.push(FunctionType::new([], [])); ++ module.functions.push(signature); ++ let module = Arc::new(module); ++ ++ let function = FunctionBodyPtr(NonNull::::dangling().as_ptr()); ++ let functions = Arc::new(PrimaryMap::from_iter([function]).into_boxed_slice()); ++ let trampolines = ++ Arc::new(PrimaryMap::from_iter([test_trampoline as VMTrampoline]).into_boxed_slice()); ++ ++ (module, function, functions, trampolines) ++ } ++ ++ #[test] ++ fn immutable_function_tables_are_shared_by_two_instances() { ++ let (module, _function, functions, trampolines) = test_module_and_function_tables(); ++ ++ let mut context = StoreObjects::default(); ++ let first = unsafe { ++ new_test_instance( ++ Arc::clone(&module), ++ &mut context, ++ Arc::clone(&functions), ++ Arc::clone(&trampolines), ++ ) ++ }; ++ let second = unsafe { ++ new_test_instance( ++ module, ++ &mut context, ++ Arc::clone(&functions), ++ Arc::clone(&trampolines), ++ ) ++ }; ++ ++ assert!(Arc::ptr_eq( ++ &first.instance().functions, ++ &second.instance().functions ++ )); ++ assert!(Arc::ptr_eq( ++ &first.instance().function_call_trampolines, ++ &second.instance().function_call_trampolines ++ )); ++ assert_eq!(Arc::strong_count(&functions), 3); ++ assert_eq!(Arc::strong_count(&trampolines), 3); ++ } ++ ++ #[test] ++ fn shared_function_tables_outlive_artifact_owner_and_peer_instance() { ++ let (module, function, functions, trampolines) = test_module_and_function_tables(); ++ let function_allocation = Arc::as_ptr(&functions); ++ let trampoline_allocation = Arc::as_ptr(&trampolines); ++ ++ let mut context = StoreObjects::default(); ++ let first = unsafe { ++ new_test_instance( ++ Arc::clone(&module), ++ &mut context, ++ Arc::clone(&functions), ++ Arc::clone(&trampolines), ++ ) ++ }; ++ let second = unsafe { ++ new_test_instance( ++ module, ++ &mut context, ++ Arc::clone(&functions), ++ Arc::clone(&trampolines), ++ ) ++ }; ++ ++ // Model dropping the artifact/module's owners and then one instance. ++ drop(functions); ++ drop(trampolines); ++ drop(first); ++ ++ assert_eq!( ++ Arc::as_ptr(&second.instance().functions), ++ function_allocation ++ ); ++ assert_eq!( ++ Arc::as_ptr(&second.instance().function_call_trampolines), ++ trampoline_allocation ++ ); ++ assert_eq!(Arc::strong_count(&second.instance().functions), 1); ++ assert_eq!( ++ second.instance().functions[LocalFunctionIndex::new(0)].0, ++ function.0 ++ ); ++ assert!(std::ptr::fn_addr_eq( ++ second.instance().function_call_trampolines[SignatureIndex::new(0)], ++ test_trampoline as VMTrampoline ++ )); ++ } ++} +diff --git a/lib/vm/src/libcalls.rs b/lib/vm/src/libcalls.rs +index 7c1e0db..532838c 100644 +--- a/lib/vm/src/libcalls.rs ++++ b/lib/vm/src/libcalls.rs +@@ -224,6 +224,49 @@ pub unsafe extern "C" fn wasmer_vm_imported_memory32_size( + } + } + ++#[unsafe(no_mangle)] ++pub unsafe extern "C" fn wasmer_vm_host_bzero(dst: *mut c_void, len: usize) { ++ unsafe { ++ std::ptr::write_bytes(dst, 0, len); ++ } ++} ++ ++#[unsafe(no_mangle)] ++pub unsafe extern "C" fn wasmer_vm_host_memset( ++ dst: *mut c_void, ++ value: i32, ++ len: usize, ++) -> *mut c_void { ++ unsafe { ++ std::ptr::write_bytes(dst, value as u8, len); ++ } ++ dst ++} ++ ++#[unsafe(no_mangle)] ++pub unsafe extern "C" fn wasmer_vm_host_memcpy( ++ dst: *mut c_void, ++ src: *const c_void, ++ len: usize, ++) -> *mut c_void { ++ unsafe { ++ std::ptr::copy_nonoverlapping(src.cast::(), dst.cast::(), len); ++ } ++ dst ++} ++ ++#[unsafe(no_mangle)] ++pub unsafe extern "C" fn wasmer_vm_host_memmove( ++ dst: *mut c_void, ++ src: *const c_void, ++ len: usize, ++) -> *mut c_void { ++ unsafe { ++ std::ptr::copy(src.cast::(), dst.cast::(), len); ++ } ++ dst ++} ++ + /// Implementation of `table.copy`. + /// + /// # Safety +@@ -983,6 +1026,10 @@ pub fn function_pointer(libcall: LibCall) -> usize { + LibCall::Memory32Init => wasmer_vm_memory32_init as *const () as usize, + LibCall::DataDrop => wasmer_vm_data_drop as *const () as usize, + LibCall::Probestack => WASMER_VM_PROBESTACK as *const () as usize, ++ LibCall::HostBzero => wasmer_vm_host_bzero as *const () as usize, ++ LibCall::HostMemset => wasmer_vm_host_memset as *const () as usize, ++ LibCall::HostMemcpy => wasmer_vm_host_memcpy as *const () as usize, ++ LibCall::HostMemmove => wasmer_vm_host_memmove as *const () as usize, + LibCall::RaiseTrap => wasmer_vm_raise_trap as *const () as usize, + LibCall::Memory32AtomicWait32 => wasmer_vm_memory32_atomic_wait32 as *const () as usize, + LibCall::ImportedMemory32AtomicWait32 => { +diff --git a/lib/vm/src/memory.rs b/lib/vm/src/memory.rs +index e7bc857..f2e730c 100644 +--- a/lib/vm/src/memory.rs ++++ b/lib/vm/src/memory.rs +@@ -33,9 +33,25 @@ struct WasmMmap { + size: Pages, + /// The owned memory definition used by the generated code + vm_memory_definition: MaybeInstanceOwned, ++ /// True only when the allocation reserves every page the memory can ever ++ /// grow to. Fixed host mappings are invalidated if the linear-memory base ++ /// relocates, so shared remaps must fail closed without this invariant. ++ stable_base: bool, + } + + impl WasmMmap { ++ fn supports_persistent_shared_fixed_remap(&self) -> bool { ++ #[cfg(not(target_os = "windows"))] ++ { ++ self.stable_base ++ } ++ ++ #[cfg(target_os = "windows")] ++ { ++ false ++ } ++ } ++ + fn get_vm_memory_definition(&self) -> NonNull { + self.vm_memory_definition.as_ptr() + } +@@ -152,8 +168,75 @@ impl WasmMmap { + Ok(()) + } + +- /// Copies the memory +- /// (in this case it performs a copy-on-write to save memory) ++ unsafe fn remap_shared_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ if !self.supports_persistent_shared_fixed_remap() { ++ return Err(MemoryError::UnsupportedOperation { ++ message: "shared fixed remaps require a supported nonmoving static linear memory" ++ .to_string(), ++ }); ++ } ++ let current_len = self.size.bytes().0; ++ if len > current_len || start > current_len - len { ++ return Err(MemoryError::InvalidMemory { ++ reason: "fixed mapping range is outside the current memory".to_string(), ++ }); ++ } ++ unsafe { ++ self.alloc ++ .remap_shared_file_fixed(start, len, file, file_offset) ++ } ++ .map_err(MemoryError::Region) ++ } ++ ++ unsafe fn remap_private_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ let current_len = self.size.bytes().0; ++ if len > current_len || start > current_len - len { ++ return Err(MemoryError::InvalidMemory { ++ reason: "fixed mapping range is outside the current memory".to_string(), ++ }); ++ } ++ unsafe { ++ self.alloc ++ .remap_private_file_fixed(start, len, file, file_offset) ++ } ++ .map_err(MemoryError::Region) ++ } ++ ++ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { ++ let current_len = self.size.bytes().0; ++ if len > current_len || start > current_len - len { ++ return Err(MemoryError::InvalidMemory { ++ reason: "fixed mapping range is outside the current memory".to_string(), ++ }); ++ } ++ unsafe { self.alloc.remap_private_fixed(start, len) }.map_err(MemoryError::Region) ++ } ++ ++ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { ++ let current_len = self.size.bytes().0; ++ if len > current_len || start > current_len - len { ++ return Err(MemoryError::InvalidMemory { ++ reason: "sync range is outside the current memory".to_string(), ++ }); ++ } ++ self.alloc ++ .msync(start, len, flags) ++ .map_err(MemoryError::Region) ++ } ++ ++ /// Copies the memory. + pub fn copy(&mut self) -> Result { + let mem_length = self.size.bytes().0; + let mut alloc = self +@@ -170,6 +253,7 @@ impl WasmMmap { + ))), + alloc, + size: self.size, ++ stable_base: self.stable_base, + }) + } + } +@@ -361,6 +445,11 @@ impl VMOwnedMemory { + }, + alloc, + size: Bytes::from(mem_length).try_into().unwrap(), ++ stable_base: matches!( ++ style, ++ MemoryStyle::Static { bound, .. } ++ if *bound >= memory.maximum.unwrap_or(Pages::max_value()) ++ ), + }; + + Ok(Self { +@@ -431,6 +520,44 @@ impl LinearMemory for VMOwnedMemory { + Ok(()) + } + ++ fn supports_persistent_shared_fixed_remap(&self) -> bool { ++ self.mmap.supports_persistent_shared_fixed_remap() ++ } ++ ++ unsafe fn remap_shared_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.mmap ++ .remap_shared_file_fixed(start, len, file, file_offset) ++ } ++ } ++ ++ unsafe fn remap_private_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.mmap ++ .remap_private_file_fixed(start, len, file, file_offset) ++ } ++ } ++ ++ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { ++ unsafe { self.mmap.remap_private_fixed(start, len) } ++ } ++ ++ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { ++ self.mmap.msync(start, len, flags) ++ } ++ + /// Return a `VMMemoryDefinition` for exposing the memory to compiled wasm code. + fn vmmemory(&self) -> NonNull { + self.mmap.vm_memory_definition.as_ptr() +@@ -586,6 +713,43 @@ impl LinearMemory for VMSharedMemory { + Ok(()) + } + ++ fn supports_persistent_shared_fixed_remap(&self) -> bool { ++ let guard = self.mmap.read().unwrap(); ++ guard.supports_persistent_shared_fixed_remap() ++ } ++ ++ unsafe fn remap_shared_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ let mut guard = self.mmap.write().unwrap(); ++ unsafe { guard.remap_shared_file_fixed(start, len, file, file_offset) } ++ } ++ ++ unsafe fn remap_private_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ let mut guard = self.mmap.write().unwrap(); ++ unsafe { guard.remap_private_file_fixed(start, len, file, file_offset) } ++ } ++ ++ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { ++ let mut guard = self.mmap.write().unwrap(); ++ unsafe { guard.remap_private_fixed(start, len) } ++ } ++ ++ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { ++ let guard = self.mmap.read().unwrap(); ++ guard.msync(start, len, flags) ++ } ++ + /// Return a `VMMemoryDefinition` for exposing the memory to compiled wasm code. + fn vmmemory(&self) -> NonNull { + let guard = self.mmap.read().unwrap(); +@@ -680,6 +844,44 @@ impl LinearMemory for VMMemory { + Ok(()) + } + ++ fn supports_persistent_shared_fixed_remap(&self) -> bool { ++ self.0.supports_persistent_shared_fixed_remap() ++ } ++ ++ unsafe fn remap_shared_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.0 ++ .remap_shared_file_fixed(start, len, file, file_offset) ++ } ++ } ++ ++ unsafe fn remap_private_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ unsafe { ++ self.0 ++ .remap_private_file_fixed(start, len, file, file_offset) ++ } ++ } ++ ++ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { ++ unsafe { self.0.remap_private_fixed(start, len) } ++ } ++ ++ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { ++ self.0.msync(start, len, flags) ++ } ++ + /// Returns the memory style for this memory. + fn style(&self) -> MemoryStyle { + self.0.style() +@@ -790,6 +992,43 @@ impl VMMemory { + } + } + ++#[cfg(test)] ++mod shared_remap_style_tests { ++ use super::*; ++ ++ #[test] ++ fn dynamic_memory_rejects_shared_fixed_remaps_before_host_mutation() { ++ let memory = MemoryType::new(Pages(1), Some(Pages(2)), false); ++ let style = MemoryStyle::Dynamic { ++ offset_guard_size: 0, ++ }; ++ let mut vm = VMOwnedMemory::new(&memory, &style).unwrap(); ++ assert!(!LinearMemory::supports_persistent_shared_fixed_remap(&vm)); ++ let file = std::fs::File::open(std::env::current_exe().unwrap()).unwrap(); ++ let error = unsafe { LinearMemory::remap_shared_file_fixed(&mut vm, 0, 4096, &file, 0) } ++ .expect_err("dynamic memory must not accept persistent host remaps"); ++ assert!(matches!(error, MemoryError::UnsupportedOperation { .. })); ++ assert!( ++ error ++ .to_string() ++ .contains("supported nonmoving static linear memory") ++ ); ++ } ++ ++ #[test] ++ #[cfg(not(target_os = "windows"))] ++ fn fully_reserved_static_memory_reports_persistent_remap_capability() { ++ let memory = MemoryType::new(Pages(1), Some(Pages(2)), false); ++ let style = MemoryStyle::Static { ++ bound: Pages(2), ++ offset_guard_size: 0, ++ }; ++ let vm = VMOwnedMemory::new(&memory, &style).unwrap(); ++ ++ assert!(LinearMemory::supports_persistent_shared_fixed_remap(&vm)); ++ } ++} ++ + #[doc(hidden)] + /// Default implementation to initialize memory with data + pub unsafe fn initialize_memory_with_data( +@@ -842,6 +1081,78 @@ where + }) + } + ++ /// Whether fixed shared-file remaps will remain valid for the lifetime of ++ /// this memory. Implementations must return false when later growth can ++ /// relocate the linear-memory base or the host cannot install such maps. ++ fn supports_persistent_shared_fixed_remap(&self) -> bool { ++ false ++ } ++ ++ /// Replace a page-aligned range inside this memory with a shared, ++ /// read-write file mapping at the same host address. ++ /// ++ /// Higher-level runtimes are responsible for translating guest mmap ++ /// requests, tracking mapping metadata, and preserving shared ranges ++ /// across process fork/copy operations. ++ /// ++ /// # Safety ++ /// No thread may access the replaced range during remapping, no Rust reference may point into ++ /// it. The backing inode must not shrink below its size validated at mapping time; only the ++ /// validated final partial page may extend beyond EOF. ++ unsafe fn remap_shared_file_fixed( ++ &mut self, ++ _start: usize, ++ _len: usize, ++ _file: &std::fs::File, ++ _file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ Err(MemoryError::UnsupportedOperation { ++ message: "remap_shared_file_fixed() is not supported".to_string(), ++ }) ++ } ++ ++ /// Replace a page-aligned range inside this memory with a private, ++ /// copy-on-write file mapping at the same host address. ++ /// ++ /// # Safety ++ /// No thread may access the replaced range during remapping, no Rust reference may point into ++ /// it, and the backing inode must not be truncated or replaced below `file_offset + len` while ++ /// accessible. ++ unsafe fn remap_private_file_fixed( ++ &mut self, ++ _start: usize, ++ _len: usize, ++ _file: &std::fs::File, ++ _file_offset: usize, ++ ) -> Result<(), MemoryError> { ++ Err(MemoryError::UnsupportedOperation { ++ message: "remap_private_file_fixed() is not supported".to_string(), ++ }) ++ } ++ ++ /// Replace a page-aligned range inside this memory with private, ++ /// zero-filled memory at the same host address. ++ /// ++ /// # Safety ++ /// No thread may access the replaced range during remapping and no Rust reference may point ++ /// into it. ++ unsafe fn remap_private_fixed( ++ &mut self, ++ _start: usize, ++ _len: usize, ++ ) -> Result<(), MemoryError> { ++ Err(MemoryError::UnsupportedOperation { ++ message: "remap_private_fixed() is not supported".to_string(), ++ }) ++ } ++ ++ /// Synchronize a page-aligned memory range with its backing file. ++ fn msync(&self, _start: usize, _len: usize, _flags: i32) -> Result<(), MemoryError> { ++ Err(MemoryError::UnsupportedOperation { ++ message: "msync() is not supported".to_string(), ++ }) ++ } ++ + /// Return a `VMMemoryDefinition` for exposing the memory to compiled wasm code. + fn vmmemory(&self) -> NonNull; + +diff --git a/lib/vm/src/mmap.rs b/lib/vm/src/mmap.rs +index 75a1852..a634ad2 100644 +--- a/lib/vm/src/mmap.rs ++++ b/lib/vm/src/mmap.rs +@@ -264,6 +264,278 @@ impl Mmap { + .map_err(|e| e.to_string()) + } + ++ /// Replace a page-aligned range inside this mapping with a shared, ++ /// read-write file mapping at the same host address. ++ /// ++ /// # Safety ++ /// ++ /// No thread may concurrently access the linear memory, and no live Rust reference may point ++ /// into the replaced range while this operation runs. The backing inode must not shrink below ++ /// the size validated by this call while the mapping is accessible; only the validated final ++ /// partial page may extend beyond EOF. ++ #[cfg(not(target_os = "windows"))] ++ pub unsafe fn remap_shared_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), String> { ++ use std::os::fd::AsRawFd; ++ ++ let page_size = region::page::size(); ++ if len == 0 { ++ return Err("mapping length must be non-zero".to_string()); ++ } ++ if start & (page_size - 1) != 0 { ++ return Err("mapping start must be page-aligned".to_string()); ++ } ++ if len & (page_size - 1) != 0 { ++ return Err("mapping length must be page-aligned".to_string()); ++ } ++ if file_offset & (page_size - 1) != 0 { ++ return Err("file offset must be page-aligned".to_string()); ++ } ++ if len > self.total_size || start > self.total_size - len { ++ return Err("mapping range is outside the reserved memory".to_string()); ++ } ++ let file_len: usize = file ++ .metadata() ++ .map_err(|err| err.to_string())? ++ .len() ++ .try_into() ++ .map_err(|_| "backing file length does not fit usize".to_string())?; ++ let available_file_bytes = file_len ++ .checked_sub(file_offset) ++ .ok_or_else(|| "file offset is beyond the backing file".to_string())?; ++ let rounded_available_file_bytes = available_file_bytes ++ .checked_add(page_size - 1) ++ .ok_or_else(|| "rounded file mapping range overflowed".to_string())? ++ & !(page_size - 1); ++ if len > rounded_available_file_bytes { ++ return Err("mapping includes a page wholly beyond the backing file".to_string()); ++ } ++ let file_offset = file_offset ++ .try_into() ++ .map_err(|_| "file offset does not fit off_t".to_string())?; ++ ++ let fixed_ptr = unsafe { (self.ptr as *mut u8).add(start) }; ++ let ptr = unsafe { ++ libc::mmap( ++ fixed_ptr.cast(), ++ len, ++ libc::PROT_READ | libc::PROT_WRITE, ++ libc::MAP_SHARED | libc::MAP_FIXED, ++ file.as_raw_fd(), ++ file_offset, ++ ) ++ }; ++ if ptr as isize == -1_isize { ++ return Err(io::Error::last_os_error().to_string()); ++ } ++ if ptr != fixed_ptr.cast() { ++ return Err("fixed mapping returned an unexpected address".to_string()); ++ } ++ ++ self.accessible_size = self.accessible_size.max(start + len); ++ Ok(()) ++ } ++ ++ /// Replace a page-aligned range inside this mapping with a private, ++ /// copy-on-write file mapping at the same host address. ++ /// ++ /// Unlike [`Self::remap_shared_file_fixed`], writes are never propagated ++ /// to the backing file. This is intended for immutable initialization ++ /// images whose clean pages should be shared between independent linear ++ /// memories while preserving normal per-instance write isolation. ++ /// ++ /// # Safety ++ /// ++ /// No thread may concurrently access the linear memory, and no live Rust reference may point ++ /// into the replaced range while this operation runs. The backing inode must not be truncated ++ /// or replaced below `file_offset + len` while the mapping is accessible. ++ #[cfg(not(target_os = "windows"))] ++ pub unsafe fn remap_private_file_fixed( ++ &mut self, ++ start: usize, ++ len: usize, ++ file: &std::fs::File, ++ file_offset: usize, ++ ) -> Result<(), String> { ++ use std::os::fd::AsRawFd; ++ ++ let page_size = region::page::size(); ++ if len == 0 { ++ return Err("mapping length must be non-zero".to_string()); ++ } ++ if start & (page_size - 1) != 0 { ++ return Err("mapping start must be page-aligned".to_string()); ++ } ++ if len & (page_size - 1) != 0 { ++ return Err("mapping length must be page-aligned".to_string()); ++ } ++ if file_offset & (page_size - 1) != 0 { ++ return Err("file offset must be page-aligned".to_string()); ++ } ++ if len > self.total_size || start > self.total_size - len { ++ return Err("mapping range is outside the reserved memory".to_string()); ++ } ++ let file_end = file_offset ++ .checked_add(len) ++ .ok_or_else(|| "file mapping range overflowed".to_string())?; ++ let file_len: usize = file ++ .metadata() ++ .map_err(|err| err.to_string())? ++ .len() ++ .try_into() ++ .map_err(|_| "backing file length does not fit usize".to_string())?; ++ if file_end > file_len { ++ return Err("mapping range extends beyond the backing file".to_string()); ++ } ++ let file_offset = file_offset ++ .try_into() ++ .map_err(|_| "file offset does not fit off_t".to_string())?; ++ ++ let fixed_ptr = unsafe { (self.ptr as *mut u8).add(start) }; ++ let ptr = unsafe { ++ libc::mmap( ++ fixed_ptr.cast(), ++ len, ++ libc::PROT_READ | libc::PROT_WRITE, ++ libc::MAP_PRIVATE | libc::MAP_FIXED, ++ file.as_raw_fd(), ++ file_offset, ++ ) ++ }; ++ if ptr as isize == -1_isize { ++ return Err(io::Error::last_os_error().to_string()); ++ } ++ if ptr != fixed_ptr.cast() { ++ return Err("fixed mapping returned an unexpected address".to_string()); ++ } ++ ++ self.accessible_size = self.accessible_size.max(start + len); ++ Ok(()) ++ } ++ ++ /// Replace a page-aligned range inside this mapping with private, ++ /// zero-filled memory at the same host address. ++ /// ++ /// # Safety ++ /// ++ /// No thread may concurrently access the linear memory, and no live Rust reference may point ++ /// into the replaced range while this operation runs. ++ #[cfg(not(target_os = "windows"))] ++ pub unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), String> { ++ let page_size = region::page::size(); ++ if len == 0 { ++ return Err("mapping length must be non-zero".to_string()); ++ } ++ if start & (page_size - 1) != 0 { ++ return Err("mapping start must be page-aligned".to_string()); ++ } ++ if len & (page_size - 1) != 0 { ++ return Err("mapping length must be page-aligned".to_string()); ++ } ++ if len > self.total_size || start > self.total_size - len { ++ return Err("mapping range is outside the reserved memory".to_string()); ++ } ++ ++ let fixed_ptr = unsafe { (self.ptr as *mut u8).add(start) }; ++ let ptr = unsafe { ++ libc::mmap( ++ fixed_ptr.cast(), ++ len, ++ libc::PROT_READ | libc::PROT_WRITE, ++ libc::MAP_PRIVATE | libc::MAP_ANON | libc::MAP_FIXED, ++ -1, ++ 0, ++ ) ++ }; ++ if ptr as isize == -1_isize { ++ return Err(io::Error::last_os_error().to_string()); ++ } ++ if ptr != fixed_ptr.cast() { ++ return Err("fixed mapping returned an unexpected address".to_string()); ++ } ++ ++ self.accessible_size = self.accessible_size.max(start + len); ++ Ok(()) ++ } ++ ++ /// Synchronize a page-aligned range in this mapping with its backing file. ++ #[cfg(not(target_os = "windows"))] ++ pub fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), String> { ++ let page_size = region::page::size(); ++ if start & (page_size - 1) != 0 { ++ return Err("mapping start must be page-aligned".to_string()); ++ } ++ if len & (page_size - 1) != 0 { ++ return Err("mapping length must be page-aligned".to_string()); ++ } ++ if len > self.total_size || start > self.total_size - len { ++ return Err("mapping range is outside the reserved memory".to_string()); ++ } ++ if len == 0 { ++ return Ok(()); ++ } ++ ++ let ptr = unsafe { (self.ptr as *mut u8).add(start) }; ++ if unsafe { libc::msync(ptr.cast(), len, flags) } != 0 { ++ return Err(io::Error::last_os_error().to_string()); ++ } ++ ++ Ok(()) ++ } ++ ++ /// Replace a page-aligned range inside this mapping with a shared, ++ /// read-write file mapping at the same host address. ++ /// ++ /// # Safety ++ /// See the non-Windows implementation; this stub performs no remap. ++ #[cfg(target_os = "windows")] ++ pub unsafe fn remap_shared_file_fixed( ++ &mut self, ++ _start: usize, ++ _len: usize, ++ _file: &std::fs::File, ++ _file_offset: usize, ++ ) -> Result<(), String> { ++ Err("fixed shared file remapping is not implemented on Windows".to_string()) ++ } ++ ++ /// Replace a page-aligned range inside this mapping with a private, ++ /// copy-on-write file mapping at the same host address. ++ /// ++ /// # Safety ++ /// See the non-Windows implementation; this stub performs no remap. ++ #[cfg(target_os = "windows")] ++ pub unsafe fn remap_private_file_fixed( ++ &mut self, ++ _start: usize, ++ _len: usize, ++ _file: &std::fs::File, ++ _file_offset: usize, ++ ) -> Result<(), String> { ++ Err("fixed private file remapping is not implemented on Windows".to_string()) ++ } ++ ++ /// Replace a page-aligned range inside this mapping with private, ++ /// zero-filled memory at the same host address. ++ /// ++ /// # Safety ++ /// See the non-Windows implementation; this stub performs no remap. ++ #[cfg(target_os = "windows")] ++ pub unsafe fn remap_private_fixed(&mut self, _start: usize, _len: usize) -> Result<(), String> { ++ Err("fixed private memory remapping is not implemented on Windows".to_string()) ++ } ++ ++ /// Synchronize a page-aligned range in this mapping with its backing file. ++ #[cfg(target_os = "windows")] ++ pub fn msync(&self, _start: usize, _len: usize, _flags: i32) -> Result<(), String> { ++ Err("memory synchronization is not implemented on Windows".to_string()) ++ } ++ + /// Make the memory starting at `start` and extending for `len` bytes accessible. + /// `start` and `len` must be native page-size multiples and describe a range within + /// `self`'s reserved memory. +@@ -364,12 +636,31 @@ impl Mmap { + + let mut new = + Self::accessible_reserved(copy_size, self.total_size, None, MmapType::Private)?; +- new.as_mut_slice_arbitary(copy_size) +- .copy_from_slice(self.as_slice_arbitary(copy_size)); ++ let src = self.as_slice_arbitary(copy_size); ++ let dst = new.as_mut_slice_arbitary(copy_size); ++ copy_sparse_range(src, dst, 0, copy_size); + Ok(new) + } + } + ++fn copy_sparse_range(src: &[u8], dst: &mut [u8], start: usize, end: usize) { ++ let page_size = region::page::size(); ++ let mut cursor = start; ++ ++ while cursor < end { ++ let next_page = cursor ++ .checked_add(page_size - (cursor % page_size)) ++ .unwrap_or(end); ++ let next = next_page.min(end); ++ let src_chunk = &src[cursor..next]; ++ ++ if src_chunk.iter().any(|byte| *byte != 0) { ++ dst[cursor..next].copy_from_slice(src_chunk); ++ } ++ cursor = next; ++ } ++} ++ + impl Drop for Mmap { + #[cfg(not(target_os = "windows"))] + fn drop(&mut self) { +@@ -404,3 +695,166 @@ fn _assert() { + fn _assert_send_sync() {} + _assert_send_sync::(); + } ++ ++#[cfg(all(test, not(target_os = "windows")))] ++mod tests { ++ use super::*; ++ use std::io::{Read, Seek, SeekFrom, Write}; ++ use std::time::{SystemTime, UNIX_EPOCH}; ++ ++ fn temp_file_path(name: &str) -> std::path::PathBuf { ++ let unique = SystemTime::now() ++ .duration_since(UNIX_EPOCH) ++ .unwrap() ++ .as_nanos(); ++ std::env::temp_dir().join(format!("wasmer-{name}-{unique}")) ++ } ++ ++ #[test] ++ fn remap_shared_file_fixed_replaces_only_requested_pages() { ++ let page_size = region::page::size(); ++ let path = temp_file_path("mmap-fixed-shared"); ++ let mut file = std::fs::OpenOptions::new() ++ .read(true) ++ .write(true) ++ .create_new(true) ++ .open(&path) ++ .unwrap(); ++ file.set_len(page_size as u64).unwrap(); ++ file.write_all(&vec![0x7a; page_size]).unwrap(); ++ file.sync_all().unwrap(); ++ ++ let mut mmap = ++ Mmap::accessible_reserved(page_size * 2, page_size * 2, None, MmapType::Private) ++ .unwrap(); ++ mmap.as_mut_slice()[0] = 0x11; ++ mmap.as_mut_slice()[page_size] = 0x22; ++ ++ unsafe { mmap.remap_shared_file_fixed(page_size, page_size, &file, 0) }.unwrap(); ++ ++ assert_eq!(mmap.as_slice()[0], 0x11); ++ assert_eq!(mmap.as_slice()[page_size], 0x7a); ++ ++ mmap.as_mut_slice()[page_size] = 0x42; ++ let r = unsafe { ++ libc::msync( ++ mmap.as_mut_ptr().add(page_size).cast(), ++ page_size, ++ libc::MS_SYNC, ++ ) ++ }; ++ assert_eq!(r, 0, "msync failed: {}", io::Error::last_os_error()); ++ ++ let mut got = [0u8; 1]; ++ file.seek(SeekFrom::Start(0)).unwrap(); ++ file.read_exact(&mut got).unwrap(); ++ assert_eq!(got[0], 0x42); ++ ++ drop(mmap); ++ drop(file); ++ let _ = std::fs::remove_file(path); ++ } ++ ++ #[test] ++ fn remap_shared_file_fixed_accepts_a_partial_final_file_page() { ++ let page_size = region::page::size(); ++ let path = temp_file_path("mmap-fixed-shared-partial-page"); ++ let mut file = std::fs::OpenOptions::new() ++ .read(true) ++ .write(true) ++ .create_new(true) ++ .open(&path) ++ .unwrap(); ++ file.set_len((page_size - 1) as u64).unwrap(); ++ file.write_all(&[0x7a]).unwrap(); ++ file.sync_all().unwrap(); ++ ++ let mut mmap = ++ Mmap::accessible_reserved(page_size, page_size, None, MmapType::Private).unwrap(); ++ unsafe { mmap.remap_shared_file_fixed(0, page_size, &file, 0) }.unwrap(); ++ assert_eq!(mmap.as_slice()[0], 0x7a); ++ assert_eq!(mmap.as_slice()[page_size - 1], 0); ++ ++ let mut oversized = ++ Mmap::accessible_reserved(page_size * 2, page_size * 2, None, MmapType::Private) ++ .unwrap(); ++ assert!( ++ unsafe { oversized.remap_shared_file_fixed(0, page_size * 2, &file, 0) } ++ .unwrap_err() ++ .contains("wholly beyond") ++ ); ++ assert_eq!(file.metadata().unwrap().len(), (page_size - 1) as u64); ++ ++ drop(mmap); ++ drop(oversized); ++ drop(file); ++ let _ = std::fs::remove_file(path); ++ } ++ ++ #[test] ++ fn remap_private_file_fixed_shares_clean_bytes_but_isolates_writes() { ++ let page_size = region::page::size(); ++ let path = temp_file_path("mmap-fixed-private-file"); ++ let mut file = std::fs::OpenOptions::new() ++ .read(true) ++ .write(true) ++ .create_new(true) ++ .open(&path) ++ .unwrap(); ++ file.set_len(page_size as u64).unwrap(); ++ file.write_all(&vec![0x7a; page_size]).unwrap(); ++ file.sync_all().unwrap(); ++ ++ let mut first = ++ Mmap::accessible_reserved(page_size, page_size, None, MmapType::Private).unwrap(); ++ let mut second = ++ Mmap::accessible_reserved(page_size, page_size, None, MmapType::Private).unwrap(); ++ unsafe { first.remap_private_file_fixed(0, page_size, &file, 0) }.unwrap(); ++ unsafe { second.remap_private_file_fixed(0, page_size, &file, 0) }.unwrap(); ++ ++ assert_eq!(first.as_slice()[0], 0x7a); ++ assert_eq!(second.as_slice()[0], 0x7a); ++ first.as_mut_slice()[0] = 0x42; ++ assert_eq!(first.as_slice()[0], 0x42); ++ assert_eq!(second.as_slice()[0], 0x7a); ++ ++ let mut got = [0u8; 1]; ++ file.seek(SeekFrom::Start(0)).unwrap(); ++ file.read_exact(&mut got).unwrap(); ++ assert_eq!(got[0], 0x7a); ++ ++ drop(first); ++ drop(second); ++ drop(file); ++ let _ = std::fs::remove_file(path); ++ } ++ ++ #[test] ++ fn copy_preserves_nonzero_and_zero_pages() { ++ let page_size = region::page::size(); ++ let mut mmap = ++ Mmap::accessible_reserved(page_size * 3, page_size * 3, None, MmapType::Private) ++ .unwrap(); ++ ++ mmap.as_mut_slice_arbitary(page_size * 3)[..page_size].fill(0x11); ++ mmap.as_mut_slice_arbitary(page_size * 3)[page_size * 2..page_size * 3].fill(0x33); ++ ++ let copied = mmap.copy(Some(page_size * 3)).unwrap(); ++ ++ assert!( ++ copied.as_slice_arbitary(page_size)[..page_size] ++ .iter() ++ .all(|byte| *byte == 0x11) ++ ); ++ assert!( ++ copied.as_slice_arbitary(page_size * 2)[page_size..page_size * 2] ++ .iter() ++ .all(|byte| *byte == 0) ++ ); ++ assert!( ++ copied.as_slice_arbitary(page_size * 3)[page_size * 2..page_size * 3] ++ .iter() ++ .all(|byte| *byte == 0x33) ++ ); ++ } ++} +diff --git a/lib/vm/src/table.rs b/lib/vm/src/table.rs +index 7eab35a..7999f02 100644 +--- a/lib/vm/src/table.rs ++++ b/lib/vm/src/table.rs +@@ -6,6 +6,7 @@ + //! `Table` is to WebAssembly tables what `Memory` is to WebAssembly linear memories. + + use crate::Trap; ++use crate::VMCallerCheckedAnyfunc; + use crate::VMExternRef; + use crate::VMFuncRef; + use crate::store::MaybeInstanceOwned; +@@ -14,6 +15,7 @@ use bytesize::ByteSize; + use std::cell::UnsafeCell; + use std::convert::TryFrom; + use std::fmt; ++use std::ptr; + use std::ptr::NonNull; + use wasmer_types::TableStyle; + use wasmer_types::{TableType, TrapCode, Type as ValType}; +@@ -75,6 +77,7 @@ const TABLE_MAX_SIZE: usize = ByteSize::mib(128).as_u64() as usize; + #[derive(Debug)] + pub struct VMTable { + vec: Vec, ++ anyfuncs: Option>, + maximum: Option, + /// The WebAssembly table description. + table: TableType, +@@ -152,9 +155,16 @@ impl VMTable { + .map_err(|_| "Table minimum is bigger than usize".to_string())?; + let mut vec = vec![RawTableElement::default(); table_minimum]; + let base = vec.as_mut_ptr(); ++ let mut anyfuncs = match table.ty { ++ ValType::FuncRef => Some(vec![VMCallerCheckedAnyfunc::null(); table_minimum]), ++ ValType::ExternRef => None, ++ _ => unreachable!("table type was already validated"), ++ }; ++ let anyfuncs_base = Self::anyfuncs_base(&mut anyfuncs); + match style { + TableStyle::CallerChecksSignature => Ok(Self { + vec, ++ anyfuncs, + maximum: table.maximum, + table: *table, + style: style.clone(), +@@ -164,12 +174,14 @@ impl VMTable { + let td = ptr.as_mut(); + td.base = base as _; + td.current_elements = table_minimum as _; ++ td.anyfuncs = anyfuncs_base; + } + MaybeInstanceOwned::Instance(table_loc) + } else { + MaybeInstanceOwned::Host(Box::new(UnsafeCell::new(VMTableDefinition { + base: base as _, + current_elements: table_minimum as _, ++ anyfuncs: anyfuncs_base, + }))) + }, + }), +@@ -177,6 +189,14 @@ impl VMTable { + } + } + ++ fn anyfuncs_base( ++ anyfuncs: &mut Option>, ++ ) -> *mut VMCallerCheckedAnyfunc { ++ anyfuncs ++ .as_mut() ++ .map_or(ptr::null_mut(), |anyfuncs| anyfuncs.as_mut_ptr()) ++ } ++ + /// Get the `VMTableDefinition`. + fn get_vm_table_definition(&self) -> NonNull { + self.vm_table_definition.as_ptr() +@@ -221,15 +241,25 @@ impl VMTable { + return Some(size); + } + ++ let init_anyfunc = match &init_value { ++ TableElement::FuncRef(funcref) => VMCallerCheckedAnyfunc::from_funcref(*funcref), ++ TableElement::ExternRef(_) => VMCallerCheckedAnyfunc::null(), ++ }; + self.vec + .resize(usize::try_from(new_len).unwrap(), init_value.into()); ++ if let Some(anyfuncs) = self.anyfuncs.as_mut() { ++ anyfuncs.resize(usize::try_from(new_len).unwrap(), init_anyfunc); ++ } ++ let base = self.vec.as_mut_ptr() as _; ++ let anyfuncs_base = Self::anyfuncs_base(&mut self.anyfuncs); + + // update table definition + unsafe { + let mut td_ptr = self.get_vm_table_definition(); + let td = td_ptr.as_mut(); + td.current_elements = new_len; +- td.base = self.vec.as_mut_ptr() as _; ++ td.base = base; ++ td.anyfuncs = anyfuncs_base; + } + Some(size) + } +@@ -271,7 +301,16 @@ impl VMTable { + *slot = r.into(); + } + (ValType::FuncRef, r @ TableElement::FuncRef(_)) => { ++ let anyfunc = match &r { ++ TableElement::FuncRef(funcref) => { ++ VMCallerCheckedAnyfunc::from_funcref(*funcref) ++ } ++ _ => unreachable!("matched TableElement::FuncRef"), ++ }; + *slot = r.into(); ++ if let Some(anyfuncs) = self.anyfuncs.as_mut() { ++ anyfuncs[index as usize] = anyfunc; ++ } + } + // This path should never be hit by the generated code due to Wasm + // validation. +@@ -279,7 +318,6 @@ impl VMTable { + panic!("Attempted to set a table of type {ty} with the value {v:?}") + } + }; +- + Ok(()) + } + None => Err(Trap::lib(TrapCode::TableAccessOutOfBounds)), +@@ -394,6 +432,11 @@ impl VMTable { + #[cfg(test)] + mod tests { + use super::{TableElement, VMTable}; ++ use crate::{ ++ VMCallerCheckedAnyfunc, VMContext, VMFunctionBody, VMFunctionContext, VMSignatureHash, ++ }; ++ use std::ptr::NonNull; ++ use wasmer_types::RawValue; + use wasmer_types::{TableStyle, TableType, Type}; + + #[test] +@@ -403,4 +446,71 @@ mod tests { + let mut table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); + assert_eq!(table.grow(0, TableElement::FuncRef(None)), None); + } ++ ++ unsafe extern "C" fn test_trampoline( ++ _vmctx: *mut VMContext, ++ _callee: *const VMFunctionBody, ++ _values: *mut RawValue, ++ ) { ++ } ++ ++ fn test_anyfunc(signature: u32) -> VMCallerCheckedAnyfunc { ++ VMCallerCheckedAnyfunc { ++ func_ptr: NonNull::::dangling().as_ptr(), ++ type_signature_hash: VMSignatureHash::new(signature), ++ vmctx: VMFunctionContext { ++ host_env: 1usize as *mut std::ffi::c_void, ++ }, ++ call_trampoline: test_trampoline, ++ } ++ } ++ ++ #[test] ++ fn funcref_table_exposes_anyfunc_shadow() { ++ let ty = TableType::new(Type::FuncRef, 1, Some(3)); ++ let mut table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); ++ ++ unsafe { ++ let definition = table.vmtable().as_ref(); ++ assert!(!definition.anyfuncs.is_null()); ++ assert!(definition.anyfuncs.read().func_ptr.is_null()); ++ } ++ ++ let anyfunc = test_anyfunc(7); ++ let funcref = crate::VMFuncRef(NonNull::from(&anyfunc)); ++ table.set(0, TableElement::FuncRef(Some(funcref))).unwrap(); ++ ++ unsafe { ++ let definition = table.vmtable().as_ref(); ++ assert_eq!(definition.anyfuncs.read(), anyfunc); ++ } ++ } ++ ++ #[test] ++ fn funcref_table_grow_updates_anyfunc_shadow_pointer_and_entries() { ++ let ty = TableType::new(Type::FuncRef, 1, Some(3)); ++ let mut table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); ++ let anyfunc = test_anyfunc(9); ++ let funcref = crate::VMFuncRef(NonNull::from(&anyfunc)); ++ ++ assert_eq!(table.grow(2, TableElement::FuncRef(Some(funcref))), Some(1)); ++ ++ unsafe { ++ let definition = table.vmtable().as_ref(); ++ assert!(!definition.anyfuncs.is_null()); ++ assert!(definition.anyfuncs.read().func_ptr.is_null()); ++ assert_eq!(definition.anyfuncs.add(1).read(), anyfunc); ++ assert_eq!(definition.anyfuncs.add(2).read(), anyfunc); ++ } ++ } ++ ++ #[test] ++ fn externref_table_has_no_anyfunc_shadow() { ++ let ty = TableType::new(Type::ExternRef, 1, Some(1)); ++ let table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); ++ ++ unsafe { ++ assert!(table.vmtable().as_ref().anyfuncs.is_null()); ++ } ++ } + } +diff --git a/lib/vm/src/trap/traphandlers.rs b/lib/vm/src/trap/traphandlers.rs +index 6e2d7cc..033c796 100644 +--- a/lib/vm/src/trap/traphandlers.rs ++++ b/lib/vm/src/trap/traphandlers.rs +@@ -87,23 +87,69 @@ pub fn get_stack_size() -> usize { + DEFAULT_STACK_SIZE.load(Ordering::Relaxed) + } + +-/// Pool of pre-allocated coroutine stacks to avoid repeated mmap syscalls. ++/// Cross-thread overflow pool for pre-allocated coroutine stacks. ++/// ++/// The common same-thread path is served from [`TLS_STACK`] without queue ++/// atomics. Nested calls and thread handoff fall back to this pool. + static STACK_POOL: LazyLock> = + LazyLock::new(crossbeam_queue::SegQueue::new); + ++/// One ready-to-use coroutine stack retained by each host thread that calls ++/// Wasm. A thread-exit destructor returns the mapping to the cross-thread pool ++/// so it remains reusable rather than leaking with the thread. ++struct StackCache(Cell>); ++ ++impl Drop for StackCache { ++ fn drop(&mut self) { ++ if let Some(stack) = self.0.take() { ++ STACK_POOL.push(stack); ++ } ++ } ++} ++ ++thread_local! { ++ static TLS_STACK: StackCache = const { StackCache(Cell::new(None)) }; ++} ++ ++/// Acquires a sufficiently large stack, preferring the atomic-free TLS slot. ++fn acquire_stack(min_size: usize) -> DefaultStack { ++ if let Some(stack) = TLS_STACK.with(|cache| cache.0.take()) { ++ if stack.size() >= min_size { ++ return stack; ++ } ++ // A stack-size increase makes the old mapping unusable. Drop it ++ // instead of circulating it through the overflow pool. ++ drop(stack); ++ } ++ ++ STACK_POOL ++ .pop() ++ .filter(|stack| stack.size() >= min_size) ++ .unwrap_or_else(|| DefaultStack::new(min_size).unwrap()) ++} ++ ++/// Returns a stack to the atomic-free TLS slot. Re-entrant execution can leave ++/// that slot occupied; in that case, move the displaced stack to the shared ++/// overflow pool. ++fn release_stack(stack: DefaultStack) { ++ if let Some(displaced) = TLS_STACK.with(|cache| cache.0.replace(Some(stack))) { ++ STACK_POOL.push(displaced); ++ } ++} ++ + /// Drains the coroutine stack pool at the moment it runs. + /// + /// This is intended to be called before retrying with a larger stack size so + /// that the pool does not keep serving cached undersized stacks. + /// +-/// Note that `STACK_POOL` is a global, concurrently used queue. Other threads +-/// may push stacks back into the pool (for example, when their Wasm execution +-/// finishes) while or after this function is running. As a result, this +-/// function provides only a best-effort drain of the pool: there is no +-/// guarantee that no undersized stacks exist immediately after it returns +-/// unless the caller ensures, via external synchronization, that no other +-/// Wasm executions can return stacks to the pool while this function runs. ++/// Note that the pool is global and each Wasm-calling thread can retain one ++/// private stack. Other threads may return stacks while or after this function ++/// runs, and their TLS slots cannot be drained here. This therefore remains a ++/// best-effort operation unless the caller externally quiesces Wasm execution. ++/// The calling thread's TLS slot is drained. + pub fn drain_stack_pool() { ++ // Drop rather than re-pool the local mapping; the queue is drained next. ++ TLS_STACK.with(|cache| cache.0.set(None)); + while STACK_POOL.pop().is_some() {} + } + +@@ -1047,17 +1093,11 @@ fn on_wasm_stack T + 'static, T: 'static>( + trap_handler: Option<*const TrapHandlerFn<'static>>, + f: F, + ) -> Result { +- // Reuse a cached stack from the pool if it is large enough, otherwise +- // allocate a fresh one. The size check prevents using undersized stacks +- // that were returned by threads still running at the old size after +- // `drain_stack_pool()` was called. `base() - limit()` is the full mmap +- // region (including guard page), which is always >= the requested size +- // for stacks allocated with that size. +- let stack = STACK_POOL +- .pop() +- .filter(|s| s.size() >= stack_size) +- .unwrap_or_else(|| DefaultStack::new(stack_size).unwrap()); +- let mut stack = scopeguard::guard(stack, |stack| STACK_POOL.push(stack)); ++ // Same-thread calls reuse the TLS mapping without touching the global ++ // queue. Nested calls and first use on a thread fall back to the overflow ++ // pool or a fresh mapping. Both cache levels reject undersized stacks. ++ let stack = acquire_stack(stack_size); ++ let mut stack = scopeguard::guard(stack, release_stack); + + // Create a coroutine with a new stack to run the function on. + let coro = ScopedCoroutine::with_stack(&mut *stack, move |yielder, ()| { +@@ -1269,6 +1309,10 @@ mod tests { + } + } + ++ fn clear_tls_stack() { ++ TLS_STACK.with(|cache| cache.0.set(None)); ++ } ++ + #[test] + fn max_stack_size_is_100mb() { + assert_eq!(MAX_STACK_SIZE, ByteSize::mib(100).as_u64() as usize); +@@ -1378,14 +1422,100 @@ mod tests { + let result = on_wasm_stack(big_size, None, || 42); + + assert_eq!(result.ok().expect("on_wasm_stack should succeed"), 42); +- // The undersized stack was discarded; the pool should now contain +- // the correctly-sized stack that was allocated for this call. +- let returned = STACK_POOL +- .pop() +- .expect("stack should have been returned to pool"); ++ // The undersized stack was discarded; the correctly-sized mapping is ++ // now held in the calling thread's fast cache. ++ let returned = TLS_STACK ++ .with(|cache| cache.0.take()) ++ .expect("stack should have been returned to the TLS cache"); + assert!( + returned.size() >= big_size, + "returned stack must be at least as large as the requested size" + ); ++ assert!(STACK_POOL.is_empty()); ++ } ++ ++ #[test] ++ fn tls_stack_reuses_mapping_without_global_queue() { ++ let _lock = GLOBAL_STATE.lock().unwrap(); ++ let _restore = RestoreStackSize(get_stack_size()); ++ drain_stack_pool(); ++ ++ let size = get_stack_size(); ++ assert!(on_wasm_stack(size, None, || ()).is_ok()); ++ let first_base = TLS_STACK.with(|cache| { ++ let stack = cache.0.take().expect("first call should populate TLS"); ++ let base = stack.base().get(); ++ cache.0.set(Some(stack)); ++ base ++ }); ++ assert!(STACK_POOL.is_empty()); ++ ++ assert!(on_wasm_stack(size, None, || ()).is_ok()); ++ let second = TLS_STACK ++ .with(|cache| cache.0.take()) ++ .expect("second call should return the stack to TLS"); ++ assert_eq!(second.base().get(), first_base); ++ assert!(STACK_POOL.is_empty()); ++ } ++ ++ #[test] ++ fn drain_stack_pool_clears_calling_thread_tls() { ++ let _lock = GLOBAL_STATE.lock().unwrap(); ++ drain_stack_pool(); ++ ++ assert!(on_wasm_stack(get_stack_size(), None, || ()).is_ok()); ++ assert!(TLS_STACK.with(|cache| cache.0.take().is_some())); ++ ++ // Re-populate TLS, then verify the public drain covers the caller's ++ // local fast slot as well as the shared queue. ++ assert!(on_wasm_stack(get_stack_size(), None, || ()).is_ok()); ++ drain_stack_pool(); ++ assert!(TLS_STACK.with(|cache| cache.0.take().is_none())); ++ assert!(STACK_POOL.is_empty()); ++ } ++ ++ #[test] ++ fn thread_exit_returns_tls_stack_to_global_pool() { ++ let _lock = GLOBAL_STATE.lock().unwrap(); ++ drain_stack_pool(); ++ clear_tls_stack(); ++ ++ let size = get_stack_size(); ++ std::thread::spawn(move || { ++ assert!(on_wasm_stack(size, None, || ()).is_ok()); ++ assert!(STACK_POOL.is_empty()); ++ }) ++ .join() ++ .unwrap(); ++ ++ let returned = STACK_POOL ++ .pop() ++ .expect("thread-exit TLS destructor should return the stack"); ++ assert!(returned.size() >= size); ++ assert!(STACK_POOL.is_empty()); ++ } ++ ++ #[test] ++ fn reentrant_calls_keep_both_stacks_reusable() { ++ let _lock = GLOBAL_STATE.lock().unwrap(); ++ let _restore = RestoreStackSize(get_stack_size()); ++ drain_stack_pool(); ++ ++ let result = on_wasm_stack(get_stack_size(), None, || { ++ on_wasm_stack(get_stack_size(), None, || 42_u32) ++ .ok() ++ .expect("inner coroutine should complete") ++ }); ++ assert_eq!(result.ok(), Some(42)); ++ ++ let local = TLS_STACK ++ .with(|cache| cache.0.take()) ++ .expect("outer stack should return to TLS"); ++ let overflow = STACK_POOL ++ .pop() ++ .expect("inner stack should remain in the overflow pool"); ++ assert!(local.size() >= get_stack_size()); ++ assert!(overflow.size() >= get_stack_size()); ++ assert!(STACK_POOL.is_empty()); + } + } +diff --git a/lib/vm/src/vmcontext.rs b/lib/vm/src/vmcontext.rs +index 50c89bf..4079bd6 100644 +--- a/lib/vm/src/vmcontext.rs ++++ b/lib/vm/src/vmcontext.rs +@@ -11,7 +11,7 @@ use crate::instance::Instance; + use crate::memory::VMMemory; + use crate::store::InternalStoreHandle; + use crate::trap::{Trap, TrapCode}; +-use crate::{VMBuiltinFunctionIndex, VMFunction}; ++use crate::{VMBuiltinFunctionIndex, VMFuncRef, VMFunction}; + use std::convert::TryFrom; + use std::hash::{Hash, Hasher}; + use std::ptr::{self, NonNull}; +@@ -461,6 +461,13 @@ pub struct VMTableDefinition { + + /// The current number of elements in the table. + pub current_elements: u32, ++ ++ /// Pointer to caller-checked function records for `funcref` tables. ++ /// ++ /// This is null for non-`funcref` tables. For `funcref` tables it mirrors ++ /// `base` element-for-element and lets compiled indirect calls load the ++ /// callable record directly while table operations keep the shadow in sync. ++ pub anyfuncs: *mut VMCallerCheckedAnyfunc, + } + + #[cfg(test)] +@@ -487,6 +494,10 @@ mod test_vmtable_definition { + offset_of!(VMTableDefinition, current_elements), + usize::from(offsets.vmtable_definition_current_elements()) + ); ++ assert_eq!( ++ offset_of!(VMTableDefinition, anyfuncs), ++ usize::from(offsets.vmtable_definition_anyfuncs()) ++ ); + } + } + +@@ -616,6 +627,14 @@ impl VMCallerCheckedAnyfunc { + call_trampoline: null_call_trampoline, + } + } ++ ++ /// Construct a caller-checked function record from a function reference. ++ pub fn from_funcref(funcref: Option) -> Self { ++ match funcref { ++ Some(funcref) => unsafe { *funcref.0.as_ptr() }, ++ None => Self::null(), ++ } ++ } + } + + impl PartialEq for VMCallerCheckedAnyfunc { +diff --git a/tests/compilers/issues.rs b/tests/compilers/issues.rs +index 271bd12..6379ce0 100644 +--- a/tests/compilers/issues.rs ++++ b/tests/compilers/issues.rs +@@ -636,6 +636,83 @@ fn compiler_debug_dir_test(mut config: crate::Config) { + assert!(Module::new(&store, wat).is_ok()); + } + ++#[cfg(feature = "llvm")] ++#[test] ++fn llvm_rotates_and_atomic_fence_emit_expected_ir() { ++ use std::path::Path; ++ use tempfile::TempDir; ++ use wasmer_compiler::{CompilerConfig, EngineBuilder}; ++ use wasmer_compiler_llvm::LLVMCallbacks; ++ ++ fn collect_postopt_ir(dir: &Path, out: &mut String) { ++ for entry in std::fs::read_dir(dir).expect("debug directory must be readable") { ++ let entry = entry.expect("debug directory entry must be readable"); ++ let path = entry.path(); ++ if path.is_dir() { ++ collect_postopt_ir(&path, out); ++ } else if path ++ .file_name() ++ .and_then(|name| name.to_str()) ++ .is_some_and(|name| name.ends_with(".postopt.ll")) ++ { ++ out.push_str( ++ &std::fs::read_to_string(&path) ++ .unwrap_or_else(|err| panic!("cannot read {}: {err}", path.display())), ++ ); ++ } ++ } ++ } ++ ++ let mut compiler_config = wasmer_compiler_llvm::LLVM::new(); ++ let temp = TempDir::new().expect("temp folder creation failed"); ++ compiler_config.callbacks(Some(LLVMCallbacks::new(temp.path().to_path_buf()).unwrap())); ++ let store = Store::new(EngineBuilder::new(compiler_config)); ++ ++ let wat = r#" ++ (module ++ (memory 1 1 shared) ++ (func $fence (export "fence") ++ atomic.fence) ++ (func $rotl32 (export "rotl32") (param i32 i32) (result i32) ++ local.get 0 ++ local.get 1 ++ i32.rotl) ++ (func $rotr32 (export "rotr32") (param i32 i32) (result i32) ++ local.get 0 ++ local.get 1 ++ i32.rotr) ++ (func $rotl64 (export "rotl64") (param i64 i64) (result i64) ++ local.get 0 ++ local.get 1 ++ i64.rotl) ++ (func $rotr64 (export "rotr64") (param i64 i64) (result i64) ++ local.get 0 ++ local.get 1 ++ i64.rotr)) ++ "#; ++ ++ Module::new(&store, wat).expect("rotate module must compile"); ++ ++ let mut ir = String::new(); ++ collect_postopt_ir(temp.path(), &mut ir); ++ ++ assert!(ir.contains("call i32 @llvm.fshl.i32")); ++ assert!(ir.contains("call i32 @llvm.fshr.i32")); ++ assert!(ir.contains("call i64 @llvm.fshl.i64")); ++ assert!(ir.contains("call i64 @llvm.fshr.i64")); ++ assert!(ir.contains("fence seq_cst")); ++ ++ let volatile_id = Box::new(wasmer_compiler_llvm::LLVM::new()) ++ .compiler() ++ .deterministic_id(); ++ let mut nonvolatile_config = wasmer_compiler_llvm::LLVM::new(); ++ nonvolatile_config.non_volatile_memops(true); ++ let nonvolatile_id = Box::new(nonvolatile_config).compiler().deterministic_id(); ++ assert_ne!(volatile_id, nonvolatile_id); ++ assert!(volatile_id.contains("-nv0-")); ++ assert!(nonvolatile_id.contains("-nv1-")); ++} ++ + #[compiler_test(issues)] + fn issue_5795_memory_reset_size(mut config: crate::Config) { + let wasm_bytes = wat2wasm( diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch deleted file mode 100644 index 0462a8b78..000000000 --- a/src/wasix/postmaster/wasmer/patches/wasmer/0001-postgres-wasix-blockers.patch +++ /dev/null @@ -1,38128 +0,0 @@ -diff --git a/Cargo.lock b/Cargo.lock -index 19464e5..8c36cb7 100644 ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -535,17 +535,6 @@ dependencies = [ - "allocator-api2", - ] - --[[package]] --name = "bus" --version = "2.4.1" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "4b7118d0221d84fada881b657c2ddb7cd55108db79c8764c9ee212c0c259b783" --dependencies = [ -- "crossbeam-channel", -- "num_cpus", -- "parking_lot_core", --] -- - [[package]] - name = "bytecheck" - version = "0.8.2" -@@ -3716,6 +3705,26 @@ dependencies = [ - "ruzstd", - ] - -+[[package]] -+name = "oliphaunt-wasix-postmaster-executor" -+version = "7.2.0-alpha.2" -+dependencies = [ -+ "anyhow", -+ "async-trait", -+ "dirs", -+ "hex", -+ "libc", -+ "serde", -+ "serde_json", -+ "sha2 0.11.0", -+ "tempfile", -+ "tracing", -+ "wasmer", -+ "wasmer-types", -+ "wasmer-vm", -+ "wasmer-wasix", -+] -+ - [[package]] - name = "once_cell" - version = "1.21.4" -@@ -7113,6 +7128,7 @@ dependencies = [ - "mio", - "normpath", - "object 0.39.1", -+ "oliphaunt-wasix-postmaster-executor", - "once_cell", - "opener", - "parking_lot", -@@ -7162,6 +7178,7 @@ dependencies = [ - "wasmer-wasix", - "wasmer-wast", - "webc", -+ "windows-sys 0.61.2", - "zip", - ] - -@@ -7578,7 +7595,6 @@ dependencies = [ - "base64 0.22.1", - "bincode 2.0.1", - "blake3", -- "bus", - "bytecheck", - "bytes", - "cfg-if", -diff --git a/Cargo.toml b/Cargo.toml -index 6ce7557..fa7d8dc 100644 ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -110,7 +111,6 @@ bindgen = "0.72.1" - bitflags = "2.11.0" - blake3 = "1.0" - build-deps = "0.1.4" --bus = "2.4.1" - bytecheck = { version = "0.8.2" } - byteorder = "1.3" - bytes = "1.11.1" -diff --git a/lib/api/src/backend/sys/entities/engine.rs b/lib/api/src/backend/sys/entities/engine.rs -index 40696a9..1fd35ea 100644 ---- a/lib/api/src/backend/sys/entities/engine.rs -+++ b/lib/api/src/backend/sys/entities/engine.rs -@@ -3,11 +3,112 @@ - use std::{path::Path, sync::Arc}; - - use shared_buffer::OwnedBuffer; --pub use wasmer_compiler::{Artifact, BaseTunables, Engine, EngineBuilder, Tunables}; --use wasmer_types::{CompilationProgressCallback, DeserializeError, Features, target::Target}; -+use wasmer_compiler::{ArtifactCreate, PendingArtifact}; -+pub use wasmer_compiler::{ -+ Artifact, BaseTunables, CodeMemoryPolicy, Engine, EngineBuilder, Tunables, -+}; -+use wasmer_types::{ -+ CompilationProgressCallback, DeserializeError, Features, ModuleHash, target::Target, -+}; -+use wasmer_vm::MemoryStyle; - - use crate::{BackendEngine, BackendModule}; - -+/// A detached module activation that still owns rollback of its exact -+/// engine-side executable-code allocation. -+/// -+/// Sealed loaders can inspect the activated module and complete fallible audit -+/// work before calling [`Self::commit`]. Dropping this value rolls the -+/// allocation back, including frame and unwind registrations. -+#[doc(hidden)] -+pub struct PendingModuleActivation { -+ artifact: PendingArtifact, -+} -+ -+/// The complete linear-memory allocation plan embedded in a serialized AOT -+/// artifact. -+/// -+/// A sealed loader can compare this plan with its admitted module contract -+/// before executable-code ownership escapes the activation transaction. -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+pub struct SerializedLinearMemoryPlan { -+ /// The module's initial memory size in WebAssembly pages. -+ pub minimum_pages: u32, -+ /// The module's declared maximum memory size in WebAssembly pages. -+ pub maximum_pages: Option, -+ /// Whether the module declares shared linear memory. -+ pub shared: bool, -+ /// The allocation strategy compiled into the artifact. -+ pub style: SerializedLinearMemoryStyle, -+} -+ -+/// A stable, inspection-only description of an AOT artifact's memory style. -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+pub enum SerializedLinearMemoryStyle { -+ /// A nonmoving reservation with a fixed bound and offset guard. -+ Static { -+ /// Reserved linear-memory capacity in WebAssembly pages. -+ bound_pages: u32, -+ /// Bytes reserved after the static bound for unchecked offsets. -+ offset_guard_bytes: u64, -+ }, -+ /// A growable allocation whose host base may need to move. -+ Dynamic { -+ /// Bytes reserved after the current allocation for unchecked offsets. -+ offset_guard_bytes: u64, -+ }, -+} -+ -+impl PendingModuleActivation { -+ fn new(artifact: PendingArtifact) -> Self { -+ Self { artifact } -+ } -+ -+ /// Return the activated module hash without committing code memory. -+ pub fn module_hash(&self) -> Option { -+ self.artifact.module_hash() -+ } -+ -+ /// Return every linear-memory type and allocation style embedded in this -+ /// artifact without committing its executable-code allocation. -+ pub fn linear_memory_plans(&self) -> Vec { -+ let artifact = self.artifact.artifact(); -+ artifact -+ .module_info() -+ .memories -+ .values() -+ .zip(artifact.memory_styles().values()) -+ .map(|(memory, style)| SerializedLinearMemoryPlan { -+ minimum_pages: memory.minimum.0, -+ maximum_pages: memory.maximum.map(|pages| pages.0), -+ shared: memory.shared, -+ style: match style { -+ MemoryStyle::Static { -+ bound, -+ offset_guard_size, -+ } => SerializedLinearMemoryStyle::Static { -+ bound_pages: bound.0, -+ offset_guard_bytes: *offset_guard_size, -+ }, -+ MemoryStyle::Dynamic { offset_guard_size } => { -+ SerializedLinearMemoryStyle::Dynamic { -+ offset_guard_bytes: *offset_guard_size, -+ } -+ } -+ }, -+ }) -+ .collect() -+ } -+ -+ /// Commit executable-code ownership and construct the public module. -+ pub fn commit(self) -> crate::Module { -+ let artifact = Arc::new(self.artifact.commit()); -+ crate::Module(BackendModule::Sys(super::module::Module::from_artifact( -+ artifact, -+ ))) -+ } -+} -+ - /// Get the default config for the sys Engine - #[allow(unreachable_code)] - #[cfg(feature = "compiler")] -@@ -83,6 +184,9 @@ pub trait NativeEngineExt { - /// Get a reference to attached Tunable of this engine - fn tunables(&self) -> &dyn Tunables; - -+ /// Select executable-memory ownership before the first artifact allocation. -+ fn set_code_memory_policy(&mut self, policy: CodeMemoryPolicy) -> Result<(), String>; -+ - /// Compile a module from bytes with a progress callback. - /// - /// The callback is invoked with progress updates during the compilation process. -@@ -120,6 +224,55 @@ pub trait NativeEngineExt { - &self, - file_ref: &Path, - ) -> Result; -+ -+ /// Load and validate a serialized WebAssembly module from an existing -+ /// memory mapping. -+ /// -+ /// This is the descriptor-bound counterpart to -+ /// [`Self::deserialize_from_mmapped_file`]: callers can verify the exact -+ /// mapping before transferring it into the checked artifact deserializer. -+ /// -+ /// # Safety -+ /// See [`Artifact::deserialize`]. -+ unsafe fn deserialize_from_mmapped_buffer( -+ &self, -+ buffer: OwnedBuffer, -+ ) -> Result; -+ -+ /// Validate a serialized universal artifact and return its embedded module -+ /// hash without allocating or publishing executable code. -+ /// -+ /// This checks the universal magic, metadata ABI, complete archived value, -+ /// and native CPU-feature requirements. The buffer is borrowed so a caller -+ /// can retain the exact verified snapshot for later lazy activation. -+ fn inspect_serialized_artifact( -+ &self, -+ buffer: &OwnedBuffer, -+ ) -> Result; -+ -+ /// Load and validate a descriptor-bound artifact, materialize only the -+ /// metadata needed for later instantiations, and release the archive. -+ /// -+ /// The resulting module cannot be re-serialized. This is intended for -+ /// sealed precompiled-only executors that value reclaimable steady-state -+ /// memory over artifact round-tripping. -+ /// -+ /// # Safety -+ /// See [`Artifact::deserialize_detached`]. -+ unsafe fn deserialize_from_mmapped_buffer_detached( -+ &self, -+ buffer: OwnedBuffer, -+ ) -> Result; -+ -+ /// Load and validate a detached descriptor-bound artifact while retaining -+ /// rollback ownership until the caller explicitly commits activation. -+ /// -+ /// # Safety -+ /// See [`Artifact::deserialize_detached_pending`]. -+ unsafe fn deserialize_from_mmapped_buffer_detached_pending( -+ &self, -+ buffer: OwnedBuffer, -+ ) -> Result; - } - - impl NativeEngineExt for crate::engine::Engine { -@@ -163,6 +316,13 @@ impl NativeEngineExt for crate::engine::Engine { - } - } - -+ fn set_code_memory_policy(&mut self, policy: CodeMemoryPolicy) -> Result<(), String> { -+ match self.be { -+ BackendEngine::Sys(ref mut engine) => engine.set_code_memory_policy(policy), -+ _ => Err("code-memory policies require a sys engine".to_string()), -+ } -+ } -+ - fn new_module_with_progress( - &self, - bytes: &[u8], -@@ -196,6 +356,39 @@ impl NativeEngineExt for crate::engine::Engine { - super::module::Module::from_artifact(artifact), - ))) - } -+ -+ unsafe fn deserialize_from_mmapped_buffer( -+ &self, -+ buffer: OwnedBuffer, -+ ) -> Result { -+ let artifact = unsafe { Arc::new(Artifact::deserialize(self.as_sys(), buffer)?) }; -+ Ok(crate::Module(BackendModule::Sys( -+ super::module::Module::from_artifact(artifact), -+ ))) -+ } -+ -+ fn inspect_serialized_artifact( -+ &self, -+ buffer: &OwnedBuffer, -+ ) -> Result { -+ Artifact::inspect_serialized(self.as_sys(), buffer.as_ref()) -+ } -+ -+ unsafe fn deserialize_from_mmapped_buffer_detached( -+ &self, -+ buffer: OwnedBuffer, -+ ) -> Result { -+ unsafe { self.deserialize_from_mmapped_buffer_detached_pending(buffer) } -+ .map(PendingModuleActivation::commit) -+ } -+ -+ unsafe fn deserialize_from_mmapped_buffer_detached_pending( -+ &self, -+ buffer: OwnedBuffer, -+ ) -> Result { -+ let artifact = unsafe { Artifact::deserialize_detached_pending(self.as_sys(), buffer)? }; -+ Ok(PendingModuleActivation::new(artifact)) -+ } - } - - impl crate::Engine { -diff --git a/lib/api/src/backend/sys/entities/instance.rs b/lib/api/src/backend/sys/entities/instance.rs -index 57c4a07..63d84c9 100644 ---- a/lib/api/src/backend/sys/entities/instance.rs -+++ b/lib/api/src/backend/sys/entities/instance.rs -@@ -44,6 +44,33 @@ impl Instance { - Ok((instance, exports)) - } - -+ /// Instantiates an instance while materializing only the named exports. -+ /// -+ /// The VM's complete `ModuleInfo::exports` remains authoritative and can -+ /// be queried later through [`Self::lookup_export`]. This is an internal -+ /// opt-in path for runtimes whose modules export very large symbol tables; -+ /// ordinary Wasmer instances stay eager. -+ #[allow(clippy::result_large_err)] -+ pub(crate) fn new_with_export_names( -+ store: &mut impl AsStoreMut, -+ module: &Module, -+ imports: &Imports, -+ export_names: &[&str], -+ ) -> Result<(Self, Exports), InstantiationError> { -+ let externs = imports -+ .imports_for_module(module) -+ .map_err(InstantiationError::Link)?; -+ let mut handle = module.as_sys().instantiate(store, &externs)?; -+ handle.unwrap_sys_mut().enable_deferred_export_cache(); -+ let exports = Self::get_named_exports(store, handle.unwrap_sys_mut(), export_names); -+ -+ let instance = Self { -+ _handle: StoreHandle::new(store.objects_mut().as_sys_mut(), handle.unwrap_sys()), -+ }; -+ -+ Ok((instance, exports)) -+ } -+ - #[allow(clippy::result_large_err)] - pub(crate) fn new_by_index( - store: &mut impl AsStoreMut, -@@ -78,6 +105,34 @@ impl Instance { - }) - .collect::() - } -+ -+ fn get_named_exports( -+ store: &mut impl AsStoreMut, -+ handle: &mut VMInstance, -+ export_names: &[&str], -+ ) -> Exports { -+ export_names -+ .iter() -+ .filter_map(|name| { -+ let export = handle.lookup(name)?; -+ let extern_ = Extern::from_vm_extern(store, crate::vm::VMExtern::Sys(export)); -+ Some(((*name).to_string(), extern_)) -+ }) -+ .collect::() -+ } -+ -+ /// Looks up one export from the VM's complete module metadata, materializing -+ /// it on demand for deferred instances. -+ pub(crate) fn lookup_export(&self, store: &mut impl AsStoreMut, name: &str) -> Option { -+ let export = self -+ ._handle -+ .get_mut(store.objects_mut().as_sys_mut()) -+ .lookup(name)?; -+ Some(Extern::from_vm_extern( -+ store, -+ crate::vm::VMExtern::Sys(export), -+ )) -+ } - } - - impl crate::BackendInstance { -diff --git a/lib/api/src/backend/sys/entities/memory/mod.rs b/lib/api/src/backend/sys/entities/memory/mod.rs -index e8ad5f5..39a84de 100644 ---- a/lib/api/src/backend/sys/entities/memory/mod.rs -+++ b/lib/api/src/backend/sys/entities/memory/mod.rs -@@ -89,6 +89,87 @@ impl Memory { - Ok(()) - } - -+ pub(crate) fn supports_persistent_shared_fixed_remap(&self, store: &impl AsStoreRef) -> bool { -+ self.handle -+ .get(store.as_store_ref().objects().as_sys()) -+ .supports_persistent_shared_fixed_remap() -+ } -+ -+ /// Replace a page-aligned range inside this memory with a shared, -+ /// read-write file mapping at the same host address. -+ /// -+ /// # Safety -+ /// No concurrent linear-memory access or live Rust reference may overlap the range during -+ /// replacement. The backing inode must not shrink below its size validated at mapping time; -+ /// only the validated final partial page may extend beyond EOF. -+ pub unsafe fn remap_shared_file_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.handle -+ .get_mut(store.as_store_mut().objects_mut().as_sys_mut()) -+ .remap_shared_file_fixed(start, len, file, file_offset) -+ } -+ } -+ -+ /// Replace a page-aligned range inside this memory with a private, -+ /// copy-on-write file mapping at the same host address. -+ /// -+ /// # Safety -+ /// No concurrent linear-memory access or live Rust reference may overlap the range during -+ /// replacement, and the backing inode must not be truncated or replaced below -+ /// `file_offset + len` while accessible. -+ pub unsafe fn remap_private_file_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.handle -+ .get_mut(store.as_store_mut().objects_mut().as_sys_mut()) -+ .remap_private_file_fixed(start, len, file, file_offset) -+ } -+ } -+ -+ /// Replace a page-aligned range inside this memory with private, -+ /// zero-filled memory at the same host address. -+ /// -+ /// # Safety -+ /// No access or Rust reference may overlap the range during replacement. -+ pub unsafe fn remap_private_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.handle -+ .get_mut(store.as_store_mut().objects_mut().as_sys_mut()) -+ .remap_private_fixed(start, len) -+ } -+ } -+ -+ /// Synchronize a page-aligned memory range with its backing file. -+ pub fn msync( -+ &self, -+ store: &impl AsStoreRef, -+ start: usize, -+ len: usize, -+ flags: i32, -+ ) -> Result<(), MemoryError> { -+ self.handle -+ .get(store.as_store_ref().objects().as_sys()) -+ .msync(start, len, flags) -+ } -+ - pub(crate) fn from_vm_extern(store: &impl AsStoreRef, vm_extern: VMExternMemory) -> Self { - Self { - handle: unsafe { -diff --git a/lib/api/src/backend/sys/entities/module.rs b/lib/api/src/backend/sys/entities/module.rs -index 4e6ae27..de6887f 100644 ---- a/lib/api/src/backend/sys/entities/module.rs -+++ b/lib/api/src/backend/sys/entities/module.rs -@@ -49,6 +49,7 @@ impl Module { - unsafe { Self::from_binary_unchecked(engine, binary) } - } - -+ #[cfg(feature = "compiler")] - pub(crate) fn from_binary_with_progress( - engine: &impl AsEngineRef, - binary: &[u8], -@@ -64,6 +65,15 @@ impl Module { - Ok(Self::from_artifact(artifact)) - } - -+ #[cfg(not(feature = "compiler"))] -+ pub(crate) fn from_binary_with_progress( -+ engine: &impl AsEngineRef, -+ binary: &[u8], -+ _callback: CompilationProgressCallback, -+ ) -> Result { -+ Self::from_binary(engine, binary) -+ } -+ - pub(crate) unsafe fn from_binary_unchecked( - engine: &impl AsEngineRef, - binary: &[u8], -diff --git a/lib/api/src/backend/sys/mod.rs b/lib/api/src/backend/sys/mod.rs -index 4e8ea87..fa6c1c5 100644 ---- a/lib/api/src/backend/sys/mod.rs -+++ b/lib/api/src/backend/sys/mod.rs -@@ -7,7 +7,10 @@ pub(crate) mod error; - pub(crate) mod tunables; - pub mod vm; - --pub use engine::NativeEngineExt; -+pub use engine::{ -+ NativeEngineExt, PendingModuleActivation, SerializedLinearMemoryPlan, -+ SerializedLinearMemoryStyle, -+}; - pub use entities::*; - pub use tunables::*; - -@@ -16,7 +19,8 @@ pub use wasmer_compiler::{ - CompilerConfig, FunctionMiddleware, MiddlewareReaderState, ModuleMiddleware, wasmparser, - }; - --pub use wasmer_compiler::{Artifact, EngineBuilder, Features, Tunables}; -+pub use shared_buffer::OwnedBuffer; -+pub use wasmer_compiler::{Artifact, CodeMemoryPolicy, EngineBuilder, Features, Tunables}; - - pub use wasmer_types::MiddlewareError; - pub use wasmer_types::target::{Architecture, CpuFeature, OperatingSystem, Target, Triple}; -diff --git a/lib/api/src/entities/exports.rs b/lib/api/src/entities/exports.rs -index 5bda728..9dc3d9b 100644 ---- a/lib/api/src/entities/exports.rs -+++ b/lib/api/src/entities/exports.rs -@@ -3,6 +3,7 @@ use crate::{Extern, Function, Global, Memory, Table, TypedFunction, WasmTypeList - use indexmap::IndexMap; - use std::fmt; - use std::iter::{ExactSizeIterator, FromIterator}; -+use std::sync::Arc; - use thiserror::Error; - - /// The `ExportError` can happen when trying to get a specific -@@ -60,7 +61,7 @@ pub enum ExportError { - #[derive(Clone, Default, PartialEq, Eq)] - #[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] - pub struct Exports { -- map: IndexMap, -+ map: Arc>, - } - - impl Exports { -@@ -72,7 +73,7 @@ impl Exports { - /// Creates a new `Exports` with capacity `n`. - pub fn with_capacity(n: usize) -> Self { - Self { -- map: IndexMap::with_capacity(n), -+ map: Arc::new(IndexMap::with_capacity(n)), - } - } - -@@ -92,7 +93,7 @@ impl Exports { - S: Into, - E: Into, - { -- self.map.insert(name.into(), value.into()); -+ Arc::make_mut(&mut self.map).insert(name.into(), value.into()); - } - - /// Get an export given a `name`. -@@ -256,7 +257,7 @@ where - impl FromIterator<(String, Extern)> for Exports { - fn from_iter>(iter: I) -> Self { - Self { -- map: IndexMap::from_iter(iter), -+ map: Arc::new(IndexMap::from_iter(iter)), - } - } - } -@@ -266,7 +267,9 @@ impl IntoIterator for Exports { - type Item = (String, Extern); - - fn into_iter(self) -> Self::IntoIter { -- self.map.into_iter() -+ Arc::try_unwrap(self.map) -+ .unwrap_or_else(|map| (*map).clone()) -+ .into_iter() - } - } - -@@ -279,6 +282,55 @@ impl<'a> IntoIterator for &'a Exports { - } - } - -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use crate::{Store, Value}; -+ -+ #[test] -+ fn clone_shares_map_until_mutated() { -+ let mut store = Store::default(); -+ let mut exports = Exports::new(); -+ exports.insert("one", Global::new(&mut store, Value::I32(1))); -+ -+ let mut cloned = exports.clone(); -+ assert!(Arc::ptr_eq(&exports.map, &cloned.map)); -+ assert_eq!(exports, cloned); -+ -+ cloned.insert("two", Global::new(&mut store, Value::I32(2))); -+ -+ assert!(!Arc::ptr_eq(&exports.map, &cloned.map)); -+ assert_eq!(exports.len(), 1); -+ assert_eq!(cloned.len(), 2); -+ assert!(exports.get_extern("two").is_none()); -+ assert!(cloned.get_extern("two").is_some()); -+ assert_ne!(exports, cloned); -+ } -+ -+ #[test] -+ fn consuming_iteration_owns_unique_and_shared_maps() { -+ let mut store = Store::default(); -+ let mut unique = Exports::new(); -+ unique.insert("one", Global::new(&mut store, Value::I32(1))); -+ let unique_names = unique.into_iter().map(|(name, _)| name).collect::>(); -+ assert_eq!(unique_names, ["one"]); -+ -+ let mut original = Exports::new(); -+ original.insert("one", Global::new(&mut store, Value::I32(1))); -+ original.insert("two", Global::new(&mut store, Value::I32(2))); -+ let shared = original.clone(); -+ assert_eq!(Arc::strong_count(&original.map), 2); -+ -+ let shared_names = shared.into_iter().map(|(name, _)| name).collect::>(); -+ -+ assert_eq!(shared_names, ["one", "two"]); -+ assert_eq!(Arc::strong_count(&original.map), 1); -+ assert_eq!(original.len(), 2); -+ assert!(original.get_extern("one").is_some()); -+ assert!(original.get_extern("two").is_some()); -+ } -+} -+ - /// This trait is used to mark types as gettable from an [`Instance`]. - /// - /// [`Instance`]: crate::Instance -diff --git a/lib/api/src/entities/instance.rs b/lib/api/src/entities/instance.rs -index edea32f..72944fe 100644 ---- a/lib/api/src/entities/instance.rs -+++ b/lib/api/src/entities/instance.rs -@@ -81,6 +81,72 @@ impl Instance { - }) - } - -+ /// Instantiates an instance with only a selected initial export set. -+ /// -+ /// This is an internal runtime optimization for modules with very large -+ /// dynamic-link symbol tables. The native backend retains the complete -+ /// module export metadata for store-aware lookup through -+ /// [`Self::lookup_export`]. Other backends remain eager to preserve their -+ /// existing semantics. -+ #[doc(hidden)] -+ #[allow(clippy::result_large_err)] -+ pub fn new_with_export_names( -+ store: &mut impl AsStoreMut, -+ module: &Module, -+ imports: &Imports, -+ export_names: &[&str], -+ ) -> Result { -+ let (_inner, exports) = match &store.as_store_mut().inner.store { -+ #[cfg(feature = "sys")] -+ crate::BackendStore::Sys(_) => { -+ let (i, e) = crate::backend::sys::instance::Instance::new_with_export_names( -+ store, -+ module, -+ imports, -+ export_names, -+ )?; -+ (crate::BackendInstance::Sys(i), e) -+ } -+ #[cfg(feature = "v8")] -+ crate::BackendStore::V8(_) => { -+ let (i, e) = crate::backend::v8::instance::Instance::new(store, module, imports)?; -+ (crate::BackendInstance::V8(i), e) -+ } -+ #[cfg(feature = "js")] -+ crate::BackendStore::Js(_) => { -+ let (i, e) = crate::backend::js::instance::Instance::new(store, module, imports)?; -+ (crate::BackendInstance::Js(i), e) -+ } -+ }; -+ -+ Ok(Self { -+ _inner, -+ module: module.clone(), -+ exports, -+ }) -+ } -+ -+ /// Resolves one export using the complete module export table. -+ /// -+ /// For selectively materialized native instances this may allocate the -+ /// corresponding runtime handle on first use. Ordinary eager instances and -+ /// non-native backends return the already materialized export. -+ #[doc(hidden)] -+ pub fn lookup_export(&self, store: &mut impl AsStoreMut, name: &str) -> Option { -+ if let Some(export) = self.exports.get_extern(name) { -+ return Some(export.clone()); -+ } -+ -+ match &self._inner { -+ #[cfg(feature = "sys")] -+ crate::BackendInstance::Sys(instance) => instance.lookup_export(store, name), -+ #[cfg(feature = "v8")] -+ crate::BackendInstance::V8(_) => self.exports.get_extern(name).cloned(), -+ #[cfg(feature = "js")] -+ crate::BackendInstance::Js(_) => self.exports.get_extern(name).cloned(), -+ } -+ } -+ - /// Creates a new `Instance` from a WebAssembly [`Module`] and a - /// vector of imports. - /// -diff --git a/lib/api/src/entities/memory/inner.rs b/lib/api/src/entities/memory/inner.rs -index 5adfc92..6526e21 100644 ---- a/lib/api/src/entities/memory/inner.rs -+++ b/lib/api/src/entities/memory/inner.rs -@@ -170,6 +170,18 @@ impl BackendMemory { - }) - } - -+ #[inline] -+ pub fn supports_persistent_shared_fixed_remap(&self, store: &impl AsStoreRef) -> bool { -+ match self { -+ #[cfg(feature = "sys")] -+ Self::Sys(s) => s.supports_persistent_shared_fixed_remap(store), -+ #[cfg(feature = "v8")] -+ Self::V8(_) => false, -+ #[cfg(feature = "js")] -+ Self::Js(_) => false, -+ } -+ } -+ - /// Attempts to duplicate this memory (if its clonable) in a new store - /// (copied memory) - #[inline] -@@ -205,6 +217,117 @@ impl BackendMemory { - } - } - -+ /// # Safety -+ /// No concurrent linear-memory access or live Rust reference may overlap the replaced range. -+ /// The backing inode must not shrink below its size validated at mapping time; only the -+ /// validated final partial page may extend beyond EOF. -+ #[inline] -+ pub unsafe fn remap_shared_file_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ match self { -+ #[cfg(feature = "sys")] -+ Self::Sys(s) => unsafe { -+ s.remap_shared_file_fixed(store, start, len, file, file_offset) -+ }, -+ #[cfg(feature = "v8")] -+ Self::V8(_) => Err(MemoryError::UnsupportedOperation { -+ message: -+ "fixed file-backed shared memory remapping is only supported by the sys backend" -+ .to_string(), -+ }), -+ #[cfg(feature = "js")] -+ Self::Js(_) => Err(MemoryError::UnsupportedOperation { -+ message: -+ "fixed file-backed shared memory remapping is only supported by the sys backend" -+ .to_string(), -+ }), -+ } -+ } -+ -+ /// # Safety -+ /// No concurrent linear-memory access or live Rust reference may overlap the replaced range, -+ /// and the backing inode must not be truncated or replaced below `file_offset + len` while -+ /// accessible. -+ #[inline] -+ pub unsafe fn remap_private_file_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ match self { -+ #[cfg(feature = "sys")] -+ Self::Sys(s) => unsafe { -+ s.remap_private_file_fixed(store, start, len, file, file_offset) -+ }, -+ #[cfg(feature = "v8")] -+ Self::V8(_) => Err(MemoryError::UnsupportedOperation { -+ message: "fixed private file remapping is only supported by the sys backend" -+ .to_string(), -+ }), -+ #[cfg(feature = "js")] -+ Self::Js(_) => Err(MemoryError::UnsupportedOperation { -+ message: "fixed private file remapping is only supported by the sys backend" -+ .to_string(), -+ }), -+ } -+ } -+ -+ /// # Safety -+ /// No access or Rust reference may overlap the replaced range. -+ #[inline] -+ pub unsafe fn remap_private_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ ) -> Result<(), MemoryError> { -+ match self { -+ #[cfg(feature = "sys")] -+ Self::Sys(s) => unsafe { s.remap_private_fixed(store, start, len) }, -+ #[cfg(feature = "v8")] -+ Self::V8(_) => Err(MemoryError::UnsupportedOperation { -+ message: "fixed private memory remapping is only supported by the sys backend" -+ .to_string(), -+ }), -+ #[cfg(feature = "js")] -+ Self::Js(_) => Err(MemoryError::UnsupportedOperation { -+ message: "fixed private memory remapping is only supported by the sys backend" -+ .to_string(), -+ }), -+ } -+ } -+ -+ #[inline] -+ pub fn msync( -+ &self, -+ store: &impl AsStoreRef, -+ start: usize, -+ len: usize, -+ flags: i32, -+ ) -> Result<(), MemoryError> { -+ match self { -+ #[cfg(feature = "sys")] -+ Self::Sys(s) => s.msync(store, start, len, flags), -+ #[cfg(feature = "v8")] -+ Self::V8(_) => Err(MemoryError::UnsupportedOperation { -+ message: "memory synchronization is only supported by the sys backend".to_string(), -+ }), -+ #[cfg(feature = "js")] -+ Self::Js(_) => Err(MemoryError::UnsupportedOperation { -+ message: "memory synchronization is only supported by the sys backend".to_string(), -+ }), -+ } -+ } -+ - #[inline] - pub(crate) fn from_vm_extern(store: &mut impl AsStoreMut, vm_extern: VMExternMemory) -> Self { - match &store.as_store_mut().inner.store { -diff --git a/lib/api/src/entities/memory/mod.rs b/lib/api/src/entities/memory/mod.rs -index e869b2f..efdfaaf 100644 ---- a/lib/api/src/entities/memory/mod.rs -+++ b/lib/api/src/entities/memory/mod.rs -@@ -151,6 +151,12 @@ impl Memory { - self.0.reset(store) - } - -+ /// Returns whether shared fixed file mappings can remain valid across all -+ /// future growth of this memory. -+ pub fn supports_persistent_shared_fixed_remap(&self, store: &impl AsStoreRef) -> bool { -+ self.0.supports_persistent_shared_fixed_remap(store) -+ } -+ - /// Attempts to duplicate this memory (if its clonable) in a new store - /// (copied memory) - pub fn copy_to_store( -@@ -161,6 +167,81 @@ impl Memory { - self.0.copy_to_store(store, new_store).map(Self) - } - -+ /// Replace an existing page-aligned range in this memory with a shared, -+ /// file-backed host mapping. -+ /// -+ /// # Safety -+ /// -+ /// No guest or host thread may concurrently access the linear memory, and no live Rust -+ /// reference may point into the replaced range while the mapping is changed. The backing inode -+ /// must not shrink below the size validated by this call while the mapping is accessible; only -+ /// the validated final partial page may extend beyond EOF. -+ pub unsafe fn remap_shared_file_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.0 -+ .remap_shared_file_fixed(store, start, len, file, file_offset) -+ } -+ } -+ -+ /// Replace an existing page-aligned range in this memory with a private, -+ /// copy-on-write file-backed host mapping. -+ /// -+ /// Clean pages may be shared by independent memories, while writes remain -+ /// private to the memory that performed them. -+ /// -+ /// # Safety -+ /// -+ /// No guest or host thread may concurrently access the linear memory, and no live Rust -+ /// reference may point into the replaced range while the mapping is changed. The backing inode -+ /// must not be truncated or replaced below `file_offset + len` while the mapping is accessible. -+ pub unsafe fn remap_private_file_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.0 -+ .remap_private_file_fixed(store, start, len, file, file_offset) -+ } -+ } -+ -+ /// Replace an existing page-aligned range in this memory with private, -+ /// zero-filled host memory. -+ /// -+ /// # Safety -+ /// -+ /// No guest or host thread may access the replaced range while the mapping is changed, and -+ /// no live Rust reference may point into it. -+ pub unsafe fn remap_private_fixed( -+ &self, -+ store: &mut impl AsStoreMut, -+ start: usize, -+ len: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { self.0.remap_private_fixed(store, start, len) } -+ } -+ -+ /// Synchronize a page-aligned memory range with its backing file. -+ pub fn msync( -+ &self, -+ store: &impl AsStoreRef, -+ start: usize, -+ len: usize, -+ flags: i32, -+ ) -> Result<(), MemoryError> { -+ self.0.msync(store, start, len, flags) -+ } -+ - pub(crate) fn from_vm_extern(store: &mut impl AsStoreMut, vm_extern: VMExternMemory) -> Self { - Self(BackendMemory::from_vm_extern(store, vm_extern)) - } -diff --git a/lib/api/tests/instance.rs b/lib/api/tests/instance.rs -index 7a57c37..bd3a664 100644 ---- a/lib/api/tests/instance.rs -+++ b/lib/api/tests/instance.rs -@@ -48,6 +48,197 @@ fn exports_work_after_multiple_instances_have_been_freed() -> Result<(), String> - Ok(()) - } - -+#[cfg(feature = "sys")] -+#[test] -+fn selectively_materialized_exports_remain_available_by_module_identity() { -+ let mut store = Store::default(); -+ let module = Module::new( -+ &store, -+ r#" -+(module -+ (func $initialize -+ i32.const 7 -+ global.set $state) -+ (start $initialize) -+ (func $answer (result i32) -+ i32.const 42) -+ (global $state (mut i32) (i32.const 0)) -+ (memory 1) -+ (export "first" (func $answer)) -+ (export "second" (func $answer)) -+ (export "state" (global $state)) -+ (export "memory" (memory 0))) -+"#, -+ ) -+ .unwrap(); -+ -+ // The ordinary API remains eager and keeps its established surface. -+ let eager = Instance::new(&mut store, &module, &Imports::new()).unwrap(); -+ assert_eq!(eager.exports.len(), 4); -+ -+ let deferred = -+ Instance::new_with_export_names(&mut store, &module, &Imports::new(), &["first"]).unwrap(); -+ assert_eq!(deferred.exports.len(), 1); -+ assert!(deferred.exports.get_extern("second").is_none()); -+ -+ let first = deferred.exports.get_function("first").unwrap().clone(); -+ let Extern::Function(second) = deferred.lookup_export(&mut store, "second").unwrap() else { -+ panic!("second must be a function") -+ }; -+ assert_eq!(first, second, "function aliases must share VM identity"); -+ assert_eq!(second.call(&mut store, &[]).unwrap()[0], Value::I32(42)); -+ -+ // A clone reaches exports that were never part of the initial materialized -+ // set, and Wasm start execution was not bypassed by selective export setup. -+ let cloned = deferred.clone(); -+ let Extern::Global(state) = cloned.lookup_export(&mut store, "state").unwrap() else { -+ panic!("state must be a global") -+ }; -+ assert_eq!(state.get(&mut store), Value::I32(7)); -+ assert!(cloned.lookup_export(&mut store, "memory").is_some()); -+ assert!(cloned.lookup_export(&mut store, "does-not-exist").is_none()); -+ -+ // Deferred lookups stay in the compact identity cache rather than growing -+ // the public string-keyed export map on every Instance clone. -+ assert_eq!(deferred.exports.len(), 1); -+} -+ -+#[engine_test] -+fn passive_data_drop_is_local_to_each_instance() -> Result<(), String> { -+ let mut store = Store::default(); -+ let module = Module::new( -+ &store, -+ r#" -+(module -+ (memory (export "memory") 1) -+ (data $payload "shared") -+ (func (export "init") (param $dst i32) (param $src i32) (param $len i32) -+ local.get $dst -+ local.get $src -+ local.get $len -+ memory.init $payload) -+ (func (export "drop") -+ data.drop $payload)) -+"#, -+ ) -+ .unwrap(); -+ -+ let first = Instance::new(&mut store, &module, &Imports::new()).unwrap(); -+ let second = Instance::new(&mut store, &module, &Imports::new()).unwrap(); -+ let third = Instance::new(&mut store, &module, &Imports::new()).unwrap(); -+ let first_init: TypedFunction<(i32, i32, i32), ()> = -+ first.exports.get_typed_function(&store, "init").unwrap(); -+ let first_drop: TypedFunction<(), ()> = -+ first.exports.get_typed_function(&store, "drop").unwrap(); -+ let second_init: TypedFunction<(i32, i32, i32), ()> = -+ second.exports.get_typed_function(&store, "init").unwrap(); -+ let second_memory = second.exports.get_memory("memory").unwrap().clone(); -+ let third_init: TypedFunction<(i32, i32, i32), ()> = -+ third.exports.get_typed_function(&store, "init").unwrap(); -+ let third_memory = third.exports.get_memory("memory").unwrap().clone(); -+ -+ first_init.call(&mut store, 0, 0, 6).unwrap(); -+ first_drop.call(&mut store).unwrap(); -+ first_drop.call(&mut store).unwrap(); -+ -+ // A dropped segment has length zero. Its empty range remains valid while -+ // a non-empty read traps, including after repeated `data.drop`s. -+ first_init.call(&mut store, 65_536, 0, 0).unwrap(); -+ first_init -+ .call(&mut store, 16, 0, 1) -+ .expect_err("memory.init after data.drop must trap"); -+ -+ // Dropping one instance must not mutate the bytes shared by its module or -+ // the independent dropped-segment state of another instance. -+ second_init.call(&mut store, 16, 0, 6).unwrap(); -+ let mut second_bytes = [0_u8; 6]; -+ second_memory -+ .view(&store) -+ .read(16, &mut second_bytes) -+ .unwrap(); -+ assert_eq!(&second_bytes, b"shared"); -+ -+ // Surviving instances own the shared module bytes. They remain usable after -+ // both the original `Module` handle and the instance that dropped its -+ // segment have gone away. -+ drop(first_init); -+ drop(first_drop); -+ drop(first); -+ drop(module); -+ third_init.call(&mut store, 24, 0, 6).unwrap(); -+ let mut third_bytes = [0_u8; 6]; -+ third_memory -+ .view(&store) -+ .read(24, &mut third_bytes) -+ .unwrap(); -+ assert_eq!(&third_bytes, b"shared"); -+ -+ Ok(()) -+} -+ -+#[engine_test] -+fn passive_data_memory_init_preserves_contents_and_bounds() -> Result<(), String> { -+ let mut store = Store::default(); -+ let module = Module::new( -+ &store, -+ r#" -+(module -+ (memory (export "memory") 1) -+ (data $payload "shared") -+ (func (export "init") (param $dst i32) (param $src i32) (param $len i32) -+ local.get $dst -+ local.get $src -+ local.get $len -+ memory.init $payload) -+ (func (export "drop") -+ data.drop $payload)) -+"#, -+ ) -+ .unwrap(); -+ -+ let instance = Instance::new(&mut store, &module, &Imports::new()).unwrap(); -+ let init: TypedFunction<(i32, i32, i32), ()> = -+ instance.exports.get_typed_function(&store, "init").unwrap(); -+ let drop_data: TypedFunction<(), ()> = -+ instance.exports.get_typed_function(&store, "drop").unwrap(); -+ let memory = instance.exports.get_memory("memory").unwrap().clone(); -+ -+ init.call(&mut store, 8, 0, 6).unwrap(); -+ init.call(&mut store, 24, 1, 4).unwrap(); -+ let mut complete = [0_u8; 6]; -+ let mut partial = [0_u8; 4]; -+ memory.view(&store).read(8, &mut complete).unwrap(); -+ memory.view(&store).read(24, &mut partial).unwrap(); -+ assert_eq!(&complete, b"shared"); -+ assert_eq!(&partial, b"hare"); -+ -+ // Empty ranges at the exact source/destination ends are valid. -+ init.call(&mut store, 65_536, 6, 0).unwrap(); -+ init.call(&mut store, 0, 6, 0).unwrap(); -+ -+ // Source and destination ranges retain the WebAssembly bounds behavior. -+ init.call(&mut store, 0, 7, 0) -+ .expect_err("an empty source range beyond the segment must trap"); -+ init.call(&mut store, 0, 5, 2) -+ .expect_err("a non-empty source range beyond the segment must trap"); -+ init.call(&mut store, 65_537, 0, 0) -+ .expect_err("an empty destination range beyond memory must trap"); -+ init.call(&mut store, 65_535, 0, 2) -+ .expect_err("a non-empty destination range beyond memory must trap"); -+ -+ drop_data.call(&mut store).unwrap(); -+ -+ // A dropped segment behaves as an empty slice: its only valid source range -+ // is (0, 0), while destination bounds are still checked normally. -+ init.call(&mut store, 65_536, 0, 0).unwrap(); -+ init.call(&mut store, 0, 1, 0) -+ .expect_err("an offset beyond a dropped segment must trap"); -+ init.call(&mut store, 0, 0, 1) -+ .expect_err("a non-empty read from a dropped segment must trap"); -+ -+ Ok(()) -+} -+ - #[engine_test] - fn unit_native_function_env() -> Result<(), String> { - let mut store = Store::default(); -diff --git a/lib/api/tests/memory.rs b/lib/api/tests/memory.rs -index 52c3c9b..1b671e4 100644 ---- a/lib/api/tests/memory.rs -+++ b/lib/api/tests/memory.rs -@@ -2,7 +2,7 @@ use std::sync::{ - Arc, - atomic::{AtomicBool, Ordering}, - }; --use wasmer::{Instance, Memory, MemoryLocation, MemoryType, Module, Store, imports}; -+use wasmer::{Instance, Memory, MemoryLocation, MemoryType, Module, Pages, Store, imports}; - - #[test] - #[allow(unused_attributes)] -@@ -115,6 +115,137 @@ fn test_wasm_slice_issue_5444() { - )) - } - -+#[cfg(all(feature = "sys", unix))] -+#[test] -+fn private_file_remap_preserves_memory_base_growth_and_mapping_lifetime() -> anyhow::Result<()> { -+ use std::io::{Read, Seek, Write}; -+ -+ const WASM_PAGE_SIZE: usize = 64 * 1024; -+ let mut store = Store::default(); -+ let memory = Memory::new(&mut store, MemoryType::new(Pages(1), Some(Pages(3)), false))?; -+ let base = memory.view(&store).data_ptr(); -+ -+ let mut backing = tempfile::tempfile()?; -+ backing.write_all(&vec![0x7a; WASM_PAGE_SIZE])?; -+ backing.sync_all()?; -+ backing.rewind()?; -+ // SAFETY: the test has exclusive access to the store and memory, and the -+ // backing remains alive and unmodified until after remapping completes. -+ unsafe { -+ memory.remap_private_file_fixed(&mut store, 0, WASM_PAGE_SIZE, &backing, 0)?; -+ } -+ -+ assert_eq!(memory.view(&store).data_ptr(), base); -+ memory.view(&store).write(0, &[0x42])?; -+ let mut backing_byte = [0_u8; 1]; -+ backing.read_exact(&mut backing_byte)?; -+ assert_eq!(backing_byte, [0x7a]); -+ -+ drop(backing); -+ let mut mapped_byte = [0_u8; 1]; -+ memory.view(&store).read(0, &mut mapped_byte)?; -+ assert_eq!(mapped_byte, [0x42]); -+ -+ memory.grow(&mut store, Pages(1))?; -+ assert_eq!(memory.view(&store).data_ptr(), base); -+ assert_eq!(memory.view(&store).size(), Pages(2)); -+ memory.view(&store).write(WASM_PAGE_SIZE as u64, &[0x55])?; -+ let mut grown_byte = [0_u8; 1]; -+ memory -+ .view(&store) -+ .read(WASM_PAGE_SIZE as u64, &mut grown_byte)?; -+ assert_eq!(grown_byte, [0x55]); -+ Ok(()) -+} -+ -+#[cfg(feature = "llvm")] -+#[test] -+fn test_llvm_imported_dynamic_memory_bulk_ops() -> anyhow::Result<()> { -+ const PAGE_SIZE: u64 = 65536; -+ -+ fn read_memory( -+ memory: &Memory, -+ store: &Store, -+ offset: u64, -+ len: usize, -+ ) -> anyhow::Result> { -+ let mut bytes = vec![0; len]; -+ memory.view(store).read(offset, &mut bytes)?; -+ Ok(bytes) -+ } -+ -+ let engine: wasmer::Engine = wasmer::sys::EngineBuilder::new(wasmer::sys::LLVM::default()) -+ .engine() -+ .into(); -+ let mut store = Store::new(engine); -+ let wat = r#" -+ (module -+ (import "env" "memory" (memory 1 1)) -+ (func (export "copy") (param i32 i32 i32) -+ local.get 0 -+ local.get 1 -+ local.get 2 -+ memory.copy) -+ (func (export "fill") (param i32 i32 i32) -+ local.get 0 -+ local.get 1 -+ local.get 2 -+ memory.fill) -+ ) -+ "#; -+ let module = Module::new(&store, wat)?; -+ let memory = Memory::new(&mut store, MemoryType::new(Pages(1), Some(Pages(1)), false))?; -+ let imports = imports! { -+ "env" => { -+ "memory" => memory.clone(), -+ } -+ }; -+ let instance = Instance::new(&mut store, &module, &imports)?; -+ let copy: wasmer::TypedFunction<(i32, i32, i32), ()> = -+ instance.exports.get_typed_function(&store, "copy")?; -+ let fill: wasmer::TypedFunction<(i32, i32, i32), ()> = -+ instance.exports.get_typed_function(&store, "fill")?; -+ -+ memory.view(&store).write(8, b"abcdef")?; -+ copy.call(&mut store, 16, 8, 6)?; -+ assert_eq!(read_memory(&memory, &store, 16, 6)?, b"abcdef"); -+ -+ memory.view(&store).write(32, b"0123456789")?; -+ copy.call(&mut store, 34, 32, 8)?; -+ assert_eq!(read_memory(&memory, &store, 32, 10)?, b"0101234567"); -+ -+ fill.call(&mut store, 48, 0x7a, 4)?; -+ assert_eq!(read_memory(&memory, &store, 48, 4)?, b"zzzz"); -+ -+ memory.view(&store).write(PAGE_SIZE - 4, &[1, 2, 3, 4])?; -+ let result = fill -+ .call(&mut store, (PAGE_SIZE - 2) as i32, 0x41, 4) -+ .unwrap_err(); -+ assert_eq!( -+ result.to_trap(), -+ Some(wasmer_types::TrapCode::HeapAccessOutOfBounds) -+ ); -+ assert_eq!( -+ read_memory(&memory, &store, PAGE_SIZE - 4, 4)?, -+ &[1, 2, 3, 4] -+ ); -+ -+ memory.view(&store).write(PAGE_SIZE - 4, &[5, 6, 7, 8])?; -+ let result = copy -+ .call(&mut store, (PAGE_SIZE - 2) as i32, 8, 4) -+ .unwrap_err(); -+ assert_eq!( -+ result.to_trap(), -+ Some(wasmer_types::TrapCode::HeapAccessOutOfBounds) -+ ); -+ assert_eq!( -+ read_memory(&memory, &store, PAGE_SIZE - 4, 4)?, -+ &[5, 6, 7, 8] -+ ); -+ -+ Ok(()) -+} -+ - #[test] - fn test_wasm_memory_size() { - let mut store = Store::default(); -diff --git a/lib/api/tests/module.rs b/lib/api/tests/module.rs -index df286eb..6ad59ec 100644 ---- a/lib/api/tests/module.rs -+++ b/lib/api/tests/module.rs -@@ -4,6 +4,27 @@ use wasm_bindgen_test::*; - - use wasmer::*; - -+#[cfg(all(feature = "sys", feature = "compiler"))] -+use std::io::{Seek, Write}; -+#[cfg(all( -+ feature = "sys", -+ feature = "compiler", -+ target_os = "linux", -+ target_arch = "x86_64" -+))] -+use std::os::unix::fs::PermissionsExt; -+#[cfg(all( -+ feature = "sys", -+ feature = "compiler", -+ target_os = "linux", -+ target_arch = "x86_64" -+))] -+use wasmer::sys::CodeMemoryPolicy; -+#[cfg(all(feature = "sys", feature = "compiler"))] -+use wasmer::sys::{NativeEngineExt, OwnedBuffer}; -+#[cfg(all(feature = "sys", feature = "compiler"))] -+use wasmer_types::MetadataHeader; -+ - #[cfg(unix)] - use std::ffi::OsStr; - #[cfg(unix)] -@@ -32,6 +53,367 @@ fn module_set_name() -> Result<(), String> { - Ok(()) - } - -+#[cfg(all(feature = "sys", feature = "compiler"))] -+fn mapped_artifact(bytes: &[u8]) -> Result { -+ let mut artifact_file = tempfile::tempfile().map_err(|error| error.to_string())?; -+ artifact_file -+ .write_all(bytes) -+ .map_err(|error| error.to_string())?; -+ artifact_file.rewind().map_err(|error| error.to_string())?; -+ OwnedBuffer::from_file(&artifact_file).map_err(|error| error.to_string()) -+} -+ -+#[test] -+#[cfg(all(feature = "sys", feature = "compiler"))] -+fn serialized_artifact_inspector_returns_embedded_hash() -> Result<(), String> { -+ let compiler_store = Store::default(); -+ let compiled = Module::new( -+ &compiler_store, -+ r#"(module (func (export "answer") (result i32) i32.const 42))"#, -+ ) -+ .map_err(|error| error.to_string())?; -+ let expected_hash = compiled -+ .info() -+ .hash -+ .ok_or_else(|| "compiled module is missing its hash".to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ let mapping = mapped_artifact(&serialized)?; -+ -+ let engine = Engine::headless(); -+ let inspected_hash = engine -+ .inspect_serialized_artifact(&mapping) -+ .map_err(|error| error.to_string())?; -+ -+ assert_eq!(inspected_hash, expected_hash); -+ Ok(()) -+} -+ -+#[test] -+#[cfg(all(feature = "sys", feature = "compiler"))] -+fn serialized_artifact_inspector_checks_magic_abi_and_archive() -> Result<(), String> { -+ let compiler_store = Store::default(); -+ let compiled = Module::new(&compiler_store, "(module)").map_err(|error| error.to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ let engine = Engine::headless(); -+ -+ let mut wrong_magic = serialized.to_vec(); -+ wrong_magic[0] ^= 0xff; -+ let error = engine -+ .inspect_serialized_artifact(&mapped_artifact(&wrong_magic)?) -+ .expect_err("an invalid universal magic must be rejected"); -+ assert!(matches!(error, DeserializeError::Incompatible(_))); -+ -+ let mut wrong_abi = serialized.to_vec(); -+ const UNIVERSAL_MAGIC_LEN: usize = 16; -+ const METADATA_VERSION_OFFSET: usize = UNIVERSAL_MAGIC_LEN + 8; -+ wrong_abi[METADATA_VERSION_OFFSET..METADATA_VERSION_OFFSET + std::mem::size_of::()] -+ .copy_from_slice(&(MetadataHeader::CURRENT_VERSION + 1).to_ne_bytes()); -+ let error = engine -+ .inspect_serialized_artifact(&mapped_artifact(&wrong_abi)?) -+ .expect_err("an incompatible metadata ABI must be rejected"); -+ assert!(matches!(error, DeserializeError::Incompatible(_))); -+ -+ let mut invalid_archive = b"wasmer-universal".to_vec(); -+ invalid_archive.extend(MetadataHeader::new(1).into_bytes()); -+ invalid_archive.push(0); -+ let error = engine -+ .inspect_serialized_artifact(&mapped_artifact(&invalid_archive)?) -+ .expect_err("a malformed rkyv archive must be rejected"); -+ assert!(matches!(error, DeserializeError::CorruptedBinary(_))); -+ -+ Ok(()) -+} -+ -+#[test] -+#[cfg(all(feature = "sys", feature = "compiler"))] -+fn detached_artifact_passive_data_has_instance_local_drop_state() -> Result<(), String> { -+ let compiler_store = Store::default(); -+ let compiled = Module::new( -+ &compiler_store, -+ r#"(module -+ (memory (export "memory") 1) -+ (data $payload "artifact") -+ (func (export "init") (param $dst i32) (param $src i32) (param $len i32) -+ local.get $dst -+ local.get $src -+ local.get $len -+ memory.init $payload) -+ (func (export "drop") -+ data.drop $payload) -+ )"#, -+ ) -+ .map_err(|error| error.to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ -+ let engine = Engine::headless(); -+ let detached = -+ unsafe { engine.deserialize_from_mmapped_buffer_detached(mapped_artifact(&serialized)?) } -+ .map_err(|error| error.to_string())?; -+ -+ let mut store = Store::new(engine); -+ let first = -+ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; -+ let second = -+ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; -+ let first_init = first -+ .exports -+ .get_typed_function::<(i32, i32, i32), ()>(&store, "init") -+ .map_err(|error| error.to_string())?; -+ let first_drop = first -+ .exports -+ .get_typed_function::<(), ()>(&store, "drop") -+ .map_err(|error| error.to_string())?; -+ let second_init = second -+ .exports -+ .get_typed_function::<(i32, i32, i32), ()>(&store, "init") -+ .map_err(|error| error.to_string())?; -+ let second_memory = second -+ .exports -+ .get_memory("memory") -+ .map_err(|error| error.to_string())? -+ .clone(); -+ -+ first_drop -+ .call(&mut store) -+ .map_err(|error| error.to_string())?; -+ first_init -+ .call(&mut store, 0, 0, 1) -+ .expect_err("the first detached instance must observe its data.drop"); -+ -+ // Instance module handles keep the detached metadata, including passive -+ // bytes, alive independently of the original detached `Module` handle. -+ drop(detached); -+ second_init -+ .call(&mut store, 32, 0, 8) -+ .map_err(|error| error.to_string())?; -+ let mut copied = [0_u8; 8]; -+ second_memory -+ .view(&store) -+ .read(32, &mut copied) -+ .map_err(|error| error.to_string())?; -+ assert_eq!(&copied, b"artifact"); -+ -+ Ok(()) -+} -+ -+#[test] -+#[cfg(all( -+ feature = "sys", -+ feature = "compiler", -+ feature = "cranelift", -+ any(target_arch = "x86_64", target_arch = "aarch64") -+))] -+fn serialized_artifact_inspector_checks_native_cpu_features() -> Result<(), String> { -+ use wasmer::sys::{CpuFeature, Cranelift, Features, Target, Triple}; -+ -+ let compiler_store = Store::default(); -+ let compiled = Module::new(&compiler_store, "(module)").map_err(|error| error.to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ let mapping = mapped_artifact(&serialized)?; -+ -+ let inspection_engine = ::new( -+ Box::new(Cranelift::default()), -+ Target::new(Triple::host(), CpuFeature::set()), -+ Features::default(), -+ ); -+ let error = inspection_engine -+ .inspect_serialized_artifact(&mapping) -+ .expect_err("an artifact requiring unavailable CPU features must be rejected"); -+ -+ assert!(matches!(error, DeserializeError::Incompatible(_))); -+ Ok(()) -+} -+ -+#[test] -+#[cfg(all(feature = "sys", feature = "compiler"))] -+fn detached_mmapped_module_executes_without_retaining_serializable_state() -> Result<(), String> { -+ let compiler_store = Store::default(); -+ let compiled = Module::new( -+ &compiler_store, -+ r#"(module -+ (memory 1) -+ (data (i32.const 1024) "sealed") -+ (func (export "answer") (result i32) i32.const 42) -+ )"#, -+ ) -+ .map_err(|error| error.to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ -+ let mapping = mapped_artifact(&serialized)?; -+ -+ let engine = Engine::headless(); -+ let detached = unsafe { engine.deserialize_from_mmapped_buffer_detached(mapping) } -+ .map_err(|error| error.to_string())?; -+ assert!( -+ detached.serialize().is_err(), -+ "a detached runtime module must not retain state only needed for re-serialization" -+ ); -+ -+ let mut store = Store::new(engine); -+ let instance = -+ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; -+ let answer = instance -+ .exports -+ .get_typed_function::<(), i32>(&store, "answer") -+ .map_err(|error| error.to_string())?; -+ assert_eq!( -+ answer.call(&mut store).map_err(|error| error.to_string())?, -+ 42 -+ ); -+ Ok(()) -+} -+ -+#[test] -+#[cfg(all( -+ feature = "sys", -+ feature = "compiler", -+ target_os = "linux", -+ target_arch = "x86_64" -+))] -+fn abandoned_pending_detached_activation_releases_strict_code_memory() -> Result<(), String> { -+ let compiler_store = Store::default(); -+ let compiled = Module::new( -+ &compiler_store, -+ r#"(module (func (export "answer") (result i32) i32.const 42))"#, -+ ) -+ .map_err(|error| error.to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ -+ let directory = tempfile::Builder::new() -+ .prefix("wasmer-pending-code-memory-module-test-") -+ .tempdir_in("/var/tmp") -+ .map_err(|error| error.to_string())?; -+ std::fs::set_permissions(directory.path(), std::fs::Permissions::from_mode(0o700)) -+ .map_err(|error| error.to_string())?; -+ let policy = CodeMemoryPolicy::strict_linux_x86_64_file_backed(directory.path())?; -+ let mut engine = Engine::headless(); -+ engine.set_code_memory_policy(policy)?; -+ -+ let baseline = engine.as_sys().code_memory_allocation_count(); -+ let pending = unsafe { -+ engine.deserialize_from_mmapped_buffer_detached_pending(mapped_artifact(&serialized)?) -+ } -+ .map_err(|error| error.to_string())?; -+ assert_eq!(engine.as_sys().code_memory_allocation_count(), baseline + 1); -+ assert!(pending.module_hash().is_some()); -+ -+ drop(pending); -+ assert_eq!(engine.as_sys().code_memory_allocation_count(), baseline); -+ Ok(()) -+} -+ -+#[test] -+#[cfg(all( -+ feature = "sys", -+ feature = "compiler", -+ target_os = "linux", -+ target_arch = "x86_64" -+))] -+fn detached_module_executes_from_strict_relocated_regular_file_code_memory() -> Result<(), String> { -+ let compiler_store = Store::default(); -+ let compiled = Module::new( -+ &compiler_store, -+ r#"(module $strict_detached_frames -+ (func $inner (unreachable)) -+ (func (export "answer") (result i32) i32.const 42) -+ (func (export "run") (call $inner)) -+ )"#, -+ ) -+ .map_err(|error| error.to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ -+ let directory = tempfile::Builder::new() -+ .prefix("wasmer-strict-code-memory-module-test-") -+ .tempdir_in("/var/tmp") -+ .map_err(|error| error.to_string())?; -+ std::fs::set_permissions(directory.path(), std::fs::Permissions::from_mode(0o700)) -+ .map_err(|error| error.to_string())?; -+ let policy = CodeMemoryPolicy::strict_linux_x86_64_file_backed(directory.path())?; -+ let mut engine = Engine::headless(); -+ engine.set_code_memory_policy(policy)?; -+ let detached = -+ unsafe { engine.deserialize_from_mmapped_buffer_detached(mapped_artifact(&serialized)?) } -+ .map_err(|error| error.to_string())?; -+ -+ let mut store = Store::new(engine); -+ let instance = -+ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; -+ let answer = instance -+ .exports -+ .get_typed_function::<(), i32>(&store, "answer") -+ .map_err(|error| error.to_string())?; -+ assert_eq!( -+ answer.call(&mut store).map_err(|error| error.to_string())?, -+ 42 -+ ); -+ -+ let run = instance -+ .exports -+ .get_typed_function::<(), ()>(&store, "run") -+ .map_err(|error| error.to_string())?; -+ let error = run -+ .call(&mut store) -+ .expect_err("strict detached code must trap"); -+ let trace = error.trace(); -+ assert_eq!(trace.len(), 2, "strict code must retain both trap frames"); -+ assert_eq!(trace[0].module_name(), "strict_detached_frames"); -+ assert_eq!(trace[0].func_index(), 0); -+ assert_eq!(trace[0].function_name(), Some("inner")); -+ assert_eq!(trace[1].module_name(), "strict_detached_frames"); -+ assert_eq!(trace[1].func_index(), 2); -+ assert_eq!(trace[1].function_name(), None); -+ assert!(error.message().contains("unreachable")); -+ Ok(()) -+} -+ -+#[test] -+#[cfg(all(feature = "sys", feature = "compiler"))] -+#[cfg_attr(target_env = "musl", ignore)] -+fn detached_mmapped_module_retains_trap_frame_metadata() -> Result<(), String> { -+ let compiler_store = Store::default(); -+ let compiled = Module::new( -+ &compiler_store, -+ r#"(module $detached_frames -+ (func $inner (unreachable)) -+ (func (export "run") (call $inner)) -+ )"#, -+ ) -+ .map_err(|error| error.to_string())?; -+ let serialized = compiled.serialize().map_err(|error| error.to_string())?; -+ -+ let engine = Engine::headless(); -+ let detached = -+ unsafe { engine.deserialize_from_mmapped_buffer_detached(mapped_artifact(&serialized)?) } -+ .map_err(|error| error.to_string())?; -+ -+ let mut store = Store::new(engine); -+ let instance = -+ Instance::new(&mut store, &detached, &imports! {}).map_err(|error| error.to_string())?; -+ let run = instance -+ .exports -+ .get_typed_function::<(), ()>(&store, "run") -+ .map_err(|error| error.to_string())?; -+ let error = run -+ .call(&mut store) -+ .expect_err("the detached module must trap"); -+ let trace = error.trace(); -+ -+ assert_eq!( -+ trace.len(), -+ 2, -+ "detached trap metadata must retain both frames" -+ ); -+ assert_eq!(trace[0].module_name(), "detached_frames"); -+ assert_eq!(trace[0].func_index(), 0); -+ assert_eq!(trace[0].function_name(), Some("inner")); -+ assert_eq!(trace[1].module_name(), "detached_frames"); -+ assert_eq!(trace[1].func_index(), 1); -+ assert_eq!(trace[1].function_name(), None); -+ assert!(error.message().contains("unreachable")); -+ -+ Ok(()) -+} -+ - #[engine_test] - fn imports() -> Result<(), String> { - let store = Store::default(); -diff --git a/lib/cli/Cargo.toml b/lib/cli/Cargo.toml -index aceecf7..6e710d3 100644 ---- a/lib/cli/Cargo.toml -+++ b/lib/cli/Cargo.toml -@@ -34,7 +34,8 @@ default = [ - "wast", - "journal", - "wasmer-artifact-create", -- "static-artifact-create" -+ "static-artifact-create", -+ "cli-signal-watcher" - ] - - # # Tun-tap client for connecting to Wasmer Edge VPNs -@@ -49,7 +50,8 @@ default = [ - journal = ["wasmer-wasix/journal"] - backend = [] - coredump = ["wasm-coredump-builder"] --sys = ["compiler", "dep:wasmer-vm"] -+native-runtime = ["dep:wasmer-vm", "wasmer/sys"] -+sys = ["compiler", "native-runtime"] - v8 = ["backend", "wasmer/v8"] - wast = ["wasmer-wast"] - host-net = ["virtual-net/host-net"] -@@ -92,10 +94,18 @@ disable-all-logging = [ - "wasmer-wasix/disable-all-logging", - "log/release_max_level_off", - ] --headless = ["dep:wasmer-vm", "wasmer/sys"] --headless-minimal = ["headless", "disable-all-logging"] -+headless = ["native-runtime"] -+headless-minimal = [ -+ "headless", -+ "disable-all-logging", -+ "oliphaunt-wasix-postmaster-executor/memory-profile-core", -+] - telemetry = [] - napi-v8 = ["dep:wasmer-napi"] -+# Preserve the interactive full CLI's historical Ctrl+C exit watcher without -+# pulling Tokio's process-global signal machinery into the compiler-free -+# headless carrier. -+cli-signal-watcher = ["tokio/signal"] - - # Optional - enable-serde = [ -@@ -134,6 +144,9 @@ wasmer-types = { version = "=7.2.0-alpha.2", path = "../types", features = [ - "detect-wasm-features", - ] } - wasmer-napi = { version = "0.702.0-alpha.2", path = "../napi", optional = true } -+oliphaunt-wasix-postmaster-executor = { version = "=7.2.0-alpha.2", path = "../oliphaunt-wasix-postmaster-executor", default-features = false, features = [ -+ "compat-cache-dir", -+] } - virtual-fs = { version = "0.702.0-alpha.2", path = "../virtual-fs", default-features = false, features = [ - "host-fs", - ] } -@@ -281,6 +294,10 @@ pretty_assertions.workspace = true - - [target.'cfg(target_os = "windows")'.dependencies] - colored.workspace = true -+windows-sys = { workspace = true, features = [ -+ "Win32_Foundation", -+ "Win32_Storage_FileSystem", -+] } - - [package.metadata.binstall] - pkg-fmt = "tgz" -diff --git a/lib/cli/src/backend.rs b/lib/cli/src/backend.rs -index 1c37513..17fe4c9 100644 ---- a/lib/cli/src/backend.rs -+++ b/lib/cli/src/backend.rs -@@ -12,7 +12,7 @@ use std::sync::Arc; - use std::{path::PathBuf, str::FromStr}; - - use anyhow::{Context, Result, bail}; --#[cfg(feature = "sys")] -+#[cfg(feature = "native-runtime")] - use wasmer::sys::*; - use wasmer::*; - use wasmer_types::{Features, target::Target}; -@@ -297,7 +297,6 @@ impl RuntimeOptions { - filtered_backends.first().unwrap().get_engine(target, self) - } - -- #[cfg(feature = "compiler")] - /// Get the enabled Wasm features. - pub fn get_features(&self, default_features: &Features) -> Result { - if self.features.all { -@@ -323,10 +322,33 @@ impl RuntimeOptions { - if self.features.reference_types { - result.reference_types(true); - } -+ if self.features.tail_call { -+ result.tail_call(true); -+ } -+ if self.features.module_linking { -+ result.module_linking(true); -+ } -+ if self.features.multi_memory { -+ result.multi_memory(true); -+ } -+ if self.features.memory64 { -+ result.memory64(true); -+ } -+ if self.features.exceptions { -+ result.exceptions(true); -+ } -+ if self.features.relaxed_simd { -+ result.relaxed_simd(true); -+ } -+ if self.features.extended_const { -+ result.extended_const(true); -+ } -+ if self.features.wide_arithmetic { -+ result.wide_arithmetic(true); -+ } - Ok(result) - } - -- #[cfg(feature = "compiler")] - /// Get a copy of the default features with user-configured options - pub fn get_configured_features(&self) -> Result { - let features = Features::default(); -@@ -531,6 +553,8 @@ impl BackendType { - Self::Singlepass, - #[cfg(feature = "v8")] - Self::V8, -+ #[cfg(all(feature = "headless", not(feature = "compiler")))] -+ Self::Headless, - ] - } - -@@ -644,7 +668,11 @@ impl BackendType { - } - #[cfg(feature = "v8")] - Self::V8 => Ok(wasmer::v8::V8::new().into()), -- Self::Headless => bail!("Headless is not a valid runtime to instantiate directly"), -+ Self::Headless => Ok(EngineBuilder::headless() -+ .set_features(Some(runtime_opts.get_configured_features()?)) -+ .set_target(Some(target.clone())) -+ .engine() -+ .into()), - #[allow(unreachable_patterns)] - _ => bail!("Unsupported backend type"), - } -@@ -663,7 +691,10 @@ impl BackendType { - Self::LLVM => wasmer::BackendKind::LLVM, - #[cfg(feature = "v8")] - Self::V8 => wasmer::BackendKind::V8, -- Self::Headless => return false, // Headless can't compile -+ // A headless engine does not compile the input module. Feature -+ // compatibility is enforced while deserializing the selected AOT -+ // artifact, so feature detection must not filter it out here. -+ Self::Headless => return true, - #[allow(unreachable_patterns)] - _ => return false, - }; -diff --git a/lib/cli/src/commands/mod.rs b/lib/cli/src/commands/mod.rs -index 5896d6f..eeeaaa9 100644 ---- a/lib/cli/src/commands/mod.rs -+++ b/lib/cli/src/commands/mod.rs -@@ -32,6 +32,7 @@ mod validate; - #[cfg(feature = "wast")] - mod wast; - use itertools::Itertools; -+#[cfg(feature = "cli-signal-watcher")] - use std::io::IsTerminal as _; - use tokio::task::JoinHandle; - -@@ -78,29 +79,38 @@ pub(crate) trait AsyncCliCommand: Send + Sync { - &self, - done: tokio::sync::oneshot::Receiver<()>, - ) -> Option>> { -- if std::io::stdin().is_terminal() { -- return Some(tokio::task::spawn(async move { -- tokio::select! { -- _ = done => {} -- -- _ = tokio::signal::ctrl_c() => { -- let term = console::Term::stdout(); -- let _ = term.show_cursor(); -- // https://learn.microsoft.com/en-us/cpp/c-runtime-library/signal-constants -- #[cfg(target_os = "windows")] -- std::process::exit(3); -- -- // POSIX compliant OSs: 128 + SIGINT (2) -- #[cfg(not(target_os = "windows"))] -- std::process::exit(130); -+ #[cfg(not(feature = "cli-signal-watcher"))] -+ { -+ let _ = done; -+ return None; -+ } -+ -+ #[cfg(feature = "cli-signal-watcher")] -+ { -+ if std::io::stdin().is_terminal() { -+ return Some(tokio::task::spawn(async move { -+ tokio::select! { -+ _ = done => {} -+ -+ _ = tokio::signal::ctrl_c() => { -+ let term = console::Term::stdout(); -+ let _ = term.show_cursor(); -+ // https://learn.microsoft.com/en-us/cpp/c-runtime-library/signal-constants -+ #[cfg(target_os = "windows")] -+ std::process::exit(3); -+ -+ // POSIX compliant OSs: 128 + SIGINT (2) -+ #[cfg(not(target_os = "windows"))] -+ std::process::exit(130); -+ } - } -- } - -- Ok::<(), anyhow::Error>(()) -- })); -- } -+ Ok::<(), anyhow::Error>(()) -+ })); -+ } - -- None -+ None -+ } - } - } - -diff --git a/lib/cli/src/commands/run/mod.rs b/lib/cli/src/commands/run/mod.rs -index cc9a3b0..8ffcb13 100644 ---- a/lib/cli/src/commands/run/mod.rs -+++ b/lib/cli/src/commands/run/mod.rs -@@ -20,15 +20,16 @@ use std::{ - time::{Duration, SystemTime, UNIX_EPOCH}, - }; - --use anyhow::{Context, Error, anyhow, bail}; -+use anyhow::{Context, Error, anyhow, bail, ensure}; - use clap::{Parser, ValueEnum}; - use colored::Colorize; - use futures::future::BoxFuture; - use indicatif::{MultiProgress, ProgressBar}; -+use oliphaunt_wasix_postmaster_executor::sealed; - use once_cell::sync::Lazy; - use tempfile::NamedTempFile; - use url::Url; --#[cfg(feature = "sys")] -+#[cfg(feature = "native-runtime")] - use wasmer::sys::NativeEngineExt; - use wasmer::{ - AsStoreMut, DeserializeError, Engine, Function, Imports, Instance, Module, RuntimeError, Store, -@@ -45,8 +46,10 @@ use wasmer_types::ModuleHash; - - #[cfg(feature = "journal")] - use wasmer_wasix::journal::{LogFileJournal, SnapshotTrigger}; -+#[cfg(any(unix, windows))] -+use wasmer_wasix::os::task::HostLifecycleSupervisor; - use wasmer_wasix::{ -- Runtime, SpawnError, WasiError, -+ ResourceLimits, Runtime, SpawnError, WasiError, - bin_factory::{BinaryPackage, BinaryPackageCommand}, - journal::CompactingLogFileJournal, - runners::{ -@@ -57,11 +60,8 @@ use wasmer_wasix::{ - wcgi::{self, AbortHandle, NoOpWcgiCallbacks, WcgiRunner}, - }, - runtime::{ -- OverriddenRuntime, -- module_cache::{CacheError, HashedModuleData}, -- package_loader::PackageLoader, -- resolver::QueryError, -- task_manager::VirtualTaskManagerExt, -+ OverriddenRuntime, module_cache::ModuleCache, package_loader::PackageLoader, -+ resolver::QueryError, task_manager::VirtualTaskManagerExt, - }, - }; - use webc::Container; -@@ -76,10 +76,54 @@ use crate::{ - }; - - use self::{ -- package_source::CliPackageSource, runtime::MonitoringRuntime, target::ExecutableTarget, -+ package_source::CliPackageSource, -+ runtime::{CliTokioRuntimePolicy, MonitoringRuntime}, -+ target::ExecutableTarget, - }; - - const TICK: Duration = Duration::from_millis(250); -+const WASIX_STACK_RLIMIT_DIVISOR: u64 = 8; -+ -+#[cfg(any(unix, windows))] -+fn attach_cli_host_lifecycle(runner: &mut WasiRunner) -> Result<(), Error> { -+ let supervisor = HostLifecycleSupervisor::install() -+ .context("Unable to install exclusive CLI host lifecycle supervision")?; -+ runner.with_host_lifecycle_supervisor(Arc::new(supervisor)); -+ Ok(()) -+} -+ -+#[cfg(not(any(unix, windows)))] -+fn attach_cli_host_lifecycle(_runner: &mut WasiRunner) -> Result<(), Error> { -+ Ok(()) -+} -+ -+fn conservative_stack_resource_limit(stack_size: usize) -> u64 { -+ ((stack_size as u64) / WASIX_STACK_RLIMIT_DIVISOR).max(1) -+} -+ -+fn resource_limits_for_engine(engine: &Engine, stack_size: Option) -> ResourceLimits { -+ ResourceLimits { -+ stack: { -+ #[cfg(feature = "native-runtime")] -+ { -+ if engine.is_sys() { -+ if let Some(stack_size) = stack_size { -+ wasmer_vm::set_stack_size(stack_size); -+ } -+ Some(conservative_stack_resource_limit( -+ wasmer_vm::get_stack_size(), -+ )) -+ } else { -+ None -+ } -+ } -+ #[cfg(not(feature = "native-runtime"))] -+ { -+ None -+ } -+ }, -+ } -+} - - /// The unstable `wasmer run` subcommand. - #[derive(Debug, Parser)] -@@ -95,6 +139,10 @@ pub struct Run { - /// Set the default stack size (default is 1048576) - #[clap(long = "stack-size")] - stack_size: Option, -+ /// Run a verified, precompiled-only executable closure. -+ #[cfg(feature = "headless")] -+ #[clap(long = "sealed-module-manifest", value_name = "PATH")] -+ sealed_module_manifest: Option, - /// The entrypoint module for webc packages. - #[clap(short, long, aliases = &["command", "command-name"])] - entrypoint: Option, -@@ -115,6 +163,17 @@ pub struct Run { - } - - impl Run { -+ fn sealed_manifest_path(&self) -> Option<&Path> { -+ #[cfg(feature = "headless")] -+ { -+ self.sealed_module_manifest.as_deref() -+ } -+ #[cfg(not(feature = "headless"))] -+ { -+ None -+ } -+ } -+ - #[cfg(feature = "napi-v8")] - fn module_needs_napi(module: &Module) -> bool { - let (napi_version, napi_extension_version) = wasmer_napi::module_needs_napi(module); -@@ -187,9 +246,28 @@ impl Run { - - pb.set_message("Initializing the WebAssembly VM"); - -- let runtime = tokio::runtime::Builder::new_multi_thread() -- .enable_all() -- .build()?; -+ let sealed_manifest_path = self.sealed_manifest_path().map(Path::to_path_buf); -+ let sealed_manifest = sealed_manifest_path -+ .as_deref() -+ .map(|manifest| { -+ let input = match &self.input { -+ CliPackageSource::File(path) => path.as_path(), -+ _ => bail!("--sealed-module-manifest requires a local executable file input"), -+ }; -+ sealed::prepare(manifest, input) -+ }) -+ .transpose()?; -+ let sealed_runtime_identity = sealed_manifest -+ .as_ref() -+ .and_then(sealed::PreparedSealedManifest::runtime_identity); -+ let tokio_runtime_policy = CliTokioRuntimePolicy::select(sealed_runtime_identity); -+ tracing::info!( -+ policy_id = tokio_runtime_policy.id(), -+ workload_id = ?tokio_runtime_policy.workload_id(), -+ configured_worker_threads = ?tokio_runtime_policy.configured_worker_threads(), -+ "Selected CLI Tokio runtime policy" -+ ); -+ let runtime = tokio_runtime_policy.build()?; - let handle = runtime.handle().clone(); - - // Check for the preferred webc version. -@@ -210,7 +288,9 @@ impl Run { - - // Try to detect WebAssembly features before selecting a backend - tracing::info!("Input source: {:?}", self.input); -- if let CliPackageSource::File(path) = &self.input { -+ if sealed_manifest_path.is_none() -+ && let CliPackageSource::File(path) = &self.input -+ { - tracing::info!("Input file path: {}", path.display()); - - // Try to read and detect any file that exists, regardless of extension -@@ -275,20 +355,24 @@ impl Run { - let engine_kind = engine.deterministic_id(); - tracing::info!("Executing on backend {engine_kind:?}"); - -- #[cfg(feature = "sys")] -- if engine.is_sys() && self.stack_size.is_some() { -- wasmer_vm::set_stack_size(self.stack_size.unwrap()); -- } -+ let sealed_modules = sealed_manifest -+ .map(|manifest| sealed::load(manifest, &engine)) -+ .transpose()?; -+ let sealed_module_cache = sealed_modules -+ .as_ref() -+ .map(|sealed| sealed.module_cache.clone() as Arc); - -- let engine = engine.clone(); -+ let resource_limits = resource_limits_for_engine(&engine, self.stack_size); - - let runtime = self.wasi.prepare_runtime( -- engine, -+ engine.clone(), - &self.env, - &capabilities::get_capability_cache_path(&self.env, &self.input)?, - runtime, - preferred_webc_version, - self.rt.compiler_debug_dir.is_some(), -+ resource_limits, -+ sealed_module_cache, - )?; - - // This is a slow operation, so let's temporarily wrap the runtime with -@@ -301,7 +385,20 @@ impl Run { - let runtime: Arc = monitoring_runtime.runtime.clone(); - let monitoring_runtime: Arc = monitoring_runtime; - -- let target = self.input.resolve_target(&monitoring_runtime, &pb)?; -+ let (target, sealed_executables) = match sealed_modules { -+ Some(sealed) => ( -+ ExecutableTarget::WebAssembly { -+ module: sealed.module, -+ module_hash: sealed.module_hash, -+ path: sealed.path, -+ }, -+ sealed.executables, -+ ), -+ None => ( -+ self.input.resolve_target(&monitoring_runtime, &pb)?, -+ Vec::new(), -+ ), -+ }; - - if let ExecutableTarget::Package(ref pkg) = target { - self.wasi -@@ -320,7 +417,13 @@ impl Run { - module, - module_hash, - path, -- } => self.execute_wasm(&path, module, module_hash, runtime.clone()), -+ } => self.execute_wasm( -+ &path, -+ module, -+ module_hash, -+ runtime.clone(), -+ sealed_executables, -+ ), - ExecutableTarget::Package(pkg) => { - // Check if we should update the engine based on the WebC package features - if let Some(cmd) = pkg.get_entrypoint_command() -@@ -359,15 +462,17 @@ impl Run { - &self.env, - &self.input, - )?; -+ let new_resource_limits = -+ resource_limits_for_engine(&new_engine, self.stack_size); - let new_runtime = self.wasi.prepare_runtime( - new_engine, - &self.env, - &capability_cache_path, -- tokio::runtime::Builder::new_multi_thread() -- .enable_all() -- .build()?, -+ tokio_runtime_policy.build()?, - preferred_webc_version, - self.rt.compiler_debug_dir.is_some(), -+ new_resource_limits, -+ None, - )?; - - let new_runtime = Arc::new(MonitoringRuntime::new( -@@ -405,9 +510,10 @@ impl Run { - module: Module, - module_hash: ModuleHash, - runtime: Arc, -+ sealed_executables: Vec<(String, ModuleHash)>, - ) -> Result<(), Error> { - if wasmer_wasix::is_wasi_module(&module) || wasmer_wasix::is_wasix_module(&module) { -- self.execute_wasi_module(path, module, module_hash, runtime) -+ self.execute_wasi_module(path, module, module_hash, runtime, sealed_executables) - } else { - self.execute_pure_wasm_module(&module) - } -@@ -495,6 +601,7 @@ impl Run { - - // Assume webcs are always WASIX - let mut runner = self.build_wasi_runner(&runtime, true)?; -+ attach_cli_host_lifecycle(&mut runner)?; - #[cfg(feature = "napi-v8")] - self.configure_wasi_runner_for_napi(&module, &mut runner); - Runner::run_command(&mut runner, command_name, pkg, runtime) -@@ -696,12 +803,17 @@ impl Run { - module: Module, - module_hash: ModuleHash, - runtime: Arc, -+ sealed_executables: Vec<(String, ModuleHash)>, - ) -> Result<(), Error> { - let program_name = wasm_path.display().to_string(); - let runtime = self.maybe_wrap_runtime_with_napi(&module, runtime)?; - - let mut runner = - self.build_wasi_runner(&runtime, wasmer_wasix::is_wasix_module(&module))?; -+ attach_cli_host_lifecycle(&mut runner)?; -+ for (guest_path, sealed_hash) in sealed_executables { -+ runner.with_sealed_module(guest_path, sealed_hash, None); -+ } - self.configure_wasi_runner_for_napi(&module, &mut runner); - runner.run_wasm( - RuntimeOrEngine::Runtime(runtime), -@@ -905,3 +1017,34 @@ fn get_exit_code( - - None - } -+ -+#[cfg(all(test, feature = "headless", not(feature = "compiler")))] -+mod tests { -+ use super::*; -+ -+ struct StackSizeRestore(usize); -+ -+ impl Drop for StackSizeRestore { -+ fn drop(&mut self) { -+ wasmer_vm::set_stack_size(self.0); -+ } -+ } -+ -+ #[test] -+ fn headless_stack_size_sets_vm_and_guest_limit() { -+ let run = Run::try_parse_from(["run", "--stack-size", "33554432", "probe.wasm"]) -+ .expect("headless run arguments should parse"); -+ let engine = run -+ .rt -+ .get_engine(&Target::default()) -+ .expect("headless engine should be available"); -+ assert_eq!(engine.deterministic_id(), "engine-headless"); -+ -+ let previous_stack_size = wasmer_vm::get_stack_size(); -+ let _restore = StackSizeRestore(previous_stack_size); -+ let limits = resource_limits_for_engine(&engine, run.stack_size); -+ -+ assert_eq!(wasmer_vm::get_stack_size(), 33_554_432); -+ assert_eq!(limits.stack, Some(4_194_304)); -+ } -+} -diff --git a/lib/cli/src/commands/run/runtime.rs b/lib/cli/src/commands/run/runtime.rs -index c562c24..d06fcc8 100644 ---- a/lib/cli/src/commands/run/runtime.rs -+++ b/lib/cli/src/commands/run/runtime.rs -@@ -1,6 +1,6 @@ - //! Provides CLI-specific Wasix components. - --use std::{sync::Arc, time::Duration}; -+use std::{io, sync::Arc, time::Duration}; - - use anyhow::Error; - use futures::future::BoxFuture; -@@ -23,6 +23,64 @@ use wasmer_wasix::{ - }; - use webc::Container; - -+use oliphaunt_wasix_postmaster_executor::sealed::SealedRuntimeIdentity; -+ -+const GENERIC_TOKIO_RUNTIME_POLICY_ID: &str = "wasmer.cli.tokio.default.v1"; -+pub(super) const SEALED_POSTMASTER_TOKIO_RUNTIME_POLICY_ID: &str = -+ "oliphaunt.wasix-postmaster.tokio.sealed-postmaster-2-worker.v1"; -+const SEALED_POSTMASTER_TOKIO_WORKER_THREADS: usize = 2; -+ -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub(super) enum CliTokioRuntimePolicy { -+ Generic, -+ SealedWasixPostmaster(SealedRuntimeIdentity), -+} -+ -+impl CliTokioRuntimePolicy { -+ pub(super) const fn select(identity: Option) -> Self { -+ match identity { -+ Some(identity) => Self::SealedWasixPostmaster(identity), -+ None => Self::Generic, -+ } -+ } -+ -+ pub(super) const fn id(self) -> &'static str { -+ match self { -+ Self::Generic => GENERIC_TOKIO_RUNTIME_POLICY_ID, -+ Self::SealedWasixPostmaster(_) => SEALED_POSTMASTER_TOKIO_RUNTIME_POLICY_ID, -+ } -+ } -+ -+ pub(super) const fn workload_id(self) -> Option<&'static str> { -+ match self { -+ Self::Generic => None, -+ Self::SealedWasixPostmaster(identity) => Some(identity.workload_id()), -+ } -+ } -+ -+ pub(super) const fn configured_worker_threads(self) -> Option { -+ match self { -+ Self::Generic => None, -+ Self::SealedWasixPostmaster(_) => Some(SEALED_POSTMASTER_TOKIO_WORKER_THREADS), -+ } -+ } -+ -+ pub(super) fn build(self) -> io::Result { -+ match self { -+ // Keep generic Wasmer behavior byte-for-byte equivalent to the -+ // historical builder: Tokio owns its platform-default worker -+ // selection and environment override semantics. -+ Self::Generic => tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build(), -+ Self::SealedWasixPostmaster(_) => tokio::runtime::Builder::new_multi_thread() -+ .worker_threads(SEALED_POSTMASTER_TOKIO_WORKER_THREADS) -+ .enable_all() -+ .build(), -+ } -+ } -+} -+ - /// Special wasix runtime implementation for the CLI. - /// - /// Wraps an undelrying runtime and adds progress monitoring for package -@@ -53,6 +111,10 @@ impl wasmer_wasix::Runtime for Monitorin - self.runtime.task_manager() - } - -+ fn resource_limits(&self) -> wasmer_wasix::ResourceLimits { -+ self.runtime.resource_limits() -+ } -+ - fn package_loader( - &self, - ) -> Arc { -@@ -264,3 +326,42 @@ impl wasmer_wasix::runtime::package_loader::PackageLoader for MonitoringPackageL - .await - } - } -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ -+ #[test] -+ fn sealed_postmaster_policy_id_and_worker_configuration_are_stable() { -+ for identity in [ -+ SealedRuntimeIdentity::WasixPostmasterInitdb, -+ SealedRuntimeIdentity::WasixPostmasterPostgres, -+ ] { -+ let policy = CliTokioRuntimePolicy::select(Some(identity)); -+ assert_eq!( -+ policy.id(), -+ "oliphaunt.wasix-postmaster.tokio.sealed-postmaster-2-worker.v1" -+ ); -+ assert_eq!(policy.workload_id(), Some(identity.workload_id())); -+ assert_eq!(policy.configured_worker_threads(), Some(2)); -+ } -+ } -+ -+ #[test] -+ fn sealed_postmaster_runtime_has_exactly_two_tokio_workers() { -+ let runtime = -+ CliTokioRuntimePolicy::select(Some(SealedRuntimeIdentity::WasixPostmasterPostgres)) -+ .build() -+ .unwrap(); -+ -+ assert_eq!(runtime.metrics().num_workers(), 2); -+ } -+ -+ #[test] -+ fn generic_runtime_policy_retains_tokio_default_worker_selection() { -+ let policy = CliTokioRuntimePolicy::select(None); -+ assert_eq!(policy.id(), "wasmer.cli.tokio.default.v1"); -+ assert_eq!(policy.workload_id(), None); -+ assert_eq!(policy.configured_worker_threads(), None); -+ } -+} -diff --git a/lib/cli/src/commands/run/wasi.rs b/lib/cli/src/commands/run/wasi.rs -index 3297890..4cfac63 100644 ---- a/lib/cli/src/commands/run/wasi.rs -+++ b/lib/cli/src/commands/run/wasi.rs -@@ -23,8 +23,8 @@ use wasmer_types::ModuleHash; - #[cfg(feature = "journal")] - use wasmer_wasix::journal::{LogFileJournal, SnapshotTrigger}; - use wasmer_wasix::{ -- PluggableRuntime, RewindState, Runtime, WasiEnv, WasiEnvBuilder, WasiError, WasiFunctionEnv, -- WasiVersion, -+ PluggableRuntime, ResourceLimits, RewindState, Runtime, WasiEnv, WasiEnvBuilder, WasiError, -+ WasiFunctionEnv, WasiVersion, - bin_factory::BinaryPackage, - capabilities::Capabilities, - get_wasi_versions, -@@ -580,12 +580,25 @@ impl Wasi { - rt_or_handle: I, - preferred_webc_version: webc::Version, - compiler_debug_dir_used: bool, -+ resource_limits: ResourceLimits, -+ sealed_module_cache: Option>, - ) -> Result> - where - I: Into, - { - let tokio_task_manager = Arc::new(TokioTaskManager::new(rt_or_handle.into())); - let mut rt = PluggableRuntime::new(tokio_task_manager.clone()); -+ rt.set_resource_limits(resource_limits); -+ if let Some(module_cache) = sealed_module_cache { -+ // This authoritative cache is the sealed carrier's exact immutable -+ // closure. It is installed before any runtime consumers are built. -+ rt.set_module_cache(module_cache); -+ } else if !self.disable_cache && !compiler_debug_dir_used { -+ let cache_dir = env.cache_dir().join("compiled"); -+ let module_cache = wasmer_wasix::runtime::module_cache::in_memory() -+ .with_fallback(FileSystemCache::new(cache_dir, tokio_task_manager)); -+ rt.set_module_cache(module_cache); -+ } - - let has_networking = self.networking.is_some() - || capabilities::get_cached_capability(pkg_cache_path) -@@ -643,13 +656,6 @@ impl Wasi { - - let registry = self.prepare_source(env, client, preferred_webc_version)?; - -- if !self.disable_cache && !compiler_debug_dir_used { -- let cache_dir = env.cache_dir().join("compiled"); -- let module_cache = wasmer_wasix::runtime::module_cache::in_memory() -- .with_fallback(FileSystemCache::new(cache_dir, tokio_task_manager)); -- rt.set_module_cache(module_cache); -- } -- - rt.set_package_loader(package_loader) - .set_source(registry) - .set_engine(engine); -diff --git a/lib/compiler-llvm/src/compiler.rs b/lib/compiler-llvm/src/compiler.rs -index b04e437..3950012 100644 ---- a/lib/compiler-llvm/src/compiler.rs -+++ b/lib/compiler-llvm/src/compiler.rs -@@ -178,13 +178,17 @@ impl Compiler for LLVMCompiler { - - fn deterministic_id(&self) -> String { - format!( -- "llvm-{}", -+ "llvm-v3-{}-nan{}-nv{}-rotable{}-pic{}", - match self.config.opt_level { - inkwell::OptimizationLevel::None => "opt0", - inkwell::OptimizationLevel::Less => "optl", - inkwell::OptimizationLevel::Default => "optd", - inkwell::OptimizationLevel::Aggressive => "opta", -- } -+ }, -+ u8::from(self.config.enable_nan_canonicalization), -+ u8::from(self.config.enable_non_volatile_memops), -+ u8::from(self.config.enable_readonly_funcref_table), -+ u8::from(self.config.is_pic), - ) - } - -diff --git a/lib/compiler-llvm/src/config.rs b/lib/compiler-llvm/src/config.rs -index d732134..58918f0 100644 ---- a/lib/compiler-llvm/src/config.rs -+++ b/lib/compiler-llvm/src/config.rs -@@ -109,7 +109,7 @@ pub struct LLVM { - pub(crate) enable_verifier: bool, - pub(crate) enable_perfmap: bool, - pub(crate) opt_level: LLVMOptLevel, -- is_pic: bool, -+ pub(crate) is_pic: bool, - pub(crate) callbacks: Option, - /// The middleware chain. - pub(crate) middlewares: Vec>, -diff --git a/lib/compiler-llvm/src/object_file.rs b/lib/compiler-llvm/src/object_file.rs -index 7265bca..816be2e 100644 ---- a/lib/compiler-llvm/src/object_file.rs -+++ b/lib/compiler-llvm/src/object_file.rs -@@ -46,6 +46,10 @@ static LIBCALLS_ELF: phf::Map<&'static str, LibCall> = phf::phf_map! { - "truncf" => LibCall::TruncF32, - "trunc" => LibCall::TruncF64, - "__chkstk" => LibCall::Probestack, -+ "bzero" => LibCall::HostBzero, -+ "memset" => LibCall::HostMemset, -+ "memcpy" => LibCall::HostMemcpy, -+ "memmove" => LibCall::HostMemmove, - "wasmer_vm_f32_ceil" => LibCall::CeilF32, - "wasmer_vm_f64_ceil" => LibCall::CeilF64, - "wasmer_vm_f32_floor" => LibCall::FloorF32, -@@ -101,6 +105,10 @@ static LIBCALLS_MACHO: phf::Map<&'static str, LibCall> = phf::phf_map! { - "_nearbyint" => LibCall::NearestF64, - "_truncf" => LibCall::TruncF32, - "_trunc" => LibCall::TruncF64, -+ "_bzero" => LibCall::HostBzero, -+ "_memset" => LibCall::HostMemset, -+ "_memcpy" => LibCall::HostMemcpy, -+ "_memmove" => LibCall::HostMemmove, - "_wasmer_vm_f32_ceil" => LibCall::CeilF32, - "_wasmer_vm_f64_ceil" => LibCall::CeilF64, - "_wasmer_vm_f32_floor" => LibCall::FloorF32, -diff --git a/lib/compiler-llvm/src/translator/code.rs b/lib/compiler-llvm/src/translator/code.rs -index fdf0509..9fc1f51 100644 ---- a/lib/compiler-llvm/src/translator/code.rs -+++ b/lib/compiler-llvm/src/translator/code.rs -@@ -361,9 +361,8 @@ impl FuncTranslator { - ); - - while fcg.state.has_control_frames() { -- let pos = reader.current_position() as u32; - let op = reader.read_operator()?; -- fcg.translate_operator(op, pos)?; -+ fcg.translate_operator(op)?; - } - - fcg.finalize(wasm_fn_type)?; -@@ -445,16 +444,25 @@ impl FuncTranslator { - target: &Triple, - ) -> Result { - let func_index = wasm_module.func_index(*local_func_index); -+ let function_body_len = function_body.data.len() as u64; - let opt_style = if Some(func_index) == self.wasm_apply_data_relocs_fn_index { - // `__wasm_apply_data_relocs` can become a very large function made up - // mostly of loads and stores, and even `-O1` can spend significant - // time optimizing it. - OptimizationStyle::Disabled -- } else if function_body.data.len() as u64 > WASM_LARGE_FUNCTION_THRESHOLD { -+ } else if function_body_len > WASM_LARGE_FUNCTION_THRESHOLD { - OptimizationStyle::ForSize - } else { - OptimizationStyle::ForSpeed - }; -+ if function_body_len > WASM_LARGE_FUNCTION_THRESHOLD { -+ tracing::debug!( -+ function = %wasm_module.get_function_name(func_index), -+ body_len = function_body_len, -+ ?opt_style, -+ "selected LLVM optimization style for large function" -+ ); -+ } - let module = self.translate_to_module( - wasm_module, - module_translation, -@@ -1247,7 +1255,7 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - // If this memory access must trap when out of bounds (i.e. it is a memory - // access written in the user program as opposed to one used by our VM) - // then mark that it can't be deleted. -- if let MemoryCache::Static { base_ptr: _ } = self.ctx.memory( -+ if let MemoryCache::Static { .. } = self.ctx.memory( - memory_index, - self.intrinsics, - self.module, -@@ -1739,7 +1747,7 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - ); - ptr_to_base - } -- MemoryCache::Static { base_ptr } => base_ptr, -+ MemoryCache::Static { base_ptr, .. } => base_ptr, - } - }; - let value_ptr = unsafe { -@@ -1752,6 +1760,160 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - ) - } - -+ fn memory_bulk_parts( -+ &mut self, -+ memory_index: MemoryIndex, -+ label: &str, -+ ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { -+ match self.ctx.memory( -+ memory_index, -+ self.intrinsics, -+ self.module, -+ self.memory_styles, -+ )? { -+ MemoryCache::Dynamic { -+ ptr_to_base_ptr, -+ ptr_to_current_length, -+ } => { -+ let base = self.build_dynamic_memory_base(ptr_to_base_ptr, memory_index, label)?; -+ Ok((base, ptr_to_current_length)) -+ } -+ MemoryCache::Static { -+ base_ptr, -+ ptr_to_current_length, -+ } => Ok((base_ptr, ptr_to_current_length)), -+ } -+ } -+ -+ fn int_value_as_u64( -+ &self, -+ value: IntValue<'ctx>, -+ label: &str, -+ ) -> Result, CompileError> { -+ match value.get_type().get_bit_width() { -+ 64 => Ok(value), -+ width if width < 64 => Ok(err!(self.builder.build_int_z_extend( -+ value, -+ self.intrinsics.i64_ty, -+ &format!("{label}_zext") -+ ))), -+ width => Err(CompileError::Codegen(format!( -+ "unsupported memory index width {width}" -+ ))), -+ } -+ } -+ -+ fn build_memory_range_check( -+ &self, -+ offset: IntValue<'ctx>, -+ len: IntValue<'ctx>, -+ ptr_to_current_length: PointerValue<'ctx>, -+ label: &str, -+ ) -> Result, CompileError> { -+ let offset = self.int_value_as_u64(offset, &format!("{label}_offset"))?; -+ let len = self.int_value_as_u64(len, &format!("{label}_len"))?; -+ let current_length = err!(self.builder.build_load( -+ self.intrinsics.i32_ty, -+ ptr_to_current_length, -+ &format!("{label}_current_length") -+ )) -+ .into_int_value(); -+ tbaa_label( -+ self.module, -+ self.intrinsics, -+ format!("{label} memory length"), -+ current_length.as_instruction_value().unwrap(), -+ ); -+ let current_length = err!(self.builder.build_int_z_extend( -+ current_length, -+ self.intrinsics.i64_ty, -+ &format!("{label}_current_length_zext") -+ )); -+ let offset_in_bounds = err!(self.builder.build_int_compare( -+ IntPredicate::ULE, -+ offset, -+ current_length, -+ &format!("{label}_offset_in_bounds") -+ )); -+ let remaining = err!(self.builder.build_int_sub( -+ current_length, -+ offset, -+ &format!("{label}_remaining") -+ )); -+ let len_in_bounds = err!(self.builder.build_int_compare( -+ IntPredicate::ULE, -+ len, -+ remaining, -+ &format!("{label}_len_in_bounds") -+ )); -+ Ok(err!(self.builder.build_and( -+ offset_in_bounds, -+ len_in_bounds, -+ &format!("{label}_in_bounds") -+ ))) -+ } -+ -+ fn build_memory_oob_trap( -+ &self, -+ in_bounds: IntValue<'ctx>, -+ label: &str, -+ ) -> Result<(), CompileError> { -+ let in_bounds = err!(self.build_call_with_param_attributes( -+ self.intrinsics.expect_i1, -+ &[ -+ in_bounds.into(), -+ self.intrinsics.i1_ty.const_int(1, true).into(), -+ ], -+ &format!("{label}_in_bounds_expect"), -+ )) -+ .try_as_basic_value() -+ .unwrap_basic() -+ .into_int_value(); -+ -+ let continue_block = self -+ .context -+ .append_basic_block(self.function, &format!("{label}_in_bounds_continue")); -+ let trap_block = self -+ .context -+ .append_basic_block(self.function, &format!("{label}_oob_trap")); -+ err!( -+ self.builder -+ .build_conditional_branch(in_bounds, continue_block, trap_block) -+ ); -+ -+ self.builder.position_at_end(trap_block); -+ err!(self.build_call_with_param_attributes( -+ self.intrinsics.throw_trap, -+ &[self.intrinsics.trap_memory_oob.into()], -+ "throw", -+ )); -+ err!(self.builder.build_unreachable()); -+ -+ self.builder.position_at_end(continue_block); -+ Ok(()) -+ } -+ -+ fn build_dynamic_memory_base( -+ &self, -+ ptr_to_base_ptr: PointerValue<'ctx>, -+ memory_index: MemoryIndex, -+ label: &str, -+ ) -> Result, CompileError> { -+ let base = err!(self.builder.build_load( -+ self.intrinsics.ptr_ty, -+ ptr_to_base_ptr, -+ &format!("{label}_base") -+ )) -+ .into_pointer_value(); -+ tbaa_label( -+ self.module, -+ self.intrinsics, -+ format!("memory base_ptr {}", memory_index.as_u32()), -+ base.as_instruction_value().unwrap(), -+ ); -+ Ok(base) -+ } -+ - fn trap_if_misaligned( - &self, - _memarg: &MemArg, -@@ -1952,6 +2114,303 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - }) - } - -+ fn emit_indirect_call_from_state( -+ &mut self, -+ sigindex: SignatureIndex, -+ table_index: TableIndex, -+ is_return_call: bool, -+ ) -> Result>, CompileError> { -+ let func_type = &self.wasm_module.signatures[sigindex]; -+ let table = self.wasm_module.tables.get(table_index).unwrap(); -+ let local_fixed_funcref_table = self -+ .wasm_module -+ .local_table_index(table_index) -+ .filter(|_| table.is_fixed_funcref_table()); -+ let expected_signature_hash = self -+ .intrinsics -+ .i32_ty -+ .const_int(u64::from(self.signature_hashes[sigindex].as_u32()), false); -+ -+ let func_index = self.state.pop1()?.into_int_value(); -+ let generic_table = if local_fixed_funcref_table.is_none() { -+ Some( -+ self.ctx -+ .table(table_index, self.intrinsics, self.module, &self.builder)?, -+ ) -+ } else { -+ None -+ }; -+ -+ let table_bound = if local_fixed_funcref_table.is_some() { -+ self.intrinsics -+ .i32_ty -+ .const_int(table.minimum.into(), false) -+ } else { -+ let (_, table_bound) = *generic_table.as_ref().unwrap(); -+ err!(self.builder.build_int_truncate( -+ table_bound, -+ self.intrinsics.i32_ty, -+ "truncated_table_bounds", -+ )) -+ }; -+ -+ let index_in_bounds = err!(self.builder.build_int_compare( -+ IntPredicate::ULT, -+ func_index, -+ table_bound, -+ "index_in_bounds", -+ )); -+ -+ let index_in_bounds = self -+ .build_call_with_param_attributes( -+ self.intrinsics.expect_i1, -+ &[ -+ index_in_bounds.into(), -+ self.intrinsics.i1_ty.const_int(1, false).into(), -+ ], -+ "index_in_bounds_expect", -+ )? -+ .try_as_basic_value() -+ .unwrap_basic() -+ .into_int_value(); -+ -+ let in_bounds_continue_block = self -+ .context -+ .append_basic_block(self.function, "in_bounds_continue_block"); -+ let not_in_bounds_block = self -+ .context -+ .append_basic_block(self.function, "not_in_bounds_block"); -+ err!(self.builder.build_conditional_branch( -+ index_in_bounds, -+ in_bounds_continue_block, -+ not_in_bounds_block, -+ )); -+ self.builder.position_at_end(not_in_bounds_block); -+ self.build_call_with_param_attributes( -+ self.intrinsics.throw_trap, -+ &[self.intrinsics.trap_table_access_oob.into()], -+ "throw", -+ )?; -+ err!(self.builder.build_unreachable()); -+ self.builder.position_at_end(in_bounds_continue_block); -+ -+ let anyfunc_struct_ptr = if let Some(local_table_index) = local_fixed_funcref_table { -+ let anyfuncs = self.ctx.fixed_funcref_table_anyfuncs( -+ local_table_index, -+ self.intrinsics, -+ &self.builder, -+ )?; -+ unsafe { -+ err!(self.builder.build_in_bounds_gep( -+ self.intrinsics.anyfunc_ty, -+ anyfuncs, -+ &[func_index], -+ "anyfunc_struct_ptr", -+ )) -+ } -+ } else if table.ty == Type::FuncRef { -+ let anyfuncs = self.ctx.table_anyfuncs( -+ table_index, -+ self.intrinsics, -+ self.module, -+ &self.builder, -+ )?; -+ unsafe { -+ err!(self.builder.build_in_bounds_gep( -+ self.intrinsics.anyfunc_ty, -+ anyfuncs, -+ &[func_index], -+ "anyfunc_struct_ptr", -+ )) -+ } -+ } else { -+ let (table_base, _) = *generic_table.as_ref().unwrap(); -+ -+ let casted_table_base = err!(self.builder.build_pointer_cast( -+ table_base, -+ self.context.ptr_type(AddressSpace::default()), -+ "casted_table_base", -+ )); -+ -+ let funcref_ptr = unsafe { -+ err!(self.builder.build_in_bounds_gep( -+ self.intrinsics.ptr_ty, -+ casted_table_base, -+ &[func_index], -+ "funcref_ptr", -+ )) -+ }; -+ -+ let anyfunc_struct_ptr = err!(self.builder.build_load( -+ self.intrinsics.ptr_ty, -+ funcref_ptr, -+ "anyfunc_struct_ptr", -+ )) -+ .into_pointer_value(); -+ -+ if !table.readonly { -+ let funcref_not_null = err!( -+ self.builder -+ .build_is_not_null(anyfunc_struct_ptr, "null_funcref_check") -+ ); -+ -+ let funcref_continue_deref_block = self -+ .context -+ .append_basic_block(self.function, "funcref_continue_deref_block"); -+ -+ let funcref_is_null_block = self -+ .context -+ .append_basic_block(self.function, "funcref_is_null_block"); -+ err!(self.builder.build_conditional_branch( -+ funcref_not_null, -+ funcref_continue_deref_block, -+ funcref_is_null_block, -+ )); -+ self.builder.position_at_end(funcref_is_null_block); -+ self.build_call_with_param_attributes( -+ self.intrinsics.throw_trap, -+ &[self.intrinsics.trap_call_indirect_null.into()], -+ "throw", -+ )?; -+ err!(self.builder.build_unreachable()); -+ self.builder.position_at_end(funcref_continue_deref_block); -+ } -+ -+ anyfunc_struct_ptr -+ }; -+ -+ let sig_hash_ptr = self -+ .builder -+ .build_struct_gep( -+ self.intrinsics.anyfunc_ty, -+ anyfunc_struct_ptr, -+ 1, -+ "sig_hash_ptr", -+ ) -+ .unwrap(); -+ let func_ptr_ptr = self -+ .builder -+ .build_struct_gep( -+ self.intrinsics.anyfunc_ty, -+ anyfunc_struct_ptr, -+ 0, -+ "func_ptr_ptr", -+ ) -+ .unwrap(); -+ let (func_ptr, found_signature_hash) = ( -+ err!( -+ self.builder -+ .build_load(self.intrinsics.ptr_ty, func_ptr_ptr, "func_ptr") -+ ) -+ .into_pointer_value(), -+ err!( -+ self.builder -+ .build_load(self.intrinsics.i32_ty, sig_hash_ptr, "sig_hash") -+ ) -+ .into_int_value(), -+ ); -+ -+ let elem_initialized = err!(self.builder.build_is_not_null(func_ptr, "")); -+ let sig_hashes_equal = err!(self.builder.build_int_compare( -+ IntPredicate::EQ, -+ expected_signature_hash, -+ found_signature_hash, -+ "sig_hashes_equal", -+ )); -+ -+ let initialized_and_sig_hashes_match = err!(self.builder.build_and( -+ elem_initialized, -+ sig_hashes_equal, -+ "" -+ )); -+ -+ let initialized_and_sig_hashes_match = self -+ .build_call_with_param_attributes( -+ self.intrinsics.expect_i1, -+ &[ -+ initialized_and_sig_hashes_match.into(), -+ self.intrinsics.i1_ty.const_int(1, false).into(), -+ ], -+ "initialized_and_sig_hashes_match_expect", -+ )? -+ .try_as_basic_value() -+ .unwrap_basic() -+ .into_int_value(); -+ -+ let continue_block = self -+ .context -+ .append_basic_block(self.function, "continue_block"); -+ let sighashes_notequal_block = self -+ .context -+ .append_basic_block(self.function, "sighashes_notequal_block"); -+ err!(self.builder.build_conditional_branch( -+ initialized_and_sig_hashes_match, -+ continue_block, -+ sighashes_notequal_block, -+ )); -+ -+ self.builder.position_at_end(sighashes_notequal_block); -+ let trap_code = err!(self.builder.build_select( -+ elem_initialized, -+ self.intrinsics.trap_call_indirect_sig, -+ self.intrinsics.trap_call_indirect_null, -+ "", -+ )); -+ self.build_call_with_param_attributes( -+ self.intrinsics.throw_trap, -+ &[trap_code.into()], -+ "throw", -+ )?; -+ err!(self.builder.build_unreachable()); -+ self.builder.position_at_end(continue_block); -+ -+ let callee_vmctx = if table.readonly { -+ self.ctx.basic().into_pointer_value() -+ } else { -+ let ctx_ptr_ptr = self -+ .builder -+ .build_struct_gep( -+ self.intrinsics.anyfunc_ty, -+ anyfunc_struct_ptr, -+ 2, -+ "ctx_ptr_ptr", -+ ) -+ .unwrap(); -+ err!( -+ self.builder -+ .build_load(self.intrinsics.ptr_ty, ctx_ptr_ptr, "ctx_ptr") -+ ) -+ .into_pointer_value() -+ }; -+ -+ let rets = if self.m0_param.is_some() { -+ self.build_m0_indirect_call( -+ table_index.as_u32(), -+ callee_vmctx, -+ func_type, -+ func_ptr, -+ func_index, -+ is_return_call, -+ )? -+ } else { -+ let (call_site, llvm_func_type) = -+ self.build_indirect_call(callee_vmctx, func_type, func_ptr, None, is_return_call)?; -+ -+ if is_return_call { -+ self.emit_return_call(call_site, llvm_func_type)?; -+ Vec::new() -+ } else { -+ self.abi -+ .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? -+ .iter() -+ .copied() -+ .collect() -+ } -+ }; -+ -+ Ok(rets) -+ } -+ - fn build_m0_indirect_call( - &mut self, - table_index: u32, -@@ -1960,14 +2419,40 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - func_ptr: PointerValue<'ctx>, - func_index: IntValue<'ctx>, - is_return_call: bool, -- ) -> Result<(), CompileError> { -- let Some(m0) = self.m0_param else { -+ ) -> Result>, CompileError> { -+ if self.m0_param.is_none() { - return Err(CompileError::Codegen( - "Call to build_m0_indirect_call without m0 parameter!".to_string(), - )); -- }; -+ } - - let params = self.state.popn_save_extra(func_type.params().len())?; -+ self.build_m0_indirect_call_with_params( -+ table_index, -+ ctx_ptr, -+ func_type, -+ func_ptr, -+ func_index, -+ is_return_call, -+ ¶ms, -+ ) -+ } -+ -+ fn build_m0_indirect_call_with_params( -+ &mut self, -+ table_index: u32, -+ ctx_ptr: PointerValue<'ctx>, -+ func_type: &FunctionType, -+ func_ptr: PointerValue<'ctx>, -+ func_index: IntValue<'ctx>, -+ is_return_call: bool, -+ params: &[(BasicValueEnum<'ctx>, ExtraInfo)], -+ ) -> Result>, CompileError> { -+ let Some(m0) = self.m0_param else { -+ return Err(CompileError::Codegen( -+ "Call to build_m0_indirect_call_with_params without m0 parameter!".to_string(), -+ )); -+ }; - - let mut local_func_indices = vec![]; - let mut foreign_func_indices = vec![]; -@@ -2082,20 +2567,22 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - }; - - if is_return_call { -- return Ok(()); -+ return Ok(Vec::new()); - } - - self.builder - .position_at_end(cont.expect("non-return call requires cont")); - -+ let mut rets = Vec::with_capacity(foreign_rets.len()); - for i in 0..foreign_rets.len() { - let f_i = foreign_rets[i]; - let l_i = local_rets[i]; - let ty = f_i.get_type(); - let v = err!(self.builder.build_phi(ty, "")); - v.add_incoming(&[(&f_i, foreign_idx_block), (&l_i, local_idx_block)]); -- self.state.push1(v.as_basic_value()); -+ rets.push(v.as_basic_value()); - } -+ Ok(rets) - } else if foreign_func_indices.is_empty() { - let (call_site, llvm_func_type) = self.build_indirect_call_with_params( - ctx_ptr, -@@ -2108,11 +2595,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - - if is_return_call { - self.emit_return_call(call_site, llvm_func_type)?; -+ Ok(Vec::new()) - } else { -- self.abi -+ Ok(self -+ .abi - .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? - .iter() -- .for_each(|ret| self.state.push1(*ret)); -+ .copied() -+ .collect()) - } - } else { - let (call_site, llvm_func_type) = self.build_indirect_call_with_params( -@@ -2125,15 +2615,16 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - )?; - if is_return_call { - self.emit_return_call(call_site, llvm_func_type)?; -+ Ok(Vec::new()) - } else { -- self.abi -+ Ok(self -+ .abi - .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? - .iter() -- .for_each(|ret| self.state.push1(*ret)); -+ .copied() -+ .collect()) - } - } -- -- Ok(()) - } - - fn build_indirect_call( -@@ -2965,8 +3456,6 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - // Basic instructions. - // https://github.com/sunfishcode/wasm-reference-manual/blob/master/WebAssembly.md#basic-instructions - fn translate_basic_operator(&mut self, op: Operator) -> Result<(), CompileError> { -- let vmctx = &self.ctx.basic().into_pointer_value(); -- - match op { - Operator::Nop => { - // Do nothing. -@@ -3476,300 +3965,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - let is_return_call = matches!(op, Operator::ReturnCallIndirect { .. }); - let sigindex = SignatureIndex::from_u32(type_index); - let table_index = TableIndex::from_u32(table_index); -- let func_type = &self.wasm_module.signatures[sigindex]; -- let table = self.wasm_module.tables.get(table_index).unwrap(); -- let local_fixed_funcref_table = self -- .wasm_module -- .local_table_index(table_index) -- .filter(|_| table.is_fixed_funcref_table()); -- let expected_signature_hash = self -- .intrinsics -- .i32_ty -- .const_int(u64::from(self.signature_hashes[sigindex].as_u32()), false); -- -- let func_index = self.state.pop1()?.into_int_value(); -- let generic_table = if local_fixed_funcref_table.is_none() { -- Some(self.ctx.table( -- table_index, -- self.intrinsics, -- self.module, -- &self.builder, -- )?) -- } else { -- None -- }; -- -- let table_bound = if local_fixed_funcref_table.is_some() { -- self.intrinsics -- .i32_ty -- .const_int(table.minimum.into(), false) -- } else { -- let (_, table_bound) = *generic_table.as_ref().unwrap(); -- err!(self.builder.build_int_truncate( -- table_bound, -- self.intrinsics.i32_ty, -- "truncated_table_bounds", -- )) -- }; -- -- // First, check if the index is outside of the table bounds. -- let index_in_bounds = err!(self.builder.build_int_compare( -- IntPredicate::ULT, -- func_index, -- table_bound, -- "index_in_bounds", -- )); -- -- let index_in_bounds = self -- .build_call_with_param_attributes( -- self.intrinsics.expect_i1, -- &[ -- index_in_bounds.into(), -- self.intrinsics.i1_ty.const_int(1, false).into(), -- ], -- "index_in_bounds_expect", -- )? -- .try_as_basic_value() -- .unwrap_basic() -- .into_int_value(); -- -- let in_bounds_continue_block = self -- .context -- .append_basic_block(self.function, "in_bounds_continue_block"); -- let not_in_bounds_block = self -- .context -- .append_basic_block(self.function, "not_in_bounds_block"); -- err!(self.builder.build_conditional_branch( -- index_in_bounds, -- in_bounds_continue_block, -- not_in_bounds_block, -- )); -- self.builder.position_at_end(not_in_bounds_block); -- self.build_call_with_param_attributes( -- self.intrinsics.throw_trap, -- &[self.intrinsics.trap_table_access_oob.into()], -- "throw", -- )?; -- err!(self.builder.build_unreachable()); -- self.builder.position_at_end(in_bounds_continue_block); -- -- let anyfunc_struct_ptr = if let Some(local_table_index) = local_fixed_funcref_table -- { -- let anyfuncs = self.ctx.fixed_funcref_table_anyfuncs( -- local_table_index, -- self.intrinsics, -- &self.builder, -- )?; -- unsafe { -- err!(self.builder.build_in_bounds_gep( -- self.intrinsics.anyfunc_ty, -- anyfuncs, -- &[func_index], -- "anyfunc_struct_ptr", -- )) -- } -- } else { -- let (table_base, _) = *generic_table.as_ref().unwrap(); -- -- // We assume the table has the `funcref` (pointer to `anyfunc`) -- // element type. -- let casted_table_base = err!(self.builder.build_pointer_cast( -- table_base, -- self.context.ptr_type(AddressSpace::default()), -- "casted_table_base", -- )); -- -- let funcref_ptr = unsafe { -- err!(self.builder.build_in_bounds_gep( -- self.intrinsics.ptr_ty, -- casted_table_base, -- &[func_index], -- "funcref_ptr", -- )) -- }; -- -- // a funcref (pointer to `anyfunc`) -- let anyfunc_struct_ptr = err!(self.builder.build_load( -- self.intrinsics.ptr_ty, -- funcref_ptr, -- "anyfunc_struct_ptr", -- )) -- .into_pointer_value(); -- -- if !table.readonly { -- // trap if we're trying to call a null funcref -- let funcref_not_null = err!( -- self.builder -- .build_is_not_null(anyfunc_struct_ptr, "null_funcref_check") -- ); -- -- let funcref_continue_deref_block = self -- .context -- .append_basic_block(self.function, "funcref_continue_deref_block"); -- -- let funcref_is_null_block = self -- .context -- .append_basic_block(self.function, "funcref_is_null_block"); -- err!(self.builder.build_conditional_branch( -- funcref_not_null, -- funcref_continue_deref_block, -- funcref_is_null_block, -- )); -- self.builder.position_at_end(funcref_is_null_block); -- self.build_call_with_param_attributes( -- self.intrinsics.throw_trap, -- &[self.intrinsics.trap_call_indirect_null.into()], -- "throw", -- )?; -- err!(self.builder.build_unreachable()); -- self.builder.position_at_end(funcref_continue_deref_block); -- } -- -- anyfunc_struct_ptr -- }; -- -- // Load things from the anyfunc data structure. -- let sig_hash_ptr = self -- .builder -- .build_struct_gep( -- self.intrinsics.anyfunc_ty, -- anyfunc_struct_ptr, -- 1, -- "sig_hash_ptr", -- ) -- .unwrap(); -- let func_ptr_ptr = self -- .builder -- .build_struct_gep( -- self.intrinsics.anyfunc_ty, -- anyfunc_struct_ptr, -- 0, -- "func_ptr_ptr", -- ) -- .unwrap(); -- let (func_ptr, found_signature_hash) = ( -- err!( -- self.builder -- .build_load(self.intrinsics.ptr_ty, func_ptr_ptr, "func_ptr") -- ) -- .into_pointer_value(), -- err!( -- self.builder -- .build_load(self.intrinsics.i32_ty, sig_hash_ptr, "sig_hash") -- ) -- .into_int_value(), -- ); -- -- // Next, check if the table element is initialized. -- -- // TODO: we may not need this check anymore -- let elem_initialized = err!(self.builder.build_is_not_null(func_ptr, "")); -- -- // Next, check if the signature id is correct. -- -- let sig_hashes_equal = err!(self.builder.build_int_compare( -- IntPredicate::EQ, -- expected_signature_hash, -- found_signature_hash, -- "sig_hashes_equal", -- )); -- -- let initialized_and_sig_hashes_match = err!(self.builder.build_and( -- elem_initialized, -- sig_hashes_equal, -- "" -- )); -- -- // Tell llvm that the expected and found signature hashes should match. -- let initialized_and_sig_hashes_match = self -- .build_call_with_param_attributes( -- self.intrinsics.expect_i1, -- &[ -- initialized_and_sig_hashes_match.into(), -- self.intrinsics.i1_ty.const_int(1, false).into(), -- ], -- "initialized_and_sig_hashes_match_expect", -- )? -- .try_as_basic_value() -- .unwrap_basic() -- .into_int_value(); -- -- let continue_block = self -- .context -- .append_basic_block(self.function, "continue_block"); -- let sighashes_notequal_block = self -- .context -- .append_basic_block(self.function, "sighashes_notequal_block"); -- err!(self.builder.build_conditional_branch( -- initialized_and_sig_hashes_match, -- continue_block, -- sighashes_notequal_block, -- )); -- -- self.builder.position_at_end(sighashes_notequal_block); -- let trap_code = err!(self.builder.build_select( -- elem_initialized, -- self.intrinsics.trap_call_indirect_sig, -- self.intrinsics.trap_call_indirect_null, -- "", -- )); -- self.build_call_with_param_attributes( -- self.intrinsics.throw_trap, -- &[trap_code.into()], -- "throw", -- )?; -- err!(self.builder.build_unreachable()); -- self.builder.position_at_end(continue_block); -- -- let callee_vmctx = if table.readonly { -- *vmctx -- } else { -- let ctx_ptr_ptr = self -- .builder -- .build_struct_gep( -- self.intrinsics.anyfunc_ty, -- anyfunc_struct_ptr, -- 2, -- "ctx_ptr_ptr", -- ) -- .unwrap(); -- err!( -- self.builder -- .build_load(self.intrinsics.ptr_ty, ctx_ptr_ptr, "ctx_ptr") -- ) -- .into_pointer_value() -- }; -- -- if self.m0_param.is_some() { -- self.build_m0_indirect_call( -- table_index.as_u32(), -- callee_vmctx, -- func_type, -- func_ptr, -- func_index, -- is_return_call, -- )?; -- } else { -- let (call_site, llvm_func_type) = self.build_indirect_call( -- callee_vmctx, -- func_type, -- func_ptr, -- None, -- is_return_call, -- )?; -- -- if is_return_call { -- self.emit_return_call(call_site, llvm_func_type)?; -- } else { -- self.abi -- .rets_from_call(&self.builder, self.intrinsics, call_site, func_type)? -- .iter() -- .for_each(|ret| self.state.push1(*ret)); -- } -- } -- -+ let rets = -+ self.emit_indirect_call_from_state(sigindex, table_index, is_return_call)?; - if is_return_call { - self.state.reachable = false; -+ } else { -+ for ret in rets { -+ self.state.push1(ret); -+ } - } - } - _ => unreachable!(), -@@ -5289,13 +5492,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); - let mask = self.intrinsics.i32_ty.const_int(31u64, false); - let v2 = err!(self.builder.build_and(v2, mask, "")); -- let lhs = err!(self.builder.build_left_shift(v1, v2, "")); -- let rhs = { -- let negv2 = err!(self.builder.build_int_neg(v2, "")); -- let rhs = err!(self.builder.build_and(negv2, mask, "")); -- err!(self.builder.build_right_shift(v1, rhs, false, "")) -- }; -- let res = err!(self.builder.build_or(lhs, rhs, "")); -+ let res = self -+ .build_call_with_param_attributes( -+ self.intrinsics.fshl_i32, -+ &[v1.into(), v1.into(), v2.into()], -+ "", -+ )? -+ .try_as_basic_value() -+ .unwrap_basic(); - self.state.push1(res); - } - Operator::I64Rotl => { -@@ -5305,13 +5509,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); - let mask = self.intrinsics.i64_ty.const_int(63u64, false); - let v2 = err!(self.builder.build_and(v2, mask, "")); -- let lhs = err!(self.builder.build_left_shift(v1, v2, "")); -- let rhs = { -- let negv2 = err!(self.builder.build_int_neg(v2, "")); -- let rhs = err!(self.builder.build_and(negv2, mask, "")); -- err!(self.builder.build_right_shift(v1, rhs, false, "")) -- }; -- let res = err!(self.builder.build_or(lhs, rhs, "")); -+ let res = self -+ .build_call_with_param_attributes( -+ self.intrinsics.fshl_i64, -+ &[v1.into(), v1.into(), v2.into()], -+ "", -+ )? -+ .try_as_basic_value() -+ .unwrap_basic(); - self.state.push1(res); - } - Operator::I32Rotr => { -@@ -5321,13 +5526,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); - let mask = self.intrinsics.i32_ty.const_int(31u64, false); - let v2 = err!(self.builder.build_and(v2, mask, "")); -- let lhs = err!(self.builder.build_right_shift(v1, v2, false, "")); -- let rhs = { -- let negv2 = err!(self.builder.build_int_neg(v2, "")); -- let rhs = err!(self.builder.build_and(negv2, mask, "")); -- err!(self.builder.build_left_shift(v1, rhs, "")) -- }; -- let res = err!(self.builder.build_or(lhs, rhs, "")); -+ let res = self -+ .build_call_with_param_attributes( -+ self.intrinsics.fshr_i32, -+ &[v1.into(), v1.into(), v2.into()], -+ "", -+ )? -+ .try_as_basic_value() -+ .unwrap_basic(); - self.state.push1(res); - } - Operator::I64Rotr => { -@@ -5337,13 +5543,14 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - let (v1, v2) = (v1.into_int_value(), v2.into_int_value()); - let mask = self.intrinsics.i64_ty.const_int(63u64, false); - let v2 = err!(self.builder.build_and(v2, mask, "")); -- let lhs = err!(self.builder.build_right_shift(v1, v2, false, "")); -- let rhs = { -- let negv2 = err!(self.builder.build_int_neg(v2, "")); -- let rhs = err!(self.builder.build_and(negv2, mask, "")); -- err!(self.builder.build_left_shift(v1, rhs, "")) -- }; -- let res = err!(self.builder.build_or(lhs, rhs, "")); -+ let res = self -+ .build_call_with_param_attributes( -+ self.intrinsics.fshr_i64, -+ &[v1.into(), v1.into(), v2.into()], -+ "", -+ )? -+ .try_as_basic_value() -+ .unwrap_basic(); - self.state.push1(res); - } - Operator::I32Clz => { -@@ -9650,7 +9857,11 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - Operator::V128Load { ref memarg } => { - let offset = self.state.pop1()?.into_int_value(); - let result = -- self.build_annotated_load(self.intrinsics.i128_ty, offset, memarg, 1)?; -+ self.build_annotated_load(self.intrinsics.i64x2_ty, offset, memarg, 1)?; -+ let result = err!( -+ self.builder -+ .build_bit_cast(result, self.intrinsics.i128_ty, "") -+ ); - self.state.push1(result); - } - Operator::V128Load8Lane { ref memarg, lane } => { -@@ -9735,8 +9946,9 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - Operator::V128Store { ref memarg } => { - let (v, i) = self.state.pop1_extra()?; - let v = self.apply_pending_canonicalization(v, i)?; -+ let v = err!(self.builder.build_bit_cast(v, self.intrinsics.i64x2_ty, "")); - let offset = self.state.pop1()?.into_int_value(); -- self.build_annotated_store(self.intrinsics.i128_ty, offset, v, memarg, 1)?; -+ self.build_annotated_store(self.intrinsics.i64x2_ty, offset, v, memarg, 1)?; - } - Operator::V128Store8Lane { ref memarg, lane } => { - let (v, i) = self.state.pop1_extra()?; -@@ -10570,54 +10782,107 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - )?; - } - Operator::MemoryCopy { dst_mem, src_mem } => { -- // ignored until we support multiple memories -- let _dst = dst_mem; -- let (memory_copy, src) = if let Some(local_memory_index) = self -- .wasm_module -- .local_memory_index(MemoryIndex::from_u32(src_mem)) -- { -- (self.intrinsics.memory_copy, local_memory_index.as_u32()) -- } else { -- (self.intrinsics.imported_memory_copy, src_mem) -- }; -- - let (dest_pos, src_pos, len) = self.state.pop3()?; -- let src_index = self.intrinsics.i32_ty.const_int(src.into(), false); -- self.build_call_with_param_attributes( -- memory_copy, -- &[ -- vmctx.as_basic_value_enum().into(), -- src_index.into(), -- dest_pos.into(), -- src_pos.into(), -- len.into(), -- ], -- "", -+ let dst_memory_index = MemoryIndex::from_u32(dst_mem); -+ let src_memory_index = MemoryIndex::from_u32(src_mem); -+ -+ let (dst_base, dst_current_length) = -+ self.memory_bulk_parts(dst_memory_index, "memory_copy_dst")?; -+ let (src_base, src_current_length) = -+ self.memory_bulk_parts(src_memory_index, "memory_copy_src")?; -+ -+ let dest_pos = dest_pos.into_int_value(); -+ let src_pos = src_pos.into_int_value(); -+ let len = len.into_int_value(); -+ -+ let dst_in_bounds = self.build_memory_range_check( -+ dest_pos, -+ len, -+ dst_current_length, -+ "memory_copy_dst", -+ )?; -+ let src_in_bounds = self.build_memory_range_check( -+ src_pos, -+ len, -+ src_current_length, -+ "memory_copy_src", - )?; -+ let in_bounds = err!(self.builder.build_and( -+ dst_in_bounds, -+ src_in_bounds, -+ "memory_copy_in_bounds" -+ )); -+ self.build_memory_oob_trap(in_bounds, "memory_copy")?; -+ -+ let dest_offset = err!(self.builder.build_int_z_extend( -+ dest_pos, -+ self.intrinsics.i64_ty, -+ "memory_copy_dst_offset" -+ )); -+ let src_offset = err!(self.builder.build_int_z_extend( -+ src_pos, -+ self.intrinsics.i64_ty, -+ "memory_copy_src_offset" -+ )); -+ let dst = unsafe { -+ err!(self.builder.build_gep( -+ self.intrinsics.i8_ty, -+ dst_base, -+ &[dest_offset], -+ "memory_copy_dst_ptr" -+ )) -+ }; -+ let src = unsafe { -+ err!(self.builder.build_gep( -+ self.intrinsics.i8_ty, -+ src_base, -+ &[src_offset], -+ "memory_copy_src_ptr" -+ )) -+ }; -+ let len = err!(self.builder.build_int_z_extend( -+ len, -+ self.intrinsics.i64_ty, -+ "memory_copy_len" -+ )); -+ err!(self.builder.build_memmove(dst, 1, src, 1, len)); - } - Operator::MemoryFill { mem } => { -- let (memory_fill, mem) = if let Some(local_memory_index) = self -- .wasm_module -- .local_memory_index(MemoryIndex::from_u32(mem)) -- { -- (self.intrinsics.memory_fill, local_memory_index.as_u32()) -- } else { -- (self.intrinsics.imported_memory_fill, mem) -- }; -- - let (dst, val, len) = self.state.pop3()?; -- let mem_index = self.intrinsics.i32_ty.const_int(mem.into(), false); -- self.build_call_with_param_attributes( -- memory_fill, -- &[ -- vmctx.as_basic_value_enum().into(), -- mem_index.into(), -- dst.into(), -- val.into(), -- len.into(), -- ], -- "", -- )?; -+ let memory_index = MemoryIndex::from_u32(mem); -+ -+ let (base, current_length) = self.memory_bulk_parts(memory_index, "memory_fill")?; -+ let dst = dst.into_int_value(); -+ let val = val.into_int_value(); -+ let len = len.into_int_value(); -+ let in_bounds = -+ self.build_memory_range_check(dst, len, current_length, "memory_fill")?; -+ self.build_memory_oob_trap(in_bounds, "memory_fill")?; -+ -+ let dst_offset = err!(self.builder.build_int_z_extend( -+ dst, -+ self.intrinsics.i64_ty, -+ "memory_fill_offset" -+ )); -+ let dst = unsafe { -+ err!(self.builder.build_gep( -+ self.intrinsics.i8_ty, -+ base, -+ &[dst_offset], -+ "memory_fill_ptr" -+ )) -+ }; -+ let val = err!(self.builder.build_int_truncate( -+ val, -+ self.intrinsics.i8_ty, -+ "memory_fill_val" -+ )); -+ let len = err!(self.builder.build_int_z_extend( -+ len, -+ self.intrinsics.i64_ty, -+ "memory_fill_len" -+ )); -+ err!(self.builder.build_memset(dst, 1, val, len)); - } - _ => unreachable!(), - } -@@ -10630,13 +10895,15 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - - match op { - Operator::AtomicFence => { -- // Fence is a nop. -- // -- // Fence was added to preserve information about fences from -- // source languages. If in the future Wasm extends the memory -- // model, and if we hadn't recorded what fences used to be there, -- // it would lead to data races that weren't present in the -- // original source language. -+ // WASIX can remap the same file-backed PostgreSQL region into -+ // multiple Wasm instances. Preserve source-language fences for -+ // that cross-instance memory, which is stronger than the -+ // ordinary single-shared-memory Wasm execution model. -+ err!(self.builder.build_fence( -+ AtomicOrdering::SequentiallyConsistent, -+ false, -+ "atomic_fence" -+ )); - } - Operator::I32AtomicLoad { ref memarg } => { - let offset = self.state.pop1()?.into_int_value(); -@@ -11984,7 +12251,7 @@ impl<'ctx> LLVMFunctionCodeGenerator<'ctx, '_> { - Ok(()) - } - -- fn translate_operator(&mut self, op: Operator, _source_loc: u32) -> Result<(), CompileError> { -+ fn translate_operator(&mut self, op: Operator) -> Result<(), CompileError> { - //let opcode_offset: Option = None; - - if !self.state.reachable { -diff --git a/lib/compiler-llvm/src/translator/intrinsics.rs b/lib/compiler-llvm/src/translator/intrinsics.rs -index 1863283..6fdb84d 100644 ---- a/lib/compiler-llvm/src/translator/intrinsics.rs -+++ b/lib/compiler-llvm/src/translator/intrinsics.rs -@@ -79,6 +79,11 @@ pub struct Intrinsics<'ctx> { - pub ctpop_i64: FunctionValue<'ctx>, - pub ctpop_i8x16: FunctionValue<'ctx>, - -+ pub fshl_i32: FunctionValue<'ctx>, -+ pub fshl_i64: FunctionValue<'ctx>, -+ pub fshr_i32: FunctionValue<'ctx>, -+ pub fshr_i64: FunctionValue<'ctx>, -+ - pub fp_rounding_md: BasicMetadataValueEnum<'ctx>, - pub fp_exception_md: BasicMetadataValueEnum<'ctx>, - pub fp_ogt_md: BasicMetadataValueEnum<'ctx>, -@@ -421,6 +426,10 @@ impl<'ctx> Intrinsics<'ctx> { - - let ret_i32_take_i32 = i32_ty.fn_type(&[i32_ty_basic_md], false); - let ret_i64_take_i64 = i64_ty.fn_type(&[i64_ty_basic_md], false); -+ let ret_i32_take_i32_i32_i32 = -+ i32_ty.fn_type(&[i32_ty_basic_md, i32_ty_basic_md, i32_ty_basic_md], false); -+ let ret_i64_take_i64_i64_i64 = -+ i64_ty.fn_type(&[i64_ty_basic_md, i64_ty_basic_md, i64_ty_basic_md], false); - - let ret_f32_take_f32 = f32_ty.fn_type(&[f32_ty_basic_md], false); - let ret_f64_take_f64 = f64_ty.fn_type(&[f64_ty_basic_md], false); -@@ -581,6 +590,11 @@ impl<'ctx> Intrinsics<'ctx> { - ctpop_i64: add_function_with_attrs("llvm.ctpop.i64", ret_i64_take_i64, None), - ctpop_i8x16: add_function_with_attrs("llvm.ctpop.v16i8", ret_i8x16_take_i8x16, None), - -+ fshl_i32: add_function_with_attrs("llvm.fshl.i32", ret_i32_take_i32_i32_i32, None), -+ fshl_i64: add_function_with_attrs("llvm.fshl.i64", ret_i64_take_i64_i64_i64, None), -+ fshr_i32: add_function_with_attrs("llvm.fshr.i32", ret_i32_take_i32_i32_i32, None), -+ fshr_i64: add_function_with_attrs("llvm.fshr.i64", ret_i64_take_i64_i64_i64, None), -+ - fp_rounding_md: context.metadata_string("round.tonearest").into(), - fp_exception_md: context.metadata_string("fpexcept.strict").into(), - -@@ -1417,13 +1431,17 @@ pub enum MemoryCache<'ctx> { - ptr_to_current_length: PointerValue<'ctx>, - }, - /// The memory is always in the same place. -- Static { base_ptr: PointerValue<'ctx> }, -+ Static { -+ base_ptr: PointerValue<'ctx>, -+ ptr_to_current_length: PointerValue<'ctx>, -+ }, - } - - #[derive(Clone)] - struct TableCache<'ctx> { - ptr_to_base_ptr: PointerValue<'ctx>, - ptr_to_bounds: PointerValue<'ctx>, -+ ptr_to_anyfuncs_ptr: PointerValue<'ctx>, - } - - #[derive(Clone, Copy)] -@@ -1567,13 +1585,13 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - intrinsics.vmmemory_definition_base_element, - "", - )); -+ let current_length_ptr = err!(cache_builder.build_struct_gep( -+ intrinsics.vmmemory_definition_ty, -+ memory_definition_ptr, -+ intrinsics.vmmemory_definition_current_length_element, -+ "", -+ )); - let value = if let MemoryStyle::Dynamic { .. } = memory_style { -- let current_length_ptr = err!(cache_builder.build_struct_gep( -- intrinsics.vmmemory_definition_ty, -- memory_definition_ptr, -- intrinsics.vmmemory_definition_current_length_element, -- "", -- )); - MemoryCache::Dynamic { - ptr_to_base_ptr: base_ptr, - ptr_to_current_length: current_length_ptr, -@@ -1587,7 +1605,10 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - format!("memory base_ptr {}", index.as_u32()), - base_ptr.as_instruction_value().unwrap(), - ); -- MemoryCache::Static { base_ptr } -+ MemoryCache::Static { -+ base_ptr, -+ ptr_to_current_length: current_length_ptr, -+ } - }; - - self.cached_memories.insert(index, value); -@@ -1604,7 +1625,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - ctx_ptr_value: PointerValue<'ctx>, - offsets: &VMOffsets, - builder: &Builder<'ctx>, -- ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { -+ ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { - if let Some(local_table_index) = wasm_module.local_table_index(table_index) { - let offset = intrinsics.i64_ty.const_int( - offsets -@@ -1627,7 +1648,18 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - unsafe { err!(builder.build_gep(intrinsics.i8_ty, ctx_ptr_value, &[offset], "")) }; - let ptr_to_bounds = err!(builder.build_bit_cast(ptr_to_bounds, intrinsics.ptr_ty, "")) - .into_pointer_value(); -- Ok((ptr_to_base_ptr, ptr_to_bounds)) -+ let offset = intrinsics.i64_ty.const_int( -+ offsets -+ .vmctx_vmtable_definition_anyfuncs(local_table_index) -+ .into(), -+ false, -+ ); -+ let ptr_to_anyfuncs_ptr = -+ unsafe { err!(builder.build_gep(intrinsics.i8_ty, ctx_ptr_value, &[offset], "")) }; -+ let ptr_to_anyfuncs_ptr = -+ err!(builder.build_bit_cast(ptr_to_anyfuncs_ptr, intrinsics.ptr_ty, "")) -+ .into_pointer_value(); -+ Ok((ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr)) - } else { - let offset = intrinsics.i64_ty.const_int( - offsets.vmctx_vmtable_import_definition(table_index).into(), -@@ -1663,7 +1695,15 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - unsafe { err!(builder.build_gep(intrinsics.i8_ty, definition_ptr, &[offset], "")) }; - let ptr_to_bounds = err!(builder.build_bit_cast(ptr_to_bounds, intrinsics.ptr_ty, "")) - .into_pointer_value(); -- Ok((ptr_to_base_ptr, ptr_to_bounds)) -+ let offset = intrinsics -+ .i64_ty -+ .const_int(offsets.vmtable_definition_anyfuncs().into(), false); -+ let ptr_to_anyfuncs_ptr = -+ unsafe { err!(builder.build_gep(intrinsics.i8_ty, definition_ptr, &[offset], "")) }; -+ let ptr_to_anyfuncs_ptr = -+ err!(builder.build_bit_cast(ptr_to_anyfuncs_ptr, intrinsics.ptr_ty, "")) -+ .into_pointer_value(); -+ Ok((ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr)) - } - } - -@@ -1673,7 +1713,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - intrinsics: &Intrinsics<'ctx>, - module: &Module<'ctx>, - body_builder: &Builder<'ctx>, -- ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { -+ ) -> Result<(PointerValue<'ctx>, PointerValue<'ctx>, PointerValue<'ctx>), CompileError> { - let (cached_tables, wasm_module, ctx_ptr_value, cache_builder, offsets) = ( - &mut self.cached_tables, - self.wasm_module, -@@ -1689,8 +1729,8 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - )) - })?; - -- // If the table is growable, it may change, so we can't cache the pointers; they need to -- // go directly in the function body at the point where they're needed -+ // If the table is growable, it may change, so keep preparing the field -+ // pointers in the function body at the point where the table is used. - if is_growable { - Self::build_table_prepare( - table_index, -@@ -1705,22 +1745,25 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - let TableCache { - ptr_to_base_ptr, - ptr_to_bounds, -+ ptr_to_anyfuncs_ptr, - } = match cached_tables.entry(table_index) { - Entry::Occupied(entry) => entry.get().clone(), - Entry::Vacant(entry) => { -- let (ptr_to_base_ptr, ptr_to_bounds) = Self::build_table_prepare( -- table_index, -- intrinsics, -- module, -- wasm_module, -- ctx_ptr_value, -- offsets, -- cache_builder, -- )?; -+ let (ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr) = -+ Self::build_table_prepare( -+ table_index, -+ intrinsics, -+ module, -+ wasm_module, -+ ctx_ptr_value, -+ offsets, -+ cache_builder, -+ )?; - - let v = TableCache { - ptr_to_base_ptr, - ptr_to_bounds, -+ ptr_to_anyfuncs_ptr, - }; - - entry.insert(v.clone()); -@@ -1729,7 +1772,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - } - }; - -- Ok((ptr_to_base_ptr, ptr_to_bounds)) -+ Ok((ptr_to_base_ptr, ptr_to_bounds, ptr_to_anyfuncs_ptr)) - } - } - -@@ -1740,7 +1783,7 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - module: &Module<'ctx>, - body_builder: &Builder<'ctx>, - ) -> Result<(PointerValue<'ctx>, IntValue<'ctx>), CompileError> { -- let (ptr_to_base_ptr, ptr_to_bounds) = -+ let (ptr_to_base_ptr, ptr_to_bounds, _) = - self.table_prepare(index, intrinsics, module, body_builder)?; - - // Safe to unwrap since an out-of-bounds index will be caught be table_prepare -@@ -1769,6 +1812,34 @@ impl<'ctx, 'a> CtxType<'ctx, 'a> { - Ok((base_ptr, bounds)) - } - -+ pub fn table_anyfuncs( -+ &mut self, -+ index: TableIndex, -+ intrinsics: &Intrinsics<'ctx>, -+ module: &Module<'ctx>, -+ body_builder: &Builder<'ctx>, -+ ) -> Result, CompileError> { -+ let (_, _, ptr_to_anyfuncs_ptr) = -+ self.table_prepare(index, intrinsics, module, body_builder)?; -+ -+ let builder = if is_table_growable(self.wasm_module, index).unwrap() { -+ &body_builder -+ } else { -+ &self.cache_builder -+ }; -+ -+ let anyfuncs = -+ err!(builder.build_load(intrinsics.ptr_ty, ptr_to_anyfuncs_ptr, "table_anyfuncs")) -+ .into_pointer_value(); -+ tbaa_label( -+ module, -+ intrinsics, -+ format!("table_anyfuncs {}", index.index()), -+ anyfuncs.as_instruction_value().unwrap(), -+ ); -+ Ok(anyfuncs) -+ } -+ - // Return a pointer to the beginning of a local funcref Table (a pointer related to vmctx). - pub fn fixed_funcref_table_anyfuncs( - &self, -diff --git a/lib/compiler-llvm/src/translator/trampoline.rs b/lib/compiler-llvm/src/translator/trampoline.rs -index e3a9eae..5075136 100644 ---- a/lib/compiler-llvm/src/translator/trampoline.rs -+++ b/lib/compiler-llvm/src/translator/trampoline.rs -@@ -358,6 +358,7 @@ impl FuncTrampoline { - compact_unwind_section_indices, - gcc_except_table_section_indices, - data_dw_ref_personality_section_indices, -+ .. - } = load_object_file( - mem_buf_slice, - &self.func_section, -diff --git a/lib/compiler/src/artifact_builders/artifact_builder.rs b/lib/compiler/src/artifact_builders/artifact_builder.rs -index 14fc0f4..8a7ff25 100644 ---- a/lib/compiler/src/artifact_builders/artifact_builder.rs -+++ b/lib/compiler/src/artifact_builders/artifact_builder.rs -@@ -106,7 +106,6 @@ impl ArtifactBuild { - translation.function_body_inputs, - progress_callback, - )?; -- - let data_initializers = translation - .data_initializers - .iter() -@@ -262,6 +261,126 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuild { - } - } - -+/// Runtime metadata materialized from a serialized artifact after its code, -+/// relocations, unwind sections, and frame registrations have been installed. -+/// -+/// Unlike [`ArtifactBuildFromArchive`], this representation does not retain the -+/// complete serialized archive. It is intended for sealed, precompiled-only -+/// runtimes where re-serialization is neither required nor desirable. Keeping -+/// only the metadata used by later instantiations lets immutable AOT pages be -+/// reclaimed once checked deserialization has completed. -+#[derive(Debug)] -+#[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] -+pub struct DetachedArtifactBuild { -+ compile_info: CompileModuleInfo, -+ data_initializers: Box<[OwnedDataInitializer]>, -+ cpu_features: u64, -+ // This is staging storage only. Registration transfers the map into the -+ // global trap registry, which then becomes its sole long-lived owner. -+ function_frame_info: Option>, -+} -+ -+impl DetachedArtifactBuild { -+ /// Materialize only metadata that remains live after allocation and linking. -+ pub fn from_archive(archive: &ArtifactBuildFromArchive) -> Result { -+ let archived = archive.cell.borrow_dependent(); -+ let data_initializers = rkyv::deserialize::<_, RkyvError>(archived.data_initializers) -+ .map_err(|e| DeserializeError::CorruptedBinary(format!("{e:?}")))?; -+ let function_frame_info = rkyv::deserialize::<_, RkyvError>( -+ &archived.compilation.function_frame_info, -+ ) -+ .map_err(|e| DeserializeError::CorruptedBinary(format!("{e:?}")))?; -+ -+ Ok(Self { -+ compile_info: archive.compile_info.clone(), -+ data_initializers, -+ cpu_features: archived.cpu_features, -+ function_frame_info: Some(function_frame_info), -+ }) -+ } -+ -+ /// Transfer frame metadata to the global trap registration. -+ /// -+ /// A detached build stages an owned map only until code registration. The -+ /// returned map must be moved into [`crate::FrameInfosVariant::Owned`] so -+ /// the detached artifact does not retain a duplicate of all trap and -+ /// address-map allocations. -+ pub(crate) fn take_frame_info_for_registration( -+ &mut self, -+ ) -> Result, DeserializeError> { -+ self.function_frame_info.take().ok_or_else(|| { -+ DeserializeError::Generic( -+ "detached artifact frame metadata was already transferred before registration" -+ .to_string(), -+ ) -+ }) -+ } -+ -+ #[cfg(test)] -+ /// Construct a detached build with frame metadata for engine ownership tests. -+ pub(crate) fn for_frame_info_test( -+ function_frame_info: PrimaryMap, -+ ) -> Self { -+ Self { -+ compile_info: CompileModuleInfo { -+ features: Features::default(), -+ module: Arc::new(ModuleInfo::default()), -+ memory_styles: PrimaryMap::new(), -+ table_styles: PrimaryMap::new(), -+ }, -+ data_initializers: Vec::new().into_boxed_slice(), -+ cpu_features: 0, -+ function_frame_info: Some(function_frame_info), -+ } -+ } -+} -+ -+impl<'a> ArtifactCreate<'a> for DetachedArtifactBuild { -+ type OwnedDataInitializer = &'a OwnedDataInitializer; -+ type OwnedDataInitializerIterator = core::slice::Iter<'a, OwnedDataInitializer>; -+ -+ fn create_module_info(&self) -> Arc { -+ self.compile_info.module.clone() -+ } -+ -+ fn set_module_info_name(&mut self, name: String) -> bool { -+ Arc::get_mut(&mut self.compile_info.module).is_some_and(|module_info| { -+ module_info.name = Some(name); -+ true -+ }) -+ } -+ -+ fn module_info(&self) -> &ModuleInfo { -+ &self.compile_info.module -+ } -+ -+ fn features(&self) -> &Features { -+ &self.compile_info.features -+ } -+ -+ fn cpu_features(&self) -> EnumSet { -+ EnumSet::from_u64(self.cpu_features) -+ } -+ -+ fn data_initializers(&'a self) -> Self::OwnedDataInitializerIterator { -+ self.data_initializers.iter() -+ } -+ -+ fn memory_styles(&self) -> &PrimaryMap { -+ &self.compile_info.memory_styles -+ } -+ -+ fn table_styles(&self) -> &PrimaryMap { -+ &self.compile_info.table_styles -+ } -+ -+ fn serialize(&self) -> Result, SerializeError> { -+ Err(SerializeError::Generic( -+ "detached runtime artifacts cannot be re-serialized".to_string(), -+ )) -+ } -+} -+ - /// Module loaded from an archive. Since `CompileModuleInfo` is part of the public - /// interface of this crate and has to be mutable, it has to be deserialized completely. - #[derive(Debug)] -diff --git a/lib/compiler/src/artifact_builders/mod.rs b/lib/compiler/src/artifact_builders/mod.rs -index acadda9..aff8e89 100644 ---- a/lib/compiler/src/artifact_builders/mod.rs -+++ b/lib/compiler/src/artifact_builders/mod.rs -@@ -3,7 +3,9 @@ - mod artifact_builder; - mod trampoline; - --pub use self::artifact_builder::{ArtifactBuild, ArtifactBuildFromArchive, ModuleFromArchive}; -+pub use self::artifact_builder::{ -+ ArtifactBuild, ArtifactBuildFromArchive, DetachedArtifactBuild, ModuleFromArchive, -+}; - pub use self::trampoline::get_libcall_trampoline; - #[cfg(feature = "compiler")] - pub use self::trampoline::*; -diff --git a/lib/compiler/src/engine/artifact.rs b/lib/compiler/src/engine/artifact.rs -index d807fdf..7847369 100644 ---- a/lib/compiler/src/engine/artifact.rs -+++ b/lib/compiler/src/engine/artifact.rs -@@ -9,16 +9,20 @@ use std::sync::{ - #[cfg(feature = "compiler")] - use crate::ModuleEnvironment; - use crate::{ -- ArtifactBuild, ArtifactBuildFromArchive, ArtifactCreate, Engine, EngineInner, Features, -- FrameInfosVariant, FunctionExtent, GlobalFrameInfoRegistration, InstantiationError, Tunables, -+ ArtifactBuild, ArtifactBuildFromArchive, ArtifactCreate, CodeMemoryId, DetachedArtifactBuild, -+ Engine, EngineInner, Features, FrameInfosVariant, FunctionExtent, GlobalFrameInfoRegistration, -+ InstantiationError, Tunables, - engine::{link::link_module, resolver::resolve_tags}, - lib::std::vec::IntoIter, - register_frame_info, resolve_imports, -- serialize::{MetadataHeader, SerializableModule}, -- types::relocation::{RelocationLike, RelocationTarget}, -+ serialize::{ArchivedSerializableModule, MetadataHeader, SerializableModule}, -+ types::{ -+ module::CompileModuleInfo, -+ relocation::{RelocationLike, RelocationTarget}, -+ }, - }; - #[cfg(feature = "static-artifact-create")] --use crate::{Compiler, FunctionBodyData, ModuleTranslationState, types::module::CompileModuleInfo}; -+use crate::{Compiler, FunctionBodyData, ModuleTranslationState}; - #[cfg(any(feature = "static-artifact-create", feature = "static-artifact-load"))] - use crate::{serialize::SerializableCompilation, types::symbols::ModuleMetadata}; - -@@ -33,11 +37,13 @@ use crate::object::{ - Object, ObjectMetadataBuilder, emit_compilation, emit_data, get_object_for_target, - }; - -+#[cfg(feature = "compiler")] -+use wasmer_types::CompilationProgressCallback; - use wasmer_types::{ -- ArchivedDataInitializerLocation, ArchivedOwnedDataInitializer, CompilationProgressCallback, -- CompileError, DataInitializer, DataInitializerLike, DataInitializerLocation, -- DataInitializerLocationLike, DeserializeError, FunctionIndex, LocalFunctionIndex, MemoryIndex, -- ModuleInfo, OwnedDataInitializer, SerializeError, SignatureIndex, TableIndex, -+ ArchivedDataInitializerLocation, ArchivedOwnedDataInitializer, CompileError, DataInitializer, -+ DataInitializerLike, DataInitializerLocation, DataInitializerLocationLike, DeserializeError, -+ FunctionIndex, LocalFunctionIndex, MemoryIndex, ModuleHash, ModuleInfo, OwnedDataInitializer, -+ SerializeError, SignatureIndex, TableIndex, VMOffsets, - entity::{BoxedSlice, PrimaryMap}, - target::{CpuFeature, Target}, - }; -@@ -58,13 +64,24 @@ pub struct AllocatedArtifact { - // using 'Artifact::take_frame_info_registration' method - // so the GloabelFrameInfo and MMap stays in sync and get dropped at the same time - frame_info_registration: Option, -- finished_functions: BoxedSlice, -+ // These immutable pointer tables are shared by every VM instance of this -+ // artifact. The executable allocation itself remains owned by EngineInner. -+ finished_functions: Arc>, - - #[cfg_attr(feature = "artifact-size", loupe(skip))] -- finished_function_call_trampolines: BoxedSlice, -+ finished_function_call_trampolines: Arc>, - finished_dynamic_function_trampolines: BoxedSlice, - signatures: BoxedSlice, - finished_function_lengths: BoxedSlice, -+ -+ /// Host-layout offsets derived from this artifact's immutable module. -+ /// -+ /// Instance creation clones this compact value instead of rescanning the -+ /// module and recomputing every VMContext offset. The module name is the -+ /// only artifact metadata that can change after allocation, and it is not -+ /// an input to `VMOffsets`. -+ #[cfg_attr(feature = "artifact-size", loupe(skip))] -+ vm_offsets: VMOffsets, - } - - #[derive(Debug, PartialEq, Eq, PartialOrd, Ord)] -@@ -101,12 +118,72 @@ impl Default for ArtifactId { - #[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] - pub struct Artifact { - id: ArtifactId, -+ code_memory_id: Option, - artifact: ArtifactBuildVariant, - // The artifact will only be allocated in memory in case we can execute it - // (that means, if the target != host then this will be None). - allocated: Option, - } - -+/// A deserialized artifact whose published executable memory has not escaped -+/// its activation transaction. -+/// -+/// Dropping this value releases the artifact and then removes its exact -+/// engine-owned `CodeMemory` allocation, which deregisters frame and unwind -+/// metadata before unmapping code. [`PendingArtifact::commit`] transfers the -+/// artifact to its caller and permanently commits that allocation. -+#[doc(hidden)] -+pub struct PendingArtifact { -+ artifact: Option, -+ engine: Engine, -+ code_memory_id: Option, -+} -+ -+impl PendingArtifact { -+ fn new(engine: &Engine, artifact: Artifact) -> Self { -+ let code_memory_id = artifact.code_memory_id; -+ Self { -+ artifact: Some(artifact), -+ engine: engine.clone(), -+ code_memory_id, -+ } -+ } -+ -+ /// Borrow the validated artifact while rollback ownership remains pending. -+ pub fn artifact(&self) -> &Artifact { -+ self.artifact -+ .as_ref() -+ .expect("pending artifact must exist before commit") -+ } -+ -+ /// Return the module hash while rollback ownership remains pending. -+ pub fn module_hash(&self) -> Option { -+ self.artifact().module_info().hash -+ } -+ -+ /// Commit executable-memory ownership and return the activated artifact. -+ pub fn commit(mut self) -> Artifact { -+ self.code_memory_id = None; -+ self.artifact -+ .take() -+ .expect("pending artifact may only be committed once") -+ } -+} -+ -+impl Drop for PendingArtifact { -+ fn drop(&mut self) { -+ let artifact = self.artifact.take(); -+ drop(artifact); -+ if let Some(code_memory_id) = self.code_memory_id.take() { -+ let removed = self.engine.rollback_code_memory_activation(code_memory_id); -+ debug_assert!( -+ removed, -+ "pending artifact lost rollback ownership of its code memory" -+ ); -+ } -+ } -+} -+ - /// Artifacts may be created as the result of the compilation of a wasm - /// module, corresponding to `ArtifactBuildVariant::Plain`, or loaded - /// from an archive, corresponding to `ArtifactBuildVariant::Archived`. -@@ -115,9 +192,64 @@ pub struct Artifact { - pub enum ArtifactBuildVariant { - Plain(ArtifactBuild), - Archived(ArtifactBuildFromArchive), -+ Detached(DetachedArtifactBuild), - } - - impl Artifact { -+ fn checked_serialized_module( -+ bytes: &[u8], -+ ) -> Result<&ArchivedSerializableModule, DeserializeError> { -+ if !ArtifactBuild::is_deserializable(bytes) { -+ return Err(DeserializeError::Incompatible( -+ "The provided bytes are not a Wasmer universal artifact".to_string(), -+ )); -+ } -+ -+ let bytes = Self::get_byte_slice(bytes, ArtifactBuild::MAGIC_HEADER.len(), bytes.len())?; -+ let metadata_len = MetadataHeader::parse(bytes)?; -+ let metadata_slice = Self::get_byte_slice(bytes, MetadataHeader::LEN, bytes.len())?; -+ let metadata_slice = Self::get_byte_slice(metadata_slice, 0, metadata_len)?; -+ SerializableModule::archive_from_slice_checked(metadata_slice) -+ } -+ -+ fn validate_cpu_features( -+ target: &Target, -+ cpu_features: EnumSet, -+ ) -> Result<(), DeserializeError> { -+ if target.is_native() && !target.cpu_features().is_superset(cpu_features) { -+ return Err(DeserializeError::Incompatible(format!( -+ "Some CPU Features needed for the artifact are missing: {:?}", -+ cpu_features.difference(*target.cpu_features()) -+ ))); -+ } -+ Ok(()) -+ } -+ -+ /// Inspect a serialized universal artifact without allocating or publishing -+ /// executable code. -+ /// -+ /// This performs the same magic, metadata ABI, checked archive, and native -+ /// CPU-feature validation as checked deserialization, then returns the -+ /// module hash embedded by the compiler. Static-object artifacts are not -+ /// accepted because inspecting them requires the static artifact loader. -+ pub fn inspect_serialized( -+ engine: &Engine, -+ bytes: &[u8], -+ ) -> Result { -+ let archived = Self::checked_serialized_module(bytes)?; -+ let compile_info: CompileModuleInfo = -+ rkyv::deserialize::<_, rkyv::rancor::Error>(&archived.compile_info) -+ .map_err(|error| DeserializeError::CorruptedBinary(error.to_string()))?; -+ let cpu_features = EnumSet::from_u64(archived.cpu_features.to_native()); -+ Self::validate_cpu_features(engine.target(), cpu_features)?; -+ -+ compile_info.module.hash.ok_or_else(|| { -+ DeserializeError::CorruptedBinary( -+ "serialized artifact does not contain an embedded module hash".to_string(), -+ ) -+ }) -+ } -+ - /// Compile a data buffer into a `ArtifactBuild`, which may then be instantiated. - #[cfg(feature = "compiler")] - pub fn new( -@@ -222,14 +354,7 @@ impl Artifact { - } - - let artifact = ArtifactBuildFromArchive::try_new(bytes, |bytes| { -- let bytes = -- Self::get_byte_slice(bytes, ArtifactBuild::MAGIC_HEADER.len(), bytes.len())?; -- -- let metadata_len = MetadataHeader::parse(bytes)?; -- let metadata_slice = Self::get_byte_slice(bytes, MetadataHeader::LEN, bytes.len())?; -- let metadata_slice = Self::get_byte_slice(metadata_slice, 0, metadata_len)?; -- -- SerializableModule::archive_from_slice_checked(metadata_slice) -+ Self::checked_serialized_module(bytes.as_ref()) - })?; - - let mut inner_engine = engine.inner_mut(); -@@ -241,6 +366,56 @@ impl Artifact { - } - } - -+ /// Deserialize and validate a serialized artifact, then detach the -+ /// serialized archive once all runtime metadata has been materialized. -+ /// -+ /// This is intended for sealed, precompiled-only executors. The resulting -+ /// artifact can be instantiated normally but cannot be re-serialized. -+ /// Releasing the archive avoids pinning code and relocation bytes that have -+ /// already been copied, linked, and published into executable memory. -+ /// -+ /// # Safety -+ /// See [`Self::deserialize`]. -+ pub unsafe fn deserialize_detached( -+ engine: &Engine, -+ bytes: OwnedBuffer, -+ ) -> Result { -+ unsafe { Self::deserialize_detached_pending(engine, bytes) }.map(PendingArtifact::commit) -+ } -+ -+ /// Deserialize a detached artifact while retaining rollback ownership of -+ /// its published executable memory. -+ /// -+ /// Callers that perform fallible admission or evidence work after -+ /// deserialization should use this entry point and call -+ /// [`PendingArtifact::commit`] only after those steps succeed. -+ /// -+ /// # Safety -+ /// See [`Self::deserialize`]. -+ pub unsafe fn deserialize_detached_pending( -+ engine: &Engine, -+ bytes: OwnedBuffer, -+ ) -> Result { -+ unsafe { -+ let artifact = if !ArtifactBuild::is_deserializable(bytes.as_ref()) { -+ Self::deserialize(engine, bytes)? -+ } else { -+ let artifact = ArtifactBuildFromArchive::try_new(bytes, |bytes| { -+ Self::checked_serialized_module(bytes.as_ref()) -+ })?; -+ -+ let mut inner_engine = engine.inner_mut(); -+ Self::from_parts_with_archive_policy( -+ &mut inner_engine, -+ ArtifactBuildVariant::Archived(artifact), -+ engine.target(), -+ true, -+ )? -+ }; -+ Ok(PendingArtifact::new(engine, artifact)) -+ } -+ } -+ - /// Deserialize a serialized artifact. - /// - /// NOTE: You should prefer [`Self::deserialize`]. -@@ -293,23 +468,34 @@ impl Artifact { - engine_inner: &mut EngineInner, - artifact: ArtifactBuildVariant, - target: &Target, -+ ) -> Result { -+ Self::from_parts_with_archive_policy(engine_inner, artifact, target, false) -+ } -+ -+ fn from_parts_with_archive_policy( -+ engine_inner: &mut EngineInner, -+ artifact: ArtifactBuildVariant, -+ target: &Target, -+ detach_archive: bool, - ) -> Result { - if !target.is_native() { - return Ok(Self { - id: Default::default(), -+ code_memory_id: None, - artifact, - allocated: None, - }); - } else { -- // check if cpu features are compatible before anything else -- let cpu_features = artifact.cpu_features(); -- if !target.cpu_features().is_superset(cpu_features) { -- return Err(DeserializeError::Incompatible(format!( -- "Some CPU Features needed for the artifact are missing: {:?}", -- cpu_features.difference(*target.cpu_features()) -- ))); -- } -+ // Check CPU compatibility before allocating executable memory. -+ Self::validate_cpu_features(target, artifact.cpu_features())?; - } -+ -+ // Allocation, relocation, publication, unwind/frame registration, and -+ // detached metadata materialization are one engine-owned transaction. -+ // Any error or panic before the final commit removes every allocation -+ // made by this artifact build. -+ let mut code_memory_transaction = engine_inner.begin_code_memory_transaction(); -+ let engine_inner = code_memory_transaction.inner_mut(); - let module_info = artifact.module_info(); - let ( - finished_functions, -@@ -331,7 +517,15 @@ impl Artifact { - a.get_dynamic_function_trampolines_ref().values(), - a.get_custom_sections_ref().values(), - )?, -+ ArtifactBuildVariant::Detached(_) => { -+ unreachable!("detached artifacts have already been allocated and linked") -+ } - }; -+ let code_memory_id = engine_inner.last_code_memory_id().ok_or_else(|| { -+ DeserializeError::Generic( -+ "artifact allocation completed without engine-owned code memory".to_string(), -+ ) -+ })?; - - let get_got_address: Box Option> = match &artifact { - ArtifactBuildVariant::Plain(p) => { -@@ -369,6 +563,9 @@ impl Artifact { - Box::new(|_: RelocationTarget| None) - } - } -+ ArtifactBuildVariant::Detached(_) => { -+ unreachable!("detached artifacts have already been allocated and linked") -+ } - }; - - match &artifact { -@@ -402,6 +599,9 @@ impl Artifact { - a.get_libcall_trampoline_len(), - &get_got_address, - ), -+ ArtifactBuildVariant::Detached(_) => { -+ unreachable!("detached artifacts have already been allocated and linked") -+ } - }; - - // Compute indices into the shared signature table. -@@ -429,6 +629,9 @@ impl Artifact { - a.get_custom_sections_ref()[v].bytes.len(), - ) - }), -+ ArtifactBuildVariant::Detached(_) => { -+ unreachable!("detached artifacts have already been allocated and linked") -+ } - }; - #[allow(unused_variables)] - let compact_unwind = match &artifact { -@@ -446,6 +649,9 @@ impl Artifact { - ) - }) - } -+ ArtifactBuildVariant::Detached(_) => { -+ unreachable!("detached artifacts have already been allocated and linked") -+ } - }; - - #[cfg(all(not(target_arch = "wasm32"), feature = "compiler"))] -@@ -454,7 +660,7 @@ impl Artifact { - } - - // Make all code compiled thus far executable. -- engine_inner.publish_compiled_code(); -+ engine_inner.publish_compiled_code()?; - - #[cfg(all(target_os = "macos", target_arch = "aarch64"))] - if let Some(compact_unwind) = compact_unwind { -@@ -476,19 +682,34 @@ impl Artifact { - .map(|extent| extent.length) - .collect::>() - .into_boxed_slice(); -- let finished_functions = finished_functions -- .values() -- .map(|extent| extent.ptr) -- .collect::>() -- .into_boxed_slice(); -+ let finished_functions = Arc::new( -+ finished_functions -+ .values() -+ .map(|extent| extent.ptr) -+ .collect::>() -+ .into_boxed_slice(), -+ ); - let finished_function_call_trampolines = -- finished_function_call_trampolines.into_boxed_slice(); -+ Arc::new(finished_function_call_trampolines.into_boxed_slice()); - let finished_dynamic_function_trampolines = - finished_dynamic_function_trampolines.into_boxed_slice(); - let signatures = signatures.into_boxed_slice(); - -+ let artifact = if detach_archive { -+ match artifact { -+ ArtifactBuildVariant::Archived(archive) => { -+ ArtifactBuildVariant::Detached(DetachedArtifactBuild::from_archive(&archive)?) -+ } -+ artifact => artifact, -+ } -+ } else { -+ artifact -+ }; -+ let vm_offsets = VMOffsets::new(std::mem::size_of::() as u8, artifact.module_info()); -+ - let mut artifact = Self { - id: Default::default(), -+ code_memory_id: Some(code_memory_id), - artifact, - allocated: Some(AllocatedArtifact { - frame_info_registered: false, -@@ -498,6 +719,7 @@ impl Artifact { - finished_dynamic_function_trampolines, - signatures, - finished_function_lengths, -+ vm_offsets, - }), - }; - -@@ -508,6 +730,7 @@ impl Artifact { - engine_inner.register_frame_info(frame_info); - } - -+ code_memory_transaction.commit(); - Ok(artifact) - } - -@@ -583,6 +806,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.create_module_info(), - Self::Archived(artifact) => artifact.create_module_info(), -+ Self::Detached(artifact) => artifact.create_module_info(), - } - } - -@@ -590,6 +814,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.set_module_info_name(name), - Self::Archived(artifact) => artifact.set_module_info_name(name), -+ Self::Detached(artifact) => artifact.set_module_info_name(name), - } - } - -@@ -597,6 +822,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.module_info(), - Self::Archived(artifact) => artifact.module_info(), -+ Self::Detached(artifact) => artifact.module_info(), - } - } - -@@ -604,6 +830,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.features(), - Self::Archived(artifact) => artifact.features(), -+ Self::Detached(artifact) => artifact.features(), - } - } - -@@ -611,6 +838,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.cpu_features(), - Self::Archived(artifact) => artifact.cpu_features(), -+ Self::Detached(artifact) => artifact.cpu_features(), - } - } - -@@ -618,6 +846,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.memory_styles(), - Self::Archived(artifact) => artifact.memory_styles(), -+ Self::Detached(artifact) => artifact.memory_styles(), - } - } - -@@ -625,6 +854,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.table_styles(), - Self::Archived(artifact) => artifact.table_styles(), -+ Self::Detached(artifact) => artifact.table_styles(), - } - } - -@@ -640,6 +870,11 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - .map(OwnedDataInitializerVariant::Archived) - .collect::>() - .into_iter(), -+ Self::Detached(artifact) => artifact -+ .data_initializers() -+ .map(OwnedDataInitializerVariant::Plain) -+ .collect::>() -+ .into_iter(), - } - } - -@@ -647,6 +882,7 @@ impl<'a> ArtifactCreate<'a> for ArtifactBuildVariant { - match self { - Self::Plain(artifact) => artifact.serialize(), - Self::Archived(artifact) => artifact.serialize(), -+ Self::Detached(artifact) => artifact.serialize(), - } - } - } -@@ -741,27 +977,22 @@ impl Artifact { - .collect::>() - .into_boxed_slice(); - -- let frame_info_registration = &mut self -- .allocated -- .as_mut() -- .expect("It must be allocated") -- .frame_info_registration; -+ let module_info = self.artifact.create_module_info(); -+ let frame_infos = match &mut self.artifact { -+ ArtifactBuildVariant::Plain(p) => { -+ FrameInfosVariant::Owned(p.get_frame_info_ref().clone()) -+ } -+ ArtifactBuildVariant::Archived(a) => FrameInfosVariant::Archived(a.clone()), -+ ArtifactBuildVariant::Detached(a) => { -+ FrameInfosVariant::Owned(a.take_frame_info_for_registration()?) -+ } -+ }; -+ let frame_info_registration = -+ register_frame_info(module_info, &finished_function_extents, frame_infos); - -- *frame_info_registration = register_frame_info( -- self.artifact.create_module_info(), -- &finished_function_extents, -- match &self.artifact { -- ArtifactBuildVariant::Plain(p) => { -- FrameInfosVariant::Owned(p.get_frame_info_ref().clone()) -- } -- ArtifactBuildVariant::Archived(a) => FrameInfosVariant::Archived(a.clone()), -- }, -- ); -- -- self.allocated -- .as_mut() -- .expect("It must be allocated") -- .frame_info_registered = true; -+ let allocated = self.allocated.as_mut().expect("It must be allocated"); -+ allocated.frame_info_registration = frame_info_registration; -+ allocated.frame_info_registered = true; - - Ok(()) - } -@@ -786,6 +1017,16 @@ impl Artifact { - .finished_functions - } - -+ fn shared_finished_functions(&self) -> Arc> { -+ Arc::clone( -+ &self -+ .allocated -+ .as_ref() -+ .expect("It must be allocated") -+ .finished_functions, -+ ) -+ } -+ - /// Returns the function call trampolines allocated in memory of this - /// `Artifact`, ready to be run. - pub fn finished_function_call_trampolines(&self) -> &BoxedSlice { -@@ -796,6 +1037,18 @@ impl Artifact { - .finished_function_call_trampolines - } - -+ fn shared_finished_function_call_trampolines( -+ &self, -+ ) -> Arc> { -+ Arc::clone( -+ &self -+ .allocated -+ .as_ref() -+ .expect("It must be allocated") -+ .finished_function_call_trampolines, -+ ) -+ } -+ - /// Returns the dynamic function trampolines allocated in memory - /// of this `Artifact`, ready to be run. - pub fn finished_dynamic_function_trampolines( -@@ -870,7 +1123,14 @@ impl Artifact { - memory_definition_locations, - table_definition_locations, - global_definition_locations, -- ) = InstanceAllocator::new(&module); -+ ) = InstanceAllocator::new_with_offsets( -+ self.allocated -+ .as_ref() -+ .expect("Artifact::instantiate called on a non-host artifact") -+ .vm_offsets -+ .clone(), -+ &module, -+ ); - let finished_memories = tunables - .create_memories( - context, -@@ -898,8 +1158,8 @@ impl Artifact { - allocator, - module, - context, -- self.finished_functions().clone(), -- self.finished_function_call_trampolines().clone(), -+ self.shared_finished_functions(), -+ self.shared_finished_function_call_trampolines(), - finished_memories, - finished_tables, - finished_globals, -@@ -1284,21 +1544,114 @@ impl Artifact { - .collect::>() - .into_boxed_slice(); - -+ // Build the variant first so its module remains available while -+ // deriving the cached host-layout offsets. -+ let artifact = ArtifactBuildVariant::Plain(artifact); -+ let vm_offsets = -+ VMOffsets::new(std::mem::size_of::() as u8, artifact.module_info()); -+ - Ok(Self { - id: Default::default(), -- artifact: ArtifactBuildVariant::Plain(artifact), -+ code_memory_id: None, -+ artifact, - allocated: Some(AllocatedArtifact { - frame_info_registered: false, - frame_info_registration: None, -- finished_functions: finished_functions.into_boxed_slice(), -- finished_function_call_trampolines: finished_function_call_trampolines -- .into_boxed_slice(), -+ finished_functions: Arc::new(finished_functions.into_boxed_slice()), -+ finished_function_call_trampolines: Arc::new( -+ finished_function_call_trampolines.into_boxed_slice(), -+ ), - finished_dynamic_function_trampolines: finished_dynamic_function_trampolines - .into_boxed_slice(), - signatures: signatures.into_boxed_slice(), - finished_function_lengths, -+ vm_offsets, - }), - }) - } - } - } -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use crate::types::function::CompiledFunctionFrameInfo; -+ use wasmer_types::{TrapCode, TrapInformation}; -+ use wasmer_vm::VMFunctionBody; -+ -+ fn empty_boxed_slice() -> BoxedSlice -+ where -+ K: wasmer_types::entity::EntityRef, -+ { -+ PrimaryMap::::new().into_boxed_slice() -+ } -+ -+ fn detached_artifact_with_one_frame(code: *const VMFunctionBody) -> Artifact { -+ let mut frame = CompiledFunctionFrameInfo::default(); -+ frame.traps.push(TrapInformation { -+ code_offset: 0, -+ trap_code: TrapCode::UnreachableCodeReached, -+ }); -+ let mut frame_infos = PrimaryMap::new(); -+ frame_infos.push(frame); -+ -+ let mut finished_functions = PrimaryMap::new(); -+ finished_functions.push(FunctionBodyPtr(code)); -+ let mut finished_function_lengths = PrimaryMap::new(); -+ finished_function_lengths.push(1); -+ -+ let artifact = -+ ArtifactBuildVariant::Detached(DetachedArtifactBuild::for_frame_info_test(frame_infos)); -+ let vm_offsets = VMOffsets::new(std::mem::size_of::() as u8, artifact.module_info()); -+ -+ Artifact { -+ id: ArtifactId::default(), -+ code_memory_id: None, -+ artifact, -+ allocated: Some(AllocatedArtifact { -+ frame_info_registered: false, -+ frame_info_registration: None, -+ finished_functions: Arc::new(finished_functions.into_boxed_slice()), -+ finished_function_call_trampolines: Arc::new(empty_boxed_slice()), -+ finished_dynamic_function_trampolines: empty_boxed_slice(), -+ signatures: empty_boxed_slice(), -+ finished_function_lengths: finished_function_lengths.into_boxed_slice(), -+ vm_offsets, -+ }), -+ } -+ } -+ -+ #[test] -+ fn detached_frame_info_registration_moves_once_and_is_idempotent() { -+ // Registration treats code addresses as opaque range keys. Back this -+ // test range with a live, uniquely allocated byte so it cannot overlap -+ // another concurrent registration. -+ let code = Box::new(0_u8); -+ let code_ptr = (&*code as *const u8).cast::(); -+ let mut artifact = detached_artifact_with_one_frame(code_ptr); -+ -+ artifact.internal_register_frame_info().unwrap(); -+ let registration = artifact -+ .internal_take_frame_info_registration() -+ .expect("one function must produce a global frame registration"); -+ -+ let detached = match &mut artifact.artifact { -+ ArtifactBuildVariant::Detached(detached) => detached, -+ _ => unreachable!("test artifact must remain detached"), -+ }; -+ let error = detached -+ .take_frame_info_for_registration() -+ .expect_err("registration must consume detached frame metadata"); -+ assert!( -+ error.to_string().contains("already transferred"), -+ "unexpected ownership error: {error}" -+ ); -+ -+ artifact -+ .internal_register_frame_info() -+ .expect("repeated registration must be a harmless no-op"); -+ assert!(artifact.internal_take_frame_info_registration().is_none()); -+ -+ drop(registration); -+ } -+} -diff --git a/lib/compiler/src/engine/code_memory.rs b/lib/compiler/src/engine/code_memory.rs -index 19166c6..c4b3739 100644 ---- a/lib/compiler/src/engine/code_memory.rs -+++ b/lib/compiler/src/engine/code_memory.rs -@@ -11,8 +11,23 @@ use crate::{ - unwind::{CompiledFunctionUnwindInfoLike, CompiledFunctionUnwindInfoReference}, - }, - }; -+use std::path::Path; -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+use std::path::PathBuf; -+use std::sync::atomic::{AtomicUsize, Ordering}; - use wasmer_vm::{Mmap, VMFunctionBody}; - -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+use std::{ -+ ffi::CStr, -+ fs::{File, OpenOptions}, -+ os::{ -+ fd::{AsRawFd, FromRawFd}, -+ unix::{fs::MetadataExt, fs::OpenOptionsExt}, -+ }, -+ sync::Arc, -+}; -+ - /// The optimal alignment for functions. - /// - /// On x86-64, this is 16 since it's what the optimizations assume. -@@ -24,26 +39,671 @@ const ARCH_FUNCTION_ALIGNMENT: usize = 16; - /// - const DATA_SECTION_ALIGNMENT: usize = 64; - -+/// Stable identity of the strict Linux x86-64 file-backed code-memory policy. -+pub const STRICT_LINUX_X86_64_CODE_MEMORY_POLICY_ID: &str = -+ "wasmer.code-memory.relocated-regular-file.linux-x86_64.v1"; -+ -+/// Process-unique identity of one engine-owned executable-code allocation. -+/// -+/// This identity exists so a sealed activation can retain rollback ownership -+/// after executable publication without relying on vector position. It is not -+/// serialized and has no cross-process meaning. -+#[doc(hidden)] -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+#[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] -+pub(crate) struct CodeMemoryId(usize); -+ -+static NEXT_CODE_MEMORY_ID: AtomicUsize = AtomicUsize::new(0); -+ -+fn next_code_memory_id() -> CodeMemoryId { -+ let id = NEXT_CODE_MEMORY_ID -+ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |id| id.checked_add(1)) -+ .expect("process exhausted unique code-memory allocation identities"); -+ CodeMemoryId(id) -+} -+ -+/// Per-engine ownership policy for relocated executable memory. -+/// -+/// Generic Wasmer engines use anonymous memory. The strict file-backed mode is -+/// deliberately opt-in and is available only on Linux x86-64. Its constructor -+/// opens and pins the selected directory; allocation never searches for a -+/// different directory or silently falls back to anonymous memory. -+#[derive(Clone, Debug)] -+pub struct CodeMemoryPolicy { -+ kind: CodeMemoryPolicyKind, -+} -+ -+#[derive(Clone, Debug)] -+enum CodeMemoryPolicyKind { -+ Anonymous, -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ StrictLinuxX86_64FileBacked(StrictLinuxX86_64FileBackedPolicy), -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+#[derive(Clone, Debug)] -+struct StrictLinuxX86_64FileBackedPolicy { -+ directory: Arc, -+ canonical_directory: Arc, -+ directory_device: u64, -+ directory_inode: u64, -+} -+ -+impl Default for CodeMemoryPolicy { -+ fn default() -> Self { -+ Self::anonymous() -+ } -+} -+ -+impl CodeMemoryPolicy { -+ /// Select ordinary anonymous executable memory. -+ pub fn anonymous() -> Self { -+ Self { -+ kind: CodeMemoryPolicyKind::Anonymous, -+ } -+ } -+ -+ /// Select strict regular-file-backed relocated code memory on Linux x86-64. -+ /// -+ /// `directory` is opened once and retained by descriptor. Every later code -+ /// allocation uses `openat(O_TMPFILE)` against that descriptor, so path -+ /// replacement after engine configuration cannot redirect allocations. -+ pub fn strict_linux_x86_64_file_backed(directory: impl AsRef) -> Result { -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ { -+ let requested_metadata = -+ std::fs::symlink_metadata(directory.as_ref()).map_err(|error| { -+ format!( -+ "inspect strict code-memory directory {}: {error}", -+ directory.as_ref().display() -+ ) -+ })?; -+ if requested_metadata.file_type().is_symlink() || !requested_metadata.is_dir() { -+ return Err(format!( -+ "strict code-memory path must be a non-symlink directory: {}", -+ directory.as_ref().display() -+ )); -+ } -+ let directory_file = OpenOptions::new() -+ .read(true) -+ .custom_flags(libc::O_CLOEXEC | libc::O_DIRECTORY | libc::O_NOFOLLOW) -+ .open(directory.as_ref()) -+ .map_err(|error| { -+ format!( -+ "open strict code-memory directory {}: {error}", -+ directory.as_ref().display() -+ ) -+ })?; -+ let metadata = directory_file.metadata().map_err(|error| { -+ format!( -+ "inspect strict code-memory directory {}: {error}", -+ directory.as_ref().display() -+ ) -+ })?; -+ if requested_metadata.dev() != metadata.dev() -+ || requested_metadata.ino() != metadata.ino() -+ { -+ return Err(format!( -+ "strict code-memory directory changed identity while being pinned: {}", -+ directory.as_ref().display() -+ )); -+ } -+ if !metadata.is_dir() { -+ return Err(format!( -+ "strict code-memory path is not a directory: {}", -+ directory.as_ref().display() -+ )); -+ } -+ if metadata.uid() != unsafe { libc::geteuid() } { -+ return Err(format!( -+ "strict code-memory directory is not owned by the effective user: {}", -+ directory.as_ref().display() -+ )); -+ } -+ if metadata.mode() & 0o7777 != 0o700 { -+ return Err(format!( -+ "strict code-memory directory must have exact mode 0700: {}", -+ directory.as_ref().display() -+ )); -+ } -+ let canonical_directory = -+ std::fs::canonicalize(directory.as_ref()).map_err(|error| { -+ format!( -+ "canonicalize pinned strict code-memory directory {}: {error}", -+ directory.as_ref().display() -+ ) -+ })?; -+ let canonical_metadata = std::fs::metadata(&canonical_directory).map_err(|error| { -+ format!( -+ "inspect canonical strict code-memory directory {}: {error}", -+ canonical_directory.display() -+ ) -+ })?; -+ if canonical_metadata.dev() != metadata.dev() -+ || canonical_metadata.ino() != metadata.ino() -+ { -+ return Err(format!( -+ "strict code-memory directory changed identity while canonicalizing: {}", -+ directory.as_ref().display() -+ )); -+ } -+ reject_memory_backed_filesystem(directory_file.as_raw_fd()).map_err(|error| { -+ format!( -+ "reject strict code-memory directory {}: {error}", -+ canonical_directory.display() -+ ) -+ })?; -+ Ok(Self { -+ kind: CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked( -+ StrictLinuxX86_64FileBackedPolicy { -+ directory: Arc::new(directory_file), -+ canonical_directory: Arc::new(canonical_directory), -+ directory_device: metadata.dev(), -+ directory_inode: metadata.ino(), -+ }, -+ ), -+ }) -+ } -+ -+ #[cfg(not(all(target_os = "linux", target_arch = "x86_64")))] -+ { -+ let _ = directory; -+ Err("strict file-backed code memory is supported only on Linux x86-64".to_string()) -+ } -+ } -+ -+ /// Return the stable policy identity. -+ pub fn id(&self) -> &'static str { -+ match self.kind { -+ CodeMemoryPolicyKind::Anonymous => "wasmer.code-memory.anonymous.v1", -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(_) => { -+ STRICT_LINUX_X86_64_CODE_MEMORY_POLICY_ID -+ } -+ } -+ } -+ -+ /// Return the pinned directory selected by a strict file-backed policy. -+ pub fn pinned_directory(&self) -> Option<&Path> { -+ match &self.kind { -+ CodeMemoryPolicyKind::Anonymous => None, -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { -+ Some(policy.canonical_directory.as_path()) -+ } -+ } -+ } -+ -+ /// Return the device containing the pinned strict-policy directory. -+ pub fn pinned_directory_device(&self) -> Option { -+ match &self.kind { -+ CodeMemoryPolicyKind::Anonymous => None, -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { -+ Some(policy.directory_device) -+ } -+ } -+ } -+ -+ /// Return the inode of the pinned strict-policy directory. -+ pub fn pinned_directory_inode(&self) -> Option { -+ match &self.kind { -+ CodeMemoryPolicyKind::Anonymous => None, -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { -+ Some(policy.directory_inode) -+ } -+ } -+ } -+} -+ -+enum CodeMemoryBacking { -+ Anonymous(Mmap), -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ StrictLinuxX86_64FileBacked(StrictLinuxX86_64CodeMapping), -+} -+ -+impl CodeMemoryBacking { -+ fn empty() -> Self { -+ Self::Anonymous(Mmap::new()) -+ } -+ -+ fn allocate(policy: &CodeMemoryPolicy, len: usize) -> Result { -+ match &policy.kind { -+ CodeMemoryPolicyKind::Anonymous => Mmap::with_at_least(len).map(Self::Anonymous), -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => { -+ StrictLinuxX86_64CodeMapping::allocate(policy, len) -+ .map(Self::StrictLinuxX86_64FileBacked) -+ } -+ } -+ } -+ -+ fn as_mut_slice(&mut self) -> &mut [u8] { -+ match self { -+ Self::Anonymous(mapping) => mapping.as_mut_slice(), -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ Self::StrictLinuxX86_64FileBacked(mapping) => mapping.as_mut_slice(), -+ } -+ } -+ -+ fn len(&self) -> usize { -+ match self { -+ Self::Anonymous(mapping) => mapping.len(), -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ Self::StrictLinuxX86_64FileBacked(mapping) => mapping.len, -+ } -+ } -+ -+ fn is_empty(&self) -> bool { -+ self.len() == 0 -+ } -+ -+ fn publish(&mut self, executable_prefix_len: usize) -> Result<(), String> { -+ match self { -+ Self::Anonymous(mapping) => { -+ if executable_prefix_len == 0 { -+ return Ok(()); -+ } -+ unsafe { -+ region::protect( -+ mapping.as_mut_ptr(), -+ executable_prefix_len, -+ region::Protection::READ_EXECUTE, -+ ) -+ } -+ .map_err(|error| format!("make anonymous code memory executable: {error}")) -+ } -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ Self::StrictLinuxX86_64FileBacked(mapping) => mapping.publish(executable_prefix_len), -+ } -+ } -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+struct StrictLinuxX86_64CodeMapping { -+ ptr: usize, -+ len: usize, -+ writable_file: Option, -+ read_only_file: Option, -+ device: u64, -+ inode: u64, -+ published: bool, -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+impl StrictLinuxX86_64CodeMapping { -+ fn allocate(policy: &StrictLinuxX86_64FileBackedPolicy, len: usize) -> Result { -+ if len == 0 { -+ return Ok(Self { -+ ptr: Vec::::new().as_ptr() as usize, -+ len: 0, -+ writable_file: None, -+ read_only_file: None, -+ device: 0, -+ inode: 0, -+ published: false, -+ }); -+ } -+ let len = len.next_multiple_of(region::page::size()); -+ -+ verify_pinned_directory(policy)?; -+ let dot = c"."; -+ let descriptor = unsafe { -+ libc::openat( -+ policy.directory.as_raw_fd(), -+ dot.as_ptr(), -+ libc::O_CLOEXEC | libc::O_RDWR | libc::O_TMPFILE | libc::O_EXCL, -+ 0o600, -+ ) -+ }; -+ if descriptor < 0 { -+ return Err(format!( -+ "create O_TMPFILE in pinned code-memory directory {}: {}", -+ policy.canonical_directory.display(), -+ std::io::Error::last_os_error() -+ )); -+ } -+ let writable_file = unsafe { File::from_raw_fd(descriptor) }; -+ let metadata = validate_anonymous_regular_file(&writable_file)?; -+ if metadata.dev() != policy.directory_device { -+ return Err("strict code-memory O_TMPFILE changed filesystem device".to_string()); -+ } -+ if metadata.uid() != unsafe { libc::geteuid() } { -+ return Err( -+ "strict code-memory O_TMPFILE is not owned by the effective user".to_string(), -+ ); -+ } -+ if metadata.mode() & 0o7777 != 0o600 { -+ return Err("strict code-memory O_TMPFILE must have exact mode 0600".to_string()); -+ } -+ reject_memory_backed_filesystem(writable_file.as_raw_fd())?; -+ -+ let length: libc::off_t = len -+ .try_into() -+ .map_err(|_| "code-memory allocation does not fit off_t".to_string())?; -+ if unsafe { libc::ftruncate(writable_file.as_raw_fd(), length) } != 0 { -+ return Err(format!( -+ "ftruncate strict code-memory file to {len} bytes: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ let allocation_result = -+ unsafe { libc::posix_fallocate(writable_file.as_raw_fd(), 0, length) }; -+ if allocation_result != 0 { -+ return Err(format!( -+ "posix_fallocate strict code-memory file to {len} bytes: {}", -+ std::io::Error::from_raw_os_error(allocation_result) -+ )); -+ } -+ -+ // Probe and retain the exact read-only inode descriptor before any -+ // relocation writes begin. A missing /proc descriptor bridge is a hard -+ // admission error, never a reason to fall back after linking starts. -+ let read_only_file = reopen_read_only(&writable_file)?; -+ validate_same_anonymous_regular_file(&writable_file, &read_only_file)?; -+ -+ let ptr = unsafe { -+ libc::mmap( -+ std::ptr::null_mut(), -+ len, -+ libc::PROT_READ | libc::PROT_WRITE, -+ libc::MAP_SHARED, -+ writable_file.as_raw_fd(), -+ 0, -+ ) -+ }; -+ if ptr == libc::MAP_FAILED { -+ return Err(format!( -+ "map strict code-memory file shared read-write: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ -+ Ok(Self { -+ ptr: ptr as usize, -+ len, -+ writable_file: Some(writable_file), -+ read_only_file: Some(read_only_file), -+ device: metadata.dev(), -+ inode: metadata.ino(), -+ published: false, -+ }) -+ } -+ -+ fn as_mut_slice(&mut self) -> &mut [u8] { -+ unsafe { std::slice::from_raw_parts_mut(self.ptr as *mut u8, self.len) } -+ } -+ -+ fn publish(&mut self, executable_prefix_len: usize) -> Result<(), String> { -+ if self.len == 0 { -+ self.published = true; -+ return Ok(()); -+ } -+ if self.published { -+ return Err("strict code-memory mapping was already published".to_string()); -+ } -+ if executable_prefix_len > self.len { -+ return Err("executable code-memory prefix exceeds its mapping".to_string()); -+ } -+ let page_size = region::page::size(); -+ let executable_pages = if executable_prefix_len == 0 { -+ 0 -+ } else { -+ executable_prefix_len.next_multiple_of(page_size) -+ }; -+ if executable_pages > self.len { -+ return Err("rounded executable code-memory prefix exceeds its mapping".to_string()); -+ } -+ -+ let writable_file = self -+ .writable_file -+ .as_ref() -+ .ok_or_else(|| "strict code-memory writable descriptor is missing".to_string())?; -+ let read_only_file = self -+ .read_only_file -+ .as_ref() -+ .ok_or_else(|| "strict code-memory read-only descriptor is missing".to_string())?; -+ validate_same_anonymous_regular_file(writable_file, read_only_file)?; -+ let current_metadata = writable_file -+ .metadata() -+ .map_err(|error| format!("inspect strict code-memory inode before publish: {error}"))?; -+ if current_metadata.dev() != self.device || current_metadata.ino() != self.inode { -+ return Err("strict code-memory inode identity changed before publish".to_string()); -+ } -+ -+ if unsafe { libc::fchmod(writable_file.as_raw_fd(), 0o400) } != 0 { -+ return Err(format!( -+ "make strict code-memory inode read-only: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ if unsafe { libc::msync(self.ptr as *mut libc::c_void, self.len, libc::MS_SYNC) } != 0 { -+ return Err(format!( -+ "synchronize relocated strict code memory: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ -+ // Remove the writable mapping before dropping the only writable file -+ // description. From this point onward there is no writable alias. -+ if unsafe { libc::mprotect(self.ptr as *mut libc::c_void, self.len, libc::PROT_READ) } != 0 -+ { -+ return Err(format!( -+ "remove write permission from relocated strict code memory: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ drop(self.writable_file.take()); -+ -+ let remapped = unsafe { -+ libc::mmap( -+ self.ptr as *mut libc::c_void, -+ self.len, -+ libc::PROT_READ, -+ libc::MAP_PRIVATE | libc::MAP_FIXED, -+ read_only_file.as_raw_fd(), -+ 0, -+ ) -+ }; -+ if remapped == libc::MAP_FAILED { -+ return Err(format!( -+ "remap relocated strict code memory private read-only: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ if remapped as usize != self.ptr { -+ return Err("fixed strict code-memory remap changed its base address".to_string()); -+ } -+ if executable_pages != 0 -+ && unsafe { -+ libc::mprotect( -+ self.ptr as *mut libc::c_void, -+ executable_pages, -+ libc::PROT_READ | libc::PROT_EXEC, -+ ) -+ } != 0 -+ { -+ return Err(format!( -+ "make relocated strict code-memory prefix executable: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ -+ // The MAP_PRIVATE view is clean and reconstructible from its unlinked -+ // regular inode. Discard residency now; demanded code pages fault back -+ // as file cache instead of permanently accounting as anonymous RSS. -+ if unsafe { libc::madvise(self.ptr as *mut libc::c_void, self.len, libc::MADV_DONTNEED) } -+ != 0 -+ { -+ return Err(format!( -+ "discard clean strict code-memory pages: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ let advice_result = unsafe { -+ libc::posix_fadvise(read_only_file.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) -+ }; -+ if advice_result != 0 { -+ return Err(format!( -+ "discard strict code-memory file cache: {}", -+ std::io::Error::from_raw_os_error(advice_result) -+ )); -+ } -+ drop(self.read_only_file.take()); -+ self.published = true; -+ Ok(()) -+ } -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+impl Drop for StrictLinuxX86_64CodeMapping { -+ fn drop(&mut self) { -+ if self.len != 0 { -+ unsafe { -+ libc::munmap(self.ptr as *mut libc::c_void, self.len); -+ } -+ } -+ } -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+fn verify_pinned_directory(policy: &StrictLinuxX86_64FileBackedPolicy) -> Result<(), String> { -+ let metadata = policy -+ .directory -+ .metadata() -+ .map_err(|error| format!("inspect pinned code-memory directory: {error}"))?; -+ if !metadata.is_dir() -+ || metadata.dev() != policy.directory_device -+ || metadata.ino() != policy.directory_inode -+ { -+ return Err("pinned code-memory directory identity changed".to_string()); -+ } -+ if metadata.uid() != unsafe { libc::geteuid() } { -+ return Err("pinned code-memory directory is not owned by the effective user".to_string()); -+ } -+ if metadata.mode() & 0o7777 != 0o700 { -+ return Err("pinned code-memory directory must retain exact mode 0700".to_string()); -+ } -+ reject_memory_backed_filesystem(policy.directory.as_raw_fd()) -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+fn validate_anonymous_regular_file(file: &File) -> Result { -+ let metadata = file -+ .metadata() -+ .map_err(|error| format!("inspect anonymous code-memory file: {error}"))?; -+ if !metadata.file_type().is_file() { -+ return Err("strict code-memory O_TMPFILE inode is not a regular file".to_string()); -+ } -+ if metadata.nlink() != 0 { -+ return Err("strict code-memory O_TMPFILE inode unexpectedly has a name".to_string()); -+ } -+ Ok(metadata) -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+fn validate_same_anonymous_regular_file(writable: &File, read_only: &File) -> Result<(), String> { -+ let writable_metadata = validate_anonymous_regular_file(writable)?; -+ let read_only_metadata = validate_anonymous_regular_file(read_only)?; -+ if writable_metadata.dev() != read_only_metadata.dev() -+ || writable_metadata.ino() != read_only_metadata.ino() -+ { -+ return Err("read-only code-memory descriptor changed inode identity".to_string()); -+ } -+ let writable_flags = unsafe { libc::fcntl(writable.as_raw_fd(), libc::F_GETFL) }; -+ if writable_flags < 0 { -+ return Err(format!( -+ "inspect writable code-memory descriptor: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ if writable_flags & libc::O_ACCMODE != libc::O_RDWR { -+ return Err("code-memory relocation descriptor is not read-write".to_string()); -+ } -+ let flags = unsafe { libc::fcntl(read_only.as_raw_fd(), libc::F_GETFL) }; -+ if flags < 0 { -+ return Err(format!( -+ "inspect read-only code-memory descriptor: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ if flags & libc::O_ACCMODE != libc::O_RDONLY { -+ return Err("code-memory descriptor bridge did not remove write access".to_string()); -+ } -+ Ok(()) -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+fn reopen_read_only(file: &File) -> Result { -+ let descriptor_path = format!("/proc/self/fd/{}\0", file.as_raw_fd()); -+ let descriptor_path = CStr::from_bytes_with_nul(descriptor_path.as_bytes()) -+ .map_err(|error| format!("construct code-memory descriptor path: {error}"))?; -+ let descriptor = -+ unsafe { libc::open(descriptor_path.as_ptr(), libc::O_RDONLY | libc::O_CLOEXEC) }; -+ if descriptor < 0 { -+ return Err(format!( -+ "reopen anonymous code-memory inode read-only: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ Ok(unsafe { File::from_raw_fd(descriptor) }) -+} -+ -+#[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+fn reject_memory_backed_filesystem(descriptor: libc::c_int) -> Result<(), String> { -+ let mut status = std::mem::MaybeUninit::::uninit(); -+ if unsafe { libc::fstatfs(descriptor, status.as_mut_ptr()) } != 0 { -+ return Err(format!( -+ "identify code-memory filesystem: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ let filesystem_type = unsafe { status.assume_init() }.f_type as u64; -+ const TMPFS_MAGIC: u64 = 0x0102_1994; -+ const RAMFS_MAGIC: u64 = 0x8584_58f6; -+ const HUGETLBFS_MAGIC: u64 = 0x9584_58f6; -+ if matches!(filesystem_type, TMPFS_MAGIC | RAMFS_MAGIC | HUGETLBFS_MAGIC) { -+ return Err(format!( -+ "code-memory filesystem is memory-backed (type 0x{filesystem_type:x})" -+ )); -+ } -+ Ok(()) -+} -+ - /// Memory manager for executable code. - pub struct CodeMemory { -+ id: CodeMemoryId, - // frame info is placed first, to ensure it's dropped before the mmap - frame_info_registration: Option, - unwind_registry: UnwindRegistry, -- mmap: Mmap, -+ backing: CodeMemoryBacking, -+ policy: CodeMemoryPolicy, - start_of_nonexecutable_pages: usize, - } - - impl CodeMemory { - /// Create a new `CodeMemory` instance. - pub fn new() -> Self { -+ Self::with_policy(CodeMemoryPolicy::anonymous()) -+ } -+ -+ pub(crate) fn with_policy(policy: CodeMemoryPolicy) -> Self { - Self { -+ id: next_code_memory_id(), - unwind_registry: UnwindRegistry::new(), -- mmap: Mmap::new(), -+ backing: CodeMemoryBacking::empty(), -+ policy, - start_of_nonexecutable_pages: 0, - frame_info_registration: None, - } - } - -+ /// Return this process-local allocation identity. -+ pub(crate) fn id(&self) -> CodeMemoryId { -+ self.id -+ } -+ - /// Mutably get the UnwindRegistry. - pub fn unwind_registry_mut(&mut self) -> &mut UnwindRegistry { - &mut self.unwind_registry -@@ -100,13 +760,13 @@ impl CodeMemory { - - // 2. Allocate the pages. Mark them all read-write. - -- self.mmap = Mmap::with_at_least(total_len)?; -+ self.backing = CodeMemoryBacking::allocate(&self.policy, total_len)?; - - // 3. Determine where the pointers to each function, executable section - // or data section are. Copy the functions. Collect the addresses of each and return them. - - let mut bytes = 0; -- let mut buf = self.mmap.as_mut_slice(); -+ let mut buf = self.backing.as_mut_slice(); - for func in functions { - let len = round_up( - Self::function_allocation_size(*func), -@@ -158,19 +818,14 @@ impl CodeMemory { - } - - /// Apply the page permissions. -- pub fn publish(&mut self) { -- if self.mmap.is_empty() || self.start_of_nonexecutable_pages == 0 { -- return; -- } -- assert!(self.mmap.len() >= self.start_of_nonexecutable_pages); -- unsafe { -- region::protect( -- self.mmap.as_mut_ptr(), -- self.start_of_nonexecutable_pages, -- region::Protection::READ_EXECUTE, -- ) -+ pub fn publish(&mut self) -> Result<(), String> { -+ if self.backing.is_empty() { -+ return Ok(()); - } -- .expect("unable to make memory readonly and executable"); -+ if self.backing.len() < self.start_of_nonexecutable_pages { -+ return Err("executable code-memory prefix exceeds allocation".to_string()); -+ } -+ self.backing.publish(self.start_of_nonexecutable_pages) - } - - /// Calculates the allocation size of the given compiled function. -@@ -243,9 +898,261 @@ fn round_up(size: usize, multiple: usize) -> usize { - - #[cfg(test)] - mod tests { -- use super::CodeMemory; -+ use super::{CodeMemory, CodeMemoryPolicy}; -+ - fn _assert() { - fn _assert_send_sync() {} - _assert_send_sync::(); - } -+ -+ #[test] -+ fn generic_code_memory_policy_remains_anonymous() { -+ let policy = CodeMemoryPolicy::default(); -+ assert_eq!(policy.id(), "wasmer.code-memory.anonymous.v1"); -+ assert!(policy.pinned_directory().is_none()); -+ assert!(policy.pinned_directory_device().is_none()); -+ assert!(policy.pinned_directory_inode().is_none()); -+ } -+ -+ #[cfg(all(target_os = "linux", target_arch = "x86_64"))] -+ mod strict_linux_x86_64 { -+ use super::super::{ -+ CodeMemoryPolicy, CodeMemoryPolicyKind, StrictLinuxX86_64CodeMapping, -+ StrictLinuxX86_64FileBackedPolicy, -+ }; -+ use std::{ -+ ffi::CString, -+ fs, -+ os::{ -+ fd::AsRawFd, -+ unix::{ -+ ffi::OsStrExt, -+ fs::{PermissionsExt, symlink}, -+ }, -+ }, -+ path::Path, -+ }; -+ -+ fn disk_directory() -> tempfile::TempDir { -+ let directory = tempfile::Builder::new() -+ .prefix("wasmer-code-memory-test-") -+ .tempdir_in("/var/tmp") -+ .expect("create strict code-memory test directory on /var/tmp"); -+ fs::set_permissions(directory.path(), fs::Permissions::from_mode(0o700)) -+ .expect("make strict code-memory test directory private"); -+ directory -+ } -+ -+ fn strict_policy(directory: &Path) -> StrictLinuxX86_64FileBackedPolicy { -+ match CodeMemoryPolicy::strict_linux_x86_64_file_backed(directory) -+ .expect("admit strict code-memory test directory") -+ .kind -+ { -+ CodeMemoryPolicyKind::StrictLinuxX86_64FileBacked(policy) => policy, -+ CodeMemoryPolicyKind::Anonymous => panic!("strict policy became anonymous"), -+ } -+ } -+ -+ fn mapping_line(address: usize) -> String { -+ fs::read_to_string("/proc/self/maps") -+ .expect("read process mappings") -+ .lines() -+ .find(|line| { -+ let range = line.split_whitespace().next().expect("mapping range"); -+ let (start, end) = range.split_once('-').expect("mapping range separator"); -+ let start = usize::from_str_radix(start, 16).expect("mapping start"); -+ let end = usize::from_str_radix(end, 16).expect("mapping end"); -+ start <= address && address < end -+ }) -+ .expect("address is present in process mappings") -+ .to_string() -+ } -+ -+ fn smaps_kib(address: usize, field: &str) -> usize { -+ let smaps = fs::read_to_string("/proc/self/smaps").expect("read process smaps"); -+ let mut selected = false; -+ for line in smaps.lines() { -+ let range = line -+ .split_whitespace() -+ .next() -+ .and_then(|value| value.split_once('-')) -+ .and_then(|(start, end)| { -+ Some(( -+ usize::from_str_radix(start, 16).ok()?, -+ usize::from_str_radix(end, 16).ok()?, -+ )) -+ }); -+ if let Some((start, end)) = range { -+ if selected { -+ break; -+ } -+ selected = start <= address && address < end; -+ continue; -+ } -+ if selected && line.starts_with(field) { -+ return line -+ .split_whitespace() -+ .nth(1) -+ .expect("smaps value") -+ .parse() -+ .expect("numeric smaps value"); -+ } -+ } -+ panic!("smaps field {field} is present for address {address:#x}"); -+ } -+ -+ #[test] -+ fn relocated_regular_file_preserves_base_bytes_permissions_and_execution() { -+ let directory = disk_directory(); -+ let policy = strict_policy(directory.path()); -+ let page_size = region::page::size(); -+ let mut mapping = StrictLinuxX86_64CodeMapping::allocate(&policy, page_size * 2) -+ .expect("allocate strict code memory"); -+ let base = mapping.ptr; -+ let inode = mapping.inode; -+ -+ // O_EXCL has a distinct meaning with O_TMPFILE: it permanently -+ // forbids turning the anonymous inode into a named file. Prove -+ // that property before publication while the writer descriptor is -+ // still available. -+ let writable_fd = mapping -+ .writable_file -+ .as_ref() -+ .expect("strict allocation retains its writer before publication") -+ .as_raw_fd(); -+ let source = CString::new(format!("/proc/self/fd/{writable_fd}")) -+ .expect("descriptor path contains no NUL"); -+ let linked_path = directory.path().join("must-remain-unnameable"); -+ let destination = CString::new(linked_path.as_os_str().as_bytes()) -+ .expect("test destination contains no NUL"); -+ let link_result = unsafe { -+ libc::linkat( -+ libc::AT_FDCWD, -+ source.as_ptr(), -+ libc::AT_FDCWD, -+ destination.as_ptr(), -+ libc::AT_SYMLINK_FOLLOW, -+ ) -+ }; -+ assert_eq!( -+ link_result, -1, -+ "O_TMPFILE|O_EXCL code image must never become nameable" -+ ); -+ assert!(!linked_path.exists()); -+ -+ // ENDBR64; mov eax, 42; ret. ENDBR64 keeps this indirect call valid -+ // on hosts enforcing Intel CET while remaining a NOP elsewhere. -+ let function = [0xf3, 0x0f, 0x1e, 0xfa, 0xb8, 42, 0, 0, 0, 0xc3]; -+ mapping.as_mut_slice()[..function.len()].copy_from_slice(&function); -+ let data = b"relocated-code-memory-byte-stability"; -+ mapping.as_mut_slice()[page_size..page_size + data.len()].copy_from_slice(data); -+ -+ mapping -+ .publish(page_size) -+ .expect("publish strict code memory"); -+ assert_eq!(mapping.ptr, base, "fixed remap must preserve the link base"); -+ assert!(mapping.writable_file.is_none()); -+ assert!(mapping.read_only_file.is_none()); -+ -+ let executable = mapping_line(base); -+ let read_only = mapping_line(base + page_size); -+ assert_eq!( -+ executable.split_whitespace().nth(1), -+ Some("r-xp"), -+ "executable prefix must be private RX: {executable}" -+ ); -+ assert_eq!( -+ read_only.split_whitespace().nth(1), -+ Some("r--p"), -+ "data suffix must be private RO: {read_only}" -+ ); -+ assert!(executable.contains("(deleted)")); -+ assert!(read_only.contains("(deleted)")); -+ assert_eq!( -+ smaps_kib(base, "Rss:"), -+ 0, -+ "MADV_DONTNEED must remove the relocated executable PTE" -+ ); -+ assert_eq!( -+ smaps_kib(base + page_size, "Rss:"), -+ 0, -+ "MADV_DONTNEED must remove the relocated data PTE" -+ ); -+ let inode = inode.to_string(); -+ for line in fs::read_to_string("/proc/self/maps") -+ .expect("read process mappings") -+ .lines() -+ { -+ let fields = line.split_whitespace().collect::>(); -+ if fields.get(4) == Some(&inode.as_str()) { -+ assert!( -+ !fields[1].contains('w'), -+ "published inode retained a writable mapping: {line}" -+ ); -+ } -+ } -+ -+ assert_eq!( -+ unsafe { std::slice::from_raw_parts(base as *const u8, function.len()) }, -+ function -+ ); -+ assert_eq!( -+ unsafe { std::slice::from_raw_parts((base + page_size) as *const u8, data.len()) }, -+ data -+ ); -+ -+ let call: unsafe extern "C" fn() -> u32 = unsafe { std::mem::transmute(base) }; -+ assert_eq!(unsafe { call() }, 42); -+ for address in [base, base + page_size] { -+ assert_eq!(smaps_kib(address, "Anonymous:"), 0); -+ assert_eq!(smaps_kib(address, "Private_Dirty:"), 0); -+ assert_eq!(smaps_kib(address, "Shared_Dirty:"), 0); -+ } -+ } -+ -+ #[test] -+ fn strict_policy_rejects_memory_backed_symlink_and_non_private_directories() { -+ if Path::new("/dev/shm").is_dir() { -+ let memory_backed = tempfile::Builder::new() -+ .prefix("wasmer-code-memory-test-") -+ .tempdir_in("/dev/shm") -+ .expect("create tmpfs rejection fixture"); -+ let error = CodeMemoryPolicy::strict_linux_x86_64_file_backed(memory_backed.path()) -+ .expect_err("tmpfs must be rejected"); -+ assert!(error.contains("memory-backed"), "unexpected error: {error}"); -+ } -+ -+ let owner = disk_directory(); -+ let target = owner.path().join("target"); -+ fs::create_dir(&target).expect("create symlink target"); -+ fs::set_permissions(&target, fs::Permissions::from_mode(0o700)) -+ .expect("set target permissions"); -+ let link = owner.path().join("link"); -+ symlink(&target, &link).expect("create directory symlink"); -+ let error = CodeMemoryPolicy::strict_linux_x86_64_file_backed(&link) -+ .expect_err("symlink must be rejected"); -+ assert!(error.contains("non-symlink"), "unexpected error: {error}"); -+ -+ fs::set_permissions(&target, fs::Permissions::from_mode(0o750)) -+ .expect("make target non-private"); -+ let error = CodeMemoryPolicy::strict_linux_x86_64_file_backed(&target) -+ .expect_err("non-0700 directory must be rejected"); -+ assert!( -+ error.contains("exact mode 0700"), -+ "unexpected error: {error}" -+ ); -+ -+ let pinned = disk_directory(); -+ let policy = strict_policy(pinned.path()); -+ fs::set_permissions(pinned.path(), fs::Permissions::from_mode(0o750)) -+ .expect("mutate pinned directory permissions"); -+ let error = StrictLinuxX86_64CodeMapping::allocate(&policy, region::page::size()) -+ .err() -+ .expect("allocation must revalidate its pinned directory"); -+ assert!( -+ error.contains("retain exact mode 0700"), -+ "unexpected error: {error}" -+ ); -+ } -+ } - } -diff --git a/lib/compiler/src/engine/inner.rs b/lib/compiler/src/engine/inner.rs -index a1e6661..a62c16d 100644 ---- a/lib/compiler/src/engine/inner.rs -+++ b/lib/compiler/src/engine/inner.rs -@@ -25,7 +25,8 @@ use wasmer_types::{ - - #[cfg(not(target_arch = "wasm32"))] - use crate::{ -- Artifact, BaseTunables, CodeMemory, FunctionExtent, GlobalFrameInfoRegistration, Tunables, -+ Artifact, BaseTunables, CodeMemory, CodeMemoryId, CodeMemoryPolicy, FunctionExtent, -+ GlobalFrameInfoRegistration, Tunables, - types::{ - function::FunctionBodyLike, - section::{CustomSectionLike, CustomSectionProtection, SectionIndex}, -@@ -69,6 +70,8 @@ impl Engine { - #[cfg(not(target_arch = "wasm32"))] - code_memory: vec![], - #[cfg(not(target_arch = "wasm32"))] -+ code_memory_policy: CodeMemoryPolicy::anonymous(), -+ #[cfg(not(target_arch = "wasm32"))] - signatures: SignatureRegistry::new(), - })), - target: Arc::new(target), -@@ -128,6 +131,8 @@ impl Engine { - #[cfg(not(target_arch = "wasm32"))] - code_memory: vec![], - #[cfg(not(target_arch = "wasm32"))] -+ code_memory_policy: CodeMemoryPolicy::anonymous(), -+ #[cfg(not(target_arch = "wasm32"))] - signatures: SignatureRegistry::new(), - })), - target: Arc::new(target), -@@ -300,6 +305,46 @@ impl Engine { - self.tunables = Arc::new(tunables); - } - -+ /// Select executable-memory ownership before this engine allocates code. -+ /// -+ /// Generic engines retain the anonymous default. A policy change after any -+ /// artifact allocation is rejected so one engine cannot mix ownership -+ /// contracts or silently downgrade a strict product policy. -+ #[cfg(not(target_arch = "wasm32"))] -+ pub fn set_code_memory_policy(&mut self, policy: CodeMemoryPolicy) -> Result<(), String> { -+ let mut inner = self.inner_mut(); -+ if !inner.code_memory.is_empty() { -+ return Err( -+ "code-memory policy must be configured before the first artifact allocation" -+ .to_string(), -+ ); -+ } -+ inner.code_memory_policy = policy; -+ Ok(()) -+ } -+ -+ /// Return the executable-memory ownership policy selected for this engine. -+ #[cfg(not(target_arch = "wasm32"))] -+ pub fn code_memory_policy(&self) -> CodeMemoryPolicy { -+ self.inner().code_memory_policy.clone() -+ } -+ -+ /// Return the number of executable-code allocations currently owned by -+ /// this engine. -+ /// -+ /// Sealed runtimes use this read-only count to prove that an abandoned -+ /// pending activation released its published allocation. -+ #[doc(hidden)] -+ #[cfg(not(target_arch = "wasm32"))] -+ pub fn code_memory_allocation_count(&self) -> usize { -+ self.inner().code_memory.len() -+ } -+ -+ #[cfg(not(target_arch = "wasm32"))] -+ pub(crate) fn rollback_code_memory_activation(&self, id: CodeMemoryId) -> bool { -+ self.inner_mut().remove_code_memory(id) -+ } -+ - /// Get a reference to attached Tunable of this engine - #[cfg(not(target_arch = "wasm32"))] - pub fn tunables(&self) -> &dyn Tunables { -@@ -347,12 +392,49 @@ pub struct EngineInner { - /// functions to memory. - #[cfg(not(target_arch = "wasm32"))] - code_memory: Vec, -+ /// Policy copied into every code-memory allocation owned by this engine. -+ #[cfg(not(target_arch = "wasm32"))] -+ code_memory_policy: CodeMemoryPolicy, - /// The signature registry is used mainly to operate with trampolines - /// performantly. - #[cfg(not(target_arch = "wasm32"))] - signatures: SignatureRegistry, - } - -+/// Rollback owner for all code-memory allocations made by one artifact build. -+/// -+/// `Artifact::from_parts` holds the engine mutex for the complete transaction, -+/// so no unrelated allocation can be appended between this checkpoint and -+/// commit. Dropping this value after any error or unwind deregisters metadata -+/// and unmaps every uncommitted allocation through `CodeMemory`'s normal drop -+/// order. -+#[cfg(not(target_arch = "wasm32"))] -+pub(crate) struct CodeMemoryTransaction<'a> { -+ engine_inner: &'a mut EngineInner, -+ checkpoint: usize, -+ committed: bool, -+} -+ -+#[cfg(not(target_arch = "wasm32"))] -+impl CodeMemoryTransaction<'_> { -+ pub(crate) fn inner_mut(&mut self) -> &mut EngineInner { -+ self.engine_inner -+ } -+ -+ pub(crate) fn commit(mut self) { -+ self.committed = true; -+ } -+} -+ -+#[cfg(not(target_arch = "wasm32"))] -+impl Drop for CodeMemoryTransaction<'_> { -+ fn drop(&mut self) { -+ if !self.committed { -+ self.engine_inner.code_memory.truncate(self.checkpoint); -+ } -+ } -+} -+ - impl std::fmt::Debug for EngineInner { - fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - let mut formatter = f.debug_struct("EngineInner"); -@@ -372,6 +454,30 @@ impl std::fmt::Debug for EngineInner { - } - - impl EngineInner { -+ #[cfg(not(target_arch = "wasm32"))] -+ pub(crate) fn begin_code_memory_transaction(&mut self) -> CodeMemoryTransaction<'_> { -+ let checkpoint = self.code_memory.len(); -+ CodeMemoryTransaction { -+ engine_inner: self, -+ checkpoint, -+ committed: false, -+ } -+ } -+ -+ #[cfg(not(target_arch = "wasm32"))] -+ pub(crate) fn last_code_memory_id(&self) -> Option { -+ self.code_memory.last().map(CodeMemory::id) -+ } -+ -+ #[cfg(not(target_arch = "wasm32"))] -+ fn remove_code_memory(&mut self, id: CodeMemoryId) -> bool { -+ let Some(index) = self.code_memory.iter().position(|memory| memory.id() == id) else { -+ return false; -+ }; -+ self.code_memory.remove(index); -+ true -+ } -+ - /// Gets the compiler associated to this engine. - #[cfg(feature = "compiler")] - pub fn compiler(&self) -> Result<&dyn Compiler, CompileError> { -@@ -429,7 +535,8 @@ impl EngineInner { - let (executable_sections, data_sections): (Vec<_>, _) = custom_sections - .clone() - .partition(|section| section.protection() == CustomSectionProtection::ReadExecute); -- self.code_memory.push(CodeMemory::new()); -+ self.code_memory -+ .push(CodeMemory::with_policy(self.code_memory_policy.clone())); - - let (mut allocated_functions, allocated_executable_sections, allocated_data_sections) = - self.code_memory -@@ -495,8 +602,22 @@ impl EngineInner { - - #[cfg(not(target_arch = "wasm32"))] - /// Make memory containing compiled code executable. -- pub(crate) fn publish_compiled_code(&mut self) { -- self.code_memory.last_mut().unwrap().publish(); -+ pub(crate) fn publish_compiled_code(&mut self) -> Result<(), CompileError> { -+ let result = self -+ .code_memory -+ .last_mut() -+ .expect("code memory must be allocated before publication") -+ .publish(); -+ if let Err(message) = result { -+ // No Artifact escapes on a publication failure. Remove the failed -+ // allocation immediately so a later deserialization cannot retain -+ // an unpublished shared mapping or writable file descriptor. -+ self.code_memory.pop(); -+ return Err(CompileError::Resource(format!( -+ "failed to publish executable code memory: {message}" -+ ))); -+ } -+ Ok(()) - } - - #[cfg(not(target_arch = "wasm32"))] -@@ -570,16 +691,31 @@ impl EngineInner { - })?; - let mut file = std::io::BufWriter::new(file); - -+ let module_label = module_info -+ .name -+ .as_deref() -+ .map(str::to_owned) -+ .or_else(|| { -+ module_info -+ .hash() -+ .map(|hash| format!("module_{}", hash.short_hash())) -+ }) -+ .unwrap_or_else(|| "module".to_string()); -+ let module_label = module_label.replace(char::is_whitespace, "_"); -+ - for (func_index, code) in finished_functions.iter() { - let func_index = module_info.func_index(func_index); -- if let Some(func_name) = module_info.function_names.get(&func_index) { -- let sanitized_name = func_name.replace(['\n', '\r'], "_"); -- let line = format!( -- "{:p} {:x} {sanitized_name}\n", -- code.ptr.0 as *const _, code.length -- ); -- write!(file, "{line}").map_err(|e| CompileError::Codegen(e.to_string()))?; -- } -+ let func_name = module_info -+ .function_names -+ .get(&func_index) -+ .cloned() -+ .unwrap_or_else(|| format!("{module_label}::function_{}", func_index.as_u32())); -+ let sanitized_name = func_name.replace(char::is_whitespace, "_"); -+ let line = format!( -+ "{:p} {:x} {sanitized_name}\n", -+ code.ptr.0 as *const _, code.length -+ ); -+ write!(file, "{line}").map_err(|e| CompileError::Codegen(e.to_string()))?; - } - - file.flush() -@@ -637,3 +773,65 @@ impl Default for EngineId { - } - } - } -+ -+#[cfg(all(test, not(target_arch = "wasm32")))] -+mod tests { -+ use super::*; -+ -+ #[test] -+ fn code_memory_policy_is_explicitly_selectable_only_before_allocation() { -+ let mut engine = Engine::headless(); -+ engine -+ .set_code_memory_policy(CodeMemoryPolicy::anonymous()) -+ .unwrap(); -+ engine.inner_mut().code_memory.push(CodeMemory::new()); -+ -+ let error = engine -+ .set_code_memory_policy(CodeMemoryPolicy::anonymous()) -+ .unwrap_err(); -+ assert!(error.contains("before the first artifact allocation")); -+ } -+ -+ #[test] -+ fn code_memory_transaction_rolls_back_uncommitted_allocations_and_keeps_committed_ones() { -+ let engine = Engine::headless(); -+ let mut inner = engine.inner_mut(); -+ -+ { -+ let mut transaction = inner.begin_code_memory_transaction(); -+ transaction.inner_mut().code_memory.push(CodeMemory::new()); -+ } -+ assert!(inner.code_memory.is_empty()); -+ -+ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ let mut transaction = inner.begin_code_memory_transaction(); -+ transaction.inner_mut().code_memory.push(CodeMemory::new()); -+ panic!("injected artifact-build panic"); -+ })); -+ assert!(panic.is_err()); -+ assert!(inner.code_memory.is_empty()); -+ -+ { -+ let mut transaction = inner.begin_code_memory_transaction(); -+ transaction.inner_mut().code_memory.push(CodeMemory::new()); -+ transaction.commit(); -+ } -+ assert_eq!(inner.code_memory.len(), 1); -+ } -+ -+ #[test] -+ fn pending_activation_rollback_removes_the_exact_allocation_out_of_order() { -+ let engine = Engine::headless(); -+ let (first, second) = { -+ let mut inner = engine.inner_mut(); -+ inner.code_memory.push(CodeMemory::new()); -+ inner.code_memory.push(CodeMemory::new()); -+ (inner.code_memory[0].id(), inner.code_memory[1].id()) -+ }; -+ -+ assert!(engine.rollback_code_memory_activation(first)); -+ let inner = engine.inner(); -+ assert_eq!(inner.code_memory.len(), 1); -+ assert_eq!(inner.code_memory[0].id(), second); -+ } -+} -diff --git a/lib/compiler/src/engine/mod.rs b/lib/compiler/src/engine/mod.rs -index ab19cc7..511d032 100644 ---- a/lib/compiler/src/engine/mod.rs -+++ b/lib/compiler/src/engine/mod.rs -@@ -28,10 +28,14 @@ pub use self::trap::*; - pub use self::tunables::{BaseTunables, Tunables}; - - #[cfg(not(target_arch = "wasm32"))] --pub use self::artifact::Artifact; -+pub use self::artifact::{Artifact, PendingArtifact}; - pub use self::builder::EngineBuilder; - #[cfg(not(target_arch = "wasm32"))] --pub use self::code_memory::CodeMemory; -+pub(crate) use self::code_memory::CodeMemoryId; -+#[cfg(not(target_arch = "wasm32"))] -+pub use self::code_memory::{ -+ CodeMemory, CodeMemoryPolicy, STRICT_LINUX_X86_64_CODE_MEMORY_POLICY_ID, -+}; - pub use self::inner::{Engine, EngineInner}; - #[cfg(not(target_arch = "wasm32"))] - pub use self::link::link_module; -diff --git a/lib/compiler/src/engine/trap/frame_info.rs b/lib/compiler/src/engine/trap/frame_info.rs -index abee089..9d51660 100644 ---- a/lib/compiler/src/engine/trap/frame_info.rs -+++ b/lib/compiler/src/engine/trap/frame_info.rs -@@ -21,7 +21,7 @@ use crate::types::function::{ArchivedCompiledFunctionFrameInfo, CompiledFunction - use rkyv::vec::ArchivedVec; - use std::collections::BTreeMap; - use std::sync::{Arc, LazyLock, RwLock}; --use wasmer_types::lib::std::{cmp, ops::Deref}; -+use wasmer_types::lib::std::ops::Deref; - use wasmer_types::{ - FrameInfo, LocalFunctionIndex, ModuleInfo, SourceLoc, TrapInformation, - entity::{BoxedSlice, EntityRef, PrimaryMap}, -@@ -61,7 +61,13 @@ pub struct GlobalFrameInfoRegistration { - #[derive(Debug)] - struct ModuleInfoFrameInfo { - start: usize, -- functions: BTreeMap, -+ /// Function address ranges sorted by their start address. -+ /// -+ /// This is a finished immutable index: a binary search is sufficient for -+ /// trap-time lookup, and contiguous storage avoids one heap node per -+ /// compiled function. Registration validates the ranges before publishing -+ /// the module to the global registry. -+ functions: Box<[FunctionInfo]>, - module: Arc, - frame_infos: FrameInfosVariant, - } -@@ -76,18 +82,16 @@ impl ModuleInfoFrameInfo { - - /// Gets a function given a pc - fn function_info(&self, pc: usize) -> Option<&FunctionInfo> { -- let (end, func) = self.functions.range(pc..).next()?; -- if func.start <= pc && pc <= *end { -- Some(func) -- } else { -- None -- } -+ let insertion = self.functions.partition_point(|func| func.start <= pc); -+ let func = self.functions.get(insertion.checked_sub(1)?)?; -+ (pc <= func.end).then_some(func) - } - } - - #[derive(Debug)] - struct FunctionInfo { - start: usize, -+ end: usize, - local_index: LocalFunctionIndex, - } - -@@ -363,9 +367,7 @@ pub fn register( - finished_functions: &BoxedSlice, - frame_infos: FrameInfosVariant, - ) -> Option { -- let mut min = usize::MAX; -- let mut max = 0; -- let mut functions = BTreeMap::new(); -+ let mut functions = Vec::with_capacity(finished_functions.len()); - for ( - i, - FunctionExtent { -@@ -376,19 +378,37 @@ pub fn register( - { - let start = **start as usize; - // end is "last byte" of the function code -- let end = start + len - 1; -- min = cmp::min(min, start); -- max = cmp::max(max, end); -- let func = FunctionInfo { -+ let end = start -+ .checked_add( -+ len.checked_sub(1) -+ .expect("compiled function extent must not be empty"), -+ ) -+ .expect("compiled function extent address overflow"); -+ functions.push(FunctionInfo { - start, -+ end, - local_index: i, -- }; -- assert!(functions.insert(end, func).is_none()); -+ }); - } - if functions.is_empty() { - return None; - } - -+ functions.sort_unstable_by_key(|func| func.start); -+ for adjacent in functions.windows(2) { -+ assert!( -+ adjacent[0].end < adjacent[1].start, -+ "compiled function extents must be disjoint" -+ ); -+ } -+ let min = functions[0].start; -+ let max = functions -+ .iter() -+ .map(|func| func.end) -+ .max() -+ .expect("nonempty function range index"); -+ let functions = functions.into_boxed_slice(); -+ - let mut info = FRAME_INFO.write().unwrap(); - // First up assert that our chunk of jit functions doesn't collide with - // any other known chunks of jit functions... -@@ -412,3 +432,83 @@ pub fn register( - assert!(prev.is_none()); - Some(GlobalFrameInfoRegistration { key: max }) - } -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use wasmer_vm::VMFunctionBody; -+ -+ fn module_with_ranges(ranges: &[(usize, usize)]) -> ModuleInfoFrameInfo { -+ ModuleInfoFrameInfo { -+ start: ranges.first().map_or(0, |range| range.0), -+ functions: ranges -+ .iter() -+ .enumerate() -+ .map(|(index, &(start, end))| FunctionInfo { -+ start, -+ end, -+ local_index: LocalFunctionIndex::new(index), -+ }) -+ .collect::>() -+ .into_boxed_slice(), -+ module: Arc::new(ModuleInfo::default()), -+ frame_infos: FrameInfosVariant::Owned(PrimaryMap::new()), -+ } -+ } -+ -+ #[test] -+ fn packed_function_ranges_preserve_boundaries_and_gaps() { -+ let module = module_with_ranges(&[(100, 109), (200, 219)]); -+ -+ assert!(module.function_info(99).is_none()); -+ assert_eq!( -+ module.function_info(100).unwrap().local_index, -+ LocalFunctionIndex::new(0) -+ ); -+ assert_eq!( -+ module.function_info(109).unwrap().local_index, -+ LocalFunctionIndex::new(0) -+ ); -+ assert!(module.function_info(110).is_none()); -+ assert!(module.function_info(199).is_none()); -+ assert_eq!( -+ module.function_info(200).unwrap().local_index, -+ LocalFunctionIndex::new(1) -+ ); -+ assert_eq!( -+ module.function_info(219).unwrap().local_index, -+ LocalFunctionIndex::new(1) -+ ); -+ assert!(module.function_info(220).is_none()); -+ } -+ -+ fn extent(start: usize, length: usize) -> FunctionExtent { -+ FunctionExtent { -+ ptr: FunctionBodyPtr(start as *const VMFunctionBody), -+ length, -+ } -+ } -+ -+ #[test] -+ #[should_panic(expected = "compiled function extent must not be empty")] -+ fn registration_rejects_empty_function_extents() { -+ let functions = PrimaryMap::from_iter([extent(0x1000, 0)]).into_boxed_slice(); -+ let _ = register( -+ Arc::new(ModuleInfo::default()), -+ &functions, -+ FrameInfosVariant::Owned(PrimaryMap::new()), -+ ); -+ } -+ -+ #[test] -+ #[should_panic(expected = "compiled function extents must be disjoint")] -+ fn registration_rejects_overlapping_function_extents() { -+ let functions = -+ PrimaryMap::from_iter([extent(0x1000, 32), extent(0x1010, 32)]).into_boxed_slice(); -+ let _ = register( -+ Arc::new(ModuleInfo::default()), -+ &functions, -+ FrameInfosVariant::Owned(PrimaryMap::new()), -+ ); -+ } -+} -diff --git a/lib/types/src/libcalls.rs b/lib/types/src/libcalls.rs -index 7aaada0..f38de45 100644 ---- a/lib/types/src/libcalls.rs -+++ b/lib/types/src/libcalls.rs -@@ -109,6 +109,18 @@ pub enum LibCall { - /// when the `enable_probestack` setting is true. - Probestack, - -+ /// Host compiler-emitted bzero. -+ HostBzero, -+ -+ /// Host compiler-emitted memset. -+ HostMemset, -+ -+ /// Host compiler-emitted memcpy. -+ HostMemcpy, -+ -+ /// Host compiler-emitted memmove. -+ HostMemmove, -+ - /// memory.atomic.wait32 for local memories - Memory32AtomicWait32, - -@@ -188,6 +200,10 @@ impl LibCall { - Self::Probestack => "_wasmer_vm_probestack", - #[cfg(not(target_vendor = "apple"))] - Self::Probestack => "wasmer_vm_probestack", -+ Self::HostBzero => "wasmer_vm_host_bzero", -+ Self::HostMemset => "wasmer_vm_host_memset", -+ Self::HostMemcpy => "wasmer_vm_host_memcpy", -+ Self::HostMemmove => "wasmer_vm_host_memmove", - Self::Memory32AtomicWait32 => "wasmer_vm_memory32_atomic_wait32", - Self::ImportedMemory32AtomicWait32 => "wasmer_vm_imported_memory32_atomic_wait32", - Self::Memory32AtomicWait64 => "wasmer_vm_memory32_atomic_wait64", -diff --git a/lib/types/src/serialize.rs b/lib/types/src/serialize.rs -index ab6c8c9..e5124e3 100644 ---- a/lib/types/src/serialize.rs -+++ b/lib/types/src/serialize.rs -@@ -13,7 +13,7 @@ pub struct MetadataHeader { - impl MetadataHeader { - /// Current ABI version. Increment this any time breaking changes are made - /// to the format of the serialized data. -- pub const CURRENT_VERSION: u32 = 20; -+ pub const CURRENT_VERSION: u32 = 21; - - /// Magic number to identify wasmer metadata. - const MAGIC: [u8; 8] = *b"WASMER\0\0"; -diff --git a/lib/types/src/vmoffsets.rs b/lib/types/src/vmoffsets.rs -index 297118e..9c370ac 100644 ---- a/lib/types/src/vmoffsets.rs -+++ b/lib/types/src/vmoffsets.rs -@@ -544,9 +544,14 @@ impl VMOffsets { - 4 - } - -+ /// The offset of the `anyfuncs` field. -+ pub const fn vmtable_definition_anyfuncs(&self) -> u8 { -+ 2 * self.pointer_size -+ } -+ - /// Return the size of `VMTableDefinition`. - pub const fn size_of_vmtable_definition(&self) -> u8 { -- 2 * self.pointer_size -+ 3 * self.pointer_size - } - } - -@@ -814,6 +819,12 @@ impl VMOffsets { - self.vmctx_vmtable_definition(index) + u32::from(self.vmtable_definition_current_elements()) - } - -+ /// Return the offset to the `anyfuncs` field in `VMTableDefinition` index `index`. -+ /// Remember updating precompute upon changes -+ pub fn vmctx_vmtable_definition_anyfuncs(&self, index: LocalTableIndex) -> u32 { -+ self.vmctx_vmtable_definition(index) + u32::from(self.vmtable_definition_anyfuncs()) -+ } -+ - /// Return the offset to the inline `VMCallerCheckedAnyfunc` array for a local fixed - /// `funcref` table. - pub fn vmctx_fixed_funcref_table_anyfuncs(&self, index: LocalTableIndex) -> Option { -diff --git a/lib/virtual-fs/src/arc_fs.rs b/lib/virtual-fs/src/arc_fs.rs -index 507bbc1..d384d2a 100644 ---- a/lib/virtual-fs/src/arc_fs.rs -+++ b/lib/virtual-fs/src/arc_fs.rs -@@ -58,6 +58,10 @@ impl FileSystem for ArcFileSystem { - self.fs.remove_file(path) - } - -+ fn open_dir(&self, path: &Path) -> Result> { -+ self.fs.open_dir(path) -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_> { - self.fs.new_open_options() - } -diff --git a/lib/virtual-fs/src/host_fs.rs b/lib/virtual-fs/src/host_fs.rs -index 687f6ce..38062d1 100644 ---- a/lib/virtual-fs/src/host_fs.rs -+++ b/lib/virtual-fs/src/host_fs.rs -@@ -1,6 +1,6 @@ - use crate::{ -- DirEntry, FileType, FsError, Metadata, OpenOptions, OpenOptionsConfig, ReadDir, Result, -- VirtualFile, -+ DirEntry, FileAdvice, FileType, FileWritebackFlags, FsError, Metadata, OpenOptions, -+ OpenOptionsConfig, ReadDir, Result, VirtualDirectory, VirtualFile, - }; - use bytes::{Buf, Bytes}; - use futures::future::BoxFuture; -@@ -9,9 +9,13 @@ use serde::{Deserialize, Serialize, de}; - use std::convert::TryInto; - use std::fs; - use std::io::{self, Seek}; -+#[cfg(not(any(unix, windows)))] -+use std::io::{Read, Write}; - use std::path::{Component, Path, PathBuf}; - use std::pin::Pin; - use std::sync::Arc; -+#[cfg(unix)] -+use std::sync::OnceLock; - use std::task::{Context, Poll}; - use std::time::{SystemTime, UNIX_EPOCH}; - use tokio::fs as tfs; -@@ -221,6 +225,28 @@ impl crate::FileSystem for FileSystem { - fs::remove_file(path).map_err(Into::into) - } - -+ fn open_dir(&self, path: &Path) -> Result> { -+ #[cfg(unix)] -+ { -+ use std::os::unix::fs::OpenOptionsExt; -+ -+ let host_path = self.prepare_path(path)?; -+ let file = fs::OpenOptions::new() -+ .read(true) -+ // O_DIRECTORY rejects non-directories while retaining a real -+ // read descriptor that supports fsync (unlike Linux O_PATH). -+ .custom_flags(libc::O_DIRECTORY) -+ .open(&host_path)?; -+ Ok(Box::new(Directory { file })) -+ } -+ -+ #[cfg(not(unix))] -+ { -+ let _ = path; -+ Err(FsError::Unsupported) -+ } -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_> { - OpenOptions::new(self) - } -@@ -242,6 +268,27 @@ impl crate::FileSystem for FileSystem { - } - } - -+#[cfg(unix)] -+#[derive(Debug)] -+struct Directory { -+ file: fs::File, -+} -+ -+#[cfg(unix)] -+impl VirtualDirectory for Directory { -+ fn sync_all_to_disk(&self) -> BoxFuture<'_, io::Result<()>> { -+ Box::pin(async move { host_sync_all(&self.file) }) -+ } -+ -+ fn has_blocking_sync_all_to_disk(&self) -> bool { -+ true -+ } -+ -+ fn sync_all_to_disk_blocking(&self) -> io::Result<()> { -+ host_sync_all(&self.file) -+ } -+} -+ - impl TryInto for std::fs::Metadata { - type Error = io::Error; - -@@ -311,6 +358,7 @@ impl crate::FileOpener for FileSystem { - let append = if conf.truncate { false } else { conf.append() }; - - let mut oo = fs::OpenOptions::new(); -+ let (sync, data_sync) = apply_sync_open_options(&mut oo, conf.sync(), conf.data_sync()); - oo.read(conf.read()) - .write(conf.write()) - .create_new(conf.create_new()) -@@ -327,11 +375,42 @@ impl crate::FileOpener for FileSystem { - read, - write, - append, -+ sync, -+ data_sync, - )) as Box - }) - } - } - -+#[cfg(unix)] -+fn apply_sync_open_options( -+ options: &mut fs::OpenOptions, -+ sync: bool, -+ data_sync: bool, -+) -> (bool, bool) { -+ use std::os::unix::fs::OpenOptionsExt; -+ -+ let mut flags = 0; -+ if sync { -+ flags |= libc::O_SYNC; -+ } else if data_sync { -+ flags |= libc::O_DSYNC; -+ } -+ if flags != 0 { -+ options.custom_flags(flags); -+ } -+ (sync, !sync && data_sync) -+} -+ -+#[cfg(not(unix))] -+fn apply_sync_open_options( -+ _options: &mut fs::OpenOptions, -+ _sync: bool, -+ _data_sync: bool, -+) -> (bool, bool) { -+ (false, false) -+} -+ - /// A thin wrapper around `std::fs::File` - #[derive(Debug)] - #[cfg_attr(feature = "enable-serde", derive(Serialize))] -@@ -339,11 +418,12 @@ pub struct File { - #[cfg_attr(feature = "enable-serde", serde(skip, default = "default_handle"))] - handle: Handle, - #[cfg_attr(feature = "enable-serde", serde(skip))] -- inner: tfs::File, -+ inner: Option, - #[cfg_attr(feature = "enable-serde", serde(skip_serializing))] - inner_std: fs::File, - pub host_path: PathBuf, -- #[cfg(feature = "enable-serde")] -+ sync: bool, -+ data_sync: bool, - flags: u16, - } - -@@ -379,17 +459,26 @@ impl<'de> Deserialize<'de> for File { - let flags = seq - .next_element()? - .ok_or_else(|| de::Error::invalid_length(1, &self))?; -- let inner = fs::OpenOptions::new() -- .read(flags & File::READ != 0) -- .write(flags & File::WRITE != 0) -- .append(flags & File::APPEND != 0) -+ let read = flags & File::READ != 0; -+ let write = flags & File::WRITE != 0; -+ let append = flags & File::APPEND != 0; -+ let sync = flags & File::SYNC != 0; -+ let data_sync = flags & File::DATA_SYNC != 0; -+ let mut open_options = fs::OpenOptions::new(); -+ let (sync, data_sync) = apply_sync_open_options(&mut open_options, sync, data_sync); -+ let inner = open_options -+ .read(read) -+ .write(write) -+ .append(append) - .open(&host_path) - .map_err(|_| de::Error::custom("Could not open file on this system"))?; - Ok(File { - handle: Handle::current(), -- inner: tokio::fs::File::from_std(inner.try_clone().unwrap()), -+ inner: None, - inner_std: inner, - host_path, -+ sync, -+ data_sync, - flags, - }) - } -@@ -418,17 +507,26 @@ impl<'de> Deserialize<'de> for File { - } - let host_path = host_path.ok_or_else(|| de::Error::missing_field("host_path"))?; - let flags = flags.ok_or_else(|| de::Error::missing_field("flags"))?; -- let inner = fs::OpenOptions::new() -- .read(flags & File::READ != 0) -- .write(flags & File::WRITE != 0) -- .append(flags & File::APPEND != 0) -+ let read = flags & File::READ != 0; -+ let write = flags & File::WRITE != 0; -+ let append = flags & File::APPEND != 0; -+ let sync = flags & File::SYNC != 0; -+ let data_sync = flags & File::DATA_SYNC != 0; -+ let mut open_options = fs::OpenOptions::new(); -+ let (sync, data_sync) = apply_sync_open_options(&mut open_options, sync, data_sync); -+ let inner = open_options -+ .read(read) -+ .write(write) -+ .append(append) - .open(&host_path) - .map_err(|_| de::Error::custom("Could not open file on this system"))?; - Ok(File { - handle: Handle::current(), -- inner: tokio::fs::File::from_std(inner.try_clone().unwrap()), -+ inner: None, - inner_std: inner, - host_path, -+ sync, -+ data_sync, - flags, - }) - } -@@ -443,6 +541,8 @@ impl File { - const READ: u16 = 1; - const WRITE: u16 = 2; - const APPEND: u16 = 4; -+ const SYNC: u16 = 8; -+ const DATA_SYNC: u16 = 16; - - /// creates a new host file from a `std::fs::File` and a path - pub fn new( -@@ -452,9 +552,10 @@ impl File { - read: bool, - write: bool, - append: bool, -+ sync: bool, -+ data_sync: bool, - ) -> Self { - let mut _flags = 0; -- - if read { - _flags |= Self::READ; - } -@@ -467,13 +568,21 @@ impl File { - _flags |= Self::APPEND; - } - -- let async_file = tfs::File::from_std(file.try_clone().unwrap()); -+ if sync { -+ _flags |= Self::SYNC; -+ } -+ -+ if data_sync { -+ _flags |= Self::DATA_SYNC; -+ } -+ - Self { - handle, - inner_std: file, -- inner: async_file, -+ inner: None, - host_path, -- #[cfg(feature = "enable-serde")] -+ sync, -+ data_sync, - flags: _flags, - } - } -@@ -482,6 +591,370 @@ impl File { - // FIXME: no unwrap! - self.inner_std.metadata().unwrap() - } -+ -+ pub fn try_clone_std_file(&self) -> io::Result { -+ self.inner_std.try_clone() -+ } -+ -+ /// Materializes the Tokio view only for callers that actually use the -+ /// asynchronous I/O traits. Most WASIX host-file operations use the -+ /// blocking/positioned fast paths and therefore need only one host fd. -+ fn async_file(&mut self) -> io::Result<&mut tfs::File> { -+ if self.inner.is_none() { -+ self.inner = Some(tfs::File::from_std(self.inner_std.try_clone()?)); -+ } -+ Ok(self.inner.as_mut().unwrap()) -+ } -+} -+ -+#[cfg(target_os = "linux")] -+fn host_file_advise(file: &fs::File, offset: u64, len: u64, advice: FileAdvice) -> io::Result<()> { -+ use std::os::fd::AsRawFd; -+ -+ let range_error = || io::Error::from_raw_os_error(libc::EOVERFLOW); -+ let _end = offset.checked_add(len).ok_or_else(range_error)?; -+ let offset: libc::off_t = offset.try_into().map_err(|_| range_error())?; -+ let len: libc::off_t = len.try_into().map_err(|_| range_error())?; -+ let advice = match advice { -+ FileAdvice::Normal => libc::POSIX_FADV_NORMAL, -+ FileAdvice::Sequential => libc::POSIX_FADV_SEQUENTIAL, -+ FileAdvice::Random => libc::POSIX_FADV_RANDOM, -+ FileAdvice::WillNeed => libc::POSIX_FADV_WILLNEED, -+ FileAdvice::DontNeed => libc::POSIX_FADV_DONTNEED, -+ FileAdvice::NoReuse => libc::POSIX_FADV_NOREUSE, -+ }; -+ -+ // Unlike most libc calls, posix_fadvise returns the errno value directly -+ // and does not communicate failures through the thread-local errno slot. -+ let result = unsafe { libc::posix_fadvise(file.as_raw_fd(), offset, len, advice) }; -+ if result == 0 { -+ Ok(()) -+ } else { -+ Err(io::Error::from_raw_os_error(result)) -+ } -+} -+ -+#[cfg(not(target_os = "linux"))] -+fn host_file_advise( -+ _file: &fs::File, -+ _offset: u64, -+ _len: u64, -+ _advice: FileAdvice, -+) -> io::Result<()> { -+ Err(io::ErrorKind::Unsupported.into()) -+} -+ -+#[cfg(target_os = "linux")] -+fn host_file_writeback_range( -+ file: &fs::File, -+ offset: u64, -+ len: u64, -+ flags: FileWritebackFlags, -+) -> io::Result<()> { -+ use std::os::fd::AsRawFd; -+ -+ let representation_error = || io::Error::from_raw_os_error(libc::EOVERFLOW); -+ let invalid_range = || io::Error::from_raw_os_error(libc::EINVAL); -+ let offset: libc::off64_t = offset.try_into().map_err(|_| representation_error())?; -+ let len: libc::off64_t = len.try_into().map_err(|_| representation_error())?; -+ -+ // Linux requires a signed, representable exclusive end. A zero length -+ // means from offset through EOF and therefore has no finite end to check. -+ if len > 0 { -+ let end = offset.checked_add(len).ok_or_else(invalid_range)?; -+ let _end: libc::off64_t = end; -+ } -+ -+ let mut host_flags = 0; -+ if flags.contains(FileWritebackFlags::WAIT_BEFORE) { -+ host_flags |= libc::SYNC_FILE_RANGE_WAIT_BEFORE; -+ } -+ if flags.contains(FileWritebackFlags::WRITE) { -+ host_flags |= libc::SYNC_FILE_RANGE_WRITE; -+ } -+ if flags.contains(FileWritebackFlags::WAIT_AFTER) { -+ host_flags |= libc::SYNC_FILE_RANGE_WAIT_AFTER; -+ } -+ -+ let result = unsafe { libc::sync_file_range(file.as_raw_fd(), offset, len, host_flags) }; -+ if result == 0 { -+ Ok(()) -+ } else { -+ Err(io::Error::last_os_error()) -+ } -+} -+ -+#[cfg(not(target_os = "linux"))] -+fn host_file_writeback_range( -+ _file: &fs::File, -+ _offset: u64, -+ _len: u64, -+ _flags: FileWritebackFlags, -+) -> io::Result<()> { -+ Err(io::ErrorKind::Unsupported.into()) -+} -+ -+#[cfg(unix)] -+fn positioned_read(file: &fs::File, buf: &mut [u8], offset: u64) -> io::Result { -+ use std::os::unix::fs::FileExt; -+ file.read_at(buf, offset) -+} -+ -+#[cfg(unix)] -+fn positioned_write(file: &fs::File, buf: &[u8], offset: u64) -> io::Result { -+ use std::os::unix::fs::FileExt; -+ file.write_at(buf, offset) -+} -+ -+#[cfg(unix)] -+fn positioned_write_vectored( -+ file: &fs::File, -+ bufs: &[io::IoSlice<'_>], -+ offset: u64, -+) -> io::Result { -+ use std::os::unix::io::AsRawFd; -+ -+ let offset: libc::off_t = offset -+ .try_into() -+ .map_err(|_| io::Error::from(io::ErrorKind::InvalidInput))?; -+ let iovcnt: libc::c_int = bufs -+ .len() -+ .try_into() -+ .map_err(|_| io::Error::from(io::ErrorKind::InvalidInput))?; -+ let raw_iov = bufs -+ .iter() -+ .map(|buf| { -+ let bytes: &[u8] = buf; -+ libc::iovec { -+ iov_base: bytes.as_ptr() as *mut libc::c_void, -+ iov_len: bytes.len(), -+ } -+ }) -+ .collect::>(); -+ -+ loop { -+ let rc = unsafe { libc::pwritev(file.as_raw_fd(), raw_iov.as_ptr(), iovcnt, offset) }; -+ if rc >= 0 { -+ return Ok(rc as usize); -+ } -+ let err = io::Error::last_os_error(); -+ if err.raw_os_error() == Some(libc::EINTR) { -+ continue; -+ } -+ return Err(err); -+ } -+} -+ -+const ZERO_WRITE_CHUNK_LEN: usize = 8192; -+static ZERO_WRITE_CHUNK: [u8; ZERO_WRITE_CHUNK_LEN] = [0; ZERO_WRITE_CHUNK_LEN]; -+ -+fn try_extend_with_zeroes(file: &fs::File, len: u64, offset: u64) -> io::Result> { -+ let written = usize::try_from(len).map_err(|_| io::Error::from(io::ErrorKind::InvalidInput))?; -+ let end = offset -+ .checked_add(len) -+ .ok_or_else(|| io::Error::from(io::ErrorKind::InvalidInput))?; -+ -+ if len == 0 { -+ return Ok(Some(0)); -+ } -+ -+ if offset >= file.metadata()?.len() { -+ file.set_len(end)?; -+ return Ok(Some(written)); -+ } -+ -+ Ok(None) -+} -+ -+#[cfg(unix)] -+fn host_iov_max() -> usize { -+ static IOV_MAX: OnceLock = OnceLock::new(); -+ *IOV_MAX.get_or_init(|| { -+ let max = unsafe { libc::sysconf(libc::_SC_IOV_MAX) }; -+ if max > 0 { max as usize } else { 1024 } -+ }) -+} -+ -+#[cfg(unix)] -+fn positioned_write_zeroes(file: &fs::File, len: u64, offset: u64) -> io::Result { -+ if let Some(written) = try_extend_with_zeroes(file, len, offset)? { -+ return Ok(written); -+ } -+ -+ let mut written = 0usize; -+ let mut remaining = len; -+ let max_batch = host_iov_max().max(1) * ZERO_WRITE_CHUNK_LEN; -+ -+ while remaining > 0 { -+ let batch_len = (remaining.min(max_batch as u64)) as usize; -+ let full_chunks = batch_len / ZERO_WRITE_CHUNK_LEN; -+ let tail = batch_len % ZERO_WRITE_CHUNK_LEN; -+ let mut bufs = Vec::with_capacity(full_chunks + usize::from(tail > 0)); -+ for _ in 0..full_chunks { -+ bufs.push(io::IoSlice::new(&ZERO_WRITE_CHUNK)); -+ } -+ if tail > 0 { -+ bufs.push(io::IoSlice::new(&ZERO_WRITE_CHUNK[..tail])); -+ } -+ -+ let local = positioned_write_vectored(file, &bufs, offset + written as u64)?; -+ if local == 0 { -+ break; -+ } -+ written = written -+ .checked_add(local) -+ .ok_or_else(|| io::Error::from(io::ErrorKind::InvalidData))?; -+ if local < batch_len { -+ break; -+ } -+ remaining -= local as u64; -+ } -+ -+ Ok(written) -+} -+ -+#[cfg(windows)] -+fn positioned_read(file: &fs::File, buf: &mut [u8], offset: u64) -> io::Result { -+ use std::os::windows::fs::FileExt; -+ file.seek_read(buf, offset) -+} -+ -+#[cfg(windows)] -+fn positioned_write(file: &fs::File, buf: &[u8], offset: u64) -> io::Result { -+ use std::os::windows::fs::FileExt; -+ file.seek_write(buf, offset) -+} -+ -+#[cfg(not(any(unix, windows)))] -+fn positioned_read(file: &mut fs::File, buf: &mut [u8], offset: u64) -> io::Result { -+ let cursor = file.stream_position()?; -+ file.seek(io::SeekFrom::Start(offset))?; -+ let read_result = file.read(buf); -+ let restore_result = file.seek(io::SeekFrom::Start(cursor)); -+ -+ match (read_result, restore_result) { -+ (Ok(read), Ok(_)) => Ok(read), -+ (Err(err), _) => Err(err), -+ (Ok(_), Err(err)) => Err(err), -+ } -+} -+ -+#[cfg(not(any(unix, windows)))] -+fn positioned_write(file: &mut fs::File, buf: &[u8], offset: u64) -> io::Result { -+ let cursor = file.stream_position()?; -+ file.seek(io::SeekFrom::Start(offset))?; -+ let write_result = file.write(buf); -+ let restore_result = file.seek(io::SeekFrom::Start(cursor)); -+ -+ match (write_result, restore_result) { -+ (Ok(written), Ok(_)) => Ok(written), -+ (Err(err), _) => Err(err), -+ (Ok(_), Err(err)) => Err(err), -+ } -+} -+ -+#[cfg(not(unix))] -+fn positioned_write_zeroes(file: &mut fs::File, len: u64, offset: u64) -> io::Result { -+ if let Some(written) = try_extend_with_zeroes(file, len, offset)? { -+ return Ok(written); -+ } -+ -+ let mut written = 0usize; -+ while (written as u64) < len { -+ let remaining = (len - written as u64) as usize; -+ let local_len = remaining.min(ZERO_WRITE_CHUNK_LEN); -+ let local = positioned_write( -+ file, -+ &ZERO_WRITE_CHUNK[..local_len], -+ offset + written as u64, -+ )?; -+ if local == 0 { -+ break; -+ } -+ written = written -+ .checked_add(local) -+ .ok_or_else(|| io::Error::from(io::ErrorKind::InvalidData))?; -+ if local < local_len { -+ break; -+ } -+ } -+ Ok(written) -+} -+ -+#[cfg(unix)] -+fn cvt_sync_result(mut sync: impl FnMut() -> libc::c_int) -> io::Result<()> { -+ loop { -+ if sync() == 0 { -+ return Ok(()); -+ } -+ let err = io::Error::last_os_error(); -+ if err.raw_os_error() == Some(libc::EINTR) { -+ continue; -+ } -+ return Err(err); -+ } -+} -+ -+#[cfg(unix)] -+fn host_sync_all(file: &fs::File) -> io::Result<()> { -+ use std::os::unix::io::AsRawFd; -+ -+ cvt_sync_result(|| unsafe { libc::fsync(file.as_raw_fd()) }) -+} -+ -+#[cfg(any( -+ target_os = "android", -+ target_os = "freebsd", -+ target_os = "fuchsia", -+ target_os = "hurd", -+ target_os = "linux", -+ target_os = "netbsd", -+ target_os = "nto", -+ target_os = "openbsd", -+))] -+fn host_sync_data(file: &fs::File) -> io::Result<()> { -+ use std::os::unix::io::AsRawFd; -+ -+ cvt_sync_result(|| unsafe { libc::fdatasync(file.as_raw_fd()) }) -+} -+ -+#[cfg(target_vendor = "apple")] -+fn host_sync_data(file: &fs::File) -> io::Result<()> { -+ use std::os::unix::io::AsRawFd; -+ -+ unsafe extern "C" { -+ fn fdatasync(fd: libc::c_int) -> libc::c_int; -+ } -+ -+ cvt_sync_result(|| unsafe { fdatasync(file.as_raw_fd()) }) -+} -+ -+#[cfg(all( -+ unix, -+ not(any( -+ target_os = "android", -+ target_os = "freebsd", -+ target_os = "fuchsia", -+ target_os = "hurd", -+ target_os = "linux", -+ target_os = "netbsd", -+ target_os = "nto", -+ target_os = "openbsd", -+ target_vendor = "apple", -+ )) -+))] -+fn host_sync_data(file: &fs::File) -> io::Result<()> { -+ host_sync_all(file) -+} -+ -+#[cfg(not(unix))] -+fn host_sync_all(file: &fs::File) -> io::Result<()> { -+ file.sync_all() -+} -+ -+#[cfg(not(unix))] -+fn host_sync_data(file: &fs::File) -> io::Result<()> { -+ file.sync_data() - } - - //#[cfg_attr(feature = "enable-serde", typetag::serde)] -@@ -538,6 +1011,14 @@ impl VirtualFile for File { - None - } - -+ fn open_read(&self) -> Option { -+ Some(self.flags & Self::READ != 0) -+ } -+ -+ fn open_write(&self) -> Option { -+ Some(self.flags & (Self::WRITE | Self::APPEND) != 0) -+ } -+ - fn poll_read_ready(mut self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll> { - let cursor = match self.inner_std.stream_position() { - Ok(a) => a, -@@ -556,6 +1037,221 @@ impl VirtualFile for File { - fn poll_write_ready(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll> { - Poll::Ready(Ok(8192)) - } -+ -+ fn has_blocking_seek(&self) -> bool { -+ true -+ } -+ -+ fn seek_blocking(&mut self, pos: io::SeekFrom) -> io::Result { -+ std::io::Seek::seek(&mut self.inner_std, pos) -+ } -+ -+ fn has_blocking_read(&self) -> bool { -+ true -+ } -+ -+ fn read_blocking(&mut self, buf: &mut [u8]) -> io::Result { -+ std::io::Read::read(&mut self.inner_std, buf) -+ } -+ -+ fn has_blocking_read_at(&self) -> bool { -+ true -+ } -+ -+ fn read_at_blocking(&mut self, buf: &mut [u8], offset: u64) -> io::Result { -+ #[cfg(any(unix, windows))] -+ { -+ positioned_read(&self.inner_std, buf, offset) -+ } -+ #[cfg(not(any(unix, windows)))] -+ { -+ positioned_read(&mut self.inner_std, buf, offset) -+ } -+ } -+ -+ fn has_blocking_read_at_shared(&self) -> bool { -+ cfg!(any(unix, windows)) -+ } -+ -+ fn read_at_blocking_shared(&self, buf: &mut [u8], offset: u64) -> io::Result { -+ #[cfg(any(unix, windows))] -+ { -+ positioned_read(&self.inner_std, buf, offset) -+ } -+ #[cfg(not(any(unix, windows)))] -+ { -+ let _ = buf; -+ let _ = offset; -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ } -+ -+ fn has_blocking_write(&self) -> bool { -+ true -+ } -+ -+ fn write_blocking(&mut self, buf: &[u8]) -> io::Result { -+ std::io::Write::write(&mut self.inner_std, buf) -+ } -+ -+ fn write_at<'a>(&'a mut self, buf: &'a [u8], offset: u64) -> BoxFuture<'a, io::Result> { -+ Box::pin(async move { -+ #[cfg(any(unix, windows))] -+ { -+ positioned_write(&self.inner_std, buf, offset) -+ } -+ #[cfg(not(any(unix, windows)))] -+ { -+ positioned_write(&mut self.inner_std, buf, offset) -+ } -+ }) -+ } -+ -+ fn has_blocking_write_at(&self) -> bool { -+ true -+ } -+ -+ fn write_at_blocking(&mut self, buf: &[u8], offset: u64) -> io::Result { -+ #[cfg(any(unix, windows))] -+ { -+ positioned_write(&self.inner_std, buf, offset) -+ } -+ #[cfg(not(any(unix, windows)))] -+ { -+ positioned_write(&mut self.inner_std, buf, offset) -+ } -+ } -+ -+ fn has_blocking_write_at_shared(&self) -> bool { -+ cfg!(any(unix, windows)) -+ } -+ -+ fn write_at_blocking_shared(&self, buf: &[u8], offset: u64) -> io::Result { -+ #[cfg(any(unix, windows))] -+ { -+ positioned_write(&self.inner_std, buf, offset) -+ } -+ #[cfg(not(any(unix, windows)))] -+ { -+ let _ = buf; -+ let _ = offset; -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ } -+ -+ fn has_blocking_write_vectored_at(&self) -> bool { -+ cfg!(unix) -+ } -+ -+ fn write_vectored_at_blocking( -+ &mut self, -+ bufs: &[io::IoSlice<'_>], -+ offset: u64, -+ ) -> io::Result { -+ #[cfg(unix)] -+ { -+ positioned_write_vectored(&self.inner_std, bufs, offset) -+ } -+ #[cfg(not(unix))] -+ { -+ let _ = bufs; -+ let _ = offset; -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ } -+ -+ fn has_blocking_write_vectored_at_shared(&self) -> bool { -+ cfg!(unix) -+ } -+ -+ fn write_vectored_at_blocking_shared( -+ &self, -+ bufs: &[io::IoSlice<'_>], -+ offset: u64, -+ ) -> io::Result { -+ #[cfg(unix)] -+ { -+ positioned_write_vectored(&self.inner_std, bufs, offset) -+ } -+ #[cfg(not(unix))] -+ { -+ let _ = bufs; -+ let _ = offset; -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ } -+ -+ fn has_blocking_write_zeroes_at(&self) -> bool { -+ true -+ } -+ -+ fn write_zeroes_at_blocking(&mut self, len: u64, offset: u64) -> io::Result { -+ #[cfg(unix)] -+ { -+ positioned_write_zeroes(&self.inner_std, len, offset) -+ } -+ #[cfg(not(unix))] -+ { -+ positioned_write_zeroes(&mut self.inner_std, len, offset) -+ } -+ } -+ -+ fn has_blocking_write_zeroes_at_shared(&self) -> bool { -+ cfg!(unix) -+ } -+ -+ fn write_zeroes_at_blocking_shared(&self, len: u64, offset: u64) -> io::Result { -+ #[cfg(unix)] -+ { -+ positioned_write_zeroes(&self.inner_std, len, offset) -+ } -+ #[cfg(not(unix))] -+ { -+ let _ = len; -+ let _ = offset; -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ } -+ -+ fn sync_data_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { -+ Box::pin(async move { host_sync_data(&self.inner_std) }) -+ } -+ -+ fn sync_all_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { -+ Box::pin(async move { host_sync_all(&self.inner_std) }) -+ } -+ -+ fn has_blocking_sync_data_to_disk(&self) -> bool { -+ true -+ } -+ -+ fn sync_data_to_disk_blocking(&mut self) -> io::Result<()> { -+ host_sync_data(&self.inner_std) -+ } -+ -+ fn has_blocking_sync_all_to_disk(&self) -> bool { -+ true -+ } -+ -+ fn sync_all_to_disk_blocking(&mut self) -> io::Result<()> { -+ host_sync_all(&self.inner_std) -+ } -+ -+ fn is_sync_on_write(&self, sync_metadata: bool) -> bool { -+ if sync_metadata { -+ self.sync -+ } else { -+ self.sync || self.data_sync -+ } -+ } -+ -+ fn advise(&self, offset: u64, len: u64, advice: FileAdvice) -> io::Result<()> { -+ host_file_advise(&self.inner_std, offset, len, advice) -+ } -+ -+ fn writeback_range(&self, offset: u64, len: u64, flags: FileWritebackFlags) -> io::Result<()> { -+ host_file_writeback_range(&self.inner_std, offset, len, flags) -+ } - } - - impl AsyncRead for File { -@@ -564,8 +1260,13 @@ impl AsyncRead for File { - cx: &mut Context<'_>, - buf: &mut tokio::io::ReadBuf<'_>, - ) -> Poll> { -- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); -- let inner = Pin::new(&mut self.inner); -+ let this = self.as_mut().get_mut(); -+ let handle = this.handle.clone(); -+ let _guard = Handle::try_current().map_err(|_| handle.enter()); -+ let inner = match this.async_file() { -+ Ok(inner) => Pin::new(inner), -+ Err(err) => return Poll::Ready(Err(err)), -+ }; - inner.poll_read(cx, buf) - } - } -@@ -576,20 +1277,35 @@ impl AsyncWrite for File { - cx: &mut Context<'_>, - buf: &[u8], - ) -> Poll> { -- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); -- let inner = Pin::new(&mut self.inner); -+ let this = self.as_mut().get_mut(); -+ let handle = this.handle.clone(); -+ let _guard = Handle::try_current().map_err(|_| handle.enter()); -+ let inner = match this.async_file() { -+ Ok(inner) => Pin::new(inner), -+ Err(err) => return Poll::Ready(Err(err)), -+ }; - inner.poll_write(cx, buf) - } - - fn poll_flush(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { -- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); -- let inner = Pin::new(&mut self.inner); -+ let this = self.as_mut().get_mut(); -+ let handle = this.handle.clone(); -+ let _guard = Handle::try_current().map_err(|_| handle.enter()); -+ let inner = match this.async_file() { -+ Ok(inner) => Pin::new(inner), -+ Err(err) => return Poll::Ready(Err(err)), -+ }; - inner.poll_flush(cx) - } - - fn poll_shutdown(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { -- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); -- let inner = Pin::new(&mut self.inner); -+ let this = self.as_mut().get_mut(); -+ let handle = this.handle.clone(); -+ let _guard = Handle::try_current().map_err(|_| handle.enter()); -+ let inner = match this.async_file() { -+ Ok(inner) => Pin::new(inner), -+ Err(err) => return Poll::Ready(Err(err)), -+ }; - inner.poll_shutdown(cx) - } - -@@ -598,26 +1314,41 @@ impl AsyncWrite for File { - cx: &mut Context<'_>, - bufs: &[io::IoSlice<'_>], - ) -> Poll> { -- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); -- let inner = Pin::new(&mut self.inner); -+ let this = self.as_mut().get_mut(); -+ let handle = this.handle.clone(); -+ let _guard = Handle::try_current().map_err(|_| handle.enter()); -+ let inner = match this.async_file() { -+ Ok(inner) => Pin::new(inner), -+ Err(err) => return Poll::Ready(Err(err)), -+ }; - inner.poll_write_vectored(cx, bufs) - } - - fn is_write_vectored(&self) -> bool { -- self.inner.is_write_vectored() -+ self.inner -+ .as_ref() -+ .map(AsyncWrite::is_write_vectored) -+ .unwrap_or(true) - } - } - - impl AsyncSeek for File { - fn start_seek(mut self: Pin<&mut Self>, position: io::SeekFrom) -> io::Result<()> { -- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); -- let inner = Pin::new(&mut self.inner); -+ let this = self.as_mut().get_mut(); -+ let handle = this.handle.clone(); -+ let _guard = Handle::try_current().map_err(|_| handle.enter()); -+ let inner = Pin::new(this.async_file()?); - inner.start_seek(position) - } - - fn poll_complete(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { -- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); -- let inner = Pin::new(&mut self.inner); -+ let this = self.as_mut().get_mut(); -+ let handle = this.handle.clone(); -+ let _guard = Handle::try_current().map_err(|_| handle.enter()); -+ let inner = match this.async_file() { -+ Ok(inner) => Pin::new(inner), -+ Err(err) => return Poll::Ready(Err(err)), -+ }; - inner.poll_complete(cx) - } - } -@@ -1026,10 +1757,220 @@ mod tests { - use tempfile::TempDir; - use tokio::runtime::Handle; - -- use super::FileSystem; -+ use super::{File, FileSystem}; -+ use crate::AsyncSeekExt; - use crate::FileSystem as FileSystemTrait; - use crate::FsError; -+ use crate::VirtualFile; -+ use crate::{FileAdvice, FileWritebackFlags}; -+ use std::io::{IoSlice, SeekFrom}; - use std::path::Path; -+ use std::sync::{Arc, Barrier}; -+ -+ #[tokio::test] -+ async fn host_file_defers_async_descriptor_until_async_io_is_requested() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("lazy-async-file.txt"); -+ std::fs::write(&path, b"contents").unwrap(); -+ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); -+ -+ let mut file = File::new( -+ Handle::current(), -+ std_file, -+ path, -+ true, -+ false, -+ false, -+ false, -+ false, -+ ); -+ -+ assert!(file.inner.is_none()); -+ file.async_file().unwrap(); -+ assert!(file.inner.is_some()); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[tokio::test] -+ async fn host_file_advice_uses_existing_descriptor_without_async_clone() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("advice.txt"); -+ std::fs::write(&path, vec![0u8; 8192]).unwrap(); -+ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); -+ -+ let file = File::new( -+ Handle::current(), -+ std_file, -+ path, -+ true, -+ false, -+ false, -+ false, -+ false, -+ ); -+ -+ assert!(file.inner.is_none()); -+ file.advise(0, 4096, FileAdvice::WillNeed).unwrap(); -+ file.advise(0, 4096, FileAdvice::DontNeed).unwrap(); -+ assert!(file.inner.is_none()); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[tokio::test] -+ async fn host_file_advice_rejects_unrepresentable_range_without_async_clone() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("advice-overflow.txt"); -+ std::fs::write(&path, b"contents").unwrap(); -+ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); -+ -+ let file = File::new( -+ Handle::current(), -+ std_file, -+ path, -+ true, -+ false, -+ false, -+ false, -+ false, -+ ); -+ -+ let beyond_off_t = (libc::off_t::MAX as u64) + 1; -+ for (offset, len) in [(u64::MAX, 1), (beyond_off_t, 0), (0, beyond_off_t)] { -+ let error = file.advise(offset, len, FileAdvice::WillNeed).unwrap_err(); -+ assert_eq!(error.raw_os_error(), Some(libc::EOVERFLOW)); -+ } -+ assert!(file.inner.is_none()); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[tokio::test] -+ async fn host_file_advice_accepts_individually_representable_boundary_ranges() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("advice-boundary.txt"); -+ std::fs::write(&path, b"contents").unwrap(); -+ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); -+ -+ let file = File::new( -+ Handle::current(), -+ std_file, -+ path, -+ true, -+ false, -+ false, -+ false, -+ false, -+ ); -+ -+ let off_max = libc::off_t::MAX as u64; -+ file.advise(off_max, 1, FileAdvice::WillNeed).unwrap(); -+ file.advise(off_max - 1, 3, FileAdvice::WillNeed).unwrap(); -+ assert!(file.inner.is_none()); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[tokio::test] -+ async fn host_file_range_writeback_uses_existing_descriptor_without_async_clone() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("writeback.txt"); -+ std::fs::write(&path, vec![0u8; 8192]).unwrap(); -+ let std_file = std::fs::OpenOptions::new() -+ .read(true) -+ .write(true) -+ .open(&path) -+ .unwrap(); -+ -+ let file = File::new( -+ Handle::current(), -+ std_file, -+ path, -+ true, -+ true, -+ false, -+ false, -+ false, -+ ); -+ -+ assert!(file.inner.is_none()); -+ file.writeback_range(0, 4096, FileWritebackFlags::WRITE) -+ .unwrap(); -+ file.writeback_range(0, 0, FileWritebackFlags::empty()) -+ .unwrap(); -+ assert!(file.inner.is_none()); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[tokio::test] -+ async fn host_file_range_writeback_rejects_unrepresentable_ranges() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("writeback-overflow.txt"); -+ std::fs::write(&path, b"contents").unwrap(); -+ let std_file = std::fs::OpenOptions::new() -+ .read(true) -+ .write(true) -+ .open(&path) -+ .unwrap(); -+ -+ let file = File::new( -+ Handle::current(), -+ std_file, -+ path, -+ true, -+ true, -+ false, -+ false, -+ false, -+ ); -+ -+ let beyond_off_t = (libc::off64_t::MAX as u64) + 1; -+ for (offset, len) in [(beyond_off_t, 0), (0, beyond_off_t)] { -+ let error = file -+ .writeback_range(offset, len, FileWritebackFlags::WRITE) -+ .unwrap_err(); -+ assert_eq!(error.raw_os_error(), Some(libc::EOVERFLOW)); -+ } -+ for (offset, len) in [ -+ (libc::off64_t::MAX as u64, 1), -+ (libc::off64_t::MAX as u64 - 1, 2), -+ ] { -+ let error = file -+ .writeback_range(offset, len, FileWritebackFlags::WRITE) -+ .unwrap_err(); -+ assert_eq!(error.raw_os_error(), Some(libc::EINVAL)); -+ } -+ assert!(file.inner.is_none()); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[tokio::test] -+ async fn host_file_range_writeback_accepts_maximum_finite_boundary() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("writeback-boundary.txt"); -+ std::fs::write(&path, b"contents").unwrap(); -+ let std_file = std::fs::OpenOptions::new() -+ .read(true) -+ .write(true) -+ .open(&path) -+ .unwrap(); -+ -+ let file = File::new( -+ Handle::current(), -+ std_file, -+ path, -+ true, -+ true, -+ false, -+ false, -+ false, -+ ); -+ -+ file.writeback_range( -+ libc::off64_t::MAX as u64 - 1, -+ 1, -+ FileWritebackFlags::empty(), -+ ) -+ .unwrap(); -+ assert!(file.inner.is_none()); -+ } - - #[tokio::test] - async fn test_new_filesystem() { -@@ -1050,6 +1991,172 @@ mod tests { - ); - } - -+ #[cfg(target_os = "linux")] -+ #[tokio::test] -+ async fn host_directory_sync_retains_identity_after_rename() { -+ let temp = TempDir::new().unwrap(); -+ let original = temp.path().join("original"); -+ let renamed = temp.path().join("renamed"); -+ std::fs::create_dir(&original).unwrap(); -+ -+ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); -+ let directory = fs.open_dir(Path::new("/original")).unwrap(); -+ assert!(directory.has_blocking_sync_all_to_disk()); -+ -+ std::fs::rename(&original, &renamed).unwrap(); -+ assert!(!original.exists()); -+ -+ // A path-reopen implementation would fail here. The retained host -+ // descriptor continues to refer to the directory opened above. -+ directory.sync_all_to_disk_blocking().unwrap(); -+ directory.sync_all_to_disk().await.unwrap(); -+ } -+ -+ #[tokio::test] -+ async fn host_file_write_at_preserves_cursor_and_syncs() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("foo.txt"); -+ std::fs::write(&path, b"abcdef").unwrap(); -+ -+ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); -+ let mut file = fs -+ .new_open_options() -+ .read(true) -+ .write(true) -+ .open(Path::new("/foo.txt")) -+ .expect("open file"); -+ -+ file.seek(SeekFrom::Start(3)).await.unwrap(); -+ assert_eq!(file.write_at(b"XY", 1).await.unwrap(), 2); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_read_at()); -+ let mut read_buf = [0; 2]; -+ assert_eq!(file.read_at_blocking(&mut read_buf, 1).unwrap(), 2); -+ assert_eq!(&read_buf, b"XY"); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_read_at_shared()); -+ let mut shared_read_buf = [0; 2]; -+ assert_eq!( -+ file.read_at_blocking_shared(&mut shared_read_buf, 1) -+ .unwrap(), -+ 2 -+ ); -+ assert_eq!(&shared_read_buf, b"XY"); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_read()); -+ file.seek_blocking(SeekFrom::Start(0)).unwrap(); -+ let mut read_buf = [0; 1]; -+ assert_eq!(file.read_blocking(&mut read_buf).unwrap(), 1); -+ assert_eq!(&read_buf, b"a"); -+ file.seek_blocking(SeekFrom::Start(3)).unwrap(); -+ assert!(file.has_blocking_write_at()); -+ assert_eq!(file.write_at_blocking(b"ZZ", 4).unwrap(), 2); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_write_at_shared()); -+ assert_eq!(file.write_at_blocking_shared(b"QQ", 4).unwrap(), 2); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_write_vectored_at()); -+ let bufs = [IoSlice::new(b"12"), IoSlice::new(b"34")]; -+ assert_eq!(file.write_vectored_at_blocking(&bufs, 1).unwrap(), 4); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_write_vectored_at_shared()); -+ let bufs = [IoSlice::new(b"56"), IoSlice::new(b"78")]; -+ assert_eq!(file.write_vectored_at_blocking_shared(&bufs, 1).unwrap(), 4); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_write_zeroes_at()); -+ assert_eq!(file.write_zeroes_at_blocking(2, 2).unwrap(), 2); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_write_zeroes_at_shared()); -+ assert_eq!(file.write_zeroes_at_blocking_shared(2, 2).unwrap(), 2); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); -+ assert!(file.has_blocking_seek()); -+ assert!(file.has_blocking_write()); -+ assert_eq!(file.seek_blocking(SeekFrom::Start(3)).unwrap(), 3); -+ assert_eq!(file.write_blocking(b"Y").unwrap(), 1); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 4); -+ assert!(file.has_blocking_sync_data_to_disk()); -+ assert!(file.has_blocking_sync_all_to_disk()); -+ file.sync_data_to_disk_blocking().unwrap(); -+ file.sync_all_to_disk_blocking().unwrap(); -+ file.sync_data_to_disk().await.unwrap(); -+ file.sync_all_to_disk().await.unwrap(); -+ assert_eq!(file.write_zeroes_at_blocking(3, 6).unwrap(), 3); -+ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 4); -+ -+ drop(file); -+ assert_eq!(std::fs::read(path).unwrap(), b"a5\0Y8Q\0\0\0"); -+ } -+ -+ #[cfg(any(unix, windows))] -+ #[tokio::test] -+ async fn host_file_shared_positioned_reads_are_concurrent_and_cursor_invariant() { -+ let temp = TempDir::new().unwrap(); -+ let path = temp.path().join("shared-positioned-read.txt"); -+ let contents = b"0123456789abcdefghijklmnopqrstuv"; -+ std::fs::write(&path, contents).unwrap(); -+ -+ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); -+ let mut file = fs -+ .new_open_options() -+ .read(true) -+ .open(Path::new("/shared-positioned-read.txt")) -+ .expect("open file"); -+ file.seek_blocking(SeekFrom::Start(7)).unwrap(); -+ assert!(file.has_blocking_read_at_shared()); -+ -+ let file = Arc::new(file); -+ let barrier = Arc::new(Barrier::new(4)); -+ let readers = [(0_u64, b"0123"), (8, b"89ab"), (16, b"ghij"), (24, b"opqr")] -+ .into_iter() -+ .map(|(offset, expected)| { -+ let file = file.clone(); -+ let barrier = barrier.clone(); -+ std::thread::spawn(move || { -+ barrier.wait(); -+ for _ in 0..256 { -+ let mut buf = [0_u8; 4]; -+ assert_eq!( -+ file.read_at_blocking_shared(&mut buf, offset).unwrap(), -+ buf.len() -+ ); -+ assert_eq!(&buf, expected); -+ } -+ }) -+ }) -+ .collect::>(); -+ -+ for reader in readers { -+ reader.join().unwrap(); -+ } -+ let mut file = Arc::try_unwrap(file).expect("reader retained the host file"); -+ assert_eq!(file.seek_blocking(SeekFrom::Current(0)).unwrap(), 7); -+ } -+ -+ #[tokio::test] -+ async fn host_file_reports_native_sync_open_modes() { -+ let temp = TempDir::new().unwrap(); -+ std::fs::write(temp.path().join("foo.txt"), b"").unwrap(); -+ -+ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); -+ let data_sync_file = fs -+ .new_open_options() -+ .write(true) -+ .data_sync(true) -+ .open(Path::new("/foo.txt")) -+ .expect("open data-sync file"); -+ assert!(data_sync_file.is_sync_on_write(false)); -+ assert!(!data_sync_file.is_sync_on_write(true)); -+ -+ let sync_file = fs -+ .new_open_options() -+ .write(true) -+ .sync(true) -+ .open(Path::new("/foo.txt")) -+ .expect("open sync file"); -+ assert!(sync_file.is_sync_on_write(false)); -+ assert!(sync_file.is_sync_on_write(true)); -+ } -+ - #[tokio::test] - async fn test_create_dir() { - let temp: TempDir = TempDir::new().unwrap(); -diff --git a/lib/virtual-fs/src/lib.rs b/lib/virtual-fs/src/lib.rs -index 4fc7c5b..4fc28c2 100644 ---- a/lib/virtual-fs/src/lib.rs -+++ b/lib/virtual-fs/src/lib.rs -@@ -9,7 +9,7 @@ use shared_buffer::OwnedBuffer; - use std::any::Any; - use std::ffi::OsString; - use std::fmt; --use std::io; -+use std::io::{self, SeekFrom}; - use std::ops::Deref; - use std::path::{Path, PathBuf}; - use std::pin::Pin; -@@ -85,6 +85,68 @@ pub trait ClonableVirtualFile: VirtualFile + Clone {} - - pub use ops::{copy_reference, copy_reference_ext, create_dir_all, walk}; - -+/// Backend-neutral file access advice. -+/// -+/// These variants correspond one-for-one with the six defined WASI -+/// `advice` values. Backends must treat advice as a range-scoped hint; a -+/// backend that cannot preserve those semantics should return -+/// [`io::ErrorKind::Unsupported`] instead of approximating them with a -+/// process-wide or persistent cache policy. -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub enum FileAdvice { -+ Normal, -+ Sequential, -+ Random, -+ WillNeed, -+ DontNeed, -+ NoReuse, -+} -+ -+/// Backend-neutral controls for range-scoped file writeback. -+/// -+/// These bits intentionally mirror Linux `sync_file_range(2)`, but this type -+/// does not imply durability: writeback does not flush file metadata or a -+/// device's volatile write cache. Backends that cannot preserve these exact -+/// advisory semantics must report [`io::ErrorKind::Unsupported`]. -+#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] -+pub struct FileWritebackFlags(u8); -+ -+impl FileWritebackFlags { -+ pub const WAIT_BEFORE: Self = Self(1); -+ pub const WRITE: Self = Self(2); -+ pub const WAIT_AFTER: Self = Self(4); -+ -+ const VALID_BITS: u32 = Self::WAIT_BEFORE.bits() | Self::WRITE.bits() | Self::WAIT_AFTER.bits(); -+ -+ pub const fn empty() -> Self { -+ Self(0) -+ } -+ -+ pub const fn from_bits(bits: u32) -> Option { -+ if bits & !Self::VALID_BITS == 0 { -+ Some(Self(bits as u8)) -+ } else { -+ None -+ } -+ } -+ -+ pub const fn bits(self) -> u32 { -+ self.0 as u32 -+ } -+ -+ pub const fn contains(self, other: Self) -> bool { -+ self.bits() & other.bits() == other.bits() -+ } -+} -+ -+impl std::ops::BitOr for FileWritebackFlags { -+ type Output = Self; -+ -+ fn bitor(self, rhs: Self) -> Self::Output { -+ Self(self.0 | rhs.0) -+ } -+} -+ - pub trait FileSystem: fmt::Debug + Send + Sync + 'static + Upcastable { - fn readlink(&self, path: &Path) -> Result; - fn read_dir(&self, path: &Path) -> Result; -@@ -101,6 +163,17 @@ pub trait FileSystem: fmt::Debug + Send + Sync + 'static + Upcastable { - fn symlink_metadata(&self, path: &Path) -> Result; - fn remove_file(&self, path: &Path) -> Result<()>; - -+ /// Open a directory as a stable object that can be synchronized later. -+ /// -+ /// This is deliberately separate from [`FileOpener`]: a directory handle -+ /// must retain the identity of the opened directory across renames and -+ /// unlinks, while reopening a path at sync time can target a different -+ /// directory. Backends without a real directory-sync primitive must fail -+ /// closed instead of reporting a successful no-op. -+ fn open_dir(&self, _path: &Path) -> Result> { -+ Err(FsError::Unsupported) -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_>; - } - -@@ -157,11 +230,34 @@ where - (**self).remove_file(path) - } - -+ fn open_dir(&self, path: &Path) -> Result> { -+ (**self).open_dir(path) -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_> { - (**self).new_open_options() - } - } - -+/// An opened directory whose identity is independent of its current path. -+/// -+/// Directory entries are metadata on POSIX filesystems, so both WASI -+/// `fd_sync` and `fd_datasync` use the full-directory sync operation exposed -+/// here. Implementations must not substitute a flush/no-op for durable sync. -+pub trait VirtualDirectory: fmt::Debug + Send + Sync + 'static { -+ fn sync_all_to_disk(&self) -> BoxFuture<'_, io::Result<()>> { -+ Box::pin(async { Err(io::ErrorKind::Unsupported.into()) }) -+ } -+ -+ fn has_blocking_sync_all_to_disk(&self) -> bool { -+ false -+ } -+ -+ fn sync_all_to_disk_blocking(&self) -> io::Result<()> { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+} -+ - pub trait FileOpener { - fn open( - &self, -@@ -178,6 +274,8 @@ pub struct OpenOptionsConfig { - pub create: bool, - pub append: bool, - pub truncate: bool, -+ pub sync: bool, -+ pub data_sync: bool, - } - - impl OpenOptionsConfig { -@@ -190,6 +288,8 @@ impl OpenOptionsConfig { - create: parent_rights.create && self.create, - append: parent_rights.append && self.append, - truncate: parent_rights.truncate && self.truncate, -+ sync: self.sync, -+ data_sync: self.data_sync, - } - } - -@@ -217,6 +317,14 @@ impl OpenOptionsConfig { - self.truncate - } - -+ pub const fn sync(&self) -> bool { -+ self.sync -+ } -+ -+ pub const fn data_sync(&self) -> bool { -+ self.data_sync -+ } -+ - /// Would a file opened with this [`OpenOptionsConfig`] change files on the - /// filesystem. - pub const fn would_mutate(&self) -> bool { -@@ -227,6 +335,8 @@ impl OpenOptionsConfig { - create, - append, - truncate, -+ sync: _, -+ data_sync: _, - } = *self; - append || write || create || create_new || truncate - } -@@ -254,6 +364,8 @@ impl<'a> OpenOptions<'a> { - create: false, - append: false, - truncate: false, -+ sync: false, -+ data_sync: false, - }, - } - } -@@ -323,6 +435,18 @@ impl<'a> OpenOptions<'a> { - self - } - -+ /// Sets synchronous write-through semantics for file data and metadata. -+ pub fn sync(&mut self, sync: bool) -> &mut Self { -+ self.conf.sync = sync; -+ self -+ } -+ -+ /// Sets synchronous write-through semantics for file data. -+ pub fn data_sync(&mut self, data_sync: bool) -> &mut Self { -+ self.conf.data_sync = data_sync; -+ self -+ } -+ - pub fn open>( - &mut self, - path: P, -@@ -374,12 +498,282 @@ pub trait VirtualFile: - None - } - -+ /// Returns whether this handle was opened with read access when the backend -+ /// can report that cheaply. -+ fn open_read(&self) -> Option { -+ None -+ } -+ -+ /// Returns whether this handle was opened with write access when the backend -+ /// can report that cheaply. -+ fn open_write(&self) -> Option { -+ None -+ } -+ - /// Writes to this file using an mmap offset and reference - /// (this method only works for mmap optimized file systems) - fn write_from_mmap(&mut self, _offset: u64, _len: u64) -> std::io::Result<()> { - Err(std::io::ErrorKind::Unsupported.into()) - } - -+ /// Advise the backing filesystem about access to one byte range. -+ /// -+ /// The default is deliberately unsupported: silently accepting advice -+ /// would prevent callers from distinguishing a real range-scoped hint -+ /// from a no-op. -+ fn advise(&self, _offset: u64, _len: u64, _advice: FileAdvice) -> io::Result<()> { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Initiate and/or wait for writeback of one file range. -+ /// -+ /// This is an advisory writeback operation, not a durability primitive. -+ /// In particular, callers must continue using data/full sync operations -+ /// wherever crash durability is required. -+ fn writeback_range( -+ &self, -+ _offset: u64, -+ _len: u64, -+ _flags: FileWritebackFlags, -+ ) -> io::Result<()> { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking seek path. -+ /// -+ /// Runtime users can use this to avoid async scheduling overhead for host -+ /// regular files while preserving the async fallback for virtual files. -+ fn has_blocking_seek(&self) -> bool { -+ false -+ } -+ -+ /// Seek using a backend-native blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_seek`] returns true. -+ fn seek_blocking(&mut self, _pos: SeekFrom) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking cursor-read path. -+ fn has_blocking_read(&self) -> bool { -+ false -+ } -+ -+ /// Read from the current file cursor using a backend-native blocking -+ /// primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_read`] returns true. -+ fn read_blocking(&mut self, _buf: &mut [u8]) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking positioned-read path. -+ fn has_blocking_read_at(&self) -> bool { -+ false -+ } -+ -+ /// Read at a file offset using a backend-native blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_read_at`] returns true. -+ fn read_at_blocking(&mut self, _buf: &mut [u8], _offset: u64) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking positioned-read path -+ /// that can be called through a shared reference. -+ /// -+ /// Backends should only return true when the operation does not mutate -+ /// backend cursor state and the backend is safe to use concurrently. -+ fn has_blocking_read_at_shared(&self) -> bool { -+ false -+ } -+ -+ /// Read at a file offset through a shared reference using a backend-native -+ /// blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_read_at_shared`] -+ /// returns true. -+ fn read_at_blocking_shared(&self, _buf: &mut [u8], _offset: u64) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking cursor-write path. -+ fn has_blocking_write(&self) -> bool { -+ false -+ } -+ -+ /// Write at the current file cursor using a backend-native blocking -+ /// primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_write`] returns true. -+ fn write_blocking(&mut self, _buf: &[u8]) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Write at a file offset without changing the file cursor. -+ /// -+ /// File-system backends should override this when they have a native -+ /// positioned-write primitive. The fallback preserves POSIX `pwrite` -+ /// cursor semantics for backends that only expose seek/write. -+ fn write_at<'a>(&'a mut self, buf: &'a [u8], offset: u64) -> BoxFuture<'a, io::Result> { -+ Box::pin(async move { -+ let cursor = self.seek(SeekFrom::Current(0)).await?; -+ self.seek(SeekFrom::Start(offset)).await?; -+ let write_result = self.write(buf).await; -+ let restore_result = self.seek(SeekFrom::Start(cursor)).await; -+ -+ match (write_result, restore_result) { -+ (Ok(written), Ok(_)) => Ok(written), -+ (Err(err), _) => Err(err), -+ (Ok(_), Err(err)) => Err(err), -+ } -+ }) -+ } -+ -+ /// Returns whether this file has a native blocking positioned-write path. -+ /// -+ /// Runtime users can use this to avoid async scheduling overhead for host -+ /// regular files while preserving the async fallback for virtual files. -+ fn has_blocking_write_at(&self) -> bool { -+ false -+ } -+ -+ /// Write at a file offset using a backend-native blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_write_at`] returns true. -+ fn write_at_blocking(&mut self, _buf: &[u8], _offset: u64) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking positioned-write path -+ /// that can be called through a shared reference. -+ /// -+ /// Backends should only return true when the operation does not mutate -+ /// backend cursor state and the backend is safe to use concurrently. -+ fn has_blocking_write_at_shared(&self) -> bool { -+ false -+ } -+ -+ /// Write at a file offset through a shared reference using a backend-native -+ /// blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_write_at_shared`] -+ /// returns true. -+ fn write_at_blocking_shared(&self, _buf: &[u8], _offset: u64) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking vectored -+ /// positioned-write path. -+ fn has_blocking_write_vectored_at(&self) -> bool { -+ false -+ } -+ -+ /// Vectored write at a file offset using a backend-native blocking -+ /// primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_write_vectored_at`] -+ /// returns true. -+ fn write_vectored_at_blocking( -+ &mut self, -+ _bufs: &[io::IoSlice<'_>], -+ _offset: u64, -+ ) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking vectored -+ /// positioned-write path that can be called through a shared reference. -+ fn has_blocking_write_vectored_at_shared(&self) -> bool { -+ false -+ } -+ -+ /// Vectored write at a file offset through a shared reference using a -+ /// backend-native blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_write_vectored_at_shared`] -+ /// returns true. -+ fn write_vectored_at_blocking_shared( -+ &self, -+ _bufs: &[io::IoSlice<'_>], -+ _offset: u64, -+ ) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking positioned zero-write -+ /// path. -+ fn has_blocking_write_zeroes_at(&self) -> bool { -+ false -+ } -+ -+ /// Write zero bytes at a file offset using a backend-native blocking -+ /// primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_write_zeroes_at`] -+ /// returns true. -+ fn write_zeroes_at_blocking(&mut self, _len: u64, _offset: u64) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking positioned zero-write -+ /// path that can be called through a shared reference. -+ fn has_blocking_write_zeroes_at_shared(&self) -> bool { -+ false -+ } -+ -+ /// Write zero bytes at a file offset through a shared reference using a -+ /// backend-native blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_write_zeroes_at_shared`] -+ /// returns true. -+ fn write_zeroes_at_blocking_shared(&self, _len: u64, _offset: u64) -> io::Result { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Synchronize file data to durable storage. -+ fn sync_data_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { -+ Box::pin(async move { self.flush().await }) -+ } -+ -+ /// Synchronize file data and metadata to durable storage. -+ fn sync_all_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { -+ Box::pin(async move { self.flush().await }) -+ } -+ -+ /// Returns whether this file has a native blocking data-only sync path. -+ fn has_blocking_sync_data_to_disk(&self) -> bool { -+ false -+ } -+ -+ /// Synchronize file data using a backend-native blocking primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_sync_data_to_disk`] -+ /// returns true. -+ fn sync_data_to_disk_blocking(&mut self) -> io::Result<()> { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file has a native blocking full sync path. -+ fn has_blocking_sync_all_to_disk(&self) -> bool { -+ false -+ } -+ -+ /// Synchronize file data and metadata using a backend-native blocking -+ /// primitive. -+ /// -+ /// Only call this when [`VirtualFile::has_blocking_sync_all_to_disk`] -+ /// returns true. -+ fn sync_all_to_disk_blocking(&mut self) -> io::Result<()> { -+ Err(io::ErrorKind::Unsupported.into()) -+ } -+ -+ /// Returns whether this file handle was opened with host-native -+ /// synchronous write-through semantics. -+ fn is_sync_on_write(&self, _sync_metadata: bool) -> bool { -+ false -+ } -+ - /// This method will copy a file from a source to this destination where - /// the default is to do a straight byte copy however file system implementors - /// may optimize this to do a zero copy -@@ -765,3 +1159,53 @@ impl Iterator for ReadDir { - None - } - } -+ -+#[cfg(test)] -+mod advice_tests { -+ use super::{FileAdvice, FileWritebackFlags, NullFile, VirtualFile}; -+ -+ #[test] -+ fn virtual_file_advice_defaults_to_unsupported() { -+ let file = NullFile::default(); -+ let error = file.advise(0, 4096, FileAdvice::WillNeed).unwrap_err(); -+ -+ assert_eq!(error.kind(), std::io::ErrorKind::Unsupported); -+ } -+ -+ #[test] -+ fn virtual_file_shared_positioned_read_defaults_to_unsupported() { -+ let file = NullFile::default(); -+ let mut buf = [0_u8; 1]; -+ -+ assert!(!file.has_blocking_read_at_shared()); -+ let error = file.read_at_blocking_shared(&mut buf, 0).unwrap_err(); -+ assert_eq!(error.kind(), std::io::ErrorKind::Unsupported); -+ } -+ -+ #[test] -+ fn file_writeback_flags_preserve_the_linux_abi_bits() { -+ assert_eq!(FileWritebackFlags::WAIT_BEFORE.bits(), 1); -+ assert_eq!(FileWritebackFlags::WRITE.bits(), 2); -+ assert_eq!(FileWritebackFlags::WAIT_AFTER.bits(), 4); -+ assert_eq!(FileWritebackFlags::empty().bits(), 0); -+ assert_eq!( -+ (FileWritebackFlags::WAIT_BEFORE -+ | FileWritebackFlags::WRITE -+ | FileWritebackFlags::WAIT_AFTER) -+ .bits(), -+ 7 -+ ); -+ assert!(FileWritebackFlags::from_bits(7).is_some()); -+ assert!(FileWritebackFlags::from_bits(8).is_none()); -+ } -+ -+ #[test] -+ fn virtual_file_writeback_defaults_to_unsupported() { -+ let file = NullFile::default(); -+ let error = file -+ .writeback_range(0, 4096, FileWritebackFlags::WRITE) -+ .unwrap_err(); -+ -+ assert_eq!(error.kind(), std::io::ErrorKind::Unsupported); -+ } -+} -diff --git a/lib/virtual-fs/src/mount_fs.rs b/lib/virtual-fs/src/mount_fs.rs -index 25b9e23..f921671 100644 ---- a/lib/virtual-fs/src/mount_fs.rs -+++ b/lib/virtual-fs/src/mount_fs.rs -@@ -695,6 +695,22 @@ impl FileSystem for MountFileSystem { - } - } - -+ fn open_dir(&self, path: &Path) -> Result> { -+ let path = self.prepare_path(path)?; -+ -+ if let Some(node) = self.exact_node(&path) -+ && node.fs.is_none() -+ { -+ // A synthetic mount-tree directory has no durable backing object. -+ return Err(FsError::Unsupported); -+ } -+ -+ match self.resolve_mount(path) { -+ Some(resolved) => resolved.fs.open_dir(&resolved.delegated_path), -+ None => Err(FsError::EntryNotFound), -+ } -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_> { - OpenOptions::new(self) - } -@@ -988,6 +1004,8 @@ mod tests { - create: true, - append: false, - truncate: false, -+ sync: false, -+ data_sync: false, - }, - ) - .unwrap(); -@@ -1001,6 +1019,8 @@ mod tests { - create: true, - append: false, - truncate: false, -+ sync: false, -+ data_sync: false, - }, - ) - .unwrap(); -diff --git a/lib/virtual-fs/src/overlay_fs.rs b/lib/virtual-fs/src/overlay_fs.rs -index 846bdf7..bd608f6 100644 ---- a/lib/virtual-fs/src/overlay_fs.rs -+++ b/lib/virtual-fs/src/overlay_fs.rs -@@ -427,6 +427,44 @@ where - self.permission_error_or_not_found(path) - } - -+ fn open_dir( -+ &self, -+ path: &Path, -+ ) -> crate::Result> { -+ if ops::is_white_out(path).is_some() { -+ return Err(FsError::EntryNotFound); -+ } -+ -+ // Sync the highest-precedence directory that is actually visible. -+ // In particular, do not bypass an unsupported writable upper layer -+ // and report success after syncing an unrelated lower directory. -+ match self.primary.metadata(path) { -+ Ok(metadata) if metadata.is_dir() => return self.primary.open_dir(path), -+ Ok(_) => return Err(FsError::NotAFile), -+ Err(error) if should_continue(error) => {} -+ Err(error) => return Err(error), -+ } -+ -+ if ops::has_white_out(&self.primary, path) { -+ return Err(FsError::EntryNotFound); -+ } -+ -+ for fs in self.secondaries.filesystems() { -+ match fs.metadata(path) { -+ // A lower-only directory can be copied up after this open. A -+ // handle to the lower inode could then return successful fsync -+ // without making the visible upper-layer mutation durable. -+ // Fail closed until the directory exists in the mutable primary. -+ Ok(metadata) if metadata.is_dir() => return Err(FsError::Unsupported), -+ Ok(_) => return Err(FsError::NotAFile), -+ Err(error) if should_continue(error) => continue, -+ Err(error) => return Err(error), -+ } -+ } -+ -+ Err(FsError::EntryNotFound) -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_> { - OpenOptions::new(self) - } -@@ -1176,6 +1214,20 @@ mod tests { - ); - } - -+ #[test] -+ fn lower_only_directory_handle_fails_closed_before_copy_up() { -+ let primary = MemFS::default(); -+ let secondary = MemFS::default(); -+ ops::create_dir_all(&secondary, "/lower-only").unwrap(); -+ let overlay = OverlayFileSystem::new(primary, [secondary]); -+ -+ assert!(overlay.metadata(Path::new("/lower-only")).unwrap().is_dir()); -+ assert!(matches!( -+ overlay.open_dir(Path::new("/lower-only")), -+ Err(FsError::Unsupported) -+ )); -+ } -+ - #[tokio::test] - async fn remove_directory() { - let primary = MemFS::default(); -diff --git a/lib/virtual-fs/src/passthru_fs.rs b/lib/virtual-fs/src/passthru_fs.rs -index a32f9da..5fe117b 100644 ---- a/lib/virtual-fs/src/passthru_fs.rs -+++ b/lib/virtual-fs/src/passthru_fs.rs -@@ -61,6 +61,10 @@ impl FileSystem for PassthruFileSystem { - self.fs.remove_file(path) - } - -+ fn open_dir(&self, path: &Path) -> Result> { -+ self.fs.open_dir(path) -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_> { - self.fs.new_open_options() - } -diff --git a/lib/virtual-fs/src/trace_fs.rs b/lib/virtual-fs/src/trace_fs.rs -index 667bc84..139034e 100644 ---- a/lib/virtual-fs/src/trace_fs.rs -+++ b/lib/virtual-fs/src/trace_fs.rs -@@ -83,6 +83,14 @@ where - self.0.remove_file(path) - } - -+ #[tracing::instrument(level = "trace", skip(self), err)] -+ fn open_dir( -+ &self, -+ path: &std::path::Path, -+ ) -> crate::Result> { -+ self.0.open_dir(path) -+ } -+ - #[tracing::instrument(level = "trace", skip(self))] - fn new_open_options(&self) -> crate::OpenOptions<'_> { - crate::OpenOptions::new(self) -diff --git a/lib/virtual-io/src/guard.rs b/lib/virtual-io/src/guard.rs -index 60d5596..f062bd9 100644 ---- a/lib/virtual-io/src/guard.rs -+++ b/lib/virtual-io/src/guard.rs -@@ -5,7 +5,9 @@ use std::{ - - use mio::Token; - --use crate::{InterestHandler, InterestType, InterestWakerMap, Selector}; -+use crate::{ -+ InterestHandler, InterestType, InterestWakerMap, MultiplexedInterestHandler, Selector, -+}; - - #[derive(Debug)] - #[must_use = "Leaking token guards will break the IO subsystem"] -@@ -58,6 +60,12 @@ impl InterestGuard { - } - } - -+ pub fn attach_waker_map(&mut self, waker_map: InterestWakerMap) { -+ if let Some(selector) = self.selector.upgrade() { -+ selector.attach_waker_map(self.token, waker_map); -+ } -+ } -+ - fn drop_internal(&mut self) { - if let Some(selector) = self.selector.upgrade() { - selector.remove(self.token, None).ok(); -@@ -70,6 +78,125 @@ pub enum HandlerGuardState { - None, - ExternalHandler(InterestGuard), - WakerMap(InterestGuard, InterestWakerMap), -+ ExternalHandlerWithWakerMap(InterestGuard, InterestWakerMap), -+} -+ -+impl HandlerGuardState { -+ pub fn set_external_handler( -+ &mut self, -+ selector: &Arc, -+ source: &mut dyn mio::event::Source, -+ interest: mio::Interest, -+ handler: Box, -+ ) -> io::Result<()> { -+ let mut current = HandlerGuardState::None; -+ std::mem::swap(self, &mut current); -+ -+ match current { -+ HandlerGuardState::None => { -+ *self = HandlerGuardState::ExternalHandler(InterestGuard::new( -+ selector, handler, source, interest, -+ )?); -+ Ok(()) -+ } -+ HandlerGuardState::ExternalHandler(mut guard) => match guard.replace_handler(handler) { -+ Ok(()) => { -+ *self = HandlerGuardState::ExternalHandler(guard); -+ Ok(()) -+ } -+ Err(handler) => { -+ guard.unregister(source).ok(); -+ *self = HandlerGuardState::ExternalHandler(InterestGuard::new( -+ selector, handler, source, interest, -+ )?); -+ Ok(()) -+ } -+ }, -+ HandlerGuardState::WakerMap(mut guard, waker_map) -+ | HandlerGuardState::ExternalHandlerWithWakerMap(mut guard, waker_map) => { -+ let handler = MultiplexedInterestHandler::new(handler, waker_map.clone()); -+ match guard.replace_handler(handler) { -+ Ok(()) => { -+ *self = HandlerGuardState::ExternalHandlerWithWakerMap(guard, waker_map); -+ Ok(()) -+ } -+ Err(handler) => { -+ guard.unregister(source).ok(); -+ *self = HandlerGuardState::ExternalHandlerWithWakerMap( -+ InterestGuard::new(selector, handler, source, interest)?, -+ waker_map, -+ ); -+ Ok(()) -+ } -+ } -+ } -+ } -+ } -+ -+ pub fn remove_external_handler( -+ &mut self, -+ source: &mut dyn mio::event::Source, -+ ) -> io::Result<()> { -+ let mut current = HandlerGuardState::None; -+ std::mem::swap(self, &mut current); -+ match current { -+ HandlerGuardState::None => Ok(()), -+ HandlerGuardState::ExternalHandler(mut guard) => guard.unregister(source), -+ HandlerGuardState::WakerMap(guard, waker_map) => { -+ *self = HandlerGuardState::WakerMap(guard, waker_map); -+ Ok(()) -+ } -+ HandlerGuardState::ExternalHandlerWithWakerMap(mut guard, waker_map) => { -+ let res = guard.replace_handler(Box::new(waker_map.clone())); -+ match res { -+ Ok(()) => { -+ *self = HandlerGuardState::WakerMap(guard, waker_map); -+ Ok(()) -+ } -+ Err(_) => { -+ guard.unregister(source).ok(); -+ Ok(()) -+ } -+ } -+ } -+ } -+ } -+ -+ pub fn remove_all_handlers(&mut self, source: &mut dyn mio::event::Source) -> io::Result<()> { -+ let mut current = HandlerGuardState::None; -+ std::mem::swap(self, &mut current); -+ match current { -+ HandlerGuardState::None => Ok(()), -+ HandlerGuardState::ExternalHandler(mut guard) -+ | HandlerGuardState::WakerMap(mut guard, _) -+ | HandlerGuardState::ExternalHandlerWithWakerMap(mut guard, _) => { -+ guard.unregister(source) -+ } -+ } -+ } -+ -+ pub fn push_interest(&mut self, interest: InterestType) { -+ match self { -+ HandlerGuardState::ExternalHandler(guard) -+ | HandlerGuardState::ExternalHandlerWithWakerMap(guard, _) => { -+ guard.interest(interest); -+ } -+ HandlerGuardState::WakerMap(_, waker_map) => { -+ waker_map.push_interest(interest); -+ } -+ HandlerGuardState::None => {} -+ } -+ } -+ -+ pub fn pop_waker_interest(&mut self, interest: InterestType) -> bool { -+ match self { -+ HandlerGuardState::WakerMap(_, waker_map) -+ | HandlerGuardState::ExternalHandlerWithWakerMap(_, waker_map) => { -+ waker_map.pop(interest) -+ } -+ HandlerGuardState::ExternalHandler(_) | HandlerGuardState::None => false, -+ } -+ } - } - - pub fn state_as_waker_map<'a>( -@@ -77,20 +204,35 @@ pub fn state_as_waker_map<'a>( - selector: &'_ Arc, - source: &'_ mut dyn mio::event::Source, - ) -> io::Result<&'a mut InterestWakerMap> { -- if !matches!(state, HandlerGuardState::WakerMap(_, _)) { -- let waker_map = InterestWakerMap::default(); -- *state = HandlerGuardState::WakerMap( -- InterestGuard::new( -- selector, -- Box::new(waker_map.clone()), -- source, -- mio::Interest::READABLE | mio::Interest::WRITABLE, -- )?, -- waker_map, -- ); -+ match state { -+ HandlerGuardState::None => { -+ let waker_map = InterestWakerMap::default(); -+ *state = HandlerGuardState::WakerMap( -+ InterestGuard::new( -+ selector, -+ Box::new(waker_map.clone()), -+ source, -+ mio::Interest::READABLE | mio::Interest::WRITABLE, -+ )?, -+ waker_map, -+ ); -+ } -+ HandlerGuardState::ExternalHandler(guard) => { -+ let waker_map = InterestWakerMap::default(); -+ guard.attach_waker_map(waker_map.clone()); -+ -+ let mut current = HandlerGuardState::None; -+ std::mem::swap(state, &mut current); -+ if let HandlerGuardState::ExternalHandler(guard) = current { -+ *state = HandlerGuardState::ExternalHandlerWithWakerMap(guard, waker_map); -+ } -+ } -+ HandlerGuardState::WakerMap(_, _) -+ | HandlerGuardState::ExternalHandlerWithWakerMap(_, _) => {} - } - Ok(match state { -- HandlerGuardState::WakerMap(_, map) => map, -+ HandlerGuardState::WakerMap(_, map) -+ | HandlerGuardState::ExternalHandlerWithWakerMap(_, map) => map, - _ => unreachable!(), - }) - } -diff --git a/lib/virtual-io/src/interest.rs b/lib/virtual-io/src/interest.rs -index 7cf0529..24ad072 100644 ---- a/lib/virtual-io/src/interest.rs -+++ b/lib/virtual-io/src/interest.rs -@@ -1,7 +1,10 @@ - use serde::{Deserialize, Serialize}; - use std::{ - collections::{HashMap, HashSet}, -- sync::{Arc, Mutex}, -+ sync::{ -+ Arc, Mutex, Weak, -+ atomic::{AtomicU64, Ordering}, -+ }, - task::{Context, RawWaker, RawWakerVTable, Waker}, - }; - -@@ -77,6 +80,151 @@ pub trait InterestHandler: Send + Sync + std::fmt::Debug { - fn has_interest(&self, interest: InterestType) -> bool; - } - -+type SharedInterestHandler = Arc>>; -+ -+#[derive(Debug, Default)] -+struct InterestHandlerFanoutInner { -+ next_id: AtomicU64, -+ handlers: Mutex>, -+} -+ -+/// A cloneable interest handler that fans each callback out to independently -+/// registered consumers. The handler map is only held while taking a snapshot; -+/// callbacks execute outside it so consumers may safely unregister themselves. -+#[derive(Debug, Clone, Default)] -+pub struct InterestHandlerFanout { -+ inner: Arc, -+} -+ -+impl InterestHandlerFanout { -+ pub fn register( -+ &self, -+ handler: Box, -+ ) -> InterestHandlerRegistration { -+ let id = self -+ .inner -+ .next_id -+ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { -+ current.checked_add(1) -+ }) -+ .expect("interest-handler fanout identity space exhausted"); -+ self.inner -+ .handlers -+ .lock() -+ .unwrap() -+ .insert(id, Arc::new(Mutex::new(handler))); -+ InterestHandlerRegistration { -+ inner: Arc::downgrade(&self.inner), -+ id, -+ } -+ } -+ -+ pub fn handler_count(&self) -> usize { -+ self.inner.handlers.lock().unwrap().len() -+ } -+ -+ fn snapshot(&self) -> InterestHandlerSnapshot { -+ let handlers = self.inner.handlers.lock().unwrap(); -+ match handlers.len() { -+ 0 => InterestHandlerSnapshot::Empty, -+ 1 => InterestHandlerSnapshot::One(handlers.values().next().unwrap().clone()), -+ _ => InterestHandlerSnapshot::Many(handlers.values().cloned().collect()), -+ } -+ } -+} -+ -+enum InterestHandlerSnapshot { -+ Empty, -+ One(SharedInterestHandler), -+ Many(Vec), -+} -+ -+impl InterestHandler for InterestHandlerFanout { -+ fn push_interest(&mut self, interest: InterestType) { -+ match self.snapshot() { -+ InterestHandlerSnapshot::Empty => {} -+ InterestHandlerSnapshot::One(handler) => { -+ handler.lock().unwrap().push_interest(interest); -+ } -+ InterestHandlerSnapshot::Many(handlers) => { -+ for handler in handlers { -+ handler.lock().unwrap().push_interest(interest); -+ } -+ } -+ } -+ } -+ -+ fn pop_interest(&mut self, interest: InterestType) -> bool { -+ match self.snapshot() { -+ InterestHandlerSnapshot::Empty => false, -+ InterestHandlerSnapshot::One(handler) => handler.lock().unwrap().pop_interest(interest), -+ InterestHandlerSnapshot::Many(handlers) => { -+ handlers.into_iter().fold(false, |seen, handler| { -+ handler.lock().unwrap().pop_interest(interest) || seen -+ }) -+ } -+ } -+ } -+ -+ fn has_interest(&self, interest: InterestType) -> bool { -+ match self.snapshot() { -+ InterestHandlerSnapshot::Empty => false, -+ InterestHandlerSnapshot::One(handler) => handler.lock().unwrap().has_interest(interest), -+ InterestHandlerSnapshot::Many(handlers) => handlers -+ .into_iter() -+ .any(|handler| handler.lock().unwrap().has_interest(interest)), -+ } -+ } -+} -+ -+#[derive(Debug)] -+pub struct InterestHandlerRegistration { -+ inner: Weak, -+ id: u64, -+} -+ -+impl Drop for InterestHandlerRegistration { -+ fn drop(&mut self) { -+ let Some(inner) = self.inner.upgrade() else { -+ return; -+ }; -+ let removed = inner.handlers.lock().unwrap().remove(&self.id); -+ drop(removed); -+ } -+} -+ -+#[derive(Debug)] -+pub struct MultiplexedInterestHandler { -+ primary: Box, -+ waker_map: InterestWakerMap, -+} -+ -+impl MultiplexedInterestHandler { -+ pub fn new( -+ primary: Box, -+ waker_map: InterestWakerMap, -+ ) -> Box { -+ Box::new(Self { primary, waker_map }) -+ } -+} -+ -+impl InterestHandler for MultiplexedInterestHandler { -+ fn push_interest(&mut self, interest: InterestType) { -+ self.primary.push_interest(interest); -+ self.waker_map.push_interest(interest); -+ } -+ -+ fn pop_interest(&mut self, interest: InterestType) -> bool { -+ let primary = self.primary.pop_interest(interest); -+ let waker_map = self.waker_map.pop_interest(interest); -+ primary || waker_map -+ } -+ -+ fn has_interest(&self, interest: InterestType) -> bool { -+ self.primary.has_interest(interest) || self.waker_map.has_interest(interest) -+ } -+} -+ - impl From<&Waker> for Box { - fn from(waker: &Waker) -> Self { - WakerInterestHandler::new(waker) -@@ -194,3 +342,99 @@ impl InterestHandler for InterestWakerMap { - state.triggered.contains(&interest) - } - } -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use std::sync::atomic::{AtomicUsize, Ordering}; -+ use std::sync::mpsc; -+ use std::time::Duration; -+ -+ #[derive(Debug)] -+ struct CountingHandler(Arc); -+ -+ impl InterestHandler for CountingHandler { -+ fn push_interest(&mut self, _interest: InterestType) { -+ self.0.fetch_add(1, Ordering::SeqCst); -+ } -+ -+ fn pop_interest(&mut self, _interest: InterestType) -> bool { -+ false -+ } -+ -+ fn has_interest(&self, _interest: InterestType) -> bool { -+ false -+ } -+ } -+ -+ #[test] -+ fn fanout_removes_only_the_dropped_registration() { -+ let mut fanout = InterestHandlerFanout::default(); -+ let first = Arc::new(AtomicUsize::new(0)); -+ let second = Arc::new(AtomicUsize::new(0)); -+ let first_registration = fanout.register(Box::new(CountingHandler(first.clone()))); -+ let _second_registration = fanout.register(Box::new(CountingHandler(second.clone()))); -+ -+ fanout.push_interest(InterestType::Readable); -+ assert_eq!(first.load(Ordering::SeqCst), 1); -+ assert_eq!(second.load(Ordering::SeqCst), 1); -+ -+ drop(first_registration); -+ fanout.push_interest(InterestType::Readable); -+ assert_eq!(first.load(Ordering::SeqCst), 1); -+ assert_eq!(second.load(Ordering::SeqCst), 2); -+ assert_eq!(fanout.handler_count(), 1); -+ } -+ -+ #[test] -+ fn singleton_fanout_uses_allocation_free_snapshot_shape() { -+ let fanout = InterestHandlerFanout::default(); -+ let _registration = -+ fanout.register(Box::new(CountingHandler(Arc::new(AtomicUsize::new(0))))); -+ assert!(matches!(fanout.snapshot(), InterestHandlerSnapshot::One(_))); -+ } -+ -+ #[derive(Debug)] -+ struct ReentrantDropHandler { -+ fanout: InterestHandlerFanout, -+ dropped: mpsc::Sender<()>, -+ } -+ -+ impl InterestHandler for ReentrantDropHandler { -+ fn push_interest(&mut self, _interest: InterestType) {} -+ -+ fn pop_interest(&mut self, _interest: InterestType) -> bool { -+ false -+ } -+ -+ fn has_interest(&self, _interest: InterestType) -> bool { -+ false -+ } -+ } -+ -+ impl Drop for ReentrantDropHandler { -+ fn drop(&mut self) { -+ let transient = self -+ .fanout -+ .register(Box::new(CountingHandler(Arc::new(AtomicUsize::new(0))))); -+ drop(transient); -+ self.dropped.send(()).unwrap(); -+ } -+ } -+ -+ #[test] -+ fn registration_drop_releases_map_lock_before_handler_drop() { -+ let fanout = InterestHandlerFanout::default(); -+ let (dropped_tx, dropped_rx) = mpsc::channel(); -+ let registration = fanout.register(Box::new(ReentrantDropHandler { -+ fanout: fanout.clone(), -+ dropped: dropped_tx, -+ })); -+ -+ std::thread::spawn(move || drop(registration)); -+ dropped_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("handler drop should re-enter the fanout without deadlocking"); -+ assert_eq!(fanout.handler_count(), 0); -+ } -+} -diff --git a/lib/virtual-io/src/selector.rs b/lib/virtual-io/src/selector.rs -index b53e018..21399a8 100644 ---- a/lib/virtual-io/src/selector.rs -+++ b/lib/virtual-io/src/selector.rs -@@ -8,7 +8,7 @@ use std::{ - }, - }; - --use crate::{InterestHandler, InterestType}; -+use crate::{InterestHandler, InterestType, InterestWakerMap, MultiplexedInterestHandler}; - - pub enum SelectorModification { - Add { -@@ -22,6 +22,10 @@ pub enum SelectorModification { - token: Token, - handler: Box, - }, -+ AttachWakerMap { -+ token: Token, -+ waker_map: InterestWakerMap, -+ }, - PushInterest { - token: Token, - interest: InterestType, -@@ -60,6 +64,13 @@ impl SelectorModification { - - lookup.insert(token, handler); - } -+ SelectorModification::AttachWakerMap { token, waker_map } => { -+ if let Some(last) = lookup.remove(&token) { -+ lookup.insert(token, MultiplexedInterestHandler::new(last, waker_map)); -+ } else { -+ lookup.insert(token, Box::new(waker_map)); -+ } -+ } - SelectorModification::PushInterest { token, interest } => { - if let Some(handler) = lookup.get_mut(&token) { - handler.push_interest(interest); -@@ -80,6 +91,10 @@ impl std::fmt::Debug for SelectorModification { - SelectorModification::Replace { token, .. } => { - f.debug_struct("Replace").field("token", token).finish() - } -+ SelectorModification::AttachWakerMap { token, .. } => f -+ .debug_struct("AttachWakerMap") -+ .field("token", token) -+ .finish(), - SelectorModification::PushInterest { token, interest } => f - .debug_struct("PushInterest") - .field("token", token) -@@ -151,14 +166,21 @@ impl Selector { - - // CONCURRENCY: This should never result in a deadlock, as long as source.deregister does not call remove or add again. - let inner_registry = self.registry.lock().unwrap(); -- match source.register(&inner_registry, token, interests) { -- Ok(()) => {} -+ let registration = match source.register(&inner_registry, token, interests) { -+ Ok(()) => Ok(()), - Err(err) if err.kind() == io::ErrorKind::AlreadyExists => { - source.deregister(&inner_registry).ok(); -- source.register(&inner_registry, token, interests)?; -+ source.register(&inner_registry, token, interests) - } -- Err(err) => return Err(err), -+ Err(err) => Err(err), - }; -+ drop(inner_registry); -+ if let Err(err) = registration { -+ // Balance the already queued Add. Remove wakes the idle poll loop, -+ // so the rejected handler is not retained waiting for another edge. -+ self.queue_modification(SelectorModification::Remove { token }); -+ return Err(err); -+ } - - Ok(token) - } -@@ -185,6 +207,10 @@ impl Selector { - self.queue_modification(SelectorModification::Replace { token, handler }); - } - -+ pub fn attach_waker_map(&self, token: Token, waker_map: InterestWakerMap) { -+ self.queue_modification(SelectorModification::AttachWakerMap { token, waker_map }); -+ } -+ - /// Generate a new unique token - #[must_use = "the token must be consumed"] - fn new_token(&self) -> Token { -@@ -193,10 +219,16 @@ impl Selector { - - /// Try to process a modification immediately, otherwise queue it up - fn queue_modification(&self, modification: SelectorModification) { -- // Replace and PushInterest can cause external code to be called so it is a good idea to process them asap so they don't get delayed too long -+ // Process selector modifications promptly. Add must wake the poll loop -+ // so the handler is installed even if the registered source is already -+ // ready and no later edge arrives. - let needs_wakeup = matches!( - &modification, -- SelectorModification::PushInterest { .. } | SelectorModification::Replace { .. } -+ SelectorModification::Add { .. } -+ | SelectorModification::Remove { .. } -+ | SelectorModification::PushInterest { .. } -+ | SelectorModification::AttachWakerMap { .. } -+ | SelectorModification::Replace { .. } - ); - - // CONCURRENCY: This will never deadlock as queued_modifications is always the innermost lock and we don't call any potentially blocking functions while holding the lock. -@@ -286,7 +318,10 @@ impl Selector { - #[cfg(all(unix, test))] - mod tests { - use super::*; -+ use crate::{HandlerGuardState, InterestGuard, state_as_waker_map}; -+ use futures::task::{ArcWake, waker}; - use std::io::Write; -+ use std::sync::Arc; - use std::sync::mpsc; - use std::thread; - use std::time::Duration; -@@ -316,6 +351,55 @@ mod tests { - token: Arc>>, - success_sender: mpsc::Sender<()>, - } -+ -+ #[derive(Debug)] -+ struct DropSignalHandler { -+ dropped: mpsc::Sender<()>, -+ } -+ -+ impl Drop for DropSignalHandler { -+ fn drop(&mut self) { -+ self.dropped.send(()).unwrap(); -+ } -+ } -+ -+ impl InterestHandler for DropSignalHandler { -+ fn push_interest(&mut self, _interest: InterestType) {} -+ -+ fn pop_interest(&mut self, _interest: InterestType) -> bool { -+ false -+ } -+ -+ fn has_interest(&self, _interest: InterestType) -> bool { -+ false -+ } -+ } -+ -+ struct FailingSource; -+ -+ impl mio::event::Source for FailingSource { -+ fn register( -+ &mut self, -+ _registry: &mio::Registry, -+ _token: Token, -+ _interests: mio::Interest, -+ ) -> io::Result<()> { -+ Err(io::Error::other("injected registration failure")) -+ } -+ -+ fn reregister( -+ &mut self, -+ _registry: &mio::Registry, -+ _token: Token, -+ _interests: mio::Interest, -+ ) -> io::Result<()> { -+ Err(io::Error::other("injected registration failure")) -+ } -+ -+ fn deregister(&mut self, _registry: &mio::Registry) -> io::Result<()> { -+ Ok(()) -+ } -+ } - impl InterestHandler for DeadlockingHandler { - fn push_interest(&mut self, _interest: InterestType) { - // This would deadlock without a queue -@@ -334,6 +418,16 @@ mod tests { - } - } - -+ struct ChannelWake { -+ sender: mpsc::Sender<()>, -+ } -+ -+ impl ArcWake for ChannelWake { -+ fn wake_by_ref(arc_self: &Arc) { -+ arc_self.sender.send(()).unwrap(); -+ } -+ } -+ - #[test] - fn test_push_interest() { - let (mut sender, mut receiver) = mio::unix::pipe::new().unwrap(); -@@ -396,6 +490,87 @@ mod tests { - selector.shutdown(); - } - -+ #[test] -+ fn selector_remove_wakes_idle_poll_and_drops_handler_promptly() { -+ let (_sender, mut receiver) = mio::unix::pipe::new().unwrap(); -+ let (dropped_tx, dropped_rx) = mpsc::channel(); -+ let selector = Selector::new(); -+ let token = selector -+ .add( -+ Box::new(DropSignalHandler { -+ dropped: dropped_tx, -+ }), -+ &mut receiver, -+ mio::Interest::READABLE, -+ ) -+ .unwrap(); -+ -+ selector.remove(token, Some(&mut receiver)).unwrap(); -+ -+ dropped_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("Remove must wake an idle selector and promptly drop its handler"); -+ selector.shutdown(); -+ } -+ -+ #[test] -+ fn selector_add_failure_balances_queued_handler_promptly() { -+ let (dropped_tx, dropped_rx) = mpsc::channel(); -+ let selector = Selector::new(); -+ let mut source = FailingSource; -+ -+ let result = selector.add( -+ Box::new(DropSignalHandler { -+ dropped: dropped_tx, -+ }), -+ &mut source, -+ mio::Interest::READABLE, -+ ); -+ assert!(result.is_err()); -+ dropped_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("failed Add must be balanced by a prompt queued Remove"); -+ selector.shutdown(); -+ } -+ -+ #[test] -+ fn external_handler_and_waker_map_both_receive_interest() { -+ let (mut sender, mut receiver) = mio::unix::pipe::new().unwrap(); -+ let (handler_sender, handler_receiver) = mpsc::channel(); -+ let (wake_sender, wake_receiver) = mpsc::channel(); -+ -+ let selector = Selector::new(); -+ let guard = InterestGuard::new( -+ &selector, -+ Box::new(TestHandler { -+ success_sender: handler_sender, -+ }), -+ &mut receiver, -+ mio::Interest::READABLE, -+ ) -+ .unwrap(); -+ let mut state = HandlerGuardState::ExternalHandler(guard); -+ let wake = waker(Arc::new(ChannelWake { -+ sender: wake_sender, -+ })); -+ -+ state_as_waker_map(&mut state, &selector, &mut receiver) -+ .unwrap() -+ .add(InterestType::Readable, &wake); -+ -+ thread::sleep(Duration::from_millis(10)); -+ sender.write_all(&[1]).unwrap(); -+ -+ handler_receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("external handler should receive readiness"); -+ wake_receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("waker map waiter should receive readiness"); -+ -+ state.remove_all_handlers(&mut receiver).unwrap(); -+ } -+ - #[test] - fn test_selector_no_deadlock_when_modifying_the_selector_from_push_interest() { - let (mut sender, mut receiver) = mio::unix::pipe::new().unwrap(); -diff --git a/lib/virtual-net/src/host.rs b/lib/virtual-net/src/host.rs -index 3f1af0f..90a1818 100644 ---- a/lib/virtual-net/src/host.rs -+++ b/lib/virtual-net/src/host.rs -@@ -25,9 +25,13 @@ use std::time::Duration; - use tokio::runtime::Handle; - #[allow(unused_imports, dead_code)] - use tracing::{debug, error, info, trace, warn}; --use virtual_mio::{ -- HandlerGuardState, InterestGuard, InterestHandler, InterestType, Selector, state_as_waker_map, --}; -+use virtual_mio::{HandlerGuardState, InterestHandler, InterestType, Selector, state_as_waker_map}; -+ -+const TCP_ACCEPT_BACKLOG_DRAIN_HARD_LIMIT: usize = 128; -+ -+fn normalize_accept_backlog_limit(backlog: usize) -> usize { -+ backlog.clamp(1, TCP_ACCEPT_BACKLOG_DRAIN_HARD_LIMIT) -+} - - #[derive(Debug)] - pub struct LocalNetworking { -@@ -75,6 +79,24 @@ impl VirtualNetworking for LocalNetworking { - only_v6: bool, - reuse_port: bool, - reuse_addr: bool, -+ ) -> Result> { -+ self.listen_tcp_with_backlog( -+ addr, -+ only_v6, -+ reuse_port, -+ reuse_addr, -+ TCP_ACCEPT_BACKLOG_DRAIN_HARD_LIMIT, -+ ) -+ .await -+ } -+ -+ async fn listen_tcp_with_backlog( -+ &self, -+ addr: SocketAddr, -+ only_v6: bool, -+ reuse_port: bool, -+ reuse_addr: bool, -+ backlog: usize, - ) -> Result> { - if let Some(ruleset) = self.ruleset.as_ref() - && !ruleset.allows_socket(addr, Direction::Inbound) -@@ -93,6 +115,7 @@ impl VirtualNetworking for LocalNetworking { - no_delay: None, - keep_alive: None, - backlog: Default::default(), -+ accept_backlog_limit: normalize_accept_backlog_limit(backlog), - ruleset: self.ruleset.clone(), - }) - }) -@@ -230,10 +253,30 @@ pub struct LocalTcpListener { - no_delay: Option, - keep_alive: Option, - backlog: VecDeque<(Box, SocketAddr)>, -+ accept_backlog_limit: usize, - ruleset: Option, - } - - impl LocalTcpListener { -+ fn prime_accept_backlog(&mut self) { -+ // Preserve level-triggered listener readiness for guests even when the -+ // host selector only delivered one edge for a burst of queued accepts. -+ while self.backlog.len() < self.accept_backlog_limit { -+ match self.try_accept_internal() { -+ Ok(child) => self.backlog.push_back(child), -+ Err(NetworkError::WouldBlock) => break, -+ Err(err) => { -+ tracing::debug!(error = ?err, "failed to prime TCP listener accept backlog"); -+ break; -+ } -+ } -+ } -+ -+ if !self.backlog.is_empty() { -+ self.handler_guard.push_interest(InterestType::Readable); -+ } -+ } -+ - fn try_accept_internal(&mut self) -> Result<(Box, SocketAddr)> { - match self.stream.accept().map_err(io_err_into_net_error) { - Ok((stream, addr)) => { -@@ -254,10 +297,10 @@ impl LocalTcpListener { - Ok((Box::new(socket), addr)) - } - Err(NetworkError::WouldBlock) => { -- if let HandlerGuardState::WakerMap(_, map) = &mut self.handler_guard { -- map.pop(InterestType::Readable); -- map.pop(InterestType::Writable); -- } -+ self.handler_guard -+ .pop_waker_interest(InterestType::Readable); -+ self.handler_guard -+ .pop_waker_interest(InterestType::Writable); - Err(NetworkError::WouldBlock) - } - Err(err) => Err(err), -@@ -267,34 +310,27 @@ impl LocalTcpListener { - - impl VirtualTcpListener for LocalTcpListener { - fn try_accept(&mut self) -> Result<(Box, SocketAddr)> { -- if let Some(child) = self.backlog.pop_front() { -- return Ok(child); -- } -- self.try_accept_internal() -- } -+ let child = match self.backlog.pop_front() { -+ Some(child) => child, -+ None => self.try_accept_internal()?, -+ }; - -- fn set_handler(&mut self, mut handler: Box) -> Result<()> { -- if let HandlerGuardState::ExternalHandler(guard) = &mut self.handler_guard { -- match guard.replace_handler(handler) { -- Ok(()) => return Ok(()), -- Err(h) => handler = h, -- } -+ self.prime_accept_backlog(); - -- // the handler could not be replaced so we need to build a new handler instead -- if let Err(err) = guard.unregister(&mut self.stream) { -- tracing::debug!("failed to unregister previous token - {}", err); -- } -- } -+ Ok(child) -+ } - -- let guard = InterestGuard::new( -- &self.selector, -- handler, -- &mut self.stream, -- mio::Interest::READABLE.add(mio::Interest::WRITABLE), -- ) -- .map_err(io_err_into_net_error)?; -+ fn set_handler(&mut self, handler: Box) -> Result<()> { -+ self.handler_guard -+ .set_external_handler( -+ &self.selector, -+ &mut self.stream, -+ mio::Interest::READABLE.add(mio::Interest::WRITABLE), -+ handler, -+ ) -+ .map_err(io_err_into_net_error)?; - -- self.handler_guard = HandlerGuardState::ExternalHandler(guard); -+ self.prime_accept_backlog(); - - Ok(()) - } -@@ -331,17 +367,9 @@ impl LocalTcpListener { - - impl VirtualIoSource for LocalTcpListener { - fn remove_handler(&mut self) { -- let mut guard = HandlerGuardState::None; -- std::mem::swap(&mut guard, &mut self.handler_guard); -- match guard { -- HandlerGuardState::ExternalHandler(mut guard) => { -- guard.unregister(&mut self.stream).ok(); -- } -- HandlerGuardState::WakerMap(mut guard, _) => { -- guard.unregister(&mut self.stream).ok(); -- } -- HandlerGuardState::None => {} -- } -+ self.handler_guard -+ .remove_external_handler(&mut self.stream) -+ .ok(); - } - - fn poll_read_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { -@@ -353,9 +381,14 @@ impl VirtualIoSource for LocalTcpListener { - let map = state_as_waker_map(state, selector, source).map_err(io_err_into_net_error)?; - map.add(InterestType::Readable, cx.waker()); - -- if let Ok(child) = self.try_accept_internal() { -- self.backlog.push_back(child); -- return Poll::Ready(Ok(1)); -+ match self.try_accept_internal() { -+ Ok(child) => { -+ self.backlog.push_back(child); -+ self.prime_accept_backlog(); -+ return Poll::Ready(Ok(self.backlog.len())); -+ } -+ Err(NetworkError::WouldBlock) => {} -+ Err(err) => return Poll::Ready(Err(err)), - } - Poll::Pending - } -@@ -369,9 +402,14 @@ impl VirtualIoSource for LocalTcpListener { - let map = state_as_waker_map(state, selector, source).map_err(io_err_into_net_error)?; - map.add(InterestType::Writable, cx.waker()); - -- if let Ok(child) = self.try_accept_internal() { -- self.backlog.push_back(child); -- return Poll::Ready(Ok(1)); -+ match self.try_accept_internal() { -+ Ok(child) => { -+ self.backlog.push_back(child); -+ self.prime_accept_backlog(); -+ return Poll::Ready(Ok(self.backlog.len())); -+ } -+ Err(NetworkError::WouldBlock) => {} -+ Err(err) => return Poll::Ready(Err(err)), - } - Poll::Pending - } -@@ -563,9 +601,8 @@ impl VirtualConnectedSocket for LocalTcpStream { - let ret = self.stream.write(data).map_err(io_err_into_net_error); - match &ret { - Ok(0) | Err(NetworkError::WouldBlock) => { -- if let HandlerGuardState::WakerMap(_, map) = &mut self.handler_guard { -- map.pop(InterestType::Writable); -- } -+ self.handler_guard -+ .pop_waker_interest(InterestType::Writable); - } - _ => {} - } -@@ -588,15 +625,37 @@ impl VirtualConnectedSocket for LocalTcpStream { - if !peek { - self.buffer.advance(amt); - } -+ if peek || !self.buffer.is_empty() { -+ self.handler_guard.push_interest(InterestType::Readable); -+ } - return Ok(amt); - } - -- if peek { -+ let ret = if peek { - self.stream.peek(buf) - } else { - self.stream.read(buf) - } -- .map_err(io_err_into_net_error) -+ .map_err(io_err_into_net_error); -+ -+ match &ret { -+ Ok(0) => self.handler_guard.push_interest(InterestType::Closed), -+ Ok(_) if peek => self.handler_guard.push_interest(InterestType::Readable), -+ Ok(_) => { -+ #[cfg(not(target_os = "windows"))] -+ prime_socket_read_readiness(&mut self.handler_guard, self.stream.as_raw_fd()); -+ } -+ Err(NetworkError::WouldBlock) => { -+ self.handler_guard -+ .pop_waker_interest(InterestType::Readable); -+ } -+ Err(NetworkError::ConnectionAborted) | Err(NetworkError::ConnectionReset) => { -+ self.handler_guard.push_interest(InterestType::Closed); -+ } -+ Err(_) => self.handler_guard.push_interest(InterestType::Error), -+ } -+ -+ ret - } - } - -@@ -651,29 +710,17 @@ impl VirtualSocket for LocalTcpStream { - } - } - -- fn set_handler(&mut self, mut handler: Box) -> Result<()> { -- if let HandlerGuardState::ExternalHandler(guard) = &mut self.handler_guard { -- match guard.replace_handler(handler) { -- Ok(()) => return Ok(()), -- Err(h) => handler = h, -- } -- -- // the handler could not be replaced so we need to build a new handler instead -- if let Err(err) = guard.unregister(&mut self.stream) { -- tracing::debug!("failed to unregister previous token - {}", err); -- } -- } -- -- let guard = InterestGuard::new( -- &self.selector, -- handler, -- &mut self.stream, -- mio::Interest::READABLE.add(mio::Interest::WRITABLE), -- ) -- .map_err(io_err_into_net_error)?; -- -- self.handler_guard = HandlerGuardState::ExternalHandler(guard); -- -+ fn set_handler(&mut self, handler: Box) -> Result<()> { -+ self.handler_guard -+ .set_external_handler( -+ &self.selector, -+ &mut self.stream, -+ mio::Interest::READABLE.add(mio::Interest::WRITABLE), -+ handler, -+ ) -+ .map_err(io_err_into_net_error)?; -+ #[cfg(not(target_os = "windows"))] -+ prime_socket_readiness(&mut self.handler_guard, self.stream.as_raw_fd()); - Ok(()) - } - } -@@ -698,17 +745,9 @@ impl LocalTcpStream { - - impl VirtualIoSource for LocalTcpStream { - fn remove_handler(&mut self) { -- let mut guard = HandlerGuardState::None; -- std::mem::swap(&mut guard, &mut self.handler_guard); -- match guard { -- HandlerGuardState::ExternalHandler(mut guard) => { -- guard.unregister(&mut self.stream).ok(); -- } -- HandlerGuardState::WakerMap(mut guard, _) => { -- guard.unregister(&mut self.stream).ok(); -- } -- HandlerGuardState::None => {} -- } -+ self.handler_guard -+ .remove_external_handler(&mut self.stream) -+ .ok(); - } - - fn poll_read_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { -@@ -740,6 +779,39 @@ impl VirtualIoSource for LocalTcpStream { - } - } - -+ fn poll_read_ready_direct(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { -+ #[cfg(target_os = "windows")] -+ { -+ return self.poll_read_ready(cx); -+ } -+ -+ if !self.buffer.is_empty() { -+ return Poll::Ready(Ok(self.buffer.len())); -+ } -+ -+ let (state, selector, stream, _) = self.split_borrow(); -+ let map = state_as_waker_map(state, selector, stream).map_err(io_err_into_net_error)?; -+ map.pop(InterestType::Readable); -+ map.add(InterestType::Readable, cx.waker()); -+ map.add(InterestType::Closed, cx.waker()); -+ -+ if map.has_interest(InterestType::Closed) { -+ return Poll::Ready(Ok(0)); -+ } -+ -+ #[cfg(not(target_os = "windows"))] -+ if let Some(revents) = libc_poll( -+ stream.as_raw_fd(), -+ libc::POLLIN | libc::POLLHUP | libc::POLLERR, -+ ) { -+ if (revents & (libc::POLLIN | libc::POLLHUP | libc::POLLERR)) != 0 { -+ return Poll::Ready(Ok(1)); -+ } -+ } -+ -+ Poll::Pending -+ } -+ - fn poll_write_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { - let (state, selector, stream, _) = self.split_borrow(); - let map = state_as_waker_map(state, selector, stream).map_err(io_err_into_net_error)?; -@@ -788,6 +860,46 @@ fn libc_poll(fd: RawFd, events: libc::c_short) -> Option { - } - } - -+#[cfg(not(target_os = "windows"))] -+fn prime_socket_readiness(handler_guard: &mut HandlerGuardState, fd: RawFd) { -+ let Some(revents) = libc_poll( -+ fd, -+ libc::POLLIN | libc::POLLOUT | libc::POLLHUP | libc::POLLERR, -+ ) else { -+ return; -+ }; -+ -+ if (revents & libc::POLLIN) != 0 { -+ handler_guard.push_interest(InterestType::Readable); -+ } -+ if (revents & libc::POLLOUT) != 0 { -+ handler_guard.push_interest(InterestType::Writable); -+ } -+ if (revents & libc::POLLHUP) != 0 { -+ handler_guard.push_interest(InterestType::Closed); -+ } -+ if (revents & libc::POLLERR) != 0 { -+ handler_guard.push_interest(InterestType::Error); -+ } -+} -+ -+#[cfg(not(target_os = "windows"))] -+fn prime_socket_read_readiness(handler_guard: &mut HandlerGuardState, fd: RawFd) { -+ let Some(revents) = libc_poll(fd, libc::POLLIN | libc::POLLHUP | libc::POLLERR) else { -+ return; -+ }; -+ -+ if (revents & libc::POLLIN) != 0 { -+ handler_guard.push_interest(InterestType::Readable); -+ } -+ if (revents & libc::POLLHUP) != 0 { -+ handler_guard.push_interest(InterestType::Closed); -+ } -+ if (revents & libc::POLLERR) != 0 { -+ handler_guard.push_interest(InterestType::Error); -+ } -+} -+ - #[derive(Debug)] - pub struct LocalUdpSocket { - socket: mio::net::UdpSocket, -@@ -910,9 +1022,8 @@ impl VirtualConnectionlessSocket for LocalUdpSocket { - .map_err(io_err_into_net_error); - match &ret { - Ok(0) | Err(NetworkError::WouldBlock) => { -- if let HandlerGuardState::WakerMap(_, map) = &mut self.handler_guard { -- map.pop(InterestType::Writable); -- } -+ self.handler_guard -+ .pop_waker_interest(InterestType::Writable); - } - _ => {} - } -@@ -925,12 +1036,30 @@ impl VirtualConnectionlessSocket for LocalUdpSocket { - peek: bool, - ) -> Result<(usize, SocketAddr)> { - let buf: &mut [u8] = unsafe { std::mem::transmute(buf) }; -- if peek { -+ let ret = if peek { - self.socket.peek_from(buf) - } else { - self.socket.recv_from(buf) - } -- .map_err(io_err_into_net_error) -+ .map_err(io_err_into_net_error); -+ -+ match &ret { -+ Ok(_) if peek => self.handler_guard.push_interest(InterestType::Readable), -+ Ok(_) => { -+ #[cfg(not(target_os = "windows"))] -+ prime_socket_read_readiness(&mut self.handler_guard, self.socket.as_raw_fd()); -+ } -+ Err(NetworkError::WouldBlock) => { -+ self.handler_guard -+ .pop_waker_interest(InterestType::Readable); -+ } -+ Err(NetworkError::ConnectionAborted) | Err(NetworkError::ConnectionReset) => { -+ self.handler_guard.push_interest(InterestType::Closed); -+ } -+ Err(_) => self.handler_guard.push_interest(InterestType::Error), -+ } -+ -+ ret - } - } - -@@ -951,31 +1080,17 @@ impl VirtualSocket for LocalUdpSocket { - Ok(SocketStatus::Opened) - } - -- fn set_handler(&mut self, mut handler: Box) -> Result<()> { -- if let HandlerGuardState::ExternalHandler(guard) = &mut self.handler_guard { -- match guard.replace_handler(handler) { -- Ok(()) => { -- return Ok(()); -- } -- Err(h) => handler = h, -- } -- -- // the handler could not be replaced so we need to build a new handler instead -- if let Err(err) = guard.unregister(&mut self.socket) { -- tracing::debug!("failed to unregister previous token - {}", err); -- } -- } -- -- let guard = InterestGuard::new( -- &self.selector, -- handler, -- &mut self.socket, -- mio::Interest::READABLE.add(mio::Interest::WRITABLE), -- ) -- .map_err(io_err_into_net_error)?; -- -- self.handler_guard = HandlerGuardState::ExternalHandler(guard); -- -+ fn set_handler(&mut self, handler: Box) -> Result<()> { -+ self.handler_guard -+ .set_external_handler( -+ &self.selector, -+ &mut self.socket, -+ mio::Interest::READABLE.add(mio::Interest::WRITABLE), -+ handler, -+ ) -+ .map_err(io_err_into_net_error)?; -+ #[cfg(not(target_os = "windows"))] -+ prime_socket_readiness(&mut self.handler_guard, self.socket.as_raw_fd()); - Ok(()) - } - } -@@ -994,17 +1109,9 @@ impl LocalUdpSocket { - - impl VirtualIoSource for LocalUdpSocket { - fn remove_handler(&mut self) { -- let mut guard = HandlerGuardState::None; -- std::mem::swap(&mut guard, &mut self.handler_guard); -- match guard { -- HandlerGuardState::ExternalHandler(mut guard) => { -- guard.unregister(&mut self.socket).ok(); -- } -- HandlerGuardState::WakerMap(mut guard, _) => { -- guard.unregister(&mut self.socket).ok(); -- } -- HandlerGuardState::None => {} -- } -+ self.handler_guard -+ .remove_external_handler(&mut self.socket) -+ .ok(); - } - - fn poll_read_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { -diff --git a/lib/virtual-net/src/lib.rs b/lib/virtual-net/src/lib.rs -index 22ae000..58cbbf4 100644 ---- a/lib/virtual-net/src/lib.rs -+++ b/lib/virtual-net/src/lib.rs -@@ -79,6 +79,11 @@ pub trait VirtualIoSource: fmt::Debug + Send + Sync + 'static { - /// Polls the source to see if there is data waiting - fn poll_read_ready(&mut self, cx: &mut Context<'_>) -> Poll>; - -+ /// Polls read readiness without eagerly buffering data when a backend can support it. -+ fn poll_read_ready_direct(&mut self, cx: &mut Context<'_>) -> Poll> { -+ self.poll_read_ready(cx) -+ } -+ - /// Polls the source to see if data can be sent - fn poll_write_ready(&mut self, cx: &mut Context<'_>) -> Poll>; - } -@@ -183,6 +188,19 @@ pub trait VirtualNetworking: fmt::Debug + Send + Sync + 'static { - Err(NetworkError::Unsupported) - } - -+ /// Listens for TCP connections and preserves the guest-requested accept -+ /// backlog when the backend can apply it. -+ async fn listen_tcp_with_backlog( -+ &self, -+ addr: SocketAddr, -+ only_v6: bool, -+ reuse_port: bool, -+ reuse_addr: bool, -+ backlog: usize, -+ ) -> Result> { -+ self.listen_tcp(addr, only_v6, reuse_port, reuse_addr).await -+ } -+ - /// Opens a UDP socket that listens on a specific IP and Port combination - /// Multiple servers (processes or threads) can bind to the same port if they each set - /// the reuse-port and-or reuse-addr flags -diff --git a/lib/virtual-net/src/tests.rs b/lib/virtual-net/src/tests.rs -index 9ed1a46..f1ce408 100644 ---- a/lib/virtual-net/src/tests.rs -+++ b/lib/virtual-net/src/tests.rs -@@ -1,7 +1,10 @@ - #![allow(unused)] - use std::{ - net::{Ipv4Addr, SocketAddrV4}, -- sync::atomic::{AtomicU16, Ordering}, -+ sync::{ -+ atomic::{AtomicU16, Ordering}, -+ mpsc, -+ }, - }; - - use tracing_test::traced_test; -@@ -13,6 +16,7 @@ use crate::{ - meta::FrameSerializationFormat, - }; - use tokio::io::{AsyncReadExt, AsyncWriteExt}; -+use virtual_mio::{InterestHandler, InterestType}; - - use super::*; - -@@ -513,6 +517,257 @@ async fn test_connect_tcp_returns_immediately_for_in_progress_connect() { - } - } - -+#[cfg(not(target_os = "windows"))] -+#[traced_test] -+#[tokio::test] -+#[serial_test::serial] -+async fn test_local_tcp_listener_reports_readable_while_accept_backlog_remains() { -+ use std::net::TcpStream; -+ use std::time::Duration; -+ -+ #[derive(Debug)] -+ struct ReadableHandler { -+ sender: mpsc::Sender, -+ } -+ -+ impl InterestHandler for ReadableHandler { -+ fn push_interest(&mut self, interest: InterestType) { -+ self.sender.send(interest).ok(); -+ } -+ -+ fn pop_interest(&mut self, _interest: InterestType) -> bool { -+ false -+ } -+ -+ fn has_interest(&self, _interest: InterestType) -> bool { -+ false -+ } -+ } -+ -+ let networking = LocalNetworking::new(); -+ let mut listener = networking -+ .listen_tcp( -+ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), -+ false, -+ false, -+ false, -+ ) -+ .await -+ .unwrap(); -+ let addr = listener.addr_local().unwrap(); -+ -+ let (sender, receiver) = mpsc::channel(); -+ listener -+ .set_handler(Box::new(ReadableHandler { sender })) -+ .unwrap(); -+ -+ let clients: Vec<_> = (0..20).map(|_| TcpStream::connect(addr).unwrap()).collect(); -+ receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("listener should report first readable accept"); -+ -+ let mut accepted = Vec::new(); -+ for idx in 0..clients.len() { -+ accepted.push(listener.try_accept().unwrap()); -+ if idx + 1 < clients.len() { -+ receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("listener should keep reporting readable across a queued burst"); -+ } -+ } -+ drop(accepted); -+ drop(clients); -+} -+ -+#[cfg(not(target_os = "windows"))] -+#[traced_test] -+#[tokio::test] -+#[serial_test::serial] -+async fn test_local_tcp_listener_reports_existing_accept_readiness_on_handler_attach() { -+ use std::net::TcpStream; -+ use std::time::Duration; -+ -+ #[derive(Debug)] -+ struct ReadableHandler { -+ sender: mpsc::Sender, -+ } -+ -+ impl InterestHandler for ReadableHandler { -+ fn push_interest(&mut self, interest: InterestType) { -+ self.sender.send(interest).ok(); -+ } -+ -+ fn pop_interest(&mut self, _interest: InterestType) -> bool { -+ false -+ } -+ -+ fn has_interest(&self, _interest: InterestType) -> bool { -+ false -+ } -+ } -+ -+ let networking = LocalNetworking::new(); -+ let mut listener = networking -+ .listen_tcp( -+ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), -+ false, -+ false, -+ false, -+ ) -+ .await -+ .unwrap(); -+ let addr = listener.addr_local().unwrap(); -+ let client = TcpStream::connect(addr).unwrap(); -+ -+ let (sender, receiver) = mpsc::channel(); -+ listener -+ .set_handler(Box::new(ReadableHandler { sender })) -+ .unwrap(); -+ assert_eq!( -+ receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("listener should report queued accept readiness"), -+ InterestType::Readable -+ ); -+ drop(client); -+} -+ -+#[cfg(not(target_os = "windows"))] -+#[traced_test] -+#[tokio::test] -+#[serial_test::serial] -+async fn test_local_tcp_stream_reports_existing_readiness_on_handler_attach() { -+ use std::io::Write; -+ use std::net::TcpStream; -+ use std::time::Duration; -+ -+ #[derive(Debug)] -+ struct ReadableHandler { -+ sender: mpsc::Sender, -+ } -+ -+ impl InterestHandler for ReadableHandler { -+ fn push_interest(&mut self, interest: InterestType) { -+ self.sender.send(interest).ok(); -+ } -+ -+ fn pop_interest(&mut self, _interest: InterestType) -> bool { -+ false -+ } -+ -+ fn has_interest(&self, _interest: InterestType) -> bool { -+ false -+ } -+ } -+ -+ let networking = LocalNetworking::new(); -+ let mut listener = networking -+ .listen_tcp( -+ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), -+ false, -+ false, -+ false, -+ ) -+ .await -+ .unwrap(); -+ let addr = listener.addr_local().unwrap(); -+ let mut client = TcpStream::connect(addr).unwrap(); -+ let (mut server, _) = listener.accept().await.unwrap(); -+ -+ client.write_all(b"ready-before-handler").unwrap(); -+ std::thread::sleep(Duration::from_millis(50)); -+ -+ let (sender, receiver) = mpsc::channel(); -+ server -+ .set_handler(Box::new(ReadableHandler { sender })) -+ .unwrap(); -+ let mut saw_readable = false; -+ for _ in 0..2 { -+ if receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("stream should report buffered host readiness") -+ == InterestType::Readable -+ { -+ saw_readable = true; -+ break; -+ } -+ } -+ assert!( -+ saw_readable, -+ "stream did not report pre-existing readable data" -+ ); -+} -+ -+#[cfg(not(target_os = "windows"))] -+#[traced_test] -+#[tokio::test] -+#[serial_test::serial] -+async fn test_local_tcp_stream_reissues_readable_after_partial_recv() { -+ use std::io::Write; -+ use std::net::TcpStream; -+ use std::time::Duration; -+ -+ #[derive(Debug)] -+ struct ReadableHandler { -+ sender: mpsc::Sender, -+ } -+ -+ impl InterestHandler for ReadableHandler { -+ fn push_interest(&mut self, interest: InterestType) { -+ self.sender.send(interest).ok(); -+ } -+ -+ fn pop_interest(&mut self, _interest: InterestType) -> bool { -+ false -+ } -+ -+ fn has_interest(&self, _interest: InterestType) -> bool { -+ false -+ } -+ } -+ -+ let networking = LocalNetworking::new(); -+ let mut listener = networking -+ .listen_tcp( -+ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), -+ false, -+ false, -+ false, -+ ) -+ .await -+ .unwrap(); -+ let addr = listener.addr_local().unwrap(); -+ let mut client = TcpStream::connect(addr).unwrap(); -+ let (mut server, _) = listener.accept().await.unwrap(); -+ -+ let (sender, receiver) = mpsc::channel(); -+ server -+ .set_handler(Box::new(ReadableHandler { sender })) -+ .unwrap(); -+ -+ client.write_all(&vec![7_u8; 64 * 1024]).unwrap(); -+ loop { -+ if receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("stream should report readable data") -+ == InterestType::Readable -+ { -+ break; -+ } -+ } -+ std::thread::sleep(Duration::from_millis(50)); -+ while receiver.try_recv().is_ok() {} -+ -+ let mut buf = [MaybeUninit::::uninit(); 8]; -+ assert_eq!(server.try_recv(&mut buf, false).unwrap(), 8); -+ assert_eq!( -+ receiver -+ .recv_timeout(Duration::from_secs(1)) -+ .expect("partial recv should leave the stream readable"), -+ InterestType::Readable -+ ); -+} -+ - #[cfg(not(target_os = "windows"))] - #[traced_test] - #[tokio::test] -diff --git a/lib/vm/src/export.rs b/lib/vm/src/export.rs -index 5d47e42..6cf092b 100644 ---- a/lib/vm/src/export.rs -+++ b/lib/vm/src/export.rs -@@ -11,6 +11,7 @@ use std::any::Any; - use wasmer_types::{FunctionType, TagKind}; - - /// The value of an export passed from one instance to another. -+#[derive(Clone, Copy, PartialEq, Eq)] - #[cfg_attr(feature = "artifact-size", derive(loupe::MemoryUsage))] - pub enum VMExtern { - /// A function export value. -diff --git a/lib/vm/src/instance/allocator.rs b/lib/vm/src/instance/allocator.rs -index c81bfd1..f040a4c 100644 ---- a/lib/vm/src/instance/allocator.rs -+++ b/lib/vm/src/instance/allocator.rs -@@ -78,6 +78,42 @@ impl InstanceAllocator { - Vec>, - ) { - let offsets = VMOffsets::new(mem::size_of::() as u8, module); -+ // SAFETY: the offsets were derived immediately above from this exact -+ // module and the host pointer width. -+ unsafe { Self::new_with_offsets(offsets, module) } -+ } -+ -+ /// Allocates instance data using precomputed host-layout offsets. -+ /// -+ /// This is the cached-offset counterpart of [`InstanceAllocator::new`]. -+ /// It avoids rescanning an immutable module when many instances are -+ /// created from the same compiled artifact. -+ /// -+ /// # Safety -+ /// -+ /// `offsets` must equal -+ /// `VMOffsets::new(size_of::() as u8, module)` for this exact -+ /// `module`. Supplying offsets from another module or pointer width can -+ /// under-allocate the dynamically sized VMContext and lead to invalid -+ /// pointer writes while the instance is initialized. -+ #[allow(clippy::type_complexity)] -+ pub unsafe fn new_with_offsets( -+ offsets: VMOffsets, -+ module: &ModuleInfo, -+ ) -> ( -+ Self, -+ Vec>, -+ Vec>, -+ Vec>, -+ ) { -+ debug_assert_eq!( -+ format!("{offsets:?}"), -+ format!( -+ "{:?}", -+ VMOffsets::new(mem::size_of::() as u8, module) -+ ), -+ "precomputed VMOffsets do not match the supplied module" -+ ); - let instance_layout = Self::instance_layout(&offsets); - - #[allow(clippy::cast_ptr_alignment)] -@@ -257,3 +293,34 @@ impl InstanceAllocator { - &self.offsets - } - } -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ -+ #[test] -+ fn vmoffsets_are_deterministic_for_a_module() { -+ let module = ModuleInfo::default(); -+ let pointer_size = mem::size_of::() as u8; -+ -+ assert_eq!( -+ format!("{:?}", VMOffsets::new(pointer_size, &module)), -+ format!("{:?}", VMOffsets::new(pointer_size, &module)) -+ ); -+ } -+ -+ #[test] -+ fn cached_offsets_produce_the_same_allocator_layout() { -+ let module = ModuleInfo::default(); -+ let (computed, _, _, _) = InstanceAllocator::new(&module); -+ let offsets = VMOffsets::new(mem::size_of::() as u8, &module); -+ // SAFETY: `offsets` was computed from this exact module and host. -+ let (cached, _, _, _) = unsafe { InstanceAllocator::new_with_offsets(offsets, &module) }; -+ -+ assert_eq!(computed.instance_layout, cached.instance_layout); -+ assert_eq!( -+ computed.offsets.size_of_vmctx(), -+ cached.offsets.size_of_vmctx() -+ ); -+ } -+} -diff --git a/lib/vm/src/instance/mod.rs b/lib/vm/src/instance/mod.rs -index 3741bbb..7a16b05 100644 ---- a/lib/vm/src/instance/mod.rs -+++ b/lib/vm/src/instance/mod.rs -@@ -74,18 +74,32 @@ pub(crate) struct Instance { - tags: BoxedSlice>, - - /// Pointers to functions in executable memory. -- functions: BoxedSlice, -+ /// -+ /// These pointers are immutable after the artifact is linked, so every -+ /// instance of the same module shares the typed pointer table. -+ functions: Arc>, - - /// Pointers to function call trampolines in executable memory. -- function_call_trampolines: BoxedSlice, -+ function_call_trampolines: Arc>, - - /// Passive elements in this instantiation. As `elem.drop`s happen, these - /// entries get removed. - passive_elements: RefCell]>>>, - -- /// Passive data segments from our module. As `data.drop`s happen, entries -- /// get removed. A missing entry is considered equivalent to an empty slice. -- passive_data: RefCell>>, -+ /// Passive data segments dropped by this instance. -+ /// -+ /// The immutable segment bytes remain owned by the shared `ModuleInfo`; -+ /// instances only need to record the segments whose logical length has -+ /// become zero after `data.drop`. -+ dropped_passive_data: RefCell>, -+ -+ /// Canonical extern handles for deferred export materialization. -+ /// -+ /// This is deliberately disabled for ordinary eager instances: populating -+ /// both their public export map and a second full cache would only increase -+ /// memory use. Deferred instances opt in, and the cache is keyed by the -+ /// underlying declaration so aliases share one local `VMFunction` handle. -+ deferred_export_cache: Option>, - - /// Mapping of function indices to their func ref backing data. `VMFuncRef`s - /// will point to elements here for functions defined by this instance. -@@ -240,7 +254,7 @@ impl Instance { - return; - }; - unsafe { -- *base.as_ptr().add(index as usize) = anyfunc_from_funcref(funcref); -+ *base.as_ptr().add(index as usize) = VMCallerCheckedAnyfunc::from_funcref(funcref); - } - } - -@@ -254,7 +268,7 @@ impl Instance { - unreachable!("fixed funcref tables cannot contain externrefs"); - }; - unsafe { -- *base.as_ptr().add(index as usize) = anyfunc_from_funcref(funcref); -+ *base.as_ptr().add(index as usize) = VMCallerCheckedAnyfunc::from_funcref(funcref); - } - } - } -@@ -859,8 +873,19 @@ impl Instance { - // https://webassembly.github.io/bulk-memory-operations/core/exec/instructions.html#exec-memory-init - - let memory = self.get_vmmemory(memory_index); -- let passive_data = self.passive_data.borrow(); -- let data = passive_data.get(&data_index).map_or(&[][..], |d| &**d); -+ let was_dropped = self -+ .dropped_passive_data -+ .borrow() -+ .binary_search(&data_index) -+ .is_ok(); -+ let data = if was_dropped { -+ &[][..] -+ } else { -+ self.module -+ .passive_data -+ .get(&data_index) -+ .map_or(&[][..], |data| &**data) -+ }; - - let current_length = unsafe { memory.vmmemory().as_ref().current_length }; - if src.checked_add(len).is_none_or(|n| n as usize > data.len()) -@@ -876,8 +901,12 @@ impl Instance { - - /// Drop the given data segment, truncating its length to zero. - pub(crate) fn data_drop(&self, data_index: DataIndex) { -- let mut passive_data = self.passive_data.borrow_mut(); -- passive_data.remove(&data_index); -+ if self.module.passive_data.contains_key(&data_index) { -+ let mut dropped = self.dropped_passive_data.borrow_mut(); -+ if let Err(position) = dropped.binary_search(&data_index) { -+ dropped.insert(position, data_index); -+ } -+ } - } - - /// Get a table by index regardless of whether it is locally-defined or an -@@ -1140,8 +1169,8 @@ impl VMInstance { - allocator: InstanceAllocator, - module: Arc, - context: &mut StoreObjects, -- finished_functions: BoxedSlice, -- finished_function_call_trampolines: BoxedSlice, -+ finished_functions: Arc>, -+ finished_function_call_trampolines: Arc>, - finished_memories: BoxedSlice>, - finished_tables: BoxedSlice>, - finished_globals: BoxedSlice>, -@@ -1155,15 +1184,6 @@ impl VMInstance { - .map(|m: &InternalStoreHandle| VMSharedTagIndex::new(m.index() as u32)) - .collect::>() - .into_boxed_slice(); -- let passive_data = RefCell::new( -- module -- .passive_data -- .clone() -- .into_iter() -- .map(|(idx, bytes)| (idx, Arc::from(bytes))) -- .collect::>(), -- ); -- - let handle = { - let offsets = allocator.offsets().clone(); - // use dummy value to create an instance so we can get the vmctx pointer -@@ -1181,7 +1201,8 @@ impl VMInstance { - functions: finished_functions, - function_call_trampolines: finished_function_call_trampolines, - passive_elements: Default::default(), -- passive_data, -+ dropped_passive_data: Default::default(), -+ deferred_export_cache: None, - funcrefs, - imported_funcrefs, - vmctx: VMContext {}, -@@ -1243,7 +1264,6 @@ impl VMInstance { - instance.builtin_functions_ptr(), - VMBuiltinFunctionsArray::initialized(), - ); -- - // Perform infallible initialization in this constructor, while fallible - // initialization is deferred to the `initialize` method. - initialize_passive_elements(instance); -@@ -1322,9 +1342,19 @@ impl VMInstance { - - /// Lookup an export with the given export declaration. - pub fn lookup_by_declaration(&mut self, export: ExportIndex) -> VMExtern { -+ if let Some(cached) = self -+ .instance() -+ .deferred_export_cache -+ .as_ref() -+ .and_then(|cache| cache.get(&export)) -+ .copied() -+ { -+ return cached; -+ } -+ - let instance = self.instance(); - -- match export { -+ let extern_ = match export { - ExportIndex::Function(index) => { - let sig_index = &instance.module.functions[index]; - let handle = if let Some(def_index) = instance.module.local_func_index(index) { -@@ -1383,6 +1413,22 @@ impl VMInstance { - let handle = instance.tags[index]; - VMExtern::Tag(handle) - } -+ }; -+ -+ if let Some(cache) = self.instance_mut().deferred_export_cache.as_mut() { -+ cache.insert(export, extern_); -+ } -+ -+ extern_ -+ } -+ -+ /// Enables identity-preserving deferred export lookup for this instance. -+ /// -+ /// Ordinary eager instances intentionally leave this disabled so they do -+ /// not retain a second copy of their complete export set. -+ pub fn enable_deferred_export_cache(&mut self) { -+ if self.instance().deferred_export_cache.is_none() { -+ self.instance_mut().deferred_export_cache = Some(HashMap::new()); - } - } - -@@ -1714,13 +1760,6 @@ fn initialize_globals(instance: &Instance) { - } - } - --fn anyfunc_from_funcref(funcref: Option) -> VMCallerCheckedAnyfunc { -- match funcref { -- Some(funcref) => unsafe { *funcref.0.as_ptr() }, -- None => VMCallerCheckedAnyfunc::null(), -- } --} -- - /// Eagerly builds all the `VMFuncRef`s for imported and local functions so that all - /// future funcref operations are just looking up this data. - fn build_funcrefs( -@@ -1763,3 +1802,161 @@ fn build_funcrefs( - imported_func_refs.into_boxed_slice(), - ) - } -+ -+#[cfg(test)] -+mod tests { -+ use super::{InstanceAllocator, VMInstance}; -+ use crate::{ -+ FunctionBodyPtr, Imports, StoreObjects, VMContext, VMFunctionBody, VMSignatureHash, -+ VMTrampoline, -+ }; -+ use std::{ptr::NonNull, sync::Arc}; -+ use wasmer_types::{ -+ FunctionType, LocalFunctionIndex, ModuleInfo, RawValue, SignatureIndex, TagIndex, -+ entity::{BoxedSlice, EntityRef, PrimaryMap}, -+ }; -+ -+ unsafe extern "C" fn test_trampoline( -+ _vmctx: *mut VMContext, -+ _callee: *const VMFunctionBody, -+ _values: *mut RawValue, -+ ) { -+ } -+ -+ fn empty_boxed_slice() -> BoxedSlice -+ where -+ K: wasmer_types::entity::EntityRef, -+ { -+ PrimaryMap::::new().into_boxed_slice() -+ } -+ -+ unsafe fn new_test_instance( -+ module: Arc, -+ context: &mut StoreObjects, -+ functions: Arc>, -+ trampolines: Arc>, -+ ) -> VMInstance { -+ let (allocator, memories, tables, globals) = InstanceAllocator::new(&module); -+ assert!(memories.is_empty()); -+ assert!(tables.is_empty()); -+ assert!(globals.is_empty()); -+ -+ unsafe { -+ VMInstance::new( -+ allocator, -+ module, -+ context, -+ functions, -+ trampolines, -+ empty_boxed_slice(), -+ empty_boxed_slice(), -+ empty_boxed_slice(), -+ empty_boxed_slice::(), -+ Imports::none(), -+ PrimaryMap::from_iter([VMSignatureHash::new(7)]).into_boxed_slice(), -+ ) -+ .unwrap() -+ } -+ } -+ -+ fn test_module_and_function_tables() -> ( -+ Arc, -+ FunctionBodyPtr, -+ Arc>, -+ Arc>, -+ ) { -+ let mut module = ModuleInfo::new(); -+ let signature = module.signatures.push(FunctionType::new([], [])); -+ module.functions.push(signature); -+ let module = Arc::new(module); -+ -+ let function = FunctionBodyPtr(NonNull::::dangling().as_ptr()); -+ let functions = Arc::new(PrimaryMap::from_iter([function]).into_boxed_slice()); -+ let trampolines = -+ Arc::new(PrimaryMap::from_iter([test_trampoline as VMTrampoline]).into_boxed_slice()); -+ -+ (module, function, functions, trampolines) -+ } -+ -+ #[test] -+ fn immutable_function_tables_are_shared_by_two_instances() { -+ let (module, _function, functions, trampolines) = test_module_and_function_tables(); -+ -+ let mut context = StoreObjects::default(); -+ let first = unsafe { -+ new_test_instance( -+ Arc::clone(&module), -+ &mut context, -+ Arc::clone(&functions), -+ Arc::clone(&trampolines), -+ ) -+ }; -+ let second = unsafe { -+ new_test_instance( -+ module, -+ &mut context, -+ Arc::clone(&functions), -+ Arc::clone(&trampolines), -+ ) -+ }; -+ -+ assert!(Arc::ptr_eq( -+ &first.instance().functions, -+ &second.instance().functions -+ )); -+ assert!(Arc::ptr_eq( -+ &first.instance().function_call_trampolines, -+ &second.instance().function_call_trampolines -+ )); -+ assert_eq!(Arc::strong_count(&functions), 3); -+ assert_eq!(Arc::strong_count(&trampolines), 3); -+ } -+ -+ #[test] -+ fn shared_function_tables_outlive_artifact_owner_and_peer_instance() { -+ let (module, function, functions, trampolines) = test_module_and_function_tables(); -+ let function_allocation = Arc::as_ptr(&functions); -+ let trampoline_allocation = Arc::as_ptr(&trampolines); -+ -+ let mut context = StoreObjects::default(); -+ let first = unsafe { -+ new_test_instance( -+ Arc::clone(&module), -+ &mut context, -+ Arc::clone(&functions), -+ Arc::clone(&trampolines), -+ ) -+ }; -+ let second = unsafe { -+ new_test_instance( -+ module, -+ &mut context, -+ Arc::clone(&functions), -+ Arc::clone(&trampolines), -+ ) -+ }; -+ -+ // Model dropping the artifact/module's owners and then one instance. -+ drop(functions); -+ drop(trampolines); -+ drop(first); -+ -+ assert_eq!( -+ Arc::as_ptr(&second.instance().functions), -+ function_allocation -+ ); -+ assert_eq!( -+ Arc::as_ptr(&second.instance().function_call_trampolines), -+ trampoline_allocation -+ ); -+ assert_eq!(Arc::strong_count(&second.instance().functions), 1); -+ assert_eq!( -+ second.instance().functions[LocalFunctionIndex::new(0)].0, -+ function.0 -+ ); -+ assert!(std::ptr::fn_addr_eq( -+ second.instance().function_call_trampolines[SignatureIndex::new(0)], -+ test_trampoline as VMTrampoline -+ )); -+ } -+} -diff --git a/lib/vm/src/libcalls.rs b/lib/vm/src/libcalls.rs -index 7c1e0db..532838c 100644 ---- a/lib/vm/src/libcalls.rs -+++ b/lib/vm/src/libcalls.rs -@@ -224,6 +224,49 @@ pub unsafe extern "C" fn wasmer_vm_imported_memory32_size( - } - } - -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn wasmer_vm_host_bzero(dst: *mut c_void, len: usize) { -+ unsafe { -+ std::ptr::write_bytes(dst, 0, len); -+ } -+} -+ -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn wasmer_vm_host_memset( -+ dst: *mut c_void, -+ value: i32, -+ len: usize, -+) -> *mut c_void { -+ unsafe { -+ std::ptr::write_bytes(dst, value as u8, len); -+ } -+ dst -+} -+ -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn wasmer_vm_host_memcpy( -+ dst: *mut c_void, -+ src: *const c_void, -+ len: usize, -+) -> *mut c_void { -+ unsafe { -+ std::ptr::copy_nonoverlapping(src.cast::(), dst.cast::(), len); -+ } -+ dst -+} -+ -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn wasmer_vm_host_memmove( -+ dst: *mut c_void, -+ src: *const c_void, -+ len: usize, -+) -> *mut c_void { -+ unsafe { -+ std::ptr::copy(src.cast::(), dst.cast::(), len); -+ } -+ dst -+} -+ - /// Implementation of `table.copy`. - /// - /// # Safety -@@ -983,6 +1026,10 @@ pub fn function_pointer(libcall: LibCall) -> usize { - LibCall::Memory32Init => wasmer_vm_memory32_init as *const () as usize, - LibCall::DataDrop => wasmer_vm_data_drop as *const () as usize, - LibCall::Probestack => WASMER_VM_PROBESTACK as *const () as usize, -+ LibCall::HostBzero => wasmer_vm_host_bzero as *const () as usize, -+ LibCall::HostMemset => wasmer_vm_host_memset as *const () as usize, -+ LibCall::HostMemcpy => wasmer_vm_host_memcpy as *const () as usize, -+ LibCall::HostMemmove => wasmer_vm_host_memmove as *const () as usize, - LibCall::RaiseTrap => wasmer_vm_raise_trap as *const () as usize, - LibCall::Memory32AtomicWait32 => wasmer_vm_memory32_atomic_wait32 as *const () as usize, - LibCall::ImportedMemory32AtomicWait32 => { -diff --git a/lib/vm/src/memory.rs b/lib/vm/src/memory.rs -index e7bc857..f2e730c 100644 ---- a/lib/vm/src/memory.rs -+++ b/lib/vm/src/memory.rs -@@ -33,9 +33,25 @@ struct WasmMmap { - size: Pages, - /// The owned memory definition used by the generated code - vm_memory_definition: MaybeInstanceOwned, -+ /// True only when the allocation reserves every page the memory can ever -+ /// grow to. Fixed host mappings are invalidated if the linear-memory base -+ /// relocates, so shared remaps must fail closed without this invariant. -+ stable_base: bool, - } - - impl WasmMmap { -+ fn supports_persistent_shared_fixed_remap(&self) -> bool { -+ #[cfg(not(target_os = "windows"))] -+ { -+ self.stable_base -+ } -+ -+ #[cfg(target_os = "windows")] -+ { -+ false -+ } -+ } -+ - fn get_vm_memory_definition(&self) -> NonNull { - self.vm_memory_definition.as_ptr() - } -@@ -152,8 +168,75 @@ impl WasmMmap { - Ok(()) - } - -- /// Copies the memory -- /// (in this case it performs a copy-on-write to save memory) -+ unsafe fn remap_shared_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ if !self.supports_persistent_shared_fixed_remap() { -+ return Err(MemoryError::UnsupportedOperation { -+ message: "shared fixed remaps require a supported nonmoving static linear memory" -+ .to_string(), -+ }); -+ } -+ let current_len = self.size.bytes().0; -+ if len > current_len || start > current_len - len { -+ return Err(MemoryError::InvalidMemory { -+ reason: "fixed mapping range is outside the current memory".to_string(), -+ }); -+ } -+ unsafe { -+ self.alloc -+ .remap_shared_file_fixed(start, len, file, file_offset) -+ } -+ .map_err(MemoryError::Region) -+ } -+ -+ unsafe fn remap_private_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ let current_len = self.size.bytes().0; -+ if len > current_len || start > current_len - len { -+ return Err(MemoryError::InvalidMemory { -+ reason: "fixed mapping range is outside the current memory".to_string(), -+ }); -+ } -+ unsafe { -+ self.alloc -+ .remap_private_file_fixed(start, len, file, file_offset) -+ } -+ .map_err(MemoryError::Region) -+ } -+ -+ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { -+ let current_len = self.size.bytes().0; -+ if len > current_len || start > current_len - len { -+ return Err(MemoryError::InvalidMemory { -+ reason: "fixed mapping range is outside the current memory".to_string(), -+ }); -+ } -+ unsafe { self.alloc.remap_private_fixed(start, len) }.map_err(MemoryError::Region) -+ } -+ -+ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { -+ let current_len = self.size.bytes().0; -+ if len > current_len || start > current_len - len { -+ return Err(MemoryError::InvalidMemory { -+ reason: "sync range is outside the current memory".to_string(), -+ }); -+ } -+ self.alloc -+ .msync(start, len, flags) -+ .map_err(MemoryError::Region) -+ } -+ -+ /// Copies the memory. - pub fn copy(&mut self) -> Result { - let mem_length = self.size.bytes().0; - let mut alloc = self -@@ -170,6 +253,7 @@ impl WasmMmap { - ))), - alloc, - size: self.size, -+ stable_base: self.stable_base, - }) - } - } -@@ -361,6 +445,11 @@ impl VMOwnedMemory { - }, - alloc, - size: Bytes::from(mem_length).try_into().unwrap(), -+ stable_base: matches!( -+ style, -+ MemoryStyle::Static { bound, .. } -+ if *bound >= memory.maximum.unwrap_or(Pages::max_value()) -+ ), - }; - - Ok(Self { -@@ -431,6 +520,44 @@ impl LinearMemory for VMOwnedMemory { - Ok(()) - } - -+ fn supports_persistent_shared_fixed_remap(&self) -> bool { -+ self.mmap.supports_persistent_shared_fixed_remap() -+ } -+ -+ unsafe fn remap_shared_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.mmap -+ .remap_shared_file_fixed(start, len, file, file_offset) -+ } -+ } -+ -+ unsafe fn remap_private_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.mmap -+ .remap_private_file_fixed(start, len, file, file_offset) -+ } -+ } -+ -+ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { -+ unsafe { self.mmap.remap_private_fixed(start, len) } -+ } -+ -+ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { -+ self.mmap.msync(start, len, flags) -+ } -+ - /// Return a `VMMemoryDefinition` for exposing the memory to compiled wasm code. - fn vmmemory(&self) -> NonNull { - self.mmap.vm_memory_definition.as_ptr() -@@ -586,6 +713,43 @@ impl LinearMemory for VMSharedMemory { - Ok(()) - } - -+ fn supports_persistent_shared_fixed_remap(&self) -> bool { -+ let guard = self.mmap.read().unwrap(); -+ guard.supports_persistent_shared_fixed_remap() -+ } -+ -+ unsafe fn remap_shared_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ let mut guard = self.mmap.write().unwrap(); -+ unsafe { guard.remap_shared_file_fixed(start, len, file, file_offset) } -+ } -+ -+ unsafe fn remap_private_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ let mut guard = self.mmap.write().unwrap(); -+ unsafe { guard.remap_private_file_fixed(start, len, file, file_offset) } -+ } -+ -+ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { -+ let mut guard = self.mmap.write().unwrap(); -+ unsafe { guard.remap_private_fixed(start, len) } -+ } -+ -+ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { -+ let guard = self.mmap.read().unwrap(); -+ guard.msync(start, len, flags) -+ } -+ - /// Return a `VMMemoryDefinition` for exposing the memory to compiled wasm code. - fn vmmemory(&self) -> NonNull { - let guard = self.mmap.read().unwrap(); -@@ -680,6 +844,44 @@ impl LinearMemory for VMMemory { - Ok(()) - } - -+ fn supports_persistent_shared_fixed_remap(&self) -> bool { -+ self.0.supports_persistent_shared_fixed_remap() -+ } -+ -+ unsafe fn remap_shared_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.0 -+ .remap_shared_file_fixed(start, len, file, file_offset) -+ } -+ } -+ -+ unsafe fn remap_private_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ unsafe { -+ self.0 -+ .remap_private_file_fixed(start, len, file, file_offset) -+ } -+ } -+ -+ unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), MemoryError> { -+ unsafe { self.0.remap_private_fixed(start, len) } -+ } -+ -+ fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), MemoryError> { -+ self.0.msync(start, len, flags) -+ } -+ - /// Returns the memory style for this memory. - fn style(&self) -> MemoryStyle { - self.0.style() -@@ -790,6 +992,43 @@ impl VMMemory { - } - } - -+#[cfg(test)] -+mod shared_remap_style_tests { -+ use super::*; -+ -+ #[test] -+ fn dynamic_memory_rejects_shared_fixed_remaps_before_host_mutation() { -+ let memory = MemoryType::new(Pages(1), Some(Pages(2)), false); -+ let style = MemoryStyle::Dynamic { -+ offset_guard_size: 0, -+ }; -+ let mut vm = VMOwnedMemory::new(&memory, &style).unwrap(); -+ assert!(!LinearMemory::supports_persistent_shared_fixed_remap(&vm)); -+ let file = std::fs::File::open(std::env::current_exe().unwrap()).unwrap(); -+ let error = unsafe { LinearMemory::remap_shared_file_fixed(&mut vm, 0, 4096, &file, 0) } -+ .expect_err("dynamic memory must not accept persistent host remaps"); -+ assert!(matches!(error, MemoryError::UnsupportedOperation { .. })); -+ assert!( -+ error -+ .to_string() -+ .contains("supported nonmoving static linear memory") -+ ); -+ } -+ -+ #[test] -+ #[cfg(not(target_os = "windows"))] -+ fn fully_reserved_static_memory_reports_persistent_remap_capability() { -+ let memory = MemoryType::new(Pages(1), Some(Pages(2)), false); -+ let style = MemoryStyle::Static { -+ bound: Pages(2), -+ offset_guard_size: 0, -+ }; -+ let vm = VMOwnedMemory::new(&memory, &style).unwrap(); -+ -+ assert!(LinearMemory::supports_persistent_shared_fixed_remap(&vm)); -+ } -+} -+ - #[doc(hidden)] - /// Default implementation to initialize memory with data - pub unsafe fn initialize_memory_with_data( -@@ -842,6 +1081,78 @@ where - }) - } - -+ /// Whether fixed shared-file remaps will remain valid for the lifetime of -+ /// this memory. Implementations must return false when later growth can -+ /// relocate the linear-memory base or the host cannot install such maps. -+ fn supports_persistent_shared_fixed_remap(&self) -> bool { -+ false -+ } -+ -+ /// Replace a page-aligned range inside this memory with a shared, -+ /// read-write file mapping at the same host address. -+ /// -+ /// Higher-level runtimes are responsible for translating guest mmap -+ /// requests, tracking mapping metadata, and preserving shared ranges -+ /// across process fork/copy operations. -+ /// -+ /// # Safety -+ /// No thread may access the replaced range during remapping, no Rust reference may point into -+ /// it. The backing inode must not shrink below its size validated at mapping time; only the -+ /// validated final partial page may extend beyond EOF. -+ unsafe fn remap_shared_file_fixed( -+ &mut self, -+ _start: usize, -+ _len: usize, -+ _file: &std::fs::File, -+ _file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ Err(MemoryError::UnsupportedOperation { -+ message: "remap_shared_file_fixed() is not supported".to_string(), -+ }) -+ } -+ -+ /// Replace a page-aligned range inside this memory with a private, -+ /// copy-on-write file mapping at the same host address. -+ /// -+ /// # Safety -+ /// No thread may access the replaced range during remapping, no Rust reference may point into -+ /// it, and the backing inode must not be truncated or replaced below `file_offset + len` while -+ /// accessible. -+ unsafe fn remap_private_file_fixed( -+ &mut self, -+ _start: usize, -+ _len: usize, -+ _file: &std::fs::File, -+ _file_offset: usize, -+ ) -> Result<(), MemoryError> { -+ Err(MemoryError::UnsupportedOperation { -+ message: "remap_private_file_fixed() is not supported".to_string(), -+ }) -+ } -+ -+ /// Replace a page-aligned range inside this memory with private, -+ /// zero-filled memory at the same host address. -+ /// -+ /// # Safety -+ /// No thread may access the replaced range during remapping and no Rust reference may point -+ /// into it. -+ unsafe fn remap_private_fixed( -+ &mut self, -+ _start: usize, -+ _len: usize, -+ ) -> Result<(), MemoryError> { -+ Err(MemoryError::UnsupportedOperation { -+ message: "remap_private_fixed() is not supported".to_string(), -+ }) -+ } -+ -+ /// Synchronize a page-aligned memory range with its backing file. -+ fn msync(&self, _start: usize, _len: usize, _flags: i32) -> Result<(), MemoryError> { -+ Err(MemoryError::UnsupportedOperation { -+ message: "msync() is not supported".to_string(), -+ }) -+ } -+ - /// Return a `VMMemoryDefinition` for exposing the memory to compiled wasm code. - fn vmmemory(&self) -> NonNull; - -diff --git a/lib/vm/src/mmap.rs b/lib/vm/src/mmap.rs -index 75a1852..a634ad2 100644 ---- a/lib/vm/src/mmap.rs -+++ b/lib/vm/src/mmap.rs -@@ -264,6 +264,278 @@ impl Mmap { - .map_err(|e| e.to_string()) - } - -+ /// Replace a page-aligned range inside this mapping with a shared, -+ /// read-write file mapping at the same host address. -+ /// -+ /// # Safety -+ /// -+ /// No thread may concurrently access the linear memory, and no live Rust reference may point -+ /// into the replaced range while this operation runs. The backing inode must not shrink below -+ /// the size validated by this call while the mapping is accessible; only the validated final -+ /// partial page may extend beyond EOF. -+ #[cfg(not(target_os = "windows"))] -+ pub unsafe fn remap_shared_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), String> { -+ use std::os::fd::AsRawFd; -+ -+ let page_size = region::page::size(); -+ if len == 0 { -+ return Err("mapping length must be non-zero".to_string()); -+ } -+ if start & (page_size - 1) != 0 { -+ return Err("mapping start must be page-aligned".to_string()); -+ } -+ if len & (page_size - 1) != 0 { -+ return Err("mapping length must be page-aligned".to_string()); -+ } -+ if file_offset & (page_size - 1) != 0 { -+ return Err("file offset must be page-aligned".to_string()); -+ } -+ if len > self.total_size || start > self.total_size - len { -+ return Err("mapping range is outside the reserved memory".to_string()); -+ } -+ let file_len: usize = file -+ .metadata() -+ .map_err(|err| err.to_string())? -+ .len() -+ .try_into() -+ .map_err(|_| "backing file length does not fit usize".to_string())?; -+ let available_file_bytes = file_len -+ .checked_sub(file_offset) -+ .ok_or_else(|| "file offset is beyond the backing file".to_string())?; -+ let rounded_available_file_bytes = available_file_bytes -+ .checked_add(page_size - 1) -+ .ok_or_else(|| "rounded file mapping range overflowed".to_string())? -+ & !(page_size - 1); -+ if len > rounded_available_file_bytes { -+ return Err("mapping includes a page wholly beyond the backing file".to_string()); -+ } -+ let file_offset = file_offset -+ .try_into() -+ .map_err(|_| "file offset does not fit off_t".to_string())?; -+ -+ let fixed_ptr = unsafe { (self.ptr as *mut u8).add(start) }; -+ let ptr = unsafe { -+ libc::mmap( -+ fixed_ptr.cast(), -+ len, -+ libc::PROT_READ | libc::PROT_WRITE, -+ libc::MAP_SHARED | libc::MAP_FIXED, -+ file.as_raw_fd(), -+ file_offset, -+ ) -+ }; -+ if ptr as isize == -1_isize { -+ return Err(io::Error::last_os_error().to_string()); -+ } -+ if ptr != fixed_ptr.cast() { -+ return Err("fixed mapping returned an unexpected address".to_string()); -+ } -+ -+ self.accessible_size = self.accessible_size.max(start + len); -+ Ok(()) -+ } -+ -+ /// Replace a page-aligned range inside this mapping with a private, -+ /// copy-on-write file mapping at the same host address. -+ /// -+ /// Unlike [`Self::remap_shared_file_fixed`], writes are never propagated -+ /// to the backing file. This is intended for immutable initialization -+ /// images whose clean pages should be shared between independent linear -+ /// memories while preserving normal per-instance write isolation. -+ /// -+ /// # Safety -+ /// -+ /// No thread may concurrently access the linear memory, and no live Rust reference may point -+ /// into the replaced range while this operation runs. The backing inode must not be truncated -+ /// or replaced below `file_offset + len` while the mapping is accessible. -+ #[cfg(not(target_os = "windows"))] -+ pub unsafe fn remap_private_file_fixed( -+ &mut self, -+ start: usize, -+ len: usize, -+ file: &std::fs::File, -+ file_offset: usize, -+ ) -> Result<(), String> { -+ use std::os::fd::AsRawFd; -+ -+ let page_size = region::page::size(); -+ if len == 0 { -+ return Err("mapping length must be non-zero".to_string()); -+ } -+ if start & (page_size - 1) != 0 { -+ return Err("mapping start must be page-aligned".to_string()); -+ } -+ if len & (page_size - 1) != 0 { -+ return Err("mapping length must be page-aligned".to_string()); -+ } -+ if file_offset & (page_size - 1) != 0 { -+ return Err("file offset must be page-aligned".to_string()); -+ } -+ if len > self.total_size || start > self.total_size - len { -+ return Err("mapping range is outside the reserved memory".to_string()); -+ } -+ let file_end = file_offset -+ .checked_add(len) -+ .ok_or_else(|| "file mapping range overflowed".to_string())?; -+ let file_len: usize = file -+ .metadata() -+ .map_err(|err| err.to_string())? -+ .len() -+ .try_into() -+ .map_err(|_| "backing file length does not fit usize".to_string())?; -+ if file_end > file_len { -+ return Err("mapping range extends beyond the backing file".to_string()); -+ } -+ let file_offset = file_offset -+ .try_into() -+ .map_err(|_| "file offset does not fit off_t".to_string())?; -+ -+ let fixed_ptr = unsafe { (self.ptr as *mut u8).add(start) }; -+ let ptr = unsafe { -+ libc::mmap( -+ fixed_ptr.cast(), -+ len, -+ libc::PROT_READ | libc::PROT_WRITE, -+ libc::MAP_PRIVATE | libc::MAP_FIXED, -+ file.as_raw_fd(), -+ file_offset, -+ ) -+ }; -+ if ptr as isize == -1_isize { -+ return Err(io::Error::last_os_error().to_string()); -+ } -+ if ptr != fixed_ptr.cast() { -+ return Err("fixed mapping returned an unexpected address".to_string()); -+ } -+ -+ self.accessible_size = self.accessible_size.max(start + len); -+ Ok(()) -+ } -+ -+ /// Replace a page-aligned range inside this mapping with private, -+ /// zero-filled memory at the same host address. -+ /// -+ /// # Safety -+ /// -+ /// No thread may concurrently access the linear memory, and no live Rust reference may point -+ /// into the replaced range while this operation runs. -+ #[cfg(not(target_os = "windows"))] -+ pub unsafe fn remap_private_fixed(&mut self, start: usize, len: usize) -> Result<(), String> { -+ let page_size = region::page::size(); -+ if len == 0 { -+ return Err("mapping length must be non-zero".to_string()); -+ } -+ if start & (page_size - 1) != 0 { -+ return Err("mapping start must be page-aligned".to_string()); -+ } -+ if len & (page_size - 1) != 0 { -+ return Err("mapping length must be page-aligned".to_string()); -+ } -+ if len > self.total_size || start > self.total_size - len { -+ return Err("mapping range is outside the reserved memory".to_string()); -+ } -+ -+ let fixed_ptr = unsafe { (self.ptr as *mut u8).add(start) }; -+ let ptr = unsafe { -+ libc::mmap( -+ fixed_ptr.cast(), -+ len, -+ libc::PROT_READ | libc::PROT_WRITE, -+ libc::MAP_PRIVATE | libc::MAP_ANON | libc::MAP_FIXED, -+ -1, -+ 0, -+ ) -+ }; -+ if ptr as isize == -1_isize { -+ return Err(io::Error::last_os_error().to_string()); -+ } -+ if ptr != fixed_ptr.cast() { -+ return Err("fixed mapping returned an unexpected address".to_string()); -+ } -+ -+ self.accessible_size = self.accessible_size.max(start + len); -+ Ok(()) -+ } -+ -+ /// Synchronize a page-aligned range in this mapping with its backing file. -+ #[cfg(not(target_os = "windows"))] -+ pub fn msync(&self, start: usize, len: usize, flags: i32) -> Result<(), String> { -+ let page_size = region::page::size(); -+ if start & (page_size - 1) != 0 { -+ return Err("mapping start must be page-aligned".to_string()); -+ } -+ if len & (page_size - 1) != 0 { -+ return Err("mapping length must be page-aligned".to_string()); -+ } -+ if len > self.total_size || start > self.total_size - len { -+ return Err("mapping range is outside the reserved memory".to_string()); -+ } -+ if len == 0 { -+ return Ok(()); -+ } -+ -+ let ptr = unsafe { (self.ptr as *mut u8).add(start) }; -+ if unsafe { libc::msync(ptr.cast(), len, flags) } != 0 { -+ return Err(io::Error::last_os_error().to_string()); -+ } -+ -+ Ok(()) -+ } -+ -+ /// Replace a page-aligned range inside this mapping with a shared, -+ /// read-write file mapping at the same host address. -+ /// -+ /// # Safety -+ /// See the non-Windows implementation; this stub performs no remap. -+ #[cfg(target_os = "windows")] -+ pub unsafe fn remap_shared_file_fixed( -+ &mut self, -+ _start: usize, -+ _len: usize, -+ _file: &std::fs::File, -+ _file_offset: usize, -+ ) -> Result<(), String> { -+ Err("fixed shared file remapping is not implemented on Windows".to_string()) -+ } -+ -+ /// Replace a page-aligned range inside this mapping with a private, -+ /// copy-on-write file mapping at the same host address. -+ /// -+ /// # Safety -+ /// See the non-Windows implementation; this stub performs no remap. -+ #[cfg(target_os = "windows")] -+ pub unsafe fn remap_private_file_fixed( -+ &mut self, -+ _start: usize, -+ _len: usize, -+ _file: &std::fs::File, -+ _file_offset: usize, -+ ) -> Result<(), String> { -+ Err("fixed private file remapping is not implemented on Windows".to_string()) -+ } -+ -+ /// Replace a page-aligned range inside this mapping with private, -+ /// zero-filled memory at the same host address. -+ /// -+ /// # Safety -+ /// See the non-Windows implementation; this stub performs no remap. -+ #[cfg(target_os = "windows")] -+ pub unsafe fn remap_private_fixed(&mut self, _start: usize, _len: usize) -> Result<(), String> { -+ Err("fixed private memory remapping is not implemented on Windows".to_string()) -+ } -+ -+ /// Synchronize a page-aligned range in this mapping with its backing file. -+ #[cfg(target_os = "windows")] -+ pub fn msync(&self, _start: usize, _len: usize, _flags: i32) -> Result<(), String> { -+ Err("memory synchronization is not implemented on Windows".to_string()) -+ } -+ - /// Make the memory starting at `start` and extending for `len` bytes accessible. - /// `start` and `len` must be native page-size multiples and describe a range within - /// `self`'s reserved memory. -@@ -364,12 +636,31 @@ impl Mmap { - - let mut new = - Self::accessible_reserved(copy_size, self.total_size, None, MmapType::Private)?; -- new.as_mut_slice_arbitary(copy_size) -- .copy_from_slice(self.as_slice_arbitary(copy_size)); -+ let src = self.as_slice_arbitary(copy_size); -+ let dst = new.as_mut_slice_arbitary(copy_size); -+ copy_sparse_range(src, dst, 0, copy_size); - Ok(new) - } - } - -+fn copy_sparse_range(src: &[u8], dst: &mut [u8], start: usize, end: usize) { -+ let page_size = region::page::size(); -+ let mut cursor = start; -+ -+ while cursor < end { -+ let next_page = cursor -+ .checked_add(page_size - (cursor % page_size)) -+ .unwrap_or(end); -+ let next = next_page.min(end); -+ let src_chunk = &src[cursor..next]; -+ -+ if src_chunk.iter().any(|byte| *byte != 0) { -+ dst[cursor..next].copy_from_slice(src_chunk); -+ } -+ cursor = next; -+ } -+} -+ - impl Drop for Mmap { - #[cfg(not(target_os = "windows"))] - fn drop(&mut self) { -@@ -404,3 +695,166 @@ fn _assert() { - fn _assert_send_sync() {} - _assert_send_sync::(); - } -+ -+#[cfg(all(test, not(target_os = "windows")))] -+mod tests { -+ use super::*; -+ use std::io::{Read, Seek, SeekFrom, Write}; -+ use std::time::{SystemTime, UNIX_EPOCH}; -+ -+ fn temp_file_path(name: &str) -> std::path::PathBuf { -+ let unique = SystemTime::now() -+ .duration_since(UNIX_EPOCH) -+ .unwrap() -+ .as_nanos(); -+ std::env::temp_dir().join(format!("wasmer-{name}-{unique}")) -+ } -+ -+ #[test] -+ fn remap_shared_file_fixed_replaces_only_requested_pages() { -+ let page_size = region::page::size(); -+ let path = temp_file_path("mmap-fixed-shared"); -+ let mut file = std::fs::OpenOptions::new() -+ .read(true) -+ .write(true) -+ .create_new(true) -+ .open(&path) -+ .unwrap(); -+ file.set_len(page_size as u64).unwrap(); -+ file.write_all(&vec![0x7a; page_size]).unwrap(); -+ file.sync_all().unwrap(); -+ -+ let mut mmap = -+ Mmap::accessible_reserved(page_size * 2, page_size * 2, None, MmapType::Private) -+ .unwrap(); -+ mmap.as_mut_slice()[0] = 0x11; -+ mmap.as_mut_slice()[page_size] = 0x22; -+ -+ unsafe { mmap.remap_shared_file_fixed(page_size, page_size, &file, 0) }.unwrap(); -+ -+ assert_eq!(mmap.as_slice()[0], 0x11); -+ assert_eq!(mmap.as_slice()[page_size], 0x7a); -+ -+ mmap.as_mut_slice()[page_size] = 0x42; -+ let r = unsafe { -+ libc::msync( -+ mmap.as_mut_ptr().add(page_size).cast(), -+ page_size, -+ libc::MS_SYNC, -+ ) -+ }; -+ assert_eq!(r, 0, "msync failed: {}", io::Error::last_os_error()); -+ -+ let mut got = [0u8; 1]; -+ file.seek(SeekFrom::Start(0)).unwrap(); -+ file.read_exact(&mut got).unwrap(); -+ assert_eq!(got[0], 0x42); -+ -+ drop(mmap); -+ drop(file); -+ let _ = std::fs::remove_file(path); -+ } -+ -+ #[test] -+ fn remap_shared_file_fixed_accepts_a_partial_final_file_page() { -+ let page_size = region::page::size(); -+ let path = temp_file_path("mmap-fixed-shared-partial-page"); -+ let mut file = std::fs::OpenOptions::new() -+ .read(true) -+ .write(true) -+ .create_new(true) -+ .open(&path) -+ .unwrap(); -+ file.set_len((page_size - 1) as u64).unwrap(); -+ file.write_all(&[0x7a]).unwrap(); -+ file.sync_all().unwrap(); -+ -+ let mut mmap = -+ Mmap::accessible_reserved(page_size, page_size, None, MmapType::Private).unwrap(); -+ unsafe { mmap.remap_shared_file_fixed(0, page_size, &file, 0) }.unwrap(); -+ assert_eq!(mmap.as_slice()[0], 0x7a); -+ assert_eq!(mmap.as_slice()[page_size - 1], 0); -+ -+ let mut oversized = -+ Mmap::accessible_reserved(page_size * 2, page_size * 2, None, MmapType::Private) -+ .unwrap(); -+ assert!( -+ unsafe { oversized.remap_shared_file_fixed(0, page_size * 2, &file, 0) } -+ .unwrap_err() -+ .contains("wholly beyond") -+ ); -+ assert_eq!(file.metadata().unwrap().len(), (page_size - 1) as u64); -+ -+ drop(mmap); -+ drop(oversized); -+ drop(file); -+ let _ = std::fs::remove_file(path); -+ } -+ -+ #[test] -+ fn remap_private_file_fixed_shares_clean_bytes_but_isolates_writes() { -+ let page_size = region::page::size(); -+ let path = temp_file_path("mmap-fixed-private-file"); -+ let mut file = std::fs::OpenOptions::new() -+ .read(true) -+ .write(true) -+ .create_new(true) -+ .open(&path) -+ .unwrap(); -+ file.set_len(page_size as u64).unwrap(); -+ file.write_all(&vec![0x7a; page_size]).unwrap(); -+ file.sync_all().unwrap(); -+ -+ let mut first = -+ Mmap::accessible_reserved(page_size, page_size, None, MmapType::Private).unwrap(); -+ let mut second = -+ Mmap::accessible_reserved(page_size, page_size, None, MmapType::Private).unwrap(); -+ unsafe { first.remap_private_file_fixed(0, page_size, &file, 0) }.unwrap(); -+ unsafe { second.remap_private_file_fixed(0, page_size, &file, 0) }.unwrap(); -+ -+ assert_eq!(first.as_slice()[0], 0x7a); -+ assert_eq!(second.as_slice()[0], 0x7a); -+ first.as_mut_slice()[0] = 0x42; -+ assert_eq!(first.as_slice()[0], 0x42); -+ assert_eq!(second.as_slice()[0], 0x7a); -+ -+ let mut got = [0u8; 1]; -+ file.seek(SeekFrom::Start(0)).unwrap(); -+ file.read_exact(&mut got).unwrap(); -+ assert_eq!(got[0], 0x7a); -+ -+ drop(first); -+ drop(second); -+ drop(file); -+ let _ = std::fs::remove_file(path); -+ } -+ -+ #[test] -+ fn copy_preserves_nonzero_and_zero_pages() { -+ let page_size = region::page::size(); -+ let mut mmap = -+ Mmap::accessible_reserved(page_size * 3, page_size * 3, None, MmapType::Private) -+ .unwrap(); -+ -+ mmap.as_mut_slice_arbitary(page_size * 3)[..page_size].fill(0x11); -+ mmap.as_mut_slice_arbitary(page_size * 3)[page_size * 2..page_size * 3].fill(0x33); -+ -+ let copied = mmap.copy(Some(page_size * 3)).unwrap(); -+ -+ assert!( -+ copied.as_slice_arbitary(page_size)[..page_size] -+ .iter() -+ .all(|byte| *byte == 0x11) -+ ); -+ assert!( -+ copied.as_slice_arbitary(page_size * 2)[page_size..page_size * 2] -+ .iter() -+ .all(|byte| *byte == 0) -+ ); -+ assert!( -+ copied.as_slice_arbitary(page_size * 3)[page_size * 2..page_size * 3] -+ .iter() -+ .all(|byte| *byte == 0x33) -+ ); -+ } -+} -diff --git a/lib/vm/src/table.rs b/lib/vm/src/table.rs -index 7eab35a..7999f02 100644 ---- a/lib/vm/src/table.rs -+++ b/lib/vm/src/table.rs -@@ -6,6 +6,7 @@ - //! `Table` is to WebAssembly tables what `Memory` is to WebAssembly linear memories. - - use crate::Trap; -+use crate::VMCallerCheckedAnyfunc; - use crate::VMExternRef; - use crate::VMFuncRef; - use crate::store::MaybeInstanceOwned; -@@ -14,6 +15,7 @@ use bytesize::ByteSize; - use std::cell::UnsafeCell; - use std::convert::TryFrom; - use std::fmt; -+use std::ptr; - use std::ptr::NonNull; - use wasmer_types::TableStyle; - use wasmer_types::{TableType, TrapCode, Type as ValType}; -@@ -75,6 +77,7 @@ const TABLE_MAX_SIZE: usize = ByteSize::mib(128).as_u64() as usize; - #[derive(Debug)] - pub struct VMTable { - vec: Vec, -+ anyfuncs: Option>, - maximum: Option, - /// The WebAssembly table description. - table: TableType, -@@ -152,9 +155,16 @@ impl VMTable { - .map_err(|_| "Table minimum is bigger than usize".to_string())?; - let mut vec = vec![RawTableElement::default(); table_minimum]; - let base = vec.as_mut_ptr(); -+ let mut anyfuncs = match table.ty { -+ ValType::FuncRef => Some(vec![VMCallerCheckedAnyfunc::null(); table_minimum]), -+ ValType::ExternRef => None, -+ _ => unreachable!("table type was already validated"), -+ }; -+ let anyfuncs_base = Self::anyfuncs_base(&mut anyfuncs); - match style { - TableStyle::CallerChecksSignature => Ok(Self { - vec, -+ anyfuncs, - maximum: table.maximum, - table: *table, - style: style.clone(), -@@ -164,12 +174,14 @@ impl VMTable { - let td = ptr.as_mut(); - td.base = base as _; - td.current_elements = table_minimum as _; -+ td.anyfuncs = anyfuncs_base; - } - MaybeInstanceOwned::Instance(table_loc) - } else { - MaybeInstanceOwned::Host(Box::new(UnsafeCell::new(VMTableDefinition { - base: base as _, - current_elements: table_minimum as _, -+ anyfuncs: anyfuncs_base, - }))) - }, - }), -@@ -177,6 +189,14 @@ impl VMTable { - } - } - -+ fn anyfuncs_base( -+ anyfuncs: &mut Option>, -+ ) -> *mut VMCallerCheckedAnyfunc { -+ anyfuncs -+ .as_mut() -+ .map_or(ptr::null_mut(), |anyfuncs| anyfuncs.as_mut_ptr()) -+ } -+ - /// Get the `VMTableDefinition`. - fn get_vm_table_definition(&self) -> NonNull { - self.vm_table_definition.as_ptr() -@@ -221,15 +241,25 @@ impl VMTable { - return Some(size); - } - -+ let init_anyfunc = match &init_value { -+ TableElement::FuncRef(funcref) => VMCallerCheckedAnyfunc::from_funcref(*funcref), -+ TableElement::ExternRef(_) => VMCallerCheckedAnyfunc::null(), -+ }; - self.vec - .resize(usize::try_from(new_len).unwrap(), init_value.into()); -+ if let Some(anyfuncs) = self.anyfuncs.as_mut() { -+ anyfuncs.resize(usize::try_from(new_len).unwrap(), init_anyfunc); -+ } -+ let base = self.vec.as_mut_ptr() as _; -+ let anyfuncs_base = Self::anyfuncs_base(&mut self.anyfuncs); - - // update table definition - unsafe { - let mut td_ptr = self.get_vm_table_definition(); - let td = td_ptr.as_mut(); - td.current_elements = new_len; -- td.base = self.vec.as_mut_ptr() as _; -+ td.base = base; -+ td.anyfuncs = anyfuncs_base; - } - Some(size) - } -@@ -271,7 +301,16 @@ impl VMTable { - *slot = r.into(); - } - (ValType::FuncRef, r @ TableElement::FuncRef(_)) => { -+ let anyfunc = match &r { -+ TableElement::FuncRef(funcref) => { -+ VMCallerCheckedAnyfunc::from_funcref(*funcref) -+ } -+ _ => unreachable!("matched TableElement::FuncRef"), -+ }; - *slot = r.into(); -+ if let Some(anyfuncs) = self.anyfuncs.as_mut() { -+ anyfuncs[index as usize] = anyfunc; -+ } - } - // This path should never be hit by the generated code due to Wasm - // validation. -@@ -279,7 +318,6 @@ impl VMTable { - panic!("Attempted to set a table of type {ty} with the value {v:?}") - } - }; -- - Ok(()) - } - None => Err(Trap::lib(TrapCode::TableAccessOutOfBounds)), -@@ -394,6 +432,11 @@ impl VMTable { - #[cfg(test)] - mod tests { - use super::{TableElement, VMTable}; -+ use crate::{ -+ VMCallerCheckedAnyfunc, VMContext, VMFunctionBody, VMFunctionContext, VMSignatureHash, -+ }; -+ use std::ptr::NonNull; -+ use wasmer_types::RawValue; - use wasmer_types::{TableStyle, TableType, Type}; - - #[test] -@@ -403,4 +446,71 @@ mod tests { - let mut table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); - assert_eq!(table.grow(0, TableElement::FuncRef(None)), None); - } -+ -+ unsafe extern "C" fn test_trampoline( -+ _vmctx: *mut VMContext, -+ _callee: *const VMFunctionBody, -+ _values: *mut RawValue, -+ ) { -+ } -+ -+ fn test_anyfunc(signature: u32) -> VMCallerCheckedAnyfunc { -+ VMCallerCheckedAnyfunc { -+ func_ptr: NonNull::::dangling().as_ptr(), -+ type_signature_hash: VMSignatureHash::new(signature), -+ vmctx: VMFunctionContext { -+ host_env: 1usize as *mut std::ffi::c_void, -+ }, -+ call_trampoline: test_trampoline, -+ } -+ } -+ -+ #[test] -+ fn funcref_table_exposes_anyfunc_shadow() { -+ let ty = TableType::new(Type::FuncRef, 1, Some(3)); -+ let mut table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); -+ -+ unsafe { -+ let definition = table.vmtable().as_ref(); -+ assert!(!definition.anyfuncs.is_null()); -+ assert!(definition.anyfuncs.read().func_ptr.is_null()); -+ } -+ -+ let anyfunc = test_anyfunc(7); -+ let funcref = crate::VMFuncRef(NonNull::from(&anyfunc)); -+ table.set(0, TableElement::FuncRef(Some(funcref))).unwrap(); -+ -+ unsafe { -+ let definition = table.vmtable().as_ref(); -+ assert_eq!(definition.anyfuncs.read(), anyfunc); -+ } -+ } -+ -+ #[test] -+ fn funcref_table_grow_updates_anyfunc_shadow_pointer_and_entries() { -+ let ty = TableType::new(Type::FuncRef, 1, Some(3)); -+ let mut table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); -+ let anyfunc = test_anyfunc(9); -+ let funcref = crate::VMFuncRef(NonNull::from(&anyfunc)); -+ -+ assert_eq!(table.grow(2, TableElement::FuncRef(Some(funcref))), Some(1)); -+ -+ unsafe { -+ let definition = table.vmtable().as_ref(); -+ assert!(!definition.anyfuncs.is_null()); -+ assert!(definition.anyfuncs.read().func_ptr.is_null()); -+ assert_eq!(definition.anyfuncs.add(1).read(), anyfunc); -+ assert_eq!(definition.anyfuncs.add(2).read(), anyfunc); -+ } -+ } -+ -+ #[test] -+ fn externref_table_has_no_anyfunc_shadow() { -+ let ty = TableType::new(Type::ExternRef, 1, Some(1)); -+ let table = VMTable::new(&ty, &TableStyle::CallerChecksSignature).unwrap(); -+ -+ unsafe { -+ assert!(table.vmtable().as_ref().anyfuncs.is_null()); -+ } -+ } - } -diff --git a/lib/vm/src/trap/traphandlers.rs b/lib/vm/src/trap/traphandlers.rs -index 6e2d7cc..033c796 100644 ---- a/lib/vm/src/trap/traphandlers.rs -+++ b/lib/vm/src/trap/traphandlers.rs -@@ -87,23 +87,69 @@ pub fn get_stack_size() -> usize { - DEFAULT_STACK_SIZE.load(Ordering::Relaxed) - } - --/// Pool of pre-allocated coroutine stacks to avoid repeated mmap syscalls. -+/// Cross-thread overflow pool for pre-allocated coroutine stacks. -+/// -+/// The common same-thread path is served from [`TLS_STACK`] without queue -+/// atomics. Nested calls and thread handoff fall back to this pool. - static STACK_POOL: LazyLock> = - LazyLock::new(crossbeam_queue::SegQueue::new); - -+/// One ready-to-use coroutine stack retained by each host thread that calls -+/// Wasm. A thread-exit destructor returns the mapping to the cross-thread pool -+/// so it remains reusable rather than leaking with the thread. -+struct StackCache(Cell>); -+ -+impl Drop for StackCache { -+ fn drop(&mut self) { -+ if let Some(stack) = self.0.take() { -+ STACK_POOL.push(stack); -+ } -+ } -+} -+ -+thread_local! { -+ static TLS_STACK: StackCache = const { StackCache(Cell::new(None)) }; -+} -+ -+/// Acquires a sufficiently large stack, preferring the atomic-free TLS slot. -+fn acquire_stack(min_size: usize) -> DefaultStack { -+ if let Some(stack) = TLS_STACK.with(|cache| cache.0.take()) { -+ if stack.size() >= min_size { -+ return stack; -+ } -+ // A stack-size increase makes the old mapping unusable. Drop it -+ // instead of circulating it through the overflow pool. -+ drop(stack); -+ } -+ -+ STACK_POOL -+ .pop() -+ .filter(|stack| stack.size() >= min_size) -+ .unwrap_or_else(|| DefaultStack::new(min_size).unwrap()) -+} -+ -+/// Returns a stack to the atomic-free TLS slot. Re-entrant execution can leave -+/// that slot occupied; in that case, move the displaced stack to the shared -+/// overflow pool. -+fn release_stack(stack: DefaultStack) { -+ if let Some(displaced) = TLS_STACK.with(|cache| cache.0.replace(Some(stack))) { -+ STACK_POOL.push(displaced); -+ } -+} -+ - /// Drains the coroutine stack pool at the moment it runs. - /// - /// This is intended to be called before retrying with a larger stack size so - /// that the pool does not keep serving cached undersized stacks. - /// --/// Note that `STACK_POOL` is a global, concurrently used queue. Other threads --/// may push stacks back into the pool (for example, when their Wasm execution --/// finishes) while or after this function is running. As a result, this --/// function provides only a best-effort drain of the pool: there is no --/// guarantee that no undersized stacks exist immediately after it returns --/// unless the caller ensures, via external synchronization, that no other --/// Wasm executions can return stacks to the pool while this function runs. -+/// Note that the pool is global and each Wasm-calling thread can retain one -+/// private stack. Other threads may return stacks while or after this function -+/// runs, and their TLS slots cannot be drained here. This therefore remains a -+/// best-effort operation unless the caller externally quiesces Wasm execution. -+/// The calling thread's TLS slot is drained. - pub fn drain_stack_pool() { -+ // Drop rather than re-pool the local mapping; the queue is drained next. -+ TLS_STACK.with(|cache| cache.0.set(None)); - while STACK_POOL.pop().is_some() {} - } - -@@ -1047,17 +1093,11 @@ fn on_wasm_stack T + 'static, T: 'static>( - trap_handler: Option<*const TrapHandlerFn<'static>>, - f: F, - ) -> Result { -- // Reuse a cached stack from the pool if it is large enough, otherwise -- // allocate a fresh one. The size check prevents using undersized stacks -- // that were returned by threads still running at the old size after -- // `drain_stack_pool()` was called. `base() - limit()` is the full mmap -- // region (including guard page), which is always >= the requested size -- // for stacks allocated with that size. -- let stack = STACK_POOL -- .pop() -- .filter(|s| s.size() >= stack_size) -- .unwrap_or_else(|| DefaultStack::new(stack_size).unwrap()); -- let mut stack = scopeguard::guard(stack, |stack| STACK_POOL.push(stack)); -+ // Same-thread calls reuse the TLS mapping without touching the global -+ // queue. Nested calls and first use on a thread fall back to the overflow -+ // pool or a fresh mapping. Both cache levels reject undersized stacks. -+ let stack = acquire_stack(stack_size); -+ let mut stack = scopeguard::guard(stack, release_stack); - - // Create a coroutine with a new stack to run the function on. - let coro = ScopedCoroutine::with_stack(&mut *stack, move |yielder, ()| { -@@ -1269,6 +1309,10 @@ mod tests { - } - } - -+ fn clear_tls_stack() { -+ TLS_STACK.with(|cache| cache.0.set(None)); -+ } -+ - #[test] - fn max_stack_size_is_100mb() { - assert_eq!(MAX_STACK_SIZE, ByteSize::mib(100).as_u64() as usize); -@@ -1378,14 +1422,100 @@ mod tests { - let result = on_wasm_stack(big_size, None, || 42); - - assert_eq!(result.ok().expect("on_wasm_stack should succeed"), 42); -- // The undersized stack was discarded; the pool should now contain -- // the correctly-sized stack that was allocated for this call. -- let returned = STACK_POOL -- .pop() -- .expect("stack should have been returned to pool"); -+ // The undersized stack was discarded; the correctly-sized mapping is -+ // now held in the calling thread's fast cache. -+ let returned = TLS_STACK -+ .with(|cache| cache.0.take()) -+ .expect("stack should have been returned to the TLS cache"); - assert!( - returned.size() >= big_size, - "returned stack must be at least as large as the requested size" - ); -+ assert!(STACK_POOL.is_empty()); -+ } -+ -+ #[test] -+ fn tls_stack_reuses_mapping_without_global_queue() { -+ let _lock = GLOBAL_STATE.lock().unwrap(); -+ let _restore = RestoreStackSize(get_stack_size()); -+ drain_stack_pool(); -+ -+ let size = get_stack_size(); -+ assert!(on_wasm_stack(size, None, || ()).is_ok()); -+ let first_base = TLS_STACK.with(|cache| { -+ let stack = cache.0.take().expect("first call should populate TLS"); -+ let base = stack.base().get(); -+ cache.0.set(Some(stack)); -+ base -+ }); -+ assert!(STACK_POOL.is_empty()); -+ -+ assert!(on_wasm_stack(size, None, || ()).is_ok()); -+ let second = TLS_STACK -+ .with(|cache| cache.0.take()) -+ .expect("second call should return the stack to TLS"); -+ assert_eq!(second.base().get(), first_base); -+ assert!(STACK_POOL.is_empty()); -+ } -+ -+ #[test] -+ fn drain_stack_pool_clears_calling_thread_tls() { -+ let _lock = GLOBAL_STATE.lock().unwrap(); -+ drain_stack_pool(); -+ -+ assert!(on_wasm_stack(get_stack_size(), None, || ()).is_ok()); -+ assert!(TLS_STACK.with(|cache| cache.0.take().is_some())); -+ -+ // Re-populate TLS, then verify the public drain covers the caller's -+ // local fast slot as well as the shared queue. -+ assert!(on_wasm_stack(get_stack_size(), None, || ()).is_ok()); -+ drain_stack_pool(); -+ assert!(TLS_STACK.with(|cache| cache.0.take().is_none())); -+ assert!(STACK_POOL.is_empty()); -+ } -+ -+ #[test] -+ fn thread_exit_returns_tls_stack_to_global_pool() { -+ let _lock = GLOBAL_STATE.lock().unwrap(); -+ drain_stack_pool(); -+ clear_tls_stack(); -+ -+ let size = get_stack_size(); -+ std::thread::spawn(move || { -+ assert!(on_wasm_stack(size, None, || ()).is_ok()); -+ assert!(STACK_POOL.is_empty()); -+ }) -+ .join() -+ .unwrap(); -+ -+ let returned = STACK_POOL -+ .pop() -+ .expect("thread-exit TLS destructor should return the stack"); -+ assert!(returned.size() >= size); -+ assert!(STACK_POOL.is_empty()); -+ } -+ -+ #[test] -+ fn reentrant_calls_keep_both_stacks_reusable() { -+ let _lock = GLOBAL_STATE.lock().unwrap(); -+ let _restore = RestoreStackSize(get_stack_size()); -+ drain_stack_pool(); -+ -+ let result = on_wasm_stack(get_stack_size(), None, || { -+ on_wasm_stack(get_stack_size(), None, || 42_u32) -+ .ok() -+ .expect("inner coroutine should complete") -+ }); -+ assert_eq!(result.ok(), Some(42)); -+ -+ let local = TLS_STACK -+ .with(|cache| cache.0.take()) -+ .expect("outer stack should return to TLS"); -+ let overflow = STACK_POOL -+ .pop() -+ .expect("inner stack should remain in the overflow pool"); -+ assert!(local.size() >= get_stack_size()); -+ assert!(overflow.size() >= get_stack_size()); -+ assert!(STACK_POOL.is_empty()); - } - } -diff --git a/lib/vm/src/vmcontext.rs b/lib/vm/src/vmcontext.rs -index 50c89bf..4079bd6 100644 ---- a/lib/vm/src/vmcontext.rs -+++ b/lib/vm/src/vmcontext.rs -@@ -11,7 +11,7 @@ use crate::instance::Instance; - use crate::memory::VMMemory; - use crate::store::InternalStoreHandle; - use crate::trap::{Trap, TrapCode}; --use crate::{VMBuiltinFunctionIndex, VMFunction}; -+use crate::{VMBuiltinFunctionIndex, VMFuncRef, VMFunction}; - use std::convert::TryFrom; - use std::hash::{Hash, Hasher}; - use std::ptr::{self, NonNull}; -@@ -461,6 +461,13 @@ pub struct VMTableDefinition { - - /// The current number of elements in the table. - pub current_elements: u32, -+ -+ /// Pointer to caller-checked function records for `funcref` tables. -+ /// -+ /// This is null for non-`funcref` tables. For `funcref` tables it mirrors -+ /// `base` element-for-element and lets compiled indirect calls load the -+ /// callable record directly while table operations keep the shadow in sync. -+ pub anyfuncs: *mut VMCallerCheckedAnyfunc, - } - - #[cfg(test)] -@@ -487,6 +494,10 @@ mod test_vmtable_definition { - offset_of!(VMTableDefinition, current_elements), - usize::from(offsets.vmtable_definition_current_elements()) - ); -+ assert_eq!( -+ offset_of!(VMTableDefinition, anyfuncs), -+ usize::from(offsets.vmtable_definition_anyfuncs()) -+ ); - } - } - -@@ -616,6 +627,14 @@ impl VMCallerCheckedAnyfunc { - call_trampoline: null_call_trampoline, - } - } -+ -+ /// Construct a caller-checked function record from a function reference. -+ pub fn from_funcref(funcref: Option) -> Self { -+ match funcref { -+ Some(funcref) => unsafe { *funcref.0.as_ptr() }, -+ None => Self::null(), -+ } -+ } - } - - impl PartialEq for VMCallerCheckedAnyfunc { -diff --git a/lib/wasix/Cargo.toml b/lib/wasix/Cargo.toml -index 9071afa..8b53f98 100644 ---- a/lib/wasix/Cargo.toml -+++ b/lib/wasix/Cargo.toml -@@ -126,7 +126,6 @@ toml.workspace = true - pin-utils.workspace = true - wasmparser.workspace = true - crossbeam-channel.workspace = true --bus.workspace = true - - [target.'cfg(not(any(target_arch = "riscv64", target_arch = "loongarch64")))'.dependencies.reqwest] - workspace = true -@@ -148,7 +147,11 @@ termios.workspace = true - - [target.'cfg(windows)'.dependencies] - windows-sys = { workspace = true, features = [ -+ "Win32_Foundation", -+ "Win32_Security", -+ "Win32_System_Console", - "Win32_System_SystemInformation", -+ "Win32_System_Threading", - ] } - - [target.'cfg(not(target_arch = "wasm32"))'.dependencies] -@@ -196,7 +199,9 @@ features = ["wasm_js"] - default = ["sys-default"] - - time = ["tokio/time"] --ctrlc = ["tokio/signal"] -+# Enables the explicit Unix host-lifecycle adapter. Merely enabling this -+# feature never installs process-global signal dispositions. -+ctrlc = [] - - webc_runner_rt_wcgi = [ - "hyper", -diff --git a/lib/wasix/src/bin_factory/exec.rs b/lib/wasix/src/bin_factory/exec.rs -index 4e60ee0..637be9e 100644 ---- a/lib/wasix/src/bin_factory/exec.rs -+++ b/lib/wasix/src/bin_factory/exec.rs -@@ -10,9 +10,11 @@ use crate::{ - ModuleInput, TaintReason, - module_cache::HashedModuleData, - task_manager::{ -- TaskWasm, TaskWasmRecycle, TaskWasmRecycleProperties, TaskWasmRunProperties, -+ TaskWasm, TaskWasmAcceptedExecutionGuard, TaskWasmRecycle, TaskWasmRecycleProperties, -+ TaskWasmRunProperties, - }, - }, -+ state::PreinitializedMemoryImageMode, - state::context_switching::ContextSwitchingEnvironment, - syscalls::rewind_ext, - }; -@@ -128,6 +130,29 @@ pub fn spawn_exec_module( - env: WasiEnv, - runtime: &Arc, - ) -> Result { -+ spawn_exec_module_with_preinitialized_memory_image(module, env, runtime, None) -+} -+ -+pub fn spawn_exec_module_with_preinitialized_memory_image( -+ module: Module, -+ env: WasiEnv, -+ runtime: &Arc, -+ preinitialized_memory_image: Option, -+) -> Result { -+ // A fresh executable image must not see inherited mappings as active in -+ // its new linear memory. Convert them transactionally into address/backing -+ // reservations before Wasmer can instantiate (and run a module start -+ // function). If task admission fails, the guard restores the old image's -+ // registry so exec failure remains non-destructive. -+ let shared_memory_exec = env -+ .state -+ .prepare_shared_memory_for_exec() -+ .map_err(|errno| { -+ SpawnError::Other(Box::new(std::io::Error::other(format!( -+ "failed to prepare shared-memory exec reservations: {errno}" -+ )))) -+ })?; -+ - // Create a new task manager - let tasks = runtime.task_manager(); - -@@ -139,20 +164,23 @@ pub fn spawn_exec_module( - // Create a thread that will run this process - let tasks_outer = tasks.clone(); - -- tasks_outer -- .task_wasm( -- TaskWasm::new(Box::new(run_exec), env, module, true, true).with_pre_run(Box::new( -- |ctx, store| { -- Box::pin(async move { -- ctx.data(store).state.fs.close_cloexec_fds().await; -- }) -- }, -- )), -- ) -- .map_err(|err| { -- error!("wasi[{}]::failed to launch module - {}", pid, err); -- SpawnError::Other(Box::new(err)) -- })? -+ let accepted_run_exec = move |props| { -+ shared_memory_exec.commit(); -+ run_exec(props); -+ }; -+ let mut task = TaskWasm::new(Box::new(accepted_run_exec), env, module, true, true) -+ .with_pre_run(Box::new(|ctx, store| { -+ Box::pin(async move { -+ ctx.data(store).state.fs.close_cloexec_fds().await; -+ }) -+ })); -+ if let Some(image) = preinitialized_memory_image { -+ task = task.with_preinitialized_memory_image(image); -+ } -+ tasks_outer.task_wasm(task).map_err(|err| { -+ error!("wasi[{}]::failed to launch module - {}", pid, err); -+ SpawnError::Other(Box::new(err)) -+ })?; - }; - - Ok(join_handle) -@@ -163,9 +191,15 @@ pub fn spawn_exec_module( - /// otherwise it will cause a panic - unsafe fn run_recycle( - callback: Option>, -+ execution_guard: Option, - ctx: WasiFunctionEnv, - mut store: Store, - ) { -+ // Terminal status must be visible before this point. Releasing the lease -+ // makes join/reinit observe real guest quiescence before the recyclable -+ // environment is published to another request. -+ ctx.data_mut(&mut store).clear_task_wasm_execution(); -+ drop(execution_guard); - if let Some(callback) = callback { - let env = ctx.data_mut(&mut store); - let memory = unsafe { env.memory() }.clone(); -@@ -186,6 +220,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { - // Create the WasiFunctionEnv - let thread = WasiThreadRunGuard::new(ctx.data(&store).thread.clone()); - let recycle = props.recycle; -+ let execution_guard = props.execution_guard; - - // Perform the initialization - // If this module exports an _initialize function, run that first. -@@ -206,7 +241,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { - thread.thread.set_status_finished(Err(err.into())); - ctx.data(&store) - .blocking_on_exit(Some(Errno::Noexec.into())); -- unsafe { run_recycle(recycle, ctx, store) }; -+ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; - return; - } - } -@@ -221,7 +256,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { - thread.thread.set_status_finished(Err(err)); - ctx.data(&store) - .blocking_on_exit(Some(Errno::Noexec.into())); -- unsafe { run_recycle(recycle, ctx, store) }; -+ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; - return; - } - }; -@@ -231,7 +266,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { - // TODO: rewrite to use crate::run_wasi_func - - // Call the module -- call_module(ctx, store, thread, rewind_state, recycle); -+ call_module(ctx, store, thread, rewind_state, recycle, execution_guard); - } - - fn get_start(ctx: &WasiFunctionEnv, store: &Store) -> Option { -@@ -252,6 +287,7 @@ fn call_module( - handle: WasiThreadRunGuard, - rewind_state: Option<(RewindState, RewindResultType)>, - recycle: Option>, -+ execution_guard: Option, - ) { - let env = ctx.data(&store); - let pid = env.pid(); -@@ -272,7 +308,15 @@ fn call_module( - ); - if res != Errno::Success { - ctx.data().blocking_on_exit(Some(res.into())); -- unsafe { run_recycle(recycle, WasiFunctionEnv { env: ctx.as_ref() }, store) }; -+ handle.thread.set_status_finished(Ok(res.into())); -+ unsafe { -+ run_recycle( -+ recycle, -+ execution_guard, -+ WasiFunctionEnv { env: ctx.as_ref() }, -+ store, -+ ) -+ }; - return; - } - } else { -@@ -285,7 +329,15 @@ fn call_module( - ); - if res != Errno::Success { - ctx.data().blocking_on_exit(Some(res.into())); -- unsafe { run_recycle(recycle, WasiFunctionEnv { env: ctx.as_ref() }, store) }; -+ handle.thread.set_status_finished(Ok(res.into())); -+ unsafe { -+ run_recycle( -+ recycle, -+ execution_guard, -+ WasiFunctionEnv { env: ctx.as_ref() }, -+ store, -+ ) -+ }; - return; - } - }; -@@ -297,7 +349,8 @@ fn call_module( - debug!("wasi[{}]::exec-failed: missing _start function", pid); - ctx.data(&store) - .blocking_on_exit(Some(Errno::Noexec.into())); -- unsafe { run_recycle(recycle, ctx, store) }; -+ handle.thread.set_status_finished(Ok(Errno::Noexec.into())); -+ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; - return; - }; - -@@ -335,8 +388,9 @@ fn call_module( - Ok(WasiError::DeepSleep(deep)) => { - // Create the callback that will be invoked when the thread respawns after a deep sleep - let rewind = deep.rewind; -+ let terminal_thread = handle.thread.clone(); - let respawn = { -- move |ctx, store, rewind_result| { -+ move |ctx, store, rewind_result, execution_guard| { - // Call the thread - call_module( - ctx, -@@ -344,15 +398,26 @@ fn call_module( - handle, - Some((rewind, RewindResultType::RewindWithResult(rewind_result))), - recycle, -+ execution_guard, - ); - } - }; - - // Spawns the WASM process after a trigger - if let Err(err) = unsafe { -- tasks.resume_wasm_after_poller(Box::new(respawn), ctx, store, deep.trigger) -+ tasks.resume_wasm_after_poller( -+ Box::new(respawn), -+ ctx, -+ store, -+ deep.trigger, -+ execution_guard, -+ ) - } { - debug!("failed to go into deep sleep - {}", err); -+ terminal_thread.set_status_finished(Err(RuntimeError::new(format!( -+ "failed to resume process after deep sleep: {err}" -+ )) -+ .into())); - } - return; - } -@@ -366,6 +431,7 @@ fn call_module( - runtime.on_taint(TaintReason::DlSymbolResolutionFailed(symbol.clone())); - Err(WasiError::DlSymbolResolutionFailed(symbol).into()) - } -+ Ok(WasiError::StoreSnapshot(err)) => Err(WasiError::StoreSnapshot(err).into()), - Err(err) => { - runtime.on_taint(TaintReason::RuntimeError(err.clone())); - Err(WasiRuntimeError::from(err)) -@@ -391,10 +457,9 @@ fn call_module( - - // Cleanup the environment - ctx.data(&store).blocking_on_exit(Some(code)); -- unsafe { run_recycle(recycle, ctx, store) }; -- - debug!("wasi[{pid}]::main() has exited with {code}"); - handle.thread.set_status_finished(ret.map(|a| a.into())); -+ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; - } - - #[allow(clippy::type_complexity)] -@@ -417,6 +482,7 @@ fn resume_vfork( - Some(WasiError::ThreadExit) => (None, wasmer_wasix_types::wasi::ExitCode::from(0u16)), - Some(WasiError::UnknownWasiVersion) => (None, Errno::Noexec.into()), - Some(WasiError::DlSymbolResolutionFailed(_)) => (None, Errno::Nolink.into()), -+ Some(WasiError::StoreSnapshot(_)) => (None, Errno::Noexec.into()), - None => ( - Some(WasiRuntimeError::from(err.clone())), - Errno::Unknown.into(), -@@ -447,10 +513,16 @@ fn resume_vfork( - vfork.env.swap_inner(ctx.data_mut(&mut store)); - std::mem::swap(vfork.env.as_mut(), ctx.data_mut(&mut store)); - let mut child_env = *vfork.env; -+ // A well-defined parent return is normally still owned by the -+ // original TaskWasm. Preserve one supplemental guard only if a -+ // continuation transferred that ownership while the child ran. -+ ctx.data_mut(&mut store) -+ .restore_parent_execution_guard(vfork.parent_execution); - child_env.owned_handles.push(vfork.handle); - -- // Terminate the child process -- child_env.process.terminate(code); -+ // Terminal status precedes release of the vfork child's execution -+ // lease; the restored parent guard remains with the active env. -+ vfork.child_execution.finish(Ok(code)); - - // If the vfork contained a context-switching environment, exit now - if ctx.data(&store).context_switching_environment.is_some() { -@@ -522,3 +594,85 @@ fn resume_vfork( - (store, Ok(None)) - } - } -+ -+#[cfg(all(test, feature = "sys-thread"))] -+mod lifecycle_tests { -+ use std::{ -+ sync::{Arc, mpsc}, -+ time::Duration, -+ }; -+ -+ use super::*; -+ use crate::{ -+ PluggableRuntime, -+ runtime::task_manager::tokio::TokioTaskManager, -+ runtime::task_manager::{TaskWasm, TaskWasmRunProperties}, -+ }; -+ use wasmer::Engine; -+ -+ #[test] -+ fn run_exec_panic_terminalizes_before_releasing_accepted_guard() { -+ let tokio_runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = tokio_runtime.enter(); -+ let engine = Engine::default(); -+ let mut runtime = PluggableRuntime::new(Arc::new(TokioTaskManager::new( -+ tokio_runtime.handle().clone(), -+ ))); -+ runtime.set_engine(engine.clone()); -+ let env = WasiEnv::builder("run-exec-panic-test") -+ .runtime(Arc::new(runtime)) -+ .build() -+ .unwrap(); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let module = Module::new( -+ &engine, -+ br#"(module (memory (export "memory") 1) (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let task = TaskWasm::new(Box::new(|_| {}), env.clone(), module, false, false); -+ let TaskWasm { -+ env: mut task_env, -+ execution_lease, -+ .. -+ } = task; -+ let accepted = execution_lease -+ .expect("test TaskWasm must acquire its pending execution guard") -+ .accept_callback(&mut task_env); -+ -+ // A raw, uninstantiated WasiFunctionEnv makes concrete run_exec panic -+ // while looking up its module handles, after both the run guard and -+ // accepted execution guard have been installed. -+ let mut store = task_env.runtime().new_store(); -+ let ctx = WasiFunctionEnv::new(&mut store, task_env); -+ let observer_process = process.clone(); -+ let observer_thread = thread.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = std::thread::spawn(move || { -+ observer_process.wait_for_execution_quiescence_blocking(); -+ let _ = observed_tx.send(observer_thread.try_join()); -+ }); -+ -+ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ run_exec(TaskWasmRunProperties { -+ ctx, -+ store, -+ trigger_result: None, -+ recycle: None, -+ execution_guard: Some(accepted), -+ }); -+ })); -+ -+ assert!(panic.is_err()); -+ let status = observed_rx -+ .recv_timeout(Duration::from_secs(2)) -+ .unwrap() -+ .expect("run_exec panic released quiescence before terminal status"); -+ assert!(status.is_err()); -+ assert_eq!(process.lock().execution_leases, 0); -+ observer.join().unwrap(); -+ } -+} -diff --git a/lib/wasix/src/bin_factory/mod.rs b/lib/wasix/src/bin_factory/mod.rs -index a290e6e..8de579c 100644 ---- a/lib/wasix/src/bin_factory/mod.rs -+++ b/lib/wasix/src/bin_factory/mod.rs -@@ -13,6 +13,7 @@ use shared_buffer::OwnedBuffer; - use virtual_fs::{AsyncReadExt, FileSystem}; - use wasmer::FunctionEnvMut; - use wasmer_package::utils::from_bytes; -+use wasmer_types::ModuleHash; - - mod binary_package; - mod exec; -@@ -21,7 +22,7 @@ pub use self::{ - binary_package::*, - exec::{ - import_package_mounts, package_command_by_name, run_exec, spawn_exec, spawn_exec_module, -- spawn_exec_wasm, spawn_load_module, -+ spawn_exec_module_with_preinitialized_memory_image, spawn_exec_wasm, spawn_load_module, - }, - }; - use crate::{ -@@ -31,6 +32,7 @@ use crate::{ - task::TaskJoinHandle, - }, - runtime::module_cache::HashedModuleData, -+ state::{PreinitializedMemoryImageHandle, PreinitializedMemoryImageMode}, - }; - - #[derive(Debug, Clone)] -@@ -38,14 +40,61 @@ pub struct BinFactory { - pub(crate) commands: Commands, - runtime: Arc, - pub(crate) local: Arc>>>>, -+ sealed_modules: Arc>, -+} -+ -+#[derive(Debug, Clone)] -+pub struct SealedExecutable { -+ module_hash: ModuleHash, -+ preinitialized_memory_image: Option, -+} -+ -+enum SealedAliasLookup<'a> { -+ Match(&'a SealedExecutable), -+ Denied, -+ Unsealed, -+} -+ -+fn lookup_sealed_alias<'a>( -+ sealed_modules: &'a HashMap, -+ name: &str, -+) -> SealedAliasLookup<'a> { -+ if let Some(executable) = sealed_modules.get(name) { -+ SealedAliasLookup::Match(executable) -+ } else if sealed_modules.is_empty() { -+ SealedAliasLookup::Unsealed -+ } else { -+ SealedAliasLookup::Denied -+ } - } - - impl BinFactory { - pub fn new(runtime: Arc) -> BinFactory { -+ Self::new_with_sealed_modules(runtime, HashMap::new()) -+ } -+ -+ pub(crate) fn new_with_sealed_modules( -+ runtime: Arc, -+ sealed_modules: HashMap)>, -+ ) -> BinFactory { - BinFactory { - commands: Commands::new_with_builtins(runtime.clone()), - runtime, - local: Arc::new(RwLock::new(HashMap::new())), -+ sealed_modules: Arc::new( -+ sealed_modules -+ .into_iter() -+ .map(|(path, (module_hash, preinitialized_memory_image))| { -+ ( -+ path, -+ SealedExecutable { -+ module_hash, -+ preinitialized_memory_image, -+ }, -+ ) -+ }) -+ .collect(), -+ ), - } - } - -@@ -101,7 +150,7 @@ impl BinFactory { - self.get_executable(name, fs) - .await - .and_then(|executable| match executable { -- Executable::Wasm(_) => None, -+ Executable::Wasm(_) | Executable::SealedModule(_) => None, - Executable::BinaryPackage(pkg) => Some(pkg), - }) - } -@@ -123,6 +172,27 @@ impl BinFactory { - - // Execute - match executable { -+ Executable::SealedModule(executable) => { -+ let mut env = env; -+ env.process.module_hash = executable.module_hash; -+ let module = self -+ .runtime -+ .module_cache() -+ .load(executable.module_hash, &self.runtime.engine()) -+ .await -+ .map_err(SpawnError::CacheError)?; -+ let image = executable -+ .preinitialized_memory_image -+ .map(|image| image.load().map(PreinitializedMemoryImageMode::Apply)) -+ .transpose() -+ .map_err(|message| SpawnError::ModuleLoad { message })?; -+ spawn_exec_module_with_preinitialized_memory_image( -+ module, -+ env, -+ &self.runtime, -+ image, -+ ) -+ } - Executable::Wasm(bytes) => { - let data = HashedModuleData::new(bytes.clone()); - spawn_exec_wasm(data, name.as_str(), env, &self.runtime).await -@@ -145,6 +215,13 @@ impl BinFactory { - parent_ctx: Option<&FunctionEnvMut<'_, WasiEnv>>, - builder: &mut Option, - ) -> Result { -+ // Syscall paths consult built-ins before `spawn`, so the sealed policy -+ // must close this resolver too. Otherwise an undeclared host command -+ // (currently `/bin/wasmer`) could bypass the exact manifest closure. -+ if !self.sealed_modules.is_empty() { -+ return Err(SpawnError::BinaryNotFound { binary: name }); -+ } -+ - // We check for built in commands - if let Some(parent_ctx) = parent_ctx { - if self.commands.exists(name.as_str()) { -@@ -164,6 +241,17 @@ impl BinFactory { - name: &str, - fs: Option<&dyn FileSystem>, - ) -> Option { -+ // A sealed registry is an exact immutable-executable policy. Once it is -+ // present, an unknown alias is denied before consulting mutable guest -+ // filesystem state. -+ match lookup_sealed_alias(&self.sealed_modules, name) { -+ SealedAliasLookup::Match(executable) => { -+ return Some(Executable::SealedModule(executable.clone())); -+ } -+ SealedAliasLookup::Denied => return None, -+ SealedAliasLookup::Unsealed => {} -+ } -+ - let name = name.to_string(); - - // Return early if the path is already cached -@@ -209,7 +297,44 @@ impl BinFactory { - } - } - -+#[cfg(test)] -+mod tests { -+ use super::*; -+ -+ #[test] -+ fn sealed_aliases_are_exact_and_close_the_executable_namespace() { -+ let module_hash = ModuleHash::from_bytes([0x5a; 32]); -+ let sealed_modules = HashMap::from([( -+ "/bin/postgres".to_string(), -+ SealedExecutable { -+ module_hash, -+ preinitialized_memory_image: None, -+ }, -+ )]); -+ -+ match lookup_sealed_alias(&sealed_modules, "/bin/postgres") { -+ SealedAliasLookup::Match(executable) => { -+ assert_eq!(executable.module_hash, module_hash) -+ } -+ _ => panic!("exact sealed alias was not resolved"), -+ } -+ assert!(matches!( -+ lookup_sealed_alias(&sealed_modules, "postgres"), -+ SealedAliasLookup::Denied -+ )); -+ assert!(matches!( -+ lookup_sealed_alias(&sealed_modules, "/bin/Postgres"), -+ SealedAliasLookup::Denied -+ )); -+ assert!(matches!( -+ lookup_sealed_alias(&HashMap::new(), "/bin/postgres"), -+ SealedAliasLookup::Unsealed -+ )); -+ } -+} -+ - pub enum Executable { -+ SealedModule(SealedExecutable), - Wasm(OwnedBuffer), - BinaryPackage(Arc), - } -diff --git a/lib/wasix/src/fs/fd.rs b/lib/wasix/src/fs/fd.rs -index 362b933..6f0a94e 100644 ---- a/lib/wasix/src/fs/fd.rs -+++ b/lib/wasix/src/fs/fd.rs -@@ -2,19 +2,202 @@ use std::{ - borrow::Cow, - collections::HashMap, - path::PathBuf, -- sync::{Arc, RwLock, RwLockReadGuard, RwLockWriteGuard, atomic::AtomicU64}, -+ sync::{ -+ Arc, Mutex, OnceLock, RwLock, RwLockReadGuard, RwLockWriteGuard, Weak, -+ atomic::{AtomicU64, Ordering}, -+ }, - }; - - #[cfg(feature = "enable-serde")] - use serde_derive::{Deserialize, Serialize}; --use virtual_fs::{Pipe, PipeRx, PipeTx, VirtualFile}; --use wasmer_wasix_types::wasi::{Fd as WasiFd, Fdflags, Fdflagsext, Filestat, Rights}; -+use virtual_fs::{Pipe, PipeRx, PipeTx, VirtualDirectory, VirtualFile}; -+use virtual_mio::InterestHandlerFanout; -+use wasmer_wasix_types::wasi::{ -+ Errno, Fd as WasiFd, Fdflags, Fdflagsext, Filestat, Filetype, Rights, -+}; - - use crate::net::socket::InodeSocket; --use crate::os::epoll::EpollState; -+use crate::os::epoll::{EpollState, EpollSubState, EpollSubscriptionKey}; - - use super::{InodeGuard, InodeWeakGuard, NotificationInner}; - -+pub type ReaddirSnapshot = Arc>; -+pub type ReaddirCache = Arc>>; -+pub(crate) type DirectorySyncHandle = Arc; -+ -+static NEXT_OPEN_FILE_DESCRIPTION_ID: AtomicU64 = AtomicU64::new(1); -+ -+#[derive(Debug)] -+struct EpollReverseRegistration { -+ state: Weak, -+ subscription: Weak, -+ key: EpollSubscriptionKey, -+} -+ -+#[derive(Debug, Default)] -+struct OpenFileDescriptionMutable { -+ descriptor_count: u32, -+ next_registration_id: u64, -+ registrations: HashMap, -+} -+ -+/// Lifecycle shared by descriptors produced by dup/fork from one open file -+/// description. Independent opens get distinct identities even when they -+/// reference the same inode. -+#[derive(Debug)] -+pub(crate) struct OpenFileDescription { -+ id: u64, -+ mutable: Mutex, -+ interest_fanout: OnceLock, -+ /// A real opened directory descriptor, or the error encountered while -+ /// acquiring one. `None` means this OFD is not a sync-capable directory. -+ /// Keeping the result on the OFD preserves identity across dup/fork and -+ /// ensures unsupported backends fail closed when sync is requested. -+ directory_sync: Option>, -+} -+ -+impl OpenFileDescription { -+ pub(crate) fn new() -> Arc { -+ Self::new_with_directory_sync(None) -+ } -+ -+ pub(crate) fn new_with_directory_sync( -+ directory_sync: Option>, -+ ) -> Arc { -+ let id = NEXT_OPEN_FILE_DESCRIPTION_ID -+ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { -+ current.checked_add(1) -+ }) -+ .expect("open-file-description identity space exhausted"); -+ Arc::new(Self { -+ id, -+ mutable: Mutex::new(OpenFileDescriptionMutable::default()), -+ interest_fanout: OnceLock::new(), -+ directory_sync, -+ }) -+ } -+ -+ pub(crate) fn directory_sync(&self) -> Option> { -+ self.directory_sync.clone() -+ } -+ -+ pub(crate) fn id(&self) -> u64 { -+ self.id -+ } -+ -+ #[cfg(test)] -+ pub(crate) fn descriptor_count(&self) -> u32 { -+ self.mutable.lock().unwrap().descriptor_count -+ } -+ -+ pub(crate) fn interest_fanout(&self) -> InterestHandlerFanout { -+ self.interest_fanout -+ .get_or_init(InterestHandlerFanout::default) -+ .clone() -+ } -+ -+ #[cfg(test)] -+ pub(crate) fn interest_fanout_initialized(&self) -> bool { -+ self.interest_fanout.get().is_some() -+ } -+ -+ #[cfg(test)] -+ pub(crate) fn epoll_registration_count(&self) -> usize { -+ self.mutable.lock().unwrap().registrations.len() -+ } -+ -+ pub(crate) fn acquire_descriptor(&self) { -+ let mut mutable = self.mutable.lock().unwrap(); -+ mutable.descriptor_count = mutable -+ .descriptor_count -+ .checked_add(1) -+ .expect("open-file-description descriptor count overflow"); -+ } -+ -+ /// Releases one user-visible descriptor. Reverse epoll records are drained -+ /// under the OFD lock and applied only after releasing it, avoiding an -+ /// OFD-registry -> epoll-state lock-order dependency. -+ pub(crate) fn release_descriptor(&self) -> bool { -+ let registrations = { -+ let mut mutable = self.mutable.lock().unwrap(); -+ mutable.descriptor_count = mutable -+ .descriptor_count -+ .checked_sub(1) -+ .expect("open-file-description descriptor dropped too many times"); -+ if mutable.descriptor_count != 0 { -+ return false; -+ } -+ std::mem::take(&mut mutable.registrations) -+ }; -+ -+ for registration in registrations.into_values() { -+ let Some(state) = registration.state.upgrade() else { -+ continue; -+ }; -+ let Some(subscription) = registration.subscription.upgrade() else { -+ continue; -+ }; -+ state.remove_if_same(registration.key, &subscription); -+ } -+ true -+ } -+ -+ pub(crate) fn register_epoll( -+ self: &Arc, -+ state: &Arc, -+ subscription: &Arc, -+ key: EpollSubscriptionKey, -+ ) -> Option { -+ let mut mutable = self.mutable.lock().unwrap(); -+ if mutable.descriptor_count == 0 { -+ return None; -+ } -+ -+ let registration_id = mutable.next_registration_id; -+ mutable.next_registration_id = mutable -+ .next_registration_id -+ .checked_add(1) -+ .expect("epoll reverse-registration identity space exhausted"); -+ mutable.registrations.insert( -+ registration_id, -+ EpollReverseRegistration { -+ state: Arc::downgrade(state), -+ subscription: Arc::downgrade(subscription), -+ key, -+ }, -+ ); -+ Some(EpollRegistrationGuard { -+ description: Arc::downgrade(self), -+ registration_id, -+ }) -+ } -+} -+ -+#[cfg(feature = "enable-serde")] -+fn default_open_file_description() -> Arc { -+ OpenFileDescription::new() -+} -+ -+#[derive(Debug)] -+pub(crate) struct EpollRegistrationGuard { -+ description: Weak, -+ registration_id: u64, -+} -+ -+impl Drop for EpollRegistrationGuard { -+ fn drop(&mut self) { -+ let Some(description) = self.description.upgrade() else { -+ return; -+ }; -+ description -+ .mutable -+ .lock() -+ .unwrap() -+ .registrations -+ .remove(&self.registration_id); -+ } -+} -+ - #[derive(Debug, Clone)] - #[cfg_attr(feature = "enable-serde", derive(Serialize, Deserialize))] - pub struct Fd { -@@ -38,7 +221,24 @@ pub struct FdInner { - pub rights_inheriting: Rights, - pub flags: Fdflags, // This is file table related flags, not fd flags - pub offset: Arc, // This also belongs in the file table -- pub fd_flags: Fdflagsext, // This is the actual FD flags that belongs here -+ /// Identity and lifecycle of the underlying open file description. This -+ /// is shared by dup/fork but not by independent opens of the same inode. -+ /// -+ /// `enable-serde` currently reconstructs this skipped field independently -+ /// for every deserialized `Fd`; it therefore does not preserve dup/fork -+ /// OFD topology, reverse epoll registrations, or live directory-sync -+ /// handles. A topology-aware snapshot representation is required before -+ /// restored descriptors can make the same lifecycle/durability guarantees -+ /// as live descriptors. The sealed PostgreSQL runtime does not enable -+ /// journaling; enabling it remains a release blocker for this path. -+ #[cfg_attr( -+ feature = "enable-serde", -+ serde(skip, default = "default_open_file_description") -+ )] -+ pub(crate) ofd: Arc, -+ #[cfg_attr(feature = "enable-serde", serde(skip, default))] -+ pub readdir_cache: ReaddirCache, -+ pub fd_flags: Fdflagsext, // This is the actual FD flags that belongs here - } - - impl Fd { -@@ -59,6 +259,19 @@ impl Fd { - /// - /// This permission is currently unused when deserializing. - pub const CREATE: u16 = 16; -+ -+ pub(crate) fn acquire_descriptor(&self) { -+ self.inner.ofd.acquire_descriptor(); -+ self.inode.acquire_handle(); -+ } -+ -+ pub(crate) fn release_descriptor(&self) { -+ let last_description_handle = self.inner.ofd.release_descriptor(); -+ if last_description_handle && let Kind::Epoll { state } = &*self.inode.read() { -+ state.close(); -+ } -+ self.inode.drop_one_handle(); -+ } - } - - /// A file that Wasi knows about that may or may not be open -diff --git a/lib/wasix/src/fs/fd_list.rs b/lib/wasix/src/fs/fd_list.rs -index 3540a55..f6d7c2b 100644 ---- a/lib/wasix/src/fs/fd_list.rs -+++ b/lib/wasix/src/fs/fd_list.rs -@@ -65,7 +65,7 @@ impl FdList { - } - - pub fn insert_first_free(&mut self, fd: Fd) -> WasiFd { -- fd.inode.acquire_handle(); -+ fd.acquire_descriptor(); - match self.first_free { - Some(free) => { - assert!(self.fds[free].is_none()); -@@ -107,7 +107,7 @@ impl FdList { - // If there's a hole but its index is too low, we need to search - Some(_) => { - // This is handled by insert or insert_first_free in every other case, but not this one -- fd.inode.acquire_handle(); -+ fd.acquire_descriptor(); - - match self.first_free_after(after_or_equal) { - // Found a suitable hole, and it's guaranteed to not be the first since -@@ -138,6 +138,25 @@ impl FdList { - } - - pub fn insert(&mut self, exclusive: bool, idx: WasiFd, fd: Fd) -> bool { -+ match self.insert_deferred(exclusive, idx, fd) { -+ Ok(displaced) => { -+ if let Some(displaced) = displaced { -+ displaced.release_descriptor(); -+ } -+ true -+ } -+ Err(_) => false, -+ } -+ } -+ -+ /// Inserts without finalizing a displaced descriptor. Callers holding the -+ /// outer fd-map lock must drop that lock before releasing the returned fd. -+ pub(crate) fn insert_deferred( -+ &mut self, -+ exclusive: bool, -+ idx: WasiFd, -+ fd: Fd, -+ ) -> Result, Fd> { - let idx = idx as usize; - - if self.fds.len() <= idx { -@@ -155,51 +174,92 @@ impl FdList { - self.fds.resize(idx + 1, None); - } - -- if let Some(ref prev_fd) = self.fds[idx] { -- if exclusive { -- return false; -- } else { -- prev_fd.inode.drop_one_handle(); -- } -+ if self.fds[idx].is_some() && exclusive { -+ return Err(fd); - } - -- fd.inode.acquire_handle(); -+ let displaced = self.fds[idx].take(); -+ fd.acquire_descriptor(); - self.fds[idx] = Some(fd); - - if self.first_free == Some(idx) { - self.first_free = self.first_free_after(idx as WasiFd + 1); - } - -- true -+ Ok(displaced) - } - - pub fn remove(&mut self, idx: WasiFd) -> Option { -+ let result = self.remove_deferred(idx); -+ if let Some(fd) = result.as_ref() { -+ fd.release_descriptor(); -+ } -+ result -+ } -+ -+ /// Removes without running last-descriptor callbacks under an outer -+ /// fd-map lock. -+ pub(crate) fn remove_deferred(&mut self, idx: WasiFd) -> Option { - let idx = idx as usize; - - let result = self.fds.get_mut(idx).and_then(|fd| fd.take()); - -- if let Some(fd) = result.as_ref() { -+ if result.is_some() { - match self.first_free { - None => self.first_free = Some(idx), - Some(x) if x > idx => self.first_free = Some(idx), - _ => (), - } -- -- fd.inode.drop_one_handle(); - } - - result - } - -+ /// Atomically moves `from` to `to` without changing the moved open-file -+ /// description's descriptor count. The displaced target is returned so -+ /// its final-close callbacks can run after the caller releases the outer -+ /// fd-map lock. -+ /// -+ /// Unlike remove followed by a normal insert, this deliberately does not -+ /// release and reacquire the source descriptor. That preserves both WASI -+ /// renumber semantics and OFD lifetime across the move. -+ pub(crate) fn renumber_deferred(&mut self, from: WasiFd, to: WasiFd) -> Result, ()> { -+ if from == to { -+ return self.get(from).map(|_| None).ok_or(()); -+ } -+ -+ let moved = self.remove_deferred(from).ok_or(())?; -+ let displaced = self.remove_deferred(to); -+ self.insert_moved(to, moved); -+ Ok(displaced) -+ } -+ -+ /// Inserts an fd whose descriptor ownership is already accounted for. -+ fn insert_moved(&mut self, idx: WasiFd, fd: Fd) { -+ let idx = idx as usize; -+ if self.fds.len() <= idx { -+ self.fds.resize(idx + 1, None); -+ } -+ debug_assert!(self.fds[idx].is_none()); -+ self.fds[idx] = Some(fd); -+ -+ if self.first_free == Some(idx) { -+ self.first_free = self.first_free_after(idx as WasiFd + 1); -+ } -+ } -+ - pub fn clear(&mut self) { -- for fd in &self.fds { -- if let Some(fd) = fd.as_ref() { -- fd.inode.drop_one_handle(); -- } -+ for fd in self.drain_deferred() { -+ fd.release_descriptor(); - } -+ } - -+ /// Drains the map without finalizing descriptor lifecycles. -+ pub(crate) fn drain_deferred(&mut self) -> Vec { -+ let drained = self.fds.iter_mut().filter_map(Option::take).collect(); - self.fds.clear(); - self.first_free = None; -+ drained - } - - pub fn iter(&self) -> FdListIterator<'_> { -@@ -225,7 +285,7 @@ impl Clone for FdList { - fn clone(&self) -> Self { - for fd in &self.fds { - if let Some(fd) = fd.as_ref() { -- fd.inode.acquire_handle(); -+ fd.acquire_descriptor(); - } - } - -@@ -292,27 +352,53 @@ impl<'a> Iterator for FdListIteratorMut<'a> { - mod tests { - use std::{ - borrow::Cow, -+ io, - sync::{ - Arc, RwLock, -- atomic::{AtomicI32, AtomicU64}, -+ atomic::{AtomicI32, AtomicU64, AtomicUsize, Ordering}, - }, - }; - - use assert_panic::assert_panic; -- use wasmer_wasix_types::wasi::{Fdflags, Fdflagsext, Rights}; -+ use wasmer_wasix_types::wasi::{EpollEventCtl, EpollType, Fdflags, Fdflagsext, Rights}; - -- use crate::fs::{Inode, InodeGuard, InodeVal, Kind, fd::FdInner}; -+ use crate::fs::{ -+ Inode, InodeGuard, InodeVal, Kind, -+ fd::{DirectorySyncHandle, FdInner, OpenFileDescription}, -+ }; -+ use crate::os::epoll::{EpollState, EpollSubscriptionKey}; - - use super::{Fd, FdList, WasiFd}; -+ use virtual_fs::VirtualDirectory; -+ -+ #[derive(Debug)] -+ struct CountingDirectory { -+ syncs: Arc, -+ } -+ -+ impl VirtualDirectory for CountingDirectory { -+ fn has_blocking_sync_all_to_disk(&self) -> bool { -+ true -+ } -+ -+ fn sync_all_to_disk_blocking(&self) -> io::Result<()> { -+ self.syncs.fetch_add(1, Ordering::Relaxed); -+ Ok(()) -+ } -+ } - - fn useless_fd(n: u16) -> Fd { -+ fd_with_kind(n, Kind::Buffer { buffer: vec![] }) -+ } -+ -+ fn fd_with_kind(n: u16, kind: Kind) -> Fd { - Fd { - open_flags: 0, - inode: InodeGuard { - ino: Inode(0), - inner: Arc::new(InodeVal { - is_preopened: false, -- kind: RwLock::new(Kind::Buffer { buffer: vec![] }), -+ kind: RwLock::new(kind), - name: RwLock::new(Cow::Borrowed("")), - stat: RwLock::new(Default::default()), - }), -@@ -321,9 +407,11 @@ mod tests { - is_stdio: false, - inner: FdInner { - offset: Arc::new(AtomicU64::new(0)), -+ ofd: OpenFileDescription::new(), - rights: Rights::empty(), - rights_inheriting: Rights::empty(), - flags: Fdflags::from_bits_preserve(n), -+ readdir_cache: Default::default(), - fd_flags: Fdflagsext::empty(), - }, - } -@@ -681,13 +769,354 @@ mod tests { - l2.clear(); - assert_eq!(fd0.inode.handle_count(), 0); - -+ // The descriptor now retains a trait-object directory handle. That -+ // handle deliberately makes the full descriptor conservative across -+ // unwind boundaries; this test is explicitly exercising a poisoned -+ // accounting invariant, so acknowledge the boundary locally instead -+ // of asserting an unwind-safety contract for every filesystem. -+ let fd0 = std::panic::AssertUnwindSafe(fd0); - assert_panic!( -- fd0.inode.drop_one_handle(), -+ fd0.0.inode.drop_one_handle(), - &str, - "InodeGuard handle dropped too many times" - ); - -- assert_panic!(drop(fd0.inode.write()), String, contains "PoisonError"); -+ assert_panic!(drop(fd0.0.inode.write()), String, contains "PoisonError"); -+ } -+ -+ #[test] -+ fn untouched_open_file_description_does_not_initialize_interest_fanout() { -+ let mut list = FdList::new(); -+ let fd = useless_fd(0); -+ let description = fd.inner.ofd.clone(); -+ -+ assert!(!description.interest_fanout_initialized()); -+ list.insert_first_free(fd); -+ assert!(!description.interest_fanout_initialized()); -+ list.remove(0).unwrap(); -+ assert!(!description.interest_fanout_initialized()); -+ } -+ -+ #[test] -+ fn remove_deferred_postpones_descriptor_finalization() { -+ let mut list = FdList::new(); -+ let fd = useless_fd(0); -+ let description = fd.inner.ofd.clone(); -+ list.insert_first_free(fd); -+ assert_eq!(description.descriptor_count(), 1); -+ -+ let removed = list.remove_deferred(0).unwrap(); -+ assert_eq!( -+ description.descriptor_count(), -+ 1, -+ "fd-map mutation must not run lifecycle callbacks while its outer lock is held" -+ ); -+ removed.release_descriptor(); -+ assert_eq!(description.descriptor_count(), 0); -+ } -+ -+ #[test] -+ fn renumber_is_an_atomic_move_and_preserves_source_descriptor_ownership() { -+ let mut list = FdList::new(); -+ let source = useless_fd(10); -+ let source_description = source.inner.ofd.clone(); -+ let target = useless_fd(20); -+ let target_description = target.inner.ofd.clone(); -+ assert!(list.insert(true, 3, source)); -+ assert!(list.insert(true, 8, target)); -+ assert_eq!(source_description.descriptor_count(), 1); -+ assert_eq!(target_description.descriptor_count(), 1); -+ -+ let displaced = list.renumber_deferred(3, 8).unwrap().unwrap(); -+ -+ assert!(list.get(3).is_none()); -+ assert!(is_useless_fd(list.get(8).unwrap(), 10)); -+ assert_eq!(source_description.descriptor_count(), 1); -+ assert_eq!(target_description.descriptor_count(), 1); -+ displaced.release_descriptor(); -+ assert_eq!(target_description.descriptor_count(), 0); -+ } -+ -+ #[test] -+ fn renumber_invalid_and_same_fd_leave_the_table_unchanged() { -+ let mut list = FdList::new(); -+ let fd = useless_fd(30); -+ let description = fd.inner.ofd.clone(); -+ assert!(list.insert(true, 4, fd)); -+ -+ assert!(list.renumber_deferred(99, 4).is_err()); -+ assert!(is_useless_fd(list.get(4).unwrap(), 30)); -+ assert_eq!(description.descriptor_count(), 1); -+ -+ assert!(list.renumber_deferred(4, 4).unwrap().is_none()); -+ assert!(is_useless_fd(list.get(4).unwrap(), 30)); -+ assert_eq!(description.descriptor_count(), 1); -+ assert!(list.renumber_deferred(99, 99).is_err()); -+ } -+ -+ #[test] -+ fn directory_sync_handle_survives_fork_clone_renumber_and_nonfinal_close() { -+ let syncs = Arc::new(AtomicUsize::new(0)); -+ let directory = Arc::new(CountingDirectory { -+ syncs: syncs.clone(), -+ }); -+ let weak_directory = Arc::downgrade(&directory); -+ let handle: DirectorySyncHandle = directory; -+ -+ let mut fd = useless_fd(50); -+ fd.inner.ofd = OpenFileDescription::new_with_directory_sync(Some(Ok(handle))); -+ let description = fd.inner.ofd.clone(); -+ let mut parent = FdList::new(); -+ assert!(parent.insert(true, 4, fd)); -+ let mut child = parent.clone(); -+ assert_eq!(description.descriptor_count(), 2); -+ -+ { -+ let parent_handle = parent -+ .get(4) -+ .unwrap() -+ .inner -+ .ofd -+ .directory_sync() -+ .unwrap() -+ .unwrap(); -+ parent_handle.sync_all_to_disk_blocking().unwrap(); -+ } -+ assert!(child.renumber_deferred(4, 9).unwrap().is_none()); -+ parent.remove(4).unwrap(); -+ assert_eq!(description.descriptor_count(), 1); -+ -+ { -+ let moved_handle = child -+ .get(9) -+ .unwrap() -+ .inner -+ .ofd -+ .directory_sync() -+ .unwrap() -+ .unwrap(); -+ moved_handle.sync_all_to_disk_blocking().unwrap(); -+ } -+ assert_eq!(syncs.load(Ordering::Relaxed), 2); -+ -+ child.remove(9).unwrap(); -+ assert_eq!(description.descriptor_count(), 0); -+ assert!(weak_directory.upgrade().is_some()); -+ drop(description); -+ assert!(weak_directory.upgrade().is_none()); -+ } -+ -+ #[test] -+ fn renumber_keeps_epoll_watch_until_moved_descriptor_final_close() { -+ let mut list = FdList::new(); -+ let fd = useless_fd(40); -+ let description = fd.inner.ofd.clone(); -+ assert!(list.insert(true, 5, fd)); -+ -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(5, description.id()); -+ let (_, subscription) = state.prepare_add(key, &readable_event(5)).unwrap(); -+ let registration = description -+ .register_epoll(&state, &subscription, key) -+ .unwrap(); -+ subscription -+ .attach_close_registration(registration) -+ .unwrap(); -+ -+ assert!(list.renumber_deferred(5, 9).unwrap().is_none()); -+ assert!(list.get(5).is_none()); -+ assert!(list.get(9).is_some()); -+ assert_eq!(description.descriptor_count(), 1); -+ assert!(state.contains_exact_subscription(key, &subscription)); -+ -+ list.remove(9).unwrap(); -+ assert_eq!(description.descriptor_count(), 0); -+ assert!(!state.contains_exact_subscription(key, &subscription)); -+ assert!(!subscription.is_active()); -+ } -+ -+ fn readable_event(fd: WasiFd) -> EpollEventCtl { -+ EpollEventCtl { -+ events: EpollType::EPOLLIN, -+ ptr: 0, -+ fd, -+ data1: 0, -+ data2: 0, -+ } -+ } -+ -+ #[test] -+ fn epoll_watch_survives_nonfinal_dup_close_and_detaches_on_final_close() { -+ let mut parent = FdList::new(); -+ let fd = useless_fd(0); -+ let description = fd.inner.ofd.clone(); -+ parent.insert_first_free(fd); -+ let mut child = parent.clone(); -+ assert_eq!(description.descriptor_count(), 2); -+ -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(0, description.id()); -+ let (_, subscription) = state.prepare_add(key, &readable_event(0)).unwrap(); -+ let registration = description -+ .register_epoll(&state, &subscription, key) -+ .unwrap(); -+ subscription -+ .attach_close_registration(registration) -+ .unwrap(); -+ -+ parent.remove(0).unwrap(); -+ assert_eq!(description.descriptor_count(), 1); -+ assert!(state.contains_exact_subscription(key, &subscription)); -+ assert!(subscription.is_active()); -+ -+ child.remove(0).unwrap(); -+ assert_eq!(description.descriptor_count(), 0); -+ assert!(!state.contains_exact_subscription(key, &subscription)); -+ assert!(!subscription.is_active()); -+ assert_eq!(description.epoll_registration_count(), 0); -+ } -+ -+ #[test] -+ fn final_close_between_reverse_registration_and_attach_rejects_add() { -+ let mut list = FdList::new(); -+ let fd = useless_fd(0); -+ let description = fd.inner.ofd.clone(); -+ list.insert_first_free(fd); -+ -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(0, description.id()); -+ let (_, subscription) = state.prepare_add(key, &readable_event(0)).unwrap(); -+ let registration = description -+ .register_epoll(&state, &subscription, key) -+ .unwrap(); -+ -+ // This is the ADD/final-close race window: reverse registration won -+ // the OFD mutex, then the last descriptor closes before ADD can attach -+ // the ownership guard to the subscription. -+ list.remove(0).unwrap(); -+ assert_eq!(description.descriptor_count(), 0); -+ assert!(!state.contains_exact_subscription(key, &subscription)); -+ assert!(!subscription.is_active()); -+ assert_eq!(description.epoll_registration_count(), 0); -+ -+ assert!( -+ subscription -+ .attach_close_registration(registration) -+ .is_err(), -+ "a final-close winner must make a late ADD attachment fail" -+ ); -+ assert_eq!(description.epoll_registration_count(), 0); -+ } -+ -+ #[test] -+ fn reused_fd_final_old_alias_close_removes_only_old_ofd_watch() { -+ let mut old_primary = FdList::new(); -+ let old_fd = useless_fd(0); -+ let old_description = old_fd.inner.ofd.clone(); -+ assert!(old_primary.insert(true, 5, old_fd)); -+ let mut old_alias = old_primary.clone(); -+ -+ let state = Arc::new(EpollState::new()); -+ let old_key = EpollSubscriptionKey::new(5, old_description.id()); -+ let (_, old_subscription) = state.prepare_add(old_key, &readable_event(5)).unwrap(); -+ let old_registration = old_description -+ .register_epoll(&state, &old_subscription, old_key) -+ .unwrap(); -+ old_subscription -+ .attach_close_registration(old_registration) -+ .unwrap(); -+ -+ old_primary.remove(5).unwrap(); -+ assert_eq!(old_description.descriptor_count(), 1); -+ -+ let mut reused_table = FdList::new(); -+ let new_fd = useless_fd(1); -+ let new_description = new_fd.inner.ofd.clone(); -+ assert!(reused_table.insert(true, 5, new_fd)); -+ let new_key = EpollSubscriptionKey::new(5, new_description.id()); -+ let (_, new_subscription) = state.prepare_add(new_key, &readable_event(5)).unwrap(); -+ let new_registration = new_description -+ .register_epoll(&state, &new_subscription, new_key) -+ .unwrap(); -+ new_subscription -+ .attach_close_registration(new_registration) -+ .unwrap(); -+ -+ assert_eq!(state.subscription_count(), 2); -+ old_alias.remove(5).unwrap(); -+ -+ assert_eq!(old_description.descriptor_count(), 0); -+ assert!(!state.contains_exact_subscription(old_key, &old_subscription)); -+ assert!(!old_subscription.is_active()); -+ assert!(state.contains_exact_subscription(new_key, &new_subscription)); -+ assert!(new_subscription.is_active()); -+ assert_eq!(new_description.epoll_registration_count(), 1); -+ -+ state.apply_del(new_key).unwrap(); -+ assert_eq!(new_description.epoll_registration_count(), 0); -+ assert!(!new_subscription.is_active()); -+ assert_eq!(new_description.descriptor_count(), 1); -+ -+ reused_table.remove(5).unwrap(); -+ assert_eq!(new_description.descriptor_count(), 0); -+ } -+ -+ #[test] -+ fn epoll_ofd_close_is_idempotent_and_waits_for_final_alias() { -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(44, 4044); -+ let (_, subscription) = state.prepare_add(key, &readable_event(44)).unwrap(); -+ let mut first = FdList::new(); -+ first.insert_first_free(fd_with_kind( -+ 0, -+ Kind::Epoll { -+ state: state.clone(), -+ }, -+ )); -+ let mut alias = first.clone(); -+ -+ first.remove(0).unwrap(); -+ assert!(!state.is_closed()); -+ assert!(state.contains_exact_subscription(key, &subscription)); -+ -+ alias.remove(0).unwrap(); -+ assert!(state.is_closed()); -+ assert_eq!(state.subscription_count(), 0); -+ assert!(!subscription.is_active()); -+ state.close(); -+ assert!(state.is_closed()); -+ } -+ -+ #[test] -+ fn closing_epoll_does_not_close_watched_open_file_description() { -+ let mut watched = FdList::new(); -+ let watched_fd = useless_fd(0); -+ let watched_description = watched_fd.inner.ofd.clone(); -+ watched.insert_first_free(watched_fd); -+ -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(0, watched_description.id()); -+ let (_, subscription) = state.prepare_add(key, &readable_event(0)).unwrap(); -+ let registration = watched_description -+ .register_epoll(&state, &subscription, key) -+ .unwrap(); -+ subscription -+ .attach_close_registration(registration) -+ .unwrap(); -+ -+ let mut epoll_fds = FdList::new(); -+ epoll_fds.insert_first_free(fd_with_kind( -+ 1, -+ Kind::Epoll { -+ state: state.clone(), -+ }, -+ )); -+ epoll_fds.remove(0).unwrap(); -+ -+ assert!(state.is_closed()); -+ assert_eq!(watched_description.descriptor_count(), 1); -+ assert_eq!(watched_description.epoll_registration_count(), 0); -+ assert!(watched.get(0).is_some()); - } - - #[test] -@@ -700,6 +1129,9 @@ mod tests { - let fd = l.get(0).unwrap(); - fd.inode.drop_one_handle(); - -+ // See `open_handles_are_updated_correctly`: only this deliberate -+ // invariant-panic test crosses the unwind boundary. -+ let l = std::panic::AssertUnwindSafe(l); - assert_panic!(drop(l), &str, "InodeGuard handle dropped too many times"); - } - } -diff --git a/lib/wasix/src/fs/inode_guard.rs b/lib/wasix/src/fs/inode_guard.rs -index 61ec5f6..9130b6b 100644 ---- a/lib/wasix/src/fs/inode_guard.rs -+++ b/lib/wasix/src/fs/inode_guard.rs -@@ -78,6 +78,26 @@ impl InodeValFilePollGuard { - subscription, - }) - } -+ -+ pub(crate) fn poll_immediate_ready(&self) -> heapless::Vec { -+ let mut join = InodeValFilePollGuardJoin { -+ mode: self.mode.clone(), -+ fd: self.fd, -+ peb: self.peb, -+ subscription: self.subscription, -+ }; -+ let waker = futures::task::noop_waker(); -+ let mut cx = Context::from_waker(&waker); -+ let mut ret = heapless::Vec::new(); -+ -+ if let Poll::Ready(events) = Future::poll(Pin::new(&mut join), &mut cx) { -+ for (_, readiness) in events { -+ ret.push(readiness).ok(); -+ } -+ } -+ -+ ret -+ } - } - - impl std::fmt::Debug for InodeValFilePollGuard { -@@ -165,7 +185,6 @@ impl Future for InodeValFilePollGuardJoin { - let mut has_write = false; - let mut has_close = false; - let mut has_hangup = false; -- - let mut ret = heapless::Vec::new(); - for in_event in iterate_poll_events(self.peb) { - match in_event { -diff --git a/lib/wasix/src/fs/mod.rs b/lib/wasix/src/fs/mod.rs -index 508dbef..35ad3b6 100644 ---- a/lib/wasix/src/fs/mod.rs -+++ b/lib/wasix/src/fs/mod.rs -@@ -46,6 +46,7 @@ use wasmer_wasix_types::{ - }, - }; - -+pub(crate) use self::fd::{DirectorySyncHandle, EpollRegistrationGuard, OpenFileDescription}; - pub use self::fd::{Fd, FdInner, InodeVal, Kind}; - pub(crate) use self::inode_guard::{ - InodeValFilePollGuard, InodeValFilePollGuardJoin, InodeValFilePollGuardMode, -@@ -134,6 +135,18 @@ pub struct InodeGuard { - // in the backing file (which may be a host file) getting closed. - open_handles: Arc, - } -+ -+#[derive(Debug)] -+pub(crate) struct InodeHandleReservation { -+ inode: InodeGuard, -+} -+ -+impl Drop for InodeHandleReservation { -+ fn drop(&mut self) { -+ self.inode.drop_one_handle(); -+ } -+} -+ - impl InodeGuard { - pub fn ino(&self) -> Inode { - self.ino -@@ -160,6 +173,17 @@ impl InodeGuard { - trace!(ino = %self.ino.0, new_count = %(prev_handles + 1), "acquiring handle for InodeGuard"); - } - -+ /// Keeps an inode's backing handle alive while a new descriptor is being -+ /// published. Call this while holding the inode lock that observed or -+ /// installed the handle so a concurrent final close cannot clear it in the -+ /// gap before `FdList` acquires the descriptor's permanent reference. -+ pub(crate) fn reserve_handle(&self) -> InodeHandleReservation { -+ self.acquire_handle(); -+ InodeHandleReservation { -+ inode: self.clone(), -+ } -+ } -+ - pub fn drop_one_handle(&self) { - let prev_handles = self.open_handles.fetch_sub(1, Ordering::SeqCst); - -@@ -516,6 +540,13 @@ impl FileSystem for WasiFsRoot { - self.root.remove_file(path) - } - -+ fn open_dir( -+ &self, -+ path: &Path, -+ ) -> virtual_fs::Result> { -+ self.root.open_dir(path) -+ } -+ - fn new_open_options(&self) -> OpenOptions<'_> { - self.root.new_open_options() - } -@@ -547,6 +578,12 @@ pub struct WasiFs { - pub(crate) init_vfs_preopens: Vec, - } - -+enum SyncTarget { -+ File(Arc>>), -+ Directory(DirectorySyncHandle), -+ Buffer, -+} -+ - impl WasiFs { - fn writable_package_mount( - fs: Arc, -@@ -674,10 +711,16 @@ impl WasiFs { - } - }); - -- if let Ok(mut map) = self.fd_map.write() { -- for fd in &to_close { -- map.remove(*fd); -- } -+ let removed = if let Ok(mut map) = self.fd_map.write() { -+ to_close -+ .iter() -+ .filter_map(|fd| map.remove_deferred(*fd)) -+ .collect::>() -+ } else { -+ Vec::new() -+ }; -+ for fd in removed { -+ fd.release_descriptor(); - } - } - -@@ -700,8 +743,13 @@ impl WasiFs { - } - }); - -- if let Ok(mut map) = self.fd_map.write() { -- map.clear(); -+ let removed = if let Ok(mut map) = self.fd_map.write() { -+ map.drain_deferred() -+ } else { -+ Vec::new() -+ }; -+ for fd in removed { -+ fd.release_descriptor(); - } - } - -@@ -1134,6 +1182,69 @@ impl WasiFs { - // loading inodes as necessary - 'symlink_resolution: while symlink_count < MAX_SYMLINKS { - let processing_cur_inode = cur_inode.clone(); -+ let component_name = component.as_os_str().to_string_lossy(); -+ -+ // The common path is walking directories already cached in the -+ // WASIX inode tree. Keep those lookups on shared locks and only -+ // fall through to the write-locked lazy-load path on misses. -+ { -+ let guard = processing_cur_inode.read(); -+ match guard.deref() { -+ Kind::Dir { -+ entries, -+ path, -+ parent, -+ .. -+ } => { -+ match component_name.borrow() { -+ ".." => { -+ if let Some(p) = parent.upgrade() { -+ cur_inode = p; -+ continue 'path_iter; -+ } else { -+ return Err(Errno::Access); -+ } -+ } -+ "." => continue 'path_iter, -+ _ => (), -+ } -+ -+ if let Some(entry) = entries.get(component_name.as_ref()) { -+ cur_inode = entry.clone(); -+ break 'symlink_resolution; -+ } -+ -+ let file = { -+ let mut cd = path.clone(); -+ cd.push(component); -+ cd -+ }; -+ if self.ephemeral_symlink_at(&file).is_none() -+ && self.root_fs.symlink_metadata(&file).is_err() -+ { -+ return Err(Errno::Noent); -+ } -+ } -+ Kind::Root { entries } => { -+ match component { -+ Component::ParentDir | Component::CurDir => continue 'path_iter, -+ _ => {} -+ } -+ -+ if let Some(entry) = entries.get(component_name.as_ref()) { -+ cur_inode = entry.clone(); -+ break 'symlink_resolution; -+ } else if let Some(root) = entries.get(&"/".to_string()) { -+ cur_inode = root.clone(); -+ continue 'symlink_resolution; -+ } else { -+ return Err(Errno::Notcapable); -+ } -+ } -+ _ => {} -+ } -+ } -+ - let mut guard = processing_cur_inode.write(); - match guard.deref_mut() { - Kind::Buffer { .. } => unimplemented!("state::get_inode_at_path for buffers"), -@@ -1170,6 +1281,7 @@ impl WasiFs { - // we want to insert newly opened dirs and files, but not transient symlinks - // TODO: explain why (think about this deeply when well rested) - let should_insert; -+ let stat; - - let kind = if let Some((base_po_dir, path_to_symlink, relative_path)) = - self.ephemeral_symlink_at(&file) -@@ -1179,6 +1291,7 @@ impl WasiFs { - should_insert = false; - loop_for_symlink = true; - symlink_count += 1; -+ stat = Filestat::default(); - Kind::Symlink { - base_po_dir, - path_to_symlink, -@@ -1191,6 +1304,16 @@ impl WasiFs { - .ok() - .ok_or(Errno::Noent)?; - let file_type = metadata.file_type(); -+ stat = Filestat { -+ st_filetype: virtual_file_type_to_wasi_file_type( -+ file_type.clone(), -+ ), -+ st_size: metadata.len(), -+ st_ctim: metadata.created(), -+ st_mtim: metadata.modified(), -+ st_atim: metadata.accessed(), -+ ..Filestat::default() -+ }; - if file_type.is_dir() { - should_insert = true; - // load DIR -@@ -1286,12 +1409,13 @@ impl WasiFs { - }; - drop(guard); - -- let new_inode = self.create_inode( -+ let new_inode = self.create_inode_with_stat( - inodes, - kind, - false, -- file.to_string_lossy().to_string(), -- )?; -+ file.to_string_lossy().to_string().into(), -+ stat, -+ ); - if should_insert { - let mut guard = processing_cur_inode.write(); - if let Kind::Dir { entries, .. } = guard.deref_mut() { -@@ -1532,6 +1656,8 @@ impl WasiFs { - rights_inheriting: ALL_RIGHTS, - flags: Fdflags::empty(), - offset: Arc::new(AtomicU64::new(0)), -+ ofd: OpenFileDescription::new(), -+ readdir_cache: Default::default(), - fd_flags: Fdflagsext::empty(), - }, - open_flags: 0, -@@ -1556,9 +1682,32 @@ impl WasiFs { - .map(|a| a.inode.clone()) - } - -+ /// Return the current backing-file length and refresh the inode cache. -+ /// Host-mounted files can be extended by another WASIX process, so the -+ /// per-inode snapshot is not authoritative for SEEK_END or filestat. -+ pub(crate) fn authoritative_fd_size(fd: &Fd) -> u64 { -+ let handle = { -+ let guard = fd.inode.read(); -+ match guard.deref() { -+ Kind::File { -+ handle: Some(handle), -+ .. -+ } => Some(handle.clone()), -+ _ => None, -+ } -+ }; -+ let Some(handle) = handle else { -+ return fd.inode.stat.read().unwrap().st_size; -+ }; -+ let size = handle.read().unwrap().size(); -+ fd.inode.stat.write().unwrap().st_size = size; -+ size -+ } -+ - pub fn filestat_fd(&self, fd: WasiFd) -> Result { -- let inode = self.get_fd_inode(fd)?; -- let guard = inode.stat.read().unwrap(); -+ let fd = self.get_fd(fd)?; -+ Self::authoritative_fd_size(&fd); -+ let guard = fd.inode.stat.read().unwrap(); - Ok(*guard.deref()) - } - -@@ -1650,6 +1799,24 @@ impl WasiFs { - } - } - -+ fn sync_target(fd: &Fd) -> Result { -+ let guard = fd.inode.read(); -+ match guard.deref() { -+ Kind::File { -+ handle: Some(file), .. -+ } => Ok(SyncTarget::File(file.clone())), -+ Kind::Dir { .. } => fd -+ .inner -+ .ofd -+ .directory_sync() -+ .ok_or(Errno::Notsup)? -+ .map(SyncTarget::Directory), -+ Kind::Root { .. } => Err(Errno::Notsup), -+ Kind::Buffer { .. } => Ok(SyncTarget::Buffer), -+ _ => Err(Errno::Io), -+ } -+ } -+ - #[allow(clippy::await_holding_lock)] - pub async fn flush(&self, fd: WasiFd) -> Result<(), Errno> { - match fd { -@@ -1702,6 +1869,108 @@ impl WasiFs { - Ok(()) - } - -+ #[allow(clippy::await_holding_lock)] -+ pub async fn sync(&self, fd: WasiFd, sync_metadata: bool) -> Result<(), Errno> { -+ match fd { -+ __WASI_STDIN_FILENO => (), -+ __WASI_STDOUT_FILENO => { -+ let mut file = -+ WasiInodes::stdout_mut(&self.fd_map).map_err(fs_error_into_wasi_err)?; -+ if sync_metadata { -+ file.sync_all_to_disk().await.map_err(map_io_err)?; -+ } else { -+ file.sync_data_to_disk().await.map_err(map_io_err)?; -+ } -+ } -+ __WASI_STDERR_FILENO => { -+ let mut file = -+ WasiInodes::stderr_mut(&self.fd_map).map_err(fs_error_into_wasi_err)?; -+ if sync_metadata { -+ file.sync_all_to_disk().await.map_err(map_io_err)?; -+ } else { -+ file.sync_data_to_disk().await.map_err(map_io_err)?; -+ } -+ } -+ _ => { -+ let fd = self.get_fd(fd)?; -+ let required_right = if sync_metadata { -+ Rights::FD_SYNC -+ } else { -+ Rights::FD_DATASYNC -+ }; -+ if !fd.inner.rights.contains(required_right) { -+ return Err(Errno::Access); -+ } -+ -+ let target = Self::sync_target(&fd)?; -+ drop(fd); -+ -+ match target { -+ SyncTarget::File(file) => { -+ let mut file = file.write().unwrap(); -+ if sync_metadata { -+ file.sync_all_to_disk().await.map_err(map_io_err)?; -+ } else { -+ file.sync_data_to_disk().await.map_err(map_io_err)?; -+ } -+ } -+ // A directory's entries are metadata, so fd_datasync must -+ // use the same full durability barrier as fd_sync. -+ SyncTarget::Directory(directory) => { -+ directory.sync_all_to_disk().await.map_err(map_io_err)?; -+ } -+ SyncTarget::Buffer => {} -+ } -+ } -+ } -+ Ok(()) -+ } -+ -+ pub fn sync_blocking(&self, fd: WasiFd, sync_metadata: bool) -> Result, Errno> { -+ match fd { -+ __WASI_STDIN_FILENO | __WASI_STDOUT_FILENO | __WASI_STDERR_FILENO => Ok(None), -+ _ => { -+ let fd = self.get_fd(fd)?; -+ let required_right = if sync_metadata { -+ Rights::FD_SYNC -+ } else { -+ Rights::FD_DATASYNC -+ }; -+ if !fd.inner.rights.contains(required_right) { -+ return Err(Errno::Access); -+ } -+ -+ let target = Self::sync_target(&fd)?; -+ drop(fd); -+ -+ match target { -+ SyncTarget::File(file) => { -+ let mut file = file.write().unwrap(); -+ if sync_metadata { -+ if !file.has_blocking_sync_all_to_disk() { -+ return Ok(None); -+ } -+ file.sync_all_to_disk_blocking().map_err(map_io_err)?; -+ } else { -+ if !file.has_blocking_sync_data_to_disk() { -+ return Ok(None); -+ } -+ file.sync_data_to_disk_blocking().map_err(map_io_err)?; -+ } -+ } -+ SyncTarget::Directory(directory) => { -+ if !directory.has_blocking_sync_all_to_disk() { -+ return Ok(None); -+ } -+ directory.sync_all_to_disk_blocking().map_err(map_io_err)?; -+ } -+ SyncTarget::Buffer => {} -+ } -+ Ok(Some(())) -+ } -+ } -+ } -+ - /// Creates an inode and inserts it given a Kind and some extra data - pub(crate) fn create_inode( - &self, -@@ -1841,6 +2110,21 @@ impl WasiFs { - idx: Option, - exclusive: bool, - ) -> Result { -+ let directory_sync: Option> = -+ if rights.contains(Rights::FD_SYNC) || rights.contains(Rights::FD_DATASYNC) { -+ let guard = inode.read(); -+ match guard.deref() { -+ Kind::Dir { path, .. } => Some( -+ self.root_fs -+ .open_dir(path) -+ .map(Arc::from) -+ .map_err(fs_error_into_wasi_err), -+ ), -+ _ => None, -+ } -+ } else { -+ None -+ }; - let is_stdio = matches!( - idx, - Some(__WASI_STDIN_FILENO) | Some(__WASI_STDOUT_FILENO) | Some(__WASI_STDERR_FILENO) -@@ -1851,6 +2135,8 @@ impl WasiFs { - rights_inheriting, - flags: fs_flags, - offset: Arc::new(AtomicU64::new(0)), -+ ofd: OpenFileDescription::new_with_directory_sync(directory_sync), -+ readdir_cache: Default::default(), - fd_flags, - }, - open_flags, -@@ -1858,17 +2144,20 @@ impl WasiFs { - is_stdio, - }; - -- let mut guard = self.fd_map.write().unwrap(); -- - match idx { - Some(idx) => { -- if guard.insert(exclusive, idx, fd) { -- Ok(idx) -- } else { -- Err(Errno::Exist) -+ let displaced = { -+ let mut guard = self.fd_map.write().unwrap(); -+ guard -+ .insert_deferred(exclusive, idx, fd) -+ .map_err(|_| Errno::Exist)? -+ }; -+ if let Some(displaced) = displaced { -+ displaced.release_descriptor(); - } -+ Ok(idx) - } -- None => Ok(guard.insert_first_free(fd)), -+ None => Ok(self.fd_map.write().unwrap().insert_first_free(fd)), - } - } - -@@ -1890,6 +2179,8 @@ impl WasiFs { - rights_inheriting: fd.inner.rights_inheriting, - flags: fd.inner.flags, - offset: fd.inner.offset.clone(), -+ ofd: fd.inner.ofd.clone(), -+ readdir_cache: fd.inner.readdir_cache.clone(), - fd_flags: match cloexec { - None => fd.inner.fd_flags, - Some(cloexec) => { -@@ -2201,23 +2492,30 @@ impl WasiFs { - kind: RwLock::new(kind), - }) - }; -- self.fd_map.write().unwrap().insert( -- false, -- raw_fd, -- Fd { -- inner: FdInner { -- rights, -- rights_inheriting: Rights::empty(), -- flags: fd_flags, -- offset: Arc::new(AtomicU64::new(0)), -- fd_flags: Fdflagsext::empty(), -- }, -- // since we're not calling open on this, we don't need open flags -- open_flags: 0, -- inode, -- is_stdio: true, -+ let new_fd = Fd { -+ inner: FdInner { -+ rights, -+ rights_inheriting: Rights::empty(), -+ flags: fd_flags, -+ offset: Arc::new(AtomicU64::new(0)), -+ ofd: OpenFileDescription::new(), -+ readdir_cache: Default::default(), -+ fd_flags: Fdflagsext::empty(), - }, -- ); -+ // since we're not calling open on this, we don't need open flags -+ open_flags: 0, -+ inode, -+ is_stdio: true, -+ }; -+ let displaced = self -+ .fd_map -+ .write() -+ .unwrap() -+ .insert_deferred(false, raw_fd, new_fd) -+ .expect("non-exclusive stdio insertion cannot fail"); -+ if let Some(displaced) = displaced { -+ displaced.release_descriptor(); -+ } - } - - pub fn get_stat_for_kind(&self, kind: &Kind) -> Result { -@@ -2291,23 +2589,22 @@ impl WasiFs { - - /// Closes an open FD, handling all details such as FD being preopen - pub(crate) fn close_fd(&self, fd: WasiFd) -> Result<(), Errno> { -- let mut fd_map = self.fd_map.write().unwrap(); -- -- let pfd = fd_map.remove(fd).ok_or(Errno::Badf); -- match pfd { -- Ok(fd_ref) => { -- let inode = fd_ref.inode.ino().as_u64(); -- let ref_cnt = fd_ref.inode.ref_cnt(); -- if ref_cnt == 1 { -- trace!(%fd, %inode, %ref_cnt, "closing file descriptor"); -- } else { -- trace!(%fd, %inode, %ref_cnt, "weakening file descriptor"); -- } -- } -- Err(err) => { -- trace!(%fd, "closing file descriptor failed - {}", err); -- } -+ let fd_ref = self -+ .fd_map -+ .write() -+ .unwrap() -+ .remove_deferred(fd) -+ .ok_or(Errno::Badf)?; -+ let inode = fd_ref.inode.ino().as_u64(); -+ let ref_cnt = fd_ref.inode.ref_cnt(); -+ if ref_cnt == 1 { -+ trace!(%fd, %inode, %ref_cnt, "closing file descriptor"); -+ } else { -+ trace!(%fd, %inode, %ref_cnt, "weakening file descriptor"); - } -+ // Finalization may acquire OFD, epoll, socket, and selector locks. The -+ // fd-map write guard above is deliberately gone before this call. -+ fd_ref.release_descriptor(); - Ok(()) - } - } -@@ -2461,6 +2758,7 @@ pub fn fs_error_into_wasi_err(fs_error: FsError) -> Errno { - mod tests { - use super::*; - use once_cell::sync::OnceCell; -+ #[cfg(feature = "host-fs")] - use tempfile::tempdir; - use virtual_fs::{RootFileSystemBuilder, TmpFileSystem}; - use wasmer::Engine; -@@ -2556,6 +2854,243 @@ mod tests { - ); - } - -+ #[cfg(all(target_os = "linux", feature = "host-fs"))] -+ #[tokio::test] -+ async fn host_directory_sync_uses_open_file_description_after_rename() { -+ let root_dir = tempdir().unwrap(); -+ let original = root_dir.path().join("original"); -+ let renamed = root_dir.path().join("renamed"); -+ std::fs::create_dir(&original).unwrap(); -+ -+ let host_fs = virtual_fs::host_fs::FileSystem::new( -+ tokio::runtime::Handle::current(), -+ root_dir.path(), -+ ) -+ .unwrap(); -+ let inodes = WasiInodes::new(); -+ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(host_fs)); -+ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); -+ let inode = wasi_fs -+ .create_inode( -+ &inodes, -+ Kind::Dir { -+ parent: wasi_fs.root_inode.downgrade(), -+ path: PathBuf::from("/original"), -+ entries: HashMap::new(), -+ }, -+ false, -+ "original".to_string(), -+ ) -+ .unwrap(); -+ let rights = Rights::FD_SYNC | Rights::FD_DATASYNC; -+ let fd = wasi_fs -+ .create_fd( -+ rights, -+ Rights::empty(), -+ Fdflags::empty(), -+ Fdflagsext::empty(), -+ 0, -+ inode, -+ ) -+ .unwrap(); -+ let duplicate = wasi_fs.clone_fd(fd).unwrap(); -+ -+ std::fs::rename(&original, &renamed).unwrap(); -+ assert!(!original.exists()); -+ wasi_fs.close_fd(fd).unwrap(); -+ -+ assert_eq!(wasi_fs.sync_blocking(duplicate, true), Ok(Some(()))); -+ assert_eq!(wasi_fs.sync_blocking(duplicate, false), Ok(Some(()))); -+ assert_eq!(wasi_fs.sync(duplicate, true).await, Ok(())); -+ } -+ -+ #[cfg(feature = "host-fs")] -+ #[tokio::test] -+ async fn host_file_size_refresh_observes_another_process_extension() { -+ let root_dir = tempdir().unwrap(); -+ let relation = root_dir.path().join("relation"); -+ std::fs::write(&relation, b"old").unwrap(); -+ -+ let host_fs = virtual_fs::host_fs::FileSystem::new( -+ tokio::runtime::Handle::current(), -+ root_dir.path(), -+ ) -+ .unwrap(); -+ let inodes = WasiInodes::new(); -+ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(host_fs)); -+ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); -+ let handle = wasi_fs -+ .root_fs -+ .new_open_options() -+ .read(true) -+ .write(true) -+ .open(Path::new("/relation")) -+ .unwrap(); -+ let inode = wasi_fs -+ .create_inode( -+ &inodes, -+ Kind::File { -+ handle: Some(Arc::new(RwLock::new(handle))), -+ path: PathBuf::from("/relation"), -+ fd: None, -+ }, -+ false, -+ "relation".to_string(), -+ ) -+ .unwrap(); -+ let fd = wasi_fs -+ .create_fd( -+ Rights::FD_FILESTAT_GET | Rights::FD_SEEK, -+ Rights::empty(), -+ Fdflags::empty(), -+ Fdflagsext::empty(), -+ 0, -+ inode, -+ ) -+ .unwrap(); -+ let fd_entry = wasi_fs.get_fd(fd).unwrap(); -+ assert_eq!(fd_entry.inode.stat.read().unwrap().st_size, 3); -+ -+ std::fs::OpenOptions::new() -+ .write(true) -+ .open(&relation) -+ .unwrap() -+ .set_len(8192) -+ .unwrap(); -+ assert_eq!(fd_entry.inode.stat.read().unwrap().st_size, 3); -+ assert_eq!(WasiFs::authoritative_fd_size(&fd_entry), 8192); -+ assert_eq!(wasi_fs.filestat_fd(fd).unwrap().st_size, 8192); -+ } -+ -+ #[cfg(feature = "host-fs")] -+ #[tokio::test] -+ async fn open_handle_reservation_bridges_final_close_to_fd_publication() { -+ let root_dir = tempdir().unwrap(); -+ std::fs::write(root_dir.path().join("relation"), b"data").unwrap(); -+ -+ let host_fs = virtual_fs::host_fs::FileSystem::new( -+ tokio::runtime::Handle::current(), -+ root_dir.path(), -+ ) -+ .unwrap(); -+ let inodes = WasiInodes::new(); -+ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(host_fs)); -+ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); -+ let handle = wasi_fs -+ .root_fs -+ .new_open_options() -+ .read(true) -+ .write(true) -+ .open(Path::new("/relation")) -+ .unwrap(); -+ let inode = wasi_fs -+ .create_inode( -+ &inodes, -+ Kind::File { -+ handle: Some(Arc::new(RwLock::new(handle))), -+ path: PathBuf::from("/relation"), -+ fd: None, -+ }, -+ false, -+ "relation".to_string(), -+ ) -+ .unwrap(); -+ let rights = Rights::FD_READ | Rights::FD_WRITE; -+ let first = wasi_fs -+ .create_fd( -+ rights, -+ Rights::empty(), -+ Fdflags::empty(), -+ Fdflagsext::empty(), -+ 0, -+ inode.clone(), -+ ) -+ .unwrap(); -+ -+ let reservation = { -+ let guard = inode.read(); -+ assert!(matches!( -+ guard.deref(), -+ Kind::File { -+ handle: Some(_), -+ .. -+ } -+ )); -+ inode.reserve_handle() -+ }; -+ wasi_fs.close_fd(first).unwrap(); -+ assert_eq!(inode.handle_count(), 1); -+ assert!(matches!( -+ inode.read().deref(), -+ Kind::File { -+ handle: Some(_), -+ .. -+ } -+ )); -+ -+ let second = wasi_fs -+ .create_fd( -+ rights, -+ Rights::empty(), -+ Fdflags::empty(), -+ Fdflagsext::empty(), -+ 0, -+ inode.clone(), -+ ) -+ .unwrap(); -+ drop(reservation); -+ assert_eq!(inode.handle_count(), 1); -+ assert!(matches!( -+ inode.read().deref(), -+ Kind::File { -+ handle: Some(_), -+ .. -+ } -+ )); -+ -+ wasi_fs.close_fd(second).unwrap(); -+ assert_eq!(inode.handle_count(), 0); -+ assert!(matches!( -+ inode.read().deref(), -+ Kind::File { handle: None, .. } -+ )); -+ } -+ -+ #[tokio::test] -+ async fn unsupported_directory_sync_fails_closed() { -+ let inodes = WasiInodes::new(); -+ let tmp = TmpFileSystem::new(); -+ tmp.create_dir(Path::new("/directory")).unwrap(); -+ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(tmp)); -+ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); -+ let inode = wasi_fs -+ .create_inode( -+ &inodes, -+ Kind::Dir { -+ parent: wasi_fs.root_inode.downgrade(), -+ path: PathBuf::from("/directory"), -+ entries: HashMap::new(), -+ }, -+ false, -+ "directory".to_string(), -+ ) -+ .unwrap(); -+ let rights = Rights::FD_SYNC | Rights::FD_DATASYNC; -+ let fd = wasi_fs -+ .create_fd( -+ rights, -+ Rights::empty(), -+ Fdflags::empty(), -+ Fdflagsext::empty(), -+ 0, -+ inode, -+ ) -+ .unwrap(); -+ -+ assert_eq!(wasi_fs.sync_blocking(fd, true), Err(Errno::Notsup)); -+ assert_eq!(wasi_fs.sync(fd, false).await, Err(Errno::Notsup)); -+ } -+ - #[cfg(feature = "host-fs")] - #[tokio::test] - async fn mapped_preopen_inode_paths_should_stay_in_guest_space() { -diff --git a/lib/wasix/src/lib.rs b/lib/wasix/src/lib.rs -index 9986bae..42a6ee3 100644 ---- a/lib/wasix/src/lib.rs -+++ b/lib/wasix/src/lib.rs -@@ -15,6 +15,9 @@ - //! [WASI plugin example](https://github.com/wasmerio/wasmer/blob/main/examples/plugin.rs) - //! for an example of how to extend WASI using the WASI FS API. - -+/// The exact `wasmer-wasix` crate version used by this runtime. -+pub const VERSION: &str = env!("CARGO_PKG_VERSION"); -+ - #[cfg(all( - not(feature = "sys"), - not(feature = "js"), -@@ -65,7 +68,7 @@ mod state; - mod syscalls; - mod utils; - --use std::sync::Arc; -+use std::{collections::HashSet, sync::Arc}; - - #[allow(unused_imports)] - use bytes::{Bytes, BytesMut}; -@@ -77,7 +80,7 @@ pub use wasmer_wasix_types; - - use wasmer::{ - AsStoreMut, Exports, FunctionEnv, Imports, Memory32, MemoryAccessError, MemorySize, -- RuntimeError, imports, namespace, -+ RuntimeError, imports, - }; - - pub use virtual_fs; -@@ -92,6 +95,28 @@ pub use virtual_net::{ - }; - use wasmer_wasix_types::wasi::{Errno, ExitCode}; - -+/// Product-specific host ABI for PostgreSQL postmaster capabilities that are -+/// not part of portable WASIX. -+pub const OLIPHAUNT_POSTMASTER_V1_NAMESPACE: &str = "oliphaunt_postmaster_v1"; -+ -+type RequiredImports<'a> = HashSet<(&'a str, &'a str)>; -+ -+macro_rules! namespace_for_required_imports { -+ ($required:expr, $namespace:expr; $( $import_name:expr => $import_item:expr ),* $(,)?) => {{ -+ let mut namespace = Exports::new(); -+ -+ $( -+ if $required.map_or(true, |required| { -+ required.contains(&($namespace, $import_name)) -+ }) { -+ namespace.insert($import_name, $import_item); -+ } -+ )* -+ -+ namespace -+ }}; -+} -+ - pub use crate::{ - fs::{Fd, VIRTUAL_ROOT_FD, WasiFs, WasiInodes, default_fs_backing}, - os::{ -@@ -99,21 +124,33 @@ pub use crate::{ - command::{BuiltinCommand, VirtualCommand}, - task::{ - control_plane::WasiControlPlane, -- process::{WasiProcess, WasiProcessId}, -+ process::{WasiProcess, WasiProcessExecutionGuard, WasiProcessId}, - thread::{WasiThread, WasiThreadError, WasiThreadHandle, WasiThreadId}, - }, - }, - rewind::*, -- runtime::{PluggableRuntime, Runtime, task_manager::VirtualTaskManager}, -+ runtime::{PluggableRuntime, ResourceLimits, Runtime, task_manager::VirtualTaskManager}, - state::{ -- ALL_RIGHTS, WasiEnv, WasiEnvBuilder, WasiEnvInit, WasiFunctionEnv, -- WasiModuleInstanceHandles, WasiModuleTreeHandles, WasiStateCreationError, -+ ALL_RIGHTS, DETERMINISTIC_START_ANALYZER_POLICY, DETERMINISTIC_START_GLOBAL_EFFECTS, -+ DETERMINISTIC_START_MEMORY_EFFECTS, DETERMINISTIC_START_MEMORY_READS, -+ DETERMINISTIC_START_PROOF_SCHEMA, DETERMINISTIC_START_TABLE_EFFECTS, -+ DeterministicStartProof, IntrinsicFileImmutability, PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, -+ PREINITIALIZED_MEMORY_IMAGE_PHASE, PreinitializedMemoryImage, -+ PreinitializedMemoryImageBacking, PreinitializedMemoryImageCapture, -+ PreinitializedMemoryImageHandle, PreinitializedMemoryImageLoadAudit, -+ PreinitializedMemoryImageLoader, PreinitializedMemoryImageMetadata, -+ PreinitializedMemoryImageMode, PreinitializedMemoryImageRuntimeAudit, WasiEnv, -+ WasiEnvBuilder, WasiEnvInit, WasiFunctionEnv, WasiModuleInstanceHandles, -+ WasiModuleTreeHandles, WasiStateCreationError, intrinsic_file_immutability, - }, - syscalls::{journal::wait_for_snapshot, rewind, rewind_ext, types, unwind}, - utils::is_wasix_module, - utils::{ - WasiVersion, get_wasi_version, get_wasi_versions, is_wasi_module, -- store::{StoreSnapshot, capture_store_snapshot, restore_store_snapshot}, -+ store::{ -+ StoreSnapshot, StoreSnapshotCaptureError, capture_store_snapshot, -+ restore_store_snapshot, -+ }, - }, - }; - -@@ -131,6 +168,8 @@ pub enum WasiError { - UnknownWasiVersion, - #[error("Dynamically-linked symbol not found or has bad type: {0}")] - DlSymbolResolutionFailed(String), -+ #[error("failed to capture WASIX store snapshot: {0}")] -+ StoreSnapshot(#[from] StoreSnapshotCaptureError), - } - - pub type WasiResult = Result, WasiError>; -@@ -332,6 +371,14 @@ pub struct WasiVFork { - /// Handle of the thread we have forked (dropping this handle - /// will signal that the thread is dead) - pub handle: WasiThreadHandle, -+ -+ /// Keeps the suspended parent non-reapable even if the child crosses a -+ /// deep-sleep TaskWasm boundary before returning control. -+ pub(crate) parent_execution: WasiProcessExecutionGuard, -+ -+ /// Owns execution admission for the child while it runs in-place inside -+ /// the parent's Wasmer instance, before exec creates a successor task. -+ pub(crate) child_execution: WasiProcessExecutionGuard, - } - - #[derive(Debug, Clone)] -@@ -350,6 +397,8 @@ impl Clone for WasiVFork { - asyncify: self.asyncify.clone(), - env: Box::new(self.env.as_ref().clone()), - handle: self.handle.clone(), -+ parent_execution: self.parent_execution.clone(), -+ child_execution: self.child_execution.clone(), - } - } - } -@@ -370,7 +419,7 @@ pub fn generate_import_object_from_env( - WasiVersion::Wasix64v1 => generate_import_object_wasix64_v1(store, ctx), - }; - -- let exports_wasi_generic = wasi_exports_generic(store, ctx); -+ let exports_wasi_generic = wasi_exports_generic(store, ctx, None); - - let imports_wasi_generic = imports! { - "wasi" => exports_wasi_generic, -@@ -378,20 +427,49 @@ pub fn generate_import_object_from_env( - - imports.extend(&imports_wasi_generic); - -+ // Explicit-version callers do not provide the module's import set, so -+ // register the small product ABI unfiltered. The main/dylink paths below -+ // still materialize only imports actually requested by each module. -+ let exports_oliphaunt_postmaster = oliphaunt_postmaster_exports(store, ctx, None); -+ let imports_oliphaunt_postmaster = imports! { -+ OLIPHAUNT_POSTMASTER_V1_NAMESPACE => exports_oliphaunt_postmaster, -+ }; -+ imports.extend(&imports_oliphaunt_postmaster); -+ - imports - } - --fn wasi_exports_generic(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { -+fn oliphaunt_postmaster_exports( -+ mut store: &mut impl AsStoreMut, -+ env: &FunctionEnv, -+ required: Option<&RequiredImports<'_>>, -+) -> Exports { -+ use syscalls::*; -+ -+ namespace_for_required_imports! { required, OLIPHAUNT_POSTMASTER_V1_NAMESPACE; -+ "fd_sync_range" => Function::new_typed_with_env(&mut store, env, fd_sync_range), -+ } -+} -+ -+fn wasi_exports_generic( -+ mut store: &mut impl AsStoreMut, -+ env: &FunctionEnv, -+ required: Option<&RequiredImports<'_>>, -+) -> Exports { - use syscalls::*; -- let namespace = namespace! { -+ let namespace = namespace_for_required_imports! { required, "wasi"; - "thread-spawn" => Function::new_typed_with_env(&mut store, env, thread_spawn::), - }; - namespace - } - --fn wasi_unstable_exports(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { -+fn wasi_unstable_exports( -+ mut store: &mut impl AsStoreMut, -+ env: &FunctionEnv, -+ required: Option<&RequiredImports<'_>>, -+) -> Exports { - use syscalls::*; -- let namespace = namespace! { -+ let namespace = namespace_for_required_imports! { required, "wasi_unstable"; - "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), - "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), - "clock_res_get" => Function::new_typed_with_env(&mut store, env, clock_res_get::), -@@ -445,9 +523,10 @@ fn wasi_unstable_exports(mut store: &mut impl AsStoreMut, env: &FunctionEnv, -+ required: Option<&RequiredImports<'_>>, - ) -> Exports { - use syscalls::*; -- let namespace = namespace! { -+ let namespace = namespace_for_required_imports! { required, "wasi_snapshot_preview1"; - "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), - "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), - "clock_res_get" => Function::new_typed_with_env(&mut store, env, clock_res_get::), -@@ -499,11 +578,15 @@ fn wasi_snapshot_preview1_exports( - namespace - } - --fn wasix_exports_32(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { -+fn wasix_exports_32( -+ mut store: &mut impl AsStoreMut, -+ env: &FunctionEnv, -+ required: Option<&RequiredImports<'_>>, -+) -> Exports { - let engine_supports_async = store.as_store_ref().engine().supports_async(); - - use syscalls::*; -- let namespace = namespace! { -+ let namespace = namespace_for_required_imports! { required, "wasix_32v1"; - "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), - "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), - "call_dynamic" => Function::new_typed_with_env(&mut store, env, call_dynamic::), -@@ -576,10 +659,14 @@ fn wasix_exports_32(mut store: &mut impl AsStoreMut, env: &FunctionEnv) - "proc_spawn2" => Function::new_typed_with_env(&mut store, env, proc_spawn2::), - "proc_id" => Function::new_typed_with_env(&mut store, env, proc_id::), - "proc_parent" => Function::new_typed_with_env(&mut store, env, proc_parent::), -+ "proc_rlimit_get" => Function::new_typed_with_env(&mut store, env, proc_rlimit_get::), - "random_get" => Function::new_typed_with_env(&mut store, env, random_get::), - "tty_get" => Function::new_typed_with_env(&mut store, env, tty_get::), - "tty_set" => Function::new_typed_with_env(&mut store, env, tty_set::), - "getcwd" => Function::new_typed_with_env(&mut store, env, getcwd::), -+ "mem_mmap" => Function::new_typed_with_env(&mut store, env, mem_mmap::), -+ "mem_munmap" => Function::new_typed_with_env(&mut store, env, mem_munmap::), -+ "mem_msync" => Function::new_typed_with_env(&mut store, env, mem_msync::), - "chdir" => Function::new_typed_with_env(&mut store, env, chdir::), - "dl_invalid_handle" => Function::new_typed_with_env(&mut store, env, dl_invalid_handle), - "dlopen" => Function::new_typed_with_env(&mut store, env, dlopen::), -@@ -646,11 +733,15 @@ fn wasix_exports_32(mut store: &mut impl AsStoreMut, env: &FunctionEnv) - namespace - } - --fn wasix_exports_64(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { -+fn wasix_exports_64( -+ mut store: &mut impl AsStoreMut, -+ env: &FunctionEnv, -+ required: Option<&RequiredImports<'_>>, -+) -> Exports { - let engine_supports_async = store.as_store_ref().engine().supports_async(); - - use syscalls::*; -- let namespace = namespace! { -+ let namespace = namespace_for_required_imports! { required, "wasix_64v1"; - "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), - "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), - "call_dynamic" => Function::new_typed_with_env(&mut store, env, call_dynamic::), -@@ -723,10 +814,14 @@ fn wasix_exports_64(mut store: &mut impl AsStoreMut, env: &FunctionEnv) - "proc_spawn2" => Function::new_typed_with_env(&mut store, env, proc_spawn2::), - "proc_id" => Function::new_typed_with_env(&mut store, env, proc_id::), - "proc_parent" => Function::new_typed_with_env(&mut store, env, proc_parent::), -+ "proc_rlimit_get" => Function::new_typed_with_env(&mut store, env, proc_rlimit_get::), - "random_get" => Function::new_typed_with_env(&mut store, env, random_get::), - "tty_get" => Function::new_typed_with_env(&mut store, env, tty_get::), - "tty_set" => Function::new_typed_with_env(&mut store, env, tty_set::), - "getcwd" => Function::new_typed_with_env(&mut store, env, getcwd::), -+ "mem_mmap" => Function::new_typed_with_env(&mut store, env, mem_mmap::), -+ "mem_munmap" => Function::new_typed_with_env(&mut store, env, mem_munmap::), -+ "mem_msync" => Function::new_typed_with_env(&mut store, env, mem_msync::), - "chdir" => Function::new_typed_with_env(&mut store, env, chdir::), - "dl_invalid_handle" => Function::new_typed_with_env(&mut store, env, dl_invalid_handle), - "dlopen" => Function::new_typed_with_env(&mut store, env, dlopen::), -@@ -796,15 +891,23 @@ fn wasix_exports_64(mut store: &mut impl AsStoreMut, env: &FunctionEnv) - // TODO: split function into two variants, one for JS and one for sys. - // (this will make code less messy) - fn import_object_for_all_wasi_versions( -- _module: &wasmer::Module, -+ module: &wasmer::Module, - store: &mut impl AsStoreMut, - env: &FunctionEnv, - ) -> Imports { -- let exports_wasi_generic = wasi_exports_generic(store, env); -- let exports_wasi_unstable = wasi_unstable_exports(store, env); -- let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env); -- let exports_wasix_32v1 = wasix_exports_32(store, env); -- let exports_wasix_64v1 = wasix_exports_64(store, env); -+ let required: RequiredImports<'_> = module -+ .info() -+ .imports -+ .keys() -+ .map(|import| (import.module.as_str(), import.field.as_str())) -+ .collect(); -+ let required = Some(&required); -+ let exports_wasi_generic = wasi_exports_generic(store, env, required); -+ let exports_wasi_unstable = wasi_unstable_exports(store, env, required); -+ let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env, required); -+ let exports_wasix_32v1 = wasix_exports_32(store, env, required); -+ let exports_wasix_64v1 = wasix_exports_64(store, env, required); -+ let exports_oliphaunt_postmaster = oliphaunt_postmaster_exports(store, env, required); - - // Allowed due to JS feature flag complications. - #[allow(unused_mut)] -@@ -814,17 +917,174 @@ fn import_object_for_all_wasi_versions( - "wasi_snapshot_preview1" => exports_wasi_snapshot_preview1, - "wasix_32v1" => exports_wasix_32v1, - "wasix_64v1" => exports_wasix_64v1, -+ OLIPHAUNT_POSTMASTER_V1_NAMESPACE => exports_oliphaunt_postmaster, - }; - - imports - } - -+#[cfg(test)] -+mod required_import_tests { -+ use super::*; -+ -+ #[test] -+ fn filtered_namespace_only_materializes_exact_module_imports() { -+ fn retained() {} -+ fn must_not_materialize() -> wasmer::Function { -+ panic!("filtered import expression was evaluated") -+ } -+ -+ let required = RequiredImports::from([("wasix_32v1", "retained")]); -+ let mut store = wasmer::Store::default(); -+ let exports = namespace_for_required_imports! { -+ Some(&required), "wasix_32v1"; -+ "retained" => wasmer::Function::new_typed(&mut store, retained), -+ "wrong-name" => must_not_materialize(), -+ }; -+ -+ assert_eq!(exports.len(), 1); -+ assert!(exports.get_function("retained").is_ok()); -+ assert!(exports.get_function("wrong-name").is_err()); -+ } -+ -+ #[test] -+ fn filtered_namespace_distinguishes_equal_names_in_other_abis() { -+ fn must_not_materialize() -> wasmer::Function { -+ panic!("an import from another ABI namespace was materialized") -+ } -+ -+ let required = RequiredImports::from([("wasi_snapshot_preview1", "fd_read")]); -+ let exports = namespace_for_required_imports! { -+ Some(&required), "wasix_32v1"; -+ "fd_read" => must_not_materialize(), -+ }; -+ -+ assert!(exports.is_empty()); -+ } -+ -+ #[tokio::test] -+ async fn oliphaunt_postmaster_fd_sync_range_has_exact_versioned_abi() { -+ use wasmer::{FunctionType, Type}; -+ -+ let engine = wasmer::Engine::default(); -+ let mut store = wasmer::Store::new(engine.clone()); -+ let func_env = WasiEnv::builder("fd-sync-range-abi-test") -+ .engine(engine) -+ .finalize(&mut store) -+ .unwrap(); -+ let required = -+ RequiredImports::from([(OLIPHAUNT_POSTMASTER_V1_NAMESPACE, "fd_sync_range")]); -+ let exports = oliphaunt_postmaster_exports(&mut store, &func_env.env, Some(&required)); -+ let function = exports.get_function("fd_sync_range").unwrap(); -+ -+ assert_eq!( -+ function.ty(&store), -+ FunctionType::new([Type::I32, Type::I64, Type::I64, Type::I32], [Type::I32]) -+ ); -+ } -+ -+ #[tokio::test] -+ async fn explicit_wasi_import_object_includes_versioned_postmaster_namespace() { -+ let engine = wasmer::Engine::default(); -+ let mut store = wasmer::Store::new(engine.clone()); -+ let func_env = WasiEnv::builder("fd-sync-range-import-test") -+ .engine(engine) -+ .finalize(&mut store) -+ .unwrap(); -+ let imports = -+ generate_import_object_from_env(&mut store, &func_env.env, WasiVersion::Wasix32v1); -+ let exports = imports -+ .get_namespace_exports(OLIPHAUNT_POSTMASTER_V1_NAMESPACE) -+ .unwrap(); -+ -+ assert_eq!(exports.len(), 1); -+ assert!(exports.get_function("fd_sync_range").is_ok()); -+ } -+ -+ #[cfg(feature = "sys")] -+ fn dynamic_mem_mmap_test_instance() -> (wasmer::Store, wasmer::Instance, WasiFunctionEnv) { -+ use wasmer::sys::{BaseTunables, NativeEngineExt}; -+ use wasmer::{Module, Pages}; -+ -+ let mut engine = wasmer::Engine::default(); -+ engine.set_tunables(BaseTunables { -+ static_memory_bound: Pages(0), -+ static_memory_offset_guard_size: 0, -+ dynamic_memory_offset_guard_size: 0, -+ }); -+ let mut store = wasmer::Store::new(engine.clone()); -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (import "wasix_32v1" "mem_mmap" -+ (func $mem_mmap -+ (param i32 i32 i32 i32 i32 i64 i32) -+ (result i32))) -+ (memory (export "memory") 1 2) -+ (func (export "map") (param $prot i32) (result i32) -+ i32.const 0 -+ i32.const 4096 -+ local.get $prot -+ i32.const 1 -+ i32.const -1 -+ i64.const 0 -+ i32.const 16 -+ call $mem_mmap))"#, -+ ) -+ .unwrap(); -+ let (instance, env) = WasiEnv::builder("direct-mem-mmap-test") -+ .engine(engine) -+ .instantiate(module, &mut store) -+ .unwrap(); -+ (store, instance, env) -+ } -+ -+ #[cfg(feature = "sys")] -+ #[tokio::test] -+ async fn direct_mem_mmap_import_rejects_weaker_protection_contracts() { -+ let (mut store, instance, _env) = dynamic_mem_mmap_test_instance(); -+ let map = instance -+ .exports -+ .get_typed_function::(&store, "map") -+ .unwrap(); -+ -+ assert_eq!( -+ map.call(&mut store, 0x01).unwrap(), -+ wasmer_wasix_types::wasi::Errno::Inval as i32 -+ ); -+ assert_eq!( -+ map.call(&mut store, 0x02).unwrap(), -+ wasmer_wasix_types::wasi::Errno::Inval as i32 -+ ); -+ } -+ -+ #[cfg(feature = "sys")] -+ #[tokio::test] -+ async fn unsupported_persistent_remap_rejects_before_grow_or_state_publication() { -+ let (mut store, instance, env) = dynamic_mem_mmap_test_instance(); -+ let memory = instance.exports.get_memory("memory").unwrap(); -+ let map = instance -+ .exports -+ .get_typed_function::(&store, "map") -+ .unwrap(); -+ let size_before = memory.size(&store); -+ -+ assert!(!memory.supports_persistent_shared_fixed_remap(&store)); -+ assert_eq!( -+ map.call(&mut store, 0x01 | 0x02).unwrap(), -+ wasmer_wasix_types::wasi::Errno::Notsup as i32 -+ ); -+ assert_eq!(memory.size(&store), size_before); -+ assert!(env.data(&store).state.shared_memory_mappings().is_empty()); -+ } -+} -+ - /// Combines a state generating function with the import list for legacy WASI - fn generate_import_object_snapshot0( - store: &mut impl AsStoreMut, - env: &FunctionEnv, - ) -> Imports { -- let exports_unstable = wasi_unstable_exports(store, env); -+ let exports_unstable = wasi_unstable_exports(store, env, None); - imports! { - "wasi_unstable" => exports_unstable - } -@@ -834,7 +1094,7 @@ fn generate_import_object_snapshot1( - store: &mut impl AsStoreMut, - env: &FunctionEnv, - ) -> Imports { -- let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env); -+ let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env, None); - imports! { - "wasi_snapshot_preview1" => exports_wasi_snapshot_preview1 - } -@@ -845,7 +1105,7 @@ fn generate_import_object_wasix32_v1( - store: &mut impl AsStoreMut, - env: &FunctionEnv, - ) -> Imports { -- let exports_wasix_32v1 = wasix_exports_32(store, env); -+ let exports_wasix_32v1 = wasix_exports_32(store, env, None); - imports! { - "wasix_32v1" => exports_wasix_32v1 - } -@@ -855,7 +1115,7 @@ fn generate_import_object_wasix64_v1( - store: &mut impl AsStoreMut, - env: &FunctionEnv, - ) -> Imports { -- let exports_wasix_64v1 = wasix_exports_64(store, env); -+ let exports_wasix_64v1 = wasix_exports_64(store, env, None); - imports! { - "wasix_64v1" => exports_wasix_64v1 - } -diff --git a/lib/wasix/src/net/socket.rs b/lib/wasix/src/net/socket.rs -index 1ca5276..d2460d4 100644 ---- a/lib/wasix/src/net/socket.rs -+++ b/lib/wasix/src/net/socket.rs -@@ -396,13 +396,10 @@ impl InodeSocket { - &self, - tasks: &dyn VirtualTaskManager, - net: &dyn VirtualNetworking, -- _backlog: usize, -+ backlog: usize, - ) -> Result, Errno> { -- let timeout = self -- .opt_time(TimeType::AcceptTimeout) -- .ok() -- .flatten() -- .unwrap_or(Duration::from_secs(30)); -+ let accept_timeout = self.opt_time(TimeType::AcceptTimeout).ok().flatten(); -+ let listen_timeout = accept_timeout.unwrap_or(Duration::from_secs(30)); - - let socket = { - let inner = self.inner.protected.read().unwrap(); -@@ -419,7 +416,7 @@ impl InodeSocket { - let reuse_addr = props.reuse_addr; - drop(inner); - -- net.listen_tcp(addr, only_v6, reuse_port, reuse_addr) -+ net.listen_tcp_with_backlog(addr, only_v6, reuse_port, reuse_addr, backlog) - } - ty => { - tracing::warn!( -@@ -441,7 +438,7 @@ impl InodeSocket { - let reuse_addr = props.reuse_addr; - drop(inner); - -- net.listen_tcp(addr, only_v6, reuse_port, reuse_addr) -+ net.listen_tcp_with_backlog(addr, only_v6, reuse_port, reuse_addr, backlog) - } - ty => { - tracing::warn!( -@@ -481,10 +478,10 @@ impl InodeSocket { - let socket = socket.map_err(net_error_into_wasi_err)?; - Ok(Some(InodeSocket::new(InodeSocketKind::TcpListener { - socket, -- accept_timeout: Some(timeout), -+ accept_timeout, - }))) - }, -- _ = tasks.sleep_now(timeout) => Err(Errno::Timedout) -+ _ = tasks.sleep_now(listen_timeout) => Err(Errno::Timedout) - } - } - -@@ -497,20 +494,11 @@ impl InodeSocket { - struct SocketAccepter<'a> { - sock: &'a InodeSocket, - nonblocking: bool, -- handler_registered: bool, -- } -- impl Drop for SocketAccepter<'_> { -- fn drop(&mut self) { -- if self.handler_registered { -- let mut inner = self.sock.inner.protected.write().unwrap(); -- inner.remove_handler(); -- } -- } - } - impl Future for SocketAccepter<'_> { - type Output = Result<(Box, SocketAddr), Errno>; - fn poll( -- mut self: Pin<&mut Self>, -+ self: Pin<&mut Self>, - cx: &mut std::task::Context<'_>, - ) -> std::task::Poll { - loop { -@@ -521,16 +509,17 @@ impl InodeSocket { - Err(NetworkError::WouldBlock) if self.nonblocking => { - Poll::Ready(Err(Errno::Again)) - } -- Err(NetworkError::WouldBlock) if !self.handler_registered => { -- let res = socket.set_handler(cx.waker().into()); -- if let Err(err) = res { -- return Poll::Ready(Err(net_error_into_wasi_err(err))); -+ Err(NetworkError::WouldBlock) => { -+ match socket.poll_read_ready(cx) { -+ Poll::Ready(Ok(_)) => {} -+ Poll::Ready(Err(err)) => { -+ return Poll::Ready(Err(net_error_into_wasi_err(err))); -+ } -+ Poll::Pending => return Poll::Pending, - } - drop(inner); -- self.handler_registered = true; - continue; - } -- Err(NetworkError::WouldBlock) => Poll::Pending, - Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), - }, - InodeSocketKind::PreSocket { .. } => Poll::Ready(Err(Errno::Notconn)), -@@ -543,7 +532,6 @@ impl InodeSocket { - let acceptor = SocketAccepter { - sock: self, - nonblocking, -- handler_registered: false, - }; - if let Some(timeout) = timeout { - tokio::select! { -@@ -1114,6 +1102,69 @@ impl InodeSocket { - } - } - -+ pub fn try_send_now(&self, buf: &[u8]) -> Result { -+ let mut inner = self.inner.protected.write().unwrap(); -+ let res = { -+ match &mut inner.kind { -+ InodeSocketKind::Raw(socket) => socket.try_send(buf), -+ InodeSocketKind::TcpStream { socket, .. } => socket.try_send(buf), -+ InodeSocketKind::UdpSocket { socket, peer } => { -+ if let Some(peer) = peer { -+ socket.try_send_to(buf, *peer) -+ } else { -+ Err(NetworkError::NotConnected) -+ } -+ } -+ InodeSocketKind::PreSocket { .. } => return Err(Errno::Notconn), -+ InodeSocketKind::RemoteSocket { is_dead, .. } => match is_dead { -+ true => return Err(Errno::Connreset), -+ false => return Ok(buf.len()), -+ }, -+ _ => return Err(Errno::Notsup), -+ } -+ }; -+ match res { -+ Ok(amt) => Ok(amt), -+ Err(NetworkError::WouldBlock) => Err(Errno::Again), -+ Err(err) => Err(net_error_into_wasi_err(err)), -+ } -+ } -+ -+ pub fn try_recv_now(&self, buf: &mut [MaybeUninit], peek: bool) -> Result { -+ let mut inner = self.inner.protected.write().unwrap(); -+ let res = { -+ match &mut inner.kind { -+ InodeSocketKind::Raw(socket) => socket.try_recv(buf, peek), -+ InodeSocketKind::TcpStream { socket, .. } => socket.try_recv(buf, peek), -+ InodeSocketKind::UdpSocket { socket, peer } => { -+ if let Some(peer) = peer { -+ match socket.try_recv_from(buf, peek) { -+ Ok((amt, addr)) if addr == *peer => Ok(amt), -+ Ok(_) => Err(NetworkError::WouldBlock), -+ Err(err) => Err(err), -+ } -+ } else { -+ match socket.try_recv_from(buf, peek) { -+ Ok((amt, _)) => Ok(amt), -+ Err(err) => Err(err), -+ } -+ } -+ } -+ InodeSocketKind::RemoteSocket { is_dead, .. } => match is_dead { -+ true => return Ok(0), -+ false => return Err(Errno::Again), -+ }, -+ InodeSocketKind::PreSocket { .. } => return Err(Errno::Notconn), -+ _ => return Err(Errno::Notsup), -+ } -+ }; -+ match res { -+ Ok(amt) => Ok(amt), -+ Err(NetworkError::WouldBlock) => Err(Errno::Again), -+ Err(err) => Err(net_error_into_wasi_err(err)), -+ } -+ } -+ - pub async fn send( - &self, - tasks: &dyn VirtualTaskManager, -@@ -1125,59 +1176,51 @@ impl InodeSocket { - inner: &'a InodeSocketInner, - data: &'b [u8], - nonblocking: bool, -- handler_registered: bool, -- } -- impl Drop for SocketSender<'_, '_> { -- fn drop(&mut self) { -- if self.handler_registered { -- let mut inner = self.inner.protected.write().unwrap(); -- inner.remove_handler(); -- } -- } - } - impl Future for SocketSender<'_, '_> { - type Output = Result; -- fn poll( -- mut self: Pin<&mut Self>, -- cx: &mut std::task::Context<'_>, -- ) -> Poll { -+ fn poll(self: Pin<&mut Self>, cx: &mut std::task::Context<'_>) -> Poll { - loop { - let mut inner = self.inner.protected.write().unwrap(); -- let res = match &mut inner.kind { -- InodeSocketKind::Raw(socket) => socket.try_send(self.data), -- InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), -- InodeSocketKind::UdpSocket { socket, peer } => { -- if let Some(peer) = peer { -- socket.try_send_to(self.data, *peer) -- } else { -- Err(NetworkError::NotConnected) -+ let res = { -+ match &mut inner.kind { -+ InodeSocketKind::Raw(socket) => socket.try_send(self.data), -+ InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), -+ InodeSocketKind::UdpSocket { socket, peer } => { -+ if let Some(peer) = peer { -+ socket.try_send_to(self.data, *peer) -+ } else { -+ Err(NetworkError::NotConnected) -+ } - } -+ InodeSocketKind::PreSocket { .. } => { -+ return Poll::Ready(Err(Errno::Notconn)); -+ } -+ InodeSocketKind::RemoteSocket { is_dead, .. } => { -+ return match is_dead { -+ true => Poll::Ready(Err(Errno::Connreset)), -+ false => Poll::Ready(Ok(self.data.len())), -+ }; -+ } -+ _ => return Poll::Ready(Err(Errno::Notsup)), - } -- InodeSocketKind::PreSocket { .. } => { -- return Poll::Ready(Err(Errno::Notconn)); -- } -- InodeSocketKind::RemoteSocket { is_dead, .. } => { -- return match is_dead { -- true => Poll::Ready(Err(Errno::Connreset)), -- false => Poll::Ready(Ok(self.data.len())), -- }; -- } -- _ => return Poll::Ready(Err(Errno::Notsup)), - }; - return match res { - Ok(amt) => Poll::Ready(Ok(amt)), - Err(NetworkError::WouldBlock) if self.nonblocking => { - Poll::Ready(Err(Errno::Again)) - } -- Err(NetworkError::WouldBlock) if !self.handler_registered => { -- inner -- .set_handler(cx.waker().into()) -- .map_err(net_error_into_wasi_err)?; -+ Err(NetworkError::WouldBlock) => { -+ match inner.poll_write_ready(cx) { -+ Poll::Ready(Ok(_)) => {} -+ Poll::Ready(Err(err)) => { -+ return Poll::Ready(Err(crate::utils::map_io_err(err))); -+ } -+ Poll::Pending => return Poll::Pending, -+ } - drop(inner); -- self.handler_registered = true; - continue; - } -- Err(NetworkError::WouldBlock) => Poll::Pending, - Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), - }; - } -@@ -1188,7 +1231,6 @@ impl InodeSocket { - inner: &self.inner, - data: buf, - nonblocking, -- handler_registered: false, - }; - if let Some(timeout) = timeout { - tokio::select! { -@@ -1213,55 +1255,49 @@ impl InodeSocket { - data: &'b [u8], - addr: SocketAddr, - nonblocking: bool, -- handler_registered: bool, -- } -- impl Drop for SocketSender<'_, '_> { -- fn drop(&mut self) { -- if self.handler_registered { -- let mut inner = self.inner.protected.write().unwrap(); -- inner.remove_handler(); -- } -- } - } - impl Future for SocketSender<'_, '_> { - type Output = Result; -- fn poll( -- mut self: Pin<&mut Self>, -- cx: &mut std::task::Context<'_>, -- ) -> Poll { -+ fn poll(self: Pin<&mut Self>, cx: &mut std::task::Context<'_>) -> Poll { - loop { - let mut inner = self.inner.protected.write().unwrap(); -- let res = match &mut inner.kind { -- InodeSocketKind::Icmp(socket) => socket.try_send_to(self.data, self.addr), -- InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), -- InodeSocketKind::UdpSocket { socket, .. } => { -- socket.try_send_to(self.data, self.addr) -- } -- InodeSocketKind::PreSocket { .. } => { -- return Poll::Ready(Err(Errno::Notconn)); -- } -- InodeSocketKind::RemoteSocket { is_dead, .. } => { -- return match is_dead { -- true => Poll::Ready(Err(Errno::Connreset)), -- false => Poll::Ready(Ok(self.data.len())), -- }; -+ let res = { -+ match &mut inner.kind { -+ InodeSocketKind::Icmp(socket) => { -+ socket.try_send_to(self.data, self.addr) -+ } -+ InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), -+ InodeSocketKind::UdpSocket { socket, .. } => { -+ socket.try_send_to(self.data, self.addr) -+ } -+ InodeSocketKind::PreSocket { .. } => { -+ return Poll::Ready(Err(Errno::Notconn)); -+ } -+ InodeSocketKind::RemoteSocket { is_dead, .. } => { -+ return match is_dead { -+ true => Poll::Ready(Err(Errno::Connreset)), -+ false => Poll::Ready(Ok(self.data.len())), -+ }; -+ } -+ _ => return Poll::Ready(Err(Errno::Notsup)), - } -- _ => return Poll::Ready(Err(Errno::Notsup)), - }; - return match res { - Ok(amt) => Poll::Ready(Ok(amt)), - Err(NetworkError::WouldBlock) if self.nonblocking => { - Poll::Ready(Err(Errno::Again)) - } -- Err(NetworkError::WouldBlock) if !self.handler_registered => { -- inner -- .set_handler(cx.waker().into()) -- .map_err(net_error_into_wasi_err)?; -- self.handler_registered = true; -+ Err(NetworkError::WouldBlock) => { -+ match inner.poll_write_ready(cx) { -+ Poll::Ready(Ok(_)) => {} -+ Poll::Ready(Err(err)) => { -+ return Poll::Ready(Err(crate::utils::map_io_err(err))); -+ } -+ Poll::Pending => return Poll::Pending, -+ } - drop(inner); - continue; - } -- Err(NetworkError::WouldBlock) => Poll::Pending, - Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), - }; - } -@@ -1273,7 +1309,6 @@ impl InodeSocket { - data: buf, - addr, - nonblocking, -- handler_registered: false, - }; - if let Some(timeout) = timeout { - tokio::select! { -@@ -1298,15 +1333,6 @@ impl InodeSocket { - data: &'b mut [MaybeUninit], - nonblocking: bool, - peek: bool, -- handler_registered: bool, -- } -- impl Drop for SocketReceiver<'_, '_> { -- fn drop(&mut self) { -- if self.handler_registered { -- let mut inner = self.inner.protected.write().unwrap(); -- inner.remove_handler(); -- } -- } - } - impl Future for SocketReceiver<'_, '_> { - type Output = Result; -@@ -1317,51 +1343,54 @@ impl InodeSocket { - loop { - let peek = self.peek; - let mut inner = self.inner.protected.write().unwrap(); -- let res = match &mut inner.kind { -- InodeSocketKind::Raw(socket) => socket.try_recv(self.data, peek), -- InodeSocketKind::TcpStream { socket, .. } => { -- socket.try_recv(self.data, peek) -- } -- InodeSocketKind::UdpSocket { socket, peer } => { -- if let Some(peer) = peer { -- match socket.try_recv_from(self.data, peek) { -- Ok((amt, addr)) if addr == *peer => Ok(amt), -- Ok(_) => Err(NetworkError::WouldBlock), -- Err(err) => Err(err), -- } -- } else { -- match socket.try_recv_from(self.data, peek) { -- Ok((amt, _)) => Ok(amt), -- Err(err) => Err(err), -+ let res = { -+ match &mut inner.kind { -+ InodeSocketKind::Raw(socket) => socket.try_recv(self.data, peek), -+ InodeSocketKind::TcpStream { socket, .. } => { -+ socket.try_recv(self.data, peek) -+ } -+ InodeSocketKind::UdpSocket { socket, peer } => { -+ if let Some(peer) = peer { -+ match socket.try_recv_from(self.data, peek) { -+ Ok((amt, addr)) if addr == *peer => Ok(amt), -+ Ok(_) => Err(NetworkError::WouldBlock), -+ Err(err) => Err(err), -+ } -+ } else { -+ match socket.try_recv_from(self.data, peek) { -+ Ok((amt, _)) => Ok(amt), -+ Err(err) => Err(err), -+ } - } - } -+ InodeSocketKind::RemoteSocket { is_dead, .. } => { -+ return match is_dead { -+ true => Poll::Ready(Ok(0)), -+ false => Poll::Pending, -+ }; -+ } -+ InodeSocketKind::PreSocket { .. } => { -+ return Poll::Ready(Err(Errno::Notconn)); -+ } -+ _ => return Poll::Ready(Err(Errno::Notsup)), - } -- InodeSocketKind::RemoteSocket { is_dead, .. } => { -- return match is_dead { -- true => Poll::Ready(Ok(0)), -- false => Poll::Pending, -- }; -- } -- InodeSocketKind::PreSocket { .. } => { -- return Poll::Ready(Err(Errno::Notconn)); -- } -- _ => return Poll::Ready(Err(Errno::Notsup)), - }; - return match res { - Ok(amt) => Poll::Ready(Ok(amt)), - Err(NetworkError::WouldBlock) if self.nonblocking => { - Poll::Ready(Err(Errno::Again)) - } -- Err(NetworkError::WouldBlock) if !self.handler_registered => { -- inner -- .set_handler(cx.waker().into()) -- .map_err(net_error_into_wasi_err)?; -- self.handler_registered = true; -+ Err(NetworkError::WouldBlock) => { -+ match inner.poll_read_ready_direct(cx) { -+ Poll::Ready(Ok(_)) => {} -+ Poll::Ready(Err(err)) => { -+ return Poll::Ready(Err(crate::utils::map_io_err(err))); -+ } -+ Poll::Pending => return Poll::Pending, -+ } - drop(inner); - continue; - } -- -- Err(NetworkError::WouldBlock) => Poll::Pending, - Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), - }; - } -@@ -1373,7 +1402,6 @@ impl InodeSocket { - data: buf, - nonblocking, - peek, -- handler_registered: false, - }; - if let Some(timeout) = timeout { - tokio::select! { -@@ -1398,15 +1426,6 @@ impl InodeSocket { - data: &'b mut [MaybeUninit], - nonblocking: bool, - peek: bool, -- handler_registered: bool, -- } -- impl Drop for SocketReceiver<'_, '_> { -- fn drop(&mut self) { -- if self.handler_registered { -- let mut inner = self.inner.protected.write().unwrap(); -- inner.remove_handler(); -- } -- } - } - impl Future for SocketReceiver<'_, '_> { - type Output = Result<(usize, SocketAddr), Errno>; -@@ -1415,8 +1434,8 @@ impl InodeSocket { - cx: &mut std::task::Context<'_>, - ) -> Poll { - let peek = self.peek; -- let mut inner = self.inner.protected.write().unwrap(); - loop { -+ let mut inner = self.inner.protected.write().unwrap(); - let res = match &mut inner.kind { - InodeSocketKind::Icmp(socket) => socket.try_recv_from(self.data, peek), - InodeSocketKind::UdpSocket { socket, .. } => { -@@ -1440,14 +1459,17 @@ impl InodeSocket { - Err(NetworkError::WouldBlock) if self.nonblocking => { - Poll::Ready(Err(Errno::Again)) - } -- Err(NetworkError::WouldBlock) if !self.handler_registered => { -- inner -- .set_handler(cx.waker().into()) -- .map_err(net_error_into_wasi_err)?; -- self.handler_registered = true; -+ Err(NetworkError::WouldBlock) => { -+ match inner.poll_read_ready(cx) { -+ Poll::Ready(Ok(_)) => {} -+ Poll::Ready(Err(err)) => { -+ return Poll::Ready(Err(crate::utils::map_io_err(err))); -+ } -+ Poll::Pending => return Poll::Pending, -+ } -+ drop(inner); - continue; - } -- Err(NetworkError::WouldBlock) => Poll::Pending, - Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), - }; - } -@@ -1459,7 +1481,6 @@ impl InodeSocket { - data: buf, - nonblocking, - peek, -- handler_registered: false, - }; - if let Some(timeout) = timeout { - tokio::select! { -@@ -1501,22 +1522,6 @@ impl InodeSocket { - } - - impl InodeSocketProtected { -- pub fn remove_handler(&mut self) { -- match &mut self.kind { -- InodeSocketKind::TcpListener { socket, .. } => socket.remove_handler(), -- InodeSocketKind::TcpStream { socket, .. } => socket.remove_handler(), -- InodeSocketKind::UdpSocket { socket, .. } => socket.remove_handler(), -- InodeSocketKind::Raw(socket) => socket.remove_handler(), -- InodeSocketKind::Icmp(socket) => socket.remove_handler(), -- InodeSocketKind::PreSocket { props, .. } => { -- props.handler.take(); -- } -- InodeSocketKind::RemoteSocket { props, .. } => { -- props.handler.take(); -- } -- } -- } -- - pub fn poll_read_ready(&mut self, cx: &mut Context<'_>) -> Poll> { - match &mut self.kind { - InodeSocketKind::TcpListener { socket, .. } => socket.poll_read_ready(cx), -@@ -1533,6 +1538,14 @@ impl InodeSocketProtected { - .map_err(net_error_into_io_err) - } - -+ pub fn poll_read_ready_direct(&mut self, cx: &mut Context<'_>) -> Poll> { -+ match &mut self.kind { -+ InodeSocketKind::TcpStream { socket, .. } => socket.poll_read_ready_direct(cx), -+ _ => return self.poll_read_ready(cx), -+ } -+ .map_err(net_error_into_io_err) -+ } -+ - pub fn poll_write_ready(&mut self, cx: &mut Context<'_>) -> Poll> { - match &mut self.kind { - InodeSocketKind::TcpListener { socket, .. } => socket.poll_write_ready(cx), -@@ -1591,7 +1604,7 @@ pub(crate) fn all_socket_rights() -> Rights { - - #[cfg(test)] - mod tests { -- use super::{InodeSocket, InodeSocketKind, WasiSocketStatus}; -+ use super::{InodeSocket, InodeSocketKind, TimeType, WasiSocketStatus}; - use std::{ - mem::MaybeUninit, - net::{Ipv4Addr, Shutdown, SocketAddr}, -@@ -1606,12 +1619,13 @@ mod tests { - use virtual_mio::InterestHandler; - use virtual_net::{ - NetworkError, Result as NetResult, SocketStatus, VirtualConnectedSocket, VirtualIoSource, -- VirtualSocket, VirtualTcpSocket, -+ VirtualSocket, VirtualTcpListener, VirtualTcpSocket, - }; - - #[derive(Debug)] - struct MockTcpSocket { - read_calls: Arc, -+ direct_read_calls: Arc, - write_calls: Arc, - status: Arc, - } -@@ -1634,6 +1648,11 @@ mod tests { - Poll::Ready(Ok(3)) - } - -+ fn poll_read_ready_direct(&mut self, _cx: &mut Context<'_>) -> Poll> { -+ self.direct_read_calls.fetch_add(1, Ordering::Relaxed); -+ Poll::Ready(Ok(11)) -+ } -+ - fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { - self.write_calls.fetch_add(1, Ordering::Relaxed); - self.status.store(MOCK_STATUS_OPENED, Ordering::Relaxed); -@@ -1746,6 +1765,46 @@ mod tests { - } - } - -+ #[derive(Debug)] -+ struct MockTcpListener; -+ -+ impl VirtualIoSource for MockTcpListener { -+ fn remove_handler(&mut self) {} -+ -+ fn poll_read_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { -+ Poll::Pending -+ } -+ -+ fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { -+ Poll::Ready(Ok(0)) -+ } -+ } -+ -+ impl VirtualTcpListener for MockTcpListener { -+ fn try_accept(&mut self) -> NetResult<(Box, SocketAddr)> { -+ Err(NetworkError::WouldBlock) -+ } -+ -+ fn set_handler( -+ &mut self, -+ _handler: Box, -+ ) -> NetResult<()> { -+ Ok(()) -+ } -+ -+ fn addr_local(&self) -> NetResult { -+ Ok(SocketAddr::from((Ipv4Addr::LOCALHOST, 0))) -+ } -+ -+ fn set_ttl(&mut self, _ttl: u8) -> NetResult<()> { -+ Ok(()) -+ } -+ -+ fn ttl(&self) -> NetResult { -+ Ok(64) -+ } -+ } -+ - #[test] - fn inode_socket_poll_write_ready_uses_write_path() { - let read_calls = Arc::new(AtomicUsize::new(0)); -@@ -1754,6 +1813,7 @@ mod tests { - let mut inode = InodeSocket::new(InodeSocketKind::TcpStream { - socket: Box::new(MockTcpSocket { - read_calls: read_calls.clone(), -+ direct_read_calls: Arc::new(AtomicUsize::new(0)), - write_calls: write_calls.clone(), - status, - }), -@@ -1770,12 +1830,39 @@ mod tests { - assert_eq!(write_calls.load(Ordering::Relaxed), 1); - } - -+ #[test] -+ fn inode_socket_poll_read_ready_direct_uses_tcp_direct_path() { -+ let read_calls = Arc::new(AtomicUsize::new(0)); -+ let direct_read_calls = Arc::new(AtomicUsize::new(0)); -+ let mut inner = super::InodeSocketProtected { -+ kind: InodeSocketKind::TcpStream { -+ socket: Box::new(MockTcpSocket { -+ read_calls: read_calls.clone(), -+ direct_read_calls: direct_read_calls.clone(), -+ write_calls: Arc::new(AtomicUsize::new(0)), -+ status: Arc::new(AtomicUsize::new(MOCK_STATUS_OPENED)), -+ }), -+ write_timeout: None, -+ read_timeout: None, -+ }, -+ }; -+ -+ let waker = futures::task::noop_waker(); -+ let mut cx = Context::from_waker(&waker); -+ let ready = inner.poll_read_ready_direct(&mut cx); -+ -+ assert!(matches!(ready, Poll::Ready(Ok(11)))); -+ assert_eq!(read_calls.load(Ordering::Relaxed), 0); -+ assert_eq!(direct_read_calls.load(Ordering::Relaxed), 1); -+ } -+ - #[test] - fn inode_socket_status_tracks_tcp_socket_status() { - let status = Arc::new(AtomicUsize::new(MOCK_STATUS_OPENING)); - let inode = InodeSocket::new(InodeSocketKind::TcpStream { - socket: Box::new(MockTcpSocket { - read_calls: Arc::new(AtomicUsize::new(0)), -+ direct_read_calls: Arc::new(AtomicUsize::new(0)), - write_calls: Arc::new(AtomicUsize::new(0)), - status: status.clone(), - }), -@@ -1787,4 +1874,14 @@ mod tests { - status.store(MOCK_STATUS_OPENED, Ordering::Relaxed); - assert!(matches!(inode.status().unwrap(), WasiSocketStatus::Opened)); - } -+ -+ #[test] -+ fn tcp_listener_accept_timeout_defaults_to_unset() { -+ let inode = InodeSocket::new(InodeSocketKind::TcpListener { -+ socket: Box::new(MockTcpListener), -+ accept_timeout: None, -+ }); -+ -+ assert_eq!(inode.opt_time(TimeType::AcceptTimeout).unwrap(), None); -+ } - } -diff --git a/lib/wasix/src/os/command/builtins/cmd_wasmer.rs b/lib/wasix/src/os/command/builtins/cmd_wasmer.rs -index 4f60993..a383abd 100644 ---- a/lib/wasix/src/os/command/builtins/cmd_wasmer.rs -+++ b/lib/wasix/src/os/command/builtins/cmd_wasmer.rs -@@ -76,10 +76,17 @@ impl CmdWasmer { - let mut env = config.take().ok_or(SpawnError::UnknownError)?; - - // Set the arguments of the environment by replacing the state -- let mut state = env.state.fork(); - args.insert(0, what.clone()); -- state.args = std::sync::Mutex::new(args); -- env.state = Arc::new(state); -+ env.state = env -+ .state -+ .fork_with(move |state| { -+ state.args = std::sync::Mutex::new(args); -+ }) -+ .map_err(|errno| { -+ SpawnError::Other(Box::new(std::io::Error::other(format!( -+ "could not fork command state during shared-memory transition: {errno}" -+ )))) -+ })?; - - let file_path = if what.starts_with('/') { - PathBuf::from(&what) -diff --git a/lib/wasix/src/os/console/mod.rs b/lib/wasix/src/os/console/mod.rs -index 1281f50..2755cdd 100644 ---- a/lib/wasix/src/os/console/mod.rs -+++ b/lib/wasix/src/os/console/mod.rs -@@ -324,7 +324,6 @@ mod tests { - /// - /// See [#4284](https://github.com/wasmerio/wasmer/issues/4284) for more. - #[test] -- #[cfg_attr(not(feature = "host-reqwest"), ignore = "Requires a HTTP client")] - #[ignore = "Unconditionally aborts (CC #4284)"] - fn test_console_dash_tty_with_args_and_env() { - let tokio_rt = tokio::runtime::Runtime::new().unwrap(); -diff --git a/lib/wasix/src/os/epoll/mod.rs b/lib/wasix/src/os/epoll/mod.rs -index 9b5ace4..7e97ee0 100644 ---- a/lib/wasix/src/os/epoll/mod.rs -+++ b/lib/wasix/src/os/epoll/mod.rs -@@ -24,7 +24,7 @@ - //! 3. Enqueue exactly one `ReadyItem` per subscription while `enqueued == true`. - //! 4. Wake one waiter via `Notify`. - //! --//! ### Consumer path (`drain_ready_events` used by `epoll_wait`) -+//! ### Consumer path (`wait_for_ready_events` used by `epoll_wait`) - //! 1. Pop `ReadyItem` from the ready queue. - //! 2. Resolve the current subscription and drop stale/missing entries. - //! 3. Atomically take readiness bits, clear `enqueued`, and map bits to output events. -@@ -40,7 +40,7 @@ - use std::{ - collections::VecDeque, - sync::{ -- Arc, Mutex as StdMutex, -+ Arc, Mutex as StdMutex, Weak, - atomic::{AtomicBool, AtomicU8, AtomicU64, Ordering}, - }, - }; -@@ -48,17 +48,16 @@ use std::{ - use fnv::FnvHashMap; - use serde::{Deserialize, Serialize}; - use tokio::sync::Notify; --use virtual_mio::{InterestHandler, InterestType}; -+use virtual_mio::{InterestHandler, InterestHandlerRegistration, InterestType}; - use virtual_net::net_error_into_io_err; - use wasmer_wasix_types::wasi::{ -- EpollEventCtl, EpollType, Errno, Eventtype, Fd as WasiFd, Subscription, -+ EpollEventCtl, EpollType, Errno, Eventtype, Fd as WasiFd, Rights, Subscription, - SubscriptionFsReadwrite, SubscriptionUnion, - }; - - use crate::{ -- fs::{InodeValFilePollGuard, InodeValFilePollGuardMode}, -- state::{PollEvent, PollEventBuilder, WasiState}, -- syscalls::poll_fd_guard, -+ fs::{EpollRegistrationGuard, Fd, InodeValFilePollGuard, InodeValFilePollGuardMode}, -+ state::{PollEvent, PollEventBuilder}, - }; - - const READABLE_BIT: u8 = 1 << 0; -@@ -66,11 +65,6 @@ const WRITABLE_BIT: u8 = 1 << 1; - const HUP_BIT: u8 = 1 << 2; - const ERR_BIT: u8 = 1 << 3; - --static EPOLL_ENQUEUE_ATTEMPTS: AtomicU64 = AtomicU64::new(0); --static EPOLL_ENQUEUE_DEDUPE_HITS: AtomicU64 = AtomicU64::new(0); --static EPOLL_STALE_GENERATION_DROPS: AtomicU64 = AtomicU64::new(0); --static EPOLL_EMPTY_DEQUEUE_ENTRIES: AtomicU64 = AtomicU64::new(0); -- - #[derive(Debug, Clone, Serialize, Deserialize)] - pub struct EpollFd { - /// Event mask configured by the caller (`epoll_ctl`). -@@ -128,55 +122,65 @@ impl EpollFd { - } - } - -+/// Linux identifies an epoll interest by the numeric fd used for ADD plus the -+/// underlying open file description. The fd can later be closed and reused -+/// while a dup/fork alias keeps the old description alive. -+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] -+pub(crate) struct EpollSubscriptionKey { -+ fd: WasiFd, -+ ofd_id: u64, -+} -+ -+impl EpollSubscriptionKey { -+ pub(crate) fn new(fd: WasiFd, ofd_id: u64) -> Self { -+ Self { fd, ofd_id } -+ } -+ -+ pub(crate) fn fd(self) -> WasiFd { -+ self.fd -+ } -+ -+ fn ofd_id(self) -> u64 { -+ self.ofd_id -+ } -+} -+ - #[derive(Debug)] - pub struct EpollJoinGuard { - /// Underlying poll registration guard. - fd_guard: InodeValFilePollGuard, -+ /// Removes only this subscription's handler from the OFD fanout. -+ _handler_registration: Option, - } - - impl EpollJoinGuard { -- fn new(fd_guard: InodeValFilePollGuard) -> Self { -- Self { fd_guard } -- } --} -- --impl Drop for EpollJoinGuard { -- fn drop(&mut self) { -- // Dropping a subscription must detach its interest handler from the source. -- match &self.fd_guard.mode { -- InodeValFilePollGuardMode::File(_) => { -- // Intentionally ignored, epoll doesn't work with files -- } -- InodeValFilePollGuardMode::Socket { inner } => { -- let mut inner = inner.protected.write().unwrap(); -- inner.remove_handler(); -- } -- InodeValFilePollGuardMode::EventNotifications(inner) => { -- inner.remove_interest_handler(); -- } -- InodeValFilePollGuardMode::DuplexPipe { pipe } => { -- let inner = pipe.write().unwrap(); -- inner.remove_interest_handler(); -- } -- InodeValFilePollGuardMode::PipeRx { rx } => { -- let inner = rx.write().unwrap(); -- inner.remove_interest_handler(); -- } -- InodeValFilePollGuardMode::PipeTx { .. } => { -- // Intentionally ignored, the sending end of a pipe can't have an interest handler -- } -+ fn new( -+ fd_guard: InodeValFilePollGuard, -+ handler_registration: Option, -+ ) -> Self { -+ Self { -+ fd_guard, -+ _handler_registration: handler_registration, - } - } - } - - #[derive(Debug)] - pub struct EpollState { -- /// Active subscriptions keyed by watched fd. -- subscriptions: StdMutex>>, -+ /// Serializes the multi-step state/source transaction performed by -+ /// epoll_ctl. Source registration can block or fail, so the subscription -+ /// map alone cannot make ADD/MOD rollback atomic with another ctl call. -+ ctl: StdMutex<()>, -+ /// Active subscriptions keyed by the ADD fd and its open-file-description. -+ subscriptions: StdMutex>>, - /// Ready queue of subscriptions with potentially pending bits. - ready: StdMutex>, - /// Wake primitive for blocked `epoll_wait`. - notify: Notify, -+ /// Final close is idempotent and prevents new registrations. -+ closed: AtomicBool, -+ /// Never-reused identity for queued readiness and replacement detection. -+ next_registration_id: AtomicU64, - } - - impl Default for EpollState { -@@ -189,34 +193,65 @@ impl EpollState { - /// Creates a fresh epoll runtime state. - pub fn new() -> Self { - Self { -+ ctl: StdMutex::new(()), - subscriptions: StdMutex::new(FnvHashMap::default()), - ready: StdMutex::new(VecDeque::new()), - notify: Notify::new(), -+ closed: AtomicBool::new(false), -+ next_registration_id: AtomicU64::new(1), - } - } - -+ fn allocate_registration_id(&self) -> u64 { -+ self.next_registration_id -+ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { -+ current.checked_add(1) -+ }) -+ .expect("epoll registration identity space exhausted") -+ } -+ -+ pub(crate) fn with_ctl_transaction(&self, transaction: impl FnOnce() -> T) -> T { -+ let _ctl = self.ctl.lock().unwrap(); -+ transaction() -+ } -+ - #[cfg(test)] -- fn insert_subscription(&self, fd: WasiFd, state: Arc) { -- self.subscriptions.lock().unwrap().insert(fd, state); -+ fn insert_subscription(&self, key: EpollSubscriptionKey, state: Arc) { -+ self.subscriptions.lock().unwrap().insert(key, state); - } - -- fn restore_subscription(&self, fd: WasiFd, previous: Option>) { -- let mut subscriptions = self.subscriptions.lock().unwrap(); -- subscriptions.remove(&fd); -- if let Some(previous) = previous { -- subscriptions.insert(fd, previous); -- } -+ fn subscription(&self, key: EpollSubscriptionKey) -> Option> { -+ self.subscriptions.lock().unwrap().get(&key).cloned() -+ } -+ -+ #[cfg(test)] -+ pub(crate) fn contains_exact_subscription( -+ &self, -+ key: EpollSubscriptionKey, -+ expected: &Arc, -+ ) -> bool { -+ self.subscription(key) -+ .is_some_and(|current| Arc::ptr_eq(¤t, expected)) - } - -- fn subscription(&self, fd: WasiFd) -> Option> { -- self.subscriptions.lock().unwrap().get(&fd).cloned() -+ #[cfg(test)] -+ pub(crate) fn subscription_count(&self) -> usize { -+ self.subscriptions.lock().unwrap().len() - } - -- fn enqueue_ready(&self, fd: WasiFd, generation: u64) { -- self.ready -- .lock() -- .unwrap() -- .push_back(ReadyItem { fd, generation }); -+ fn enqueue_ready(&self, key: EpollSubscriptionKey, registration_id: u64) { -+ let mut ready = self.ready.lock().unwrap(); -+ // Linearize enqueue against final close. If enqueue wins this lock, -+ // close clears the item afterward; if close publishes first, no new -+ // item may enter a permanently closed epoll queue. -+ if self.closed.load(Ordering::Acquire) { -+ return; -+ } -+ ready.push_back(ReadyItem { -+ key, -+ registration_id, -+ }); -+ drop(ready); - self.notify.notify_one(); - } - -@@ -224,90 +259,215 @@ impl EpollState { - self.ready.lock().unwrap().pop_front() - } - -- /// Waits until a producer enqueues readiness and notifies. -- pub async fn wait(&self) { -- self.notify.notified().await; -+ fn ready_queue_snapshot(&self) -> Vec { -+ self.ready.lock().unwrap().iter().copied().collect() -+ } -+ -+ fn refresh_level_readiness(&self) -> bool { -+ if self.is_closed() { -+ return false; -+ } -+ let ready_items = self.ready_queue_snapshot(); -+ let refreshed_ready_queue = !ready_items.is_empty(); -+ let subscriptions: Vec<_> = { -+ let subscriptions = self.subscriptions.lock().unwrap(); -+ if refreshed_ready_queue { -+ ready_items -+ .iter() -+ .filter_map(|item| { -+ let sub_state = subscriptions.get(&item.key)?.clone(); -+ (sub_state.registration_id() == item.registration_id) -+ .then_some((item.key, sub_state)) -+ }) -+ .collect() -+ } else { -+ subscriptions -+ .iter() -+ .map(|(&key, sub_state)| (key, sub_state.clone())) -+ .collect() -+ } -+ }; -+ for (key, sub_state) in subscriptions { -+ let current_bits = if refreshed_ready_queue { -+ let pending_bits = sub_state.pending_bits(); -+ if pending_bits != 0 { -+ pending_bits -+ } else { -+ sub_state.poll_current_level_bits() -+ } -+ } else { -+ sub_state.poll_current_level_bits() -+ }; -+ if current_bits == 0 { -+ continue; -+ } -+ -+ sub_state -+ .pending_bits -+ .fetch_or(current_bits, Ordering::AcqRel); -+ if sub_state.mark_enqueued() { -+ self.enqueue_ready(key, sub_state.registration_id()); -+ } -+ } -+ refreshed_ready_queue - } - - pub(crate) fn prepare_add( - &self, -- fd: WasiFd, -+ key: EpollSubscriptionKey, - event: &EpollEventCtl, - ) -> Result<(EpollFd, Arc), Errno> { - let mut subscriptions = self.subscriptions.lock().unwrap(); -- if subscriptions.contains_key(&fd) { -+ if self.closed.load(Ordering::Acquire) { -+ return Err(Errno::Badf); -+ } -+ if subscriptions.contains_key(&key) { - return Err(Errno::Exist); - } - -- let (epoll_fd, sub_state) = self.build_pending_subscription(fd, event, 1); -- subscriptions.insert(fd, sub_state.clone()); -+ let (epoll_fd, sub_state) = self.build_pending_subscription(key, event); -+ subscriptions.insert(key, sub_state.clone()); - Ok((epoll_fd, sub_state)) - } - - pub(crate) fn prepare_mod( - &self, -- fd: WasiFd, -+ key: EpollSubscriptionKey, - event: &EpollEventCtl, - ) -> Result<(EpollFd, Arc, Arc), Errno> { - let mut subscriptions = self.subscriptions.lock().unwrap(); -- let Some(previous) = subscriptions.remove(&fd) else { -+ if self.closed.load(Ordering::Acquire) { -+ return Err(Errno::Badf); -+ } -+ let Some(previous) = subscriptions.remove(&key) else { - return Err(Errno::Noent); - }; -- tracing::trace!(fd, "unregistering waker"); -+ tracing::trace!(fd = key.fd(), ofd_id = key.ofd_id(), "unregistering waker"); - -- let (epoll_fd, sub_state) = -- self.build_pending_subscription(fd, event, previous.next_generation()); -- subscriptions.insert(fd, sub_state.clone()); -+ let (epoll_fd, sub_state) = self.build_pending_subscription(key, event); -+ subscriptions.insert(key, sub_state.clone()); - Ok((epoll_fd, sub_state, previous)) - } - -- pub(crate) fn apply_del(&self, fd: WasiFd) -> Result<(), Errno> { -- let removed = self -- .subscriptions -- .lock() -- .unwrap() -- .remove(&fd) -- .ok_or(Errno::Noent)?; -- removed.detach_joins(); -+ pub(crate) fn prepare_restore( -+ &self, -+ key: EpollSubscriptionKey, -+ epoll_fd: EpollFd, -+ ) -> Result, Errno> { -+ let mut subscriptions = self.subscriptions.lock().unwrap(); -+ if self.closed.load(Ordering::Acquire) { -+ return Err(Errno::Badf); -+ } -+ if subscriptions.contains_key(&key) { -+ return Err(Errno::Exist); -+ } -+ let registration_id = self.allocate_registration_id(); -+ let sub_state = Arc::new(EpollSubState::new(epoll_fd, registration_id)); -+ subscriptions.insert(key, sub_state.clone()); -+ Ok(sub_state) -+ } -+ -+ pub(crate) fn apply_del(&self, key: EpollSubscriptionKey) -> Result<(), Errno> { -+ let removed = self.subscriptions.lock().unwrap().remove(&key); -+ let Some(removed) = removed else { -+ return Err(Errno::Noent); -+ }; -+ removed.deactivate_and_detach(); - Ok(()) - } - -- pub(crate) fn rollback_registration(&self, fd: WasiFd, previous: Option>) { -- self.restore_subscription(fd, previous); -+ pub(crate) fn rollback_registration( -+ &self, -+ key: EpollSubscriptionKey, -+ expected: &Arc, -+ ) -> bool { -+ self.remove_if_same(key, expected) - } - - fn build_pending_subscription( - &self, -- fd: WasiFd, -+ key: EpollSubscriptionKey, - event: &EpollEventCtl, -- generation: u64, - ) -> (EpollFd, Arc) { -- let epoll_fd = EpollFd::from_event_ctl(fd, event); -+ let registration_id = self.allocate_registration_id(); -+ let epoll_fd = EpollFd::from_event_ctl(key.fd(), event); - tracing::trace!( - peb = ?event.events, - ptr = ?event.ptr, - data1 = event.data1, - data2 = event.data2, -- fd, -+ fd = key.fd(), -+ ofd_id = key.ofd_id(), - "registering waker" - ); -- let sub_state = Arc::new(EpollSubState::new(epoll_fd.clone(), generation)); -+ let sub_state = Arc::new(EpollSubState::new(epoll_fd.clone(), registration_id)); - (epoll_fd, sub_state) - } -+ -+ pub(crate) fn remove_if_same( -+ &self, -+ key: EpollSubscriptionKey, -+ expected: &Arc, -+ ) -> bool { -+ let removed = { -+ let mut subscriptions = self.subscriptions.lock().unwrap(); -+ let is_same = subscriptions -+ .get(&key) -+ .is_some_and(|current| Arc::ptr_eq(current, expected)); -+ is_same.then(|| subscriptions.remove(&key)).flatten() -+ }; -+ let Some(removed) = removed else { -+ return false; -+ }; -+ removed.deactivate_and_detach(); -+ true -+ } -+ -+ /// Closes the epoll open file description. This is idempotent and drops -+ /// subscription resources outside the state locks. -+ pub(crate) fn close(&self) { -+ if self.closed.swap(true, Ordering::AcqRel) { -+ return; -+ } -+ let subscriptions = { -+ let mut subscriptions = self.subscriptions.lock().unwrap(); -+ subscriptions -+ .drain() -+ .map(|(_, sub)| sub) -+ .collect::>() -+ }; -+ for subscription in subscriptions { -+ subscription.deactivate_and_detach(); -+ } -+ self.ready.lock().unwrap().clear(); -+ self.notify.notify_waiters(); -+ } -+ -+ pub(crate) fn is_closed(&self) -> bool { -+ self.closed.load(Ordering::Acquire) -+ } - } - - #[derive(Debug)] - pub struct EpollSubState { - /// Snapshot of user-visible metadata. - fd_meta: StdMutex, -- /// Guard ownership for all attached handlers. -- joins: StdMutex>, -+ /// Lifecycle-owned resources. Active is checked under the same lock used -+ /// to attach resources so final close cannot race in a dead handler. -+ resources: StdMutex, - /// Atomic readiness bitset (EPOLLIN/OUT/HUP/ERR). - pending_bits: AtomicU8, - /// Queue dedupe flag: whether this sub already has a ready-queue entry. - enqueued: AtomicBool, -- /// Generation used to invalidate stale queue entries after DEL/MOD. -- generation: AtomicU64, -+ /// Never-reused identity used to invalidate stale queue entries. -+ registration_id: AtomicU64, -+} -+ -+#[derive(Debug)] -+struct EpollSubResources { -+ active: bool, -+ joins: Vec, -+ close_registration: Option, - } - - impl EpollSubState { -@@ -315,32 +475,68 @@ impl EpollSubState { - pub fn new(fd_meta: EpollFd, generation: u64) -> Self { - Self { - fd_meta: StdMutex::new(fd_meta), -- joins: StdMutex::new(Vec::new()), -+ resources: StdMutex::new(EpollSubResources { -+ active: true, -+ joins: Vec::new(), -+ close_registration: None, -+ }), - pending_bits: AtomicU8::new(0), - enqueued: AtomicBool::new(false), -- generation: AtomicU64::new(generation), -+ registration_id: AtomicU64::new(generation), - } - } - -- /// Returns `generation + 1` without mutating the current subscription. -- /// -- /// Callers use this to seed the generation of a replacement subscription. -- pub fn next_generation(&self) -> u64 { -- self.generation.load(Ordering::Acquire).saturating_add(1) -+ /// Adds a registration guard that will detach handlers when dropped. -+ pub fn add_join(&self, join: EpollJoinGuard) -> Result<(), EpollJoinGuard> { -+ let mut resources = self.resources.lock().unwrap(); -+ if !resources.active { -+ return Err(join); -+ } -+ resources.joins.push(join); -+ Ok(()) - } - -- /// Adds a registration guard that will detach handlers when dropped. -- pub fn add_join(&self, join: EpollJoinGuard) { -- self.joins.lock().unwrap().push(join); -+ pub(crate) fn attach_close_registration( -+ &self, -+ registration: EpollRegistrationGuard, -+ ) -> Result<(), EpollRegistrationGuard> { -+ let mut resources = self.resources.lock().unwrap(); -+ if !resources.active { -+ return Err(registration); -+ } -+ resources.close_registration = Some(registration); -+ Ok(()) -+ } -+ -+ /// Marks this subscription inactive, extracts owned resources under the -+ /// lifecycle lock, then drops them after releasing it. -+ pub(crate) fn deactivate_and_detach(&self) { -+ let (joins, close_registration) = { -+ let mut resources = self.resources.lock().unwrap(); -+ if !resources.active && resources.joins.is_empty() { -+ return; -+ } -+ resources.active = false; -+ ( -+ std::mem::take(&mut resources.joins), -+ resources.close_registration.take(), -+ ) -+ }; -+ drop(joins); -+ drop(close_registration); - } - -- /// Detaches and drops all registered handlers for this subscription. -- pub fn detach_joins(&self) { -- self.joins.lock().unwrap().clear(); -+ pub(crate) fn is_active(&self) -> bool { -+ self.resources.lock().unwrap().active - } - -- fn generation(&self) -> u64 { -- self.generation.load(Ordering::Acquire) -+ #[cfg(test)] -+ fn join_count(&self) -> usize { -+ self.resources.lock().unwrap().joins.len() -+ } -+ -+ fn registration_id(&self) -> u64 { -+ self.registration_id.load(Ordering::Acquire) - } - - pub(crate) fn fd_meta(&self) -> EpollFd { -@@ -369,14 +565,53 @@ impl EpollSubState { - fn clear_enqueued(&self) { - self.enqueued.store(false, Ordering::Release); - } -+ -+ fn poll_current_level_bits(&self) -> u8 { -+ let event = self.fd_meta(); -+ let mask_bits = epoll_mask_to_pending_bits(event.events()); -+ let mut bits = 0; -+ -+ let resources = self.resources.lock().unwrap(); -+ for join in resources.joins.iter() { -+ for readiness in join.fd_guard.poll_immediate_ready() { -+ if let Some(bit) = epoll_type_to_pending_bit(readiness) { -+ bits |= bit; -+ } -+ } -+ } -+ -+ bits & mask_bits -+ } -+ -+ fn has_tcp_listener_join(&self) -> bool { -+ self.resources.lock().unwrap().joins.iter().any(|join| { -+ let InodeValFilePollGuardMode::Socket { inner } = &join.fd_guard.mode else { -+ return false; -+ }; -+ matches!( -+ &inner.protected.read().unwrap().kind, -+ crate::net::socket::InodeSocketKind::TcpListener { .. } -+ ) -+ }) -+ } -+ -+ fn validate_current_level_bits(&self, bits: u8) -> u8 { -+ if !self.has_tcp_listener_join() { -+ return bits; -+ } -+ -+ let current_bits = self.poll_current_level_bits(); -+ let edge_bits = bits & (HUP_BIT | ERR_BIT); -+ (bits & current_bits) | edge_bits -+ } - } - - #[derive(Debug, Clone, Copy)] - struct ReadyItem { -- /// Watched fd key used to resolve current subscription state. -- fd: WasiFd, -- /// Generation snapshot captured when enqueued. -- generation: u64, -+ /// Exact subscription key used to resolve current state. -+ key: EpollSubscriptionKey, -+ /// Never-reused registration identity captured when enqueued. -+ registration_id: u64, - } - - /// Maps epoll readiness flags into internal pending-bit positions. -@@ -447,6 +682,7 @@ fn epoll_mask_to_pending_bits(mask: EpollType) -> u8 { - } - - fn prime_immediate_writable_if_applicable( -+ key: EpollSubscriptionKey, - event: &EpollFd, - fd_guard: &InodeValFilePollGuard, - epoll_state: &Arc, -@@ -472,18 +708,18 @@ fn prime_immediate_writable_if_applicable( - .pending_bits - .fetch_or(WRITABLE_BIT, Ordering::AcqRel); - if sub_state.mark_enqueued() { -- epoll_state.enqueue_ready(event.fd(), sub_state.generation()); -+ epoll_state.enqueue_ready(key, sub_state.registration_id()); - } - } - - /// Re-enqueues a subscription if new pending bits arrived during/after consumer drain. - fn repair_ready_queue_after_drain( - epoll_state: &Arc, -- fd: WasiFd, -+ key: EpollSubscriptionKey, - sub_state: &Arc, - ) { - if sub_state.pending_bits() != 0 && sub_state.mark_enqueued() { -- epoll_state.enqueue_ready(fd, sub_state.generation()); -+ epoll_state.enqueue_ready(key, sub_state.registration_id()); - } - } - -@@ -502,14 +738,11 @@ pub(crate) fn drain_ready_events( - let Some(item) = epoll_state.dequeue_ready() else { - break; - }; -- -- let Some(sub_state) = epoll_state.subscription(item.fd) else { -- epoll_empty_dequeue_entry(); -+ let Some(sub_state) = epoll_state.subscription(item.key) else { - continue; - }; - -- if sub_state.generation() != item.generation { -- epoll_stale_generation_drop(); -+ if sub_state.registration_id() != item.registration_id { - continue; - } - -@@ -517,8 +750,13 @@ pub(crate) fn drain_ready_events( - sub_state.clear_enqueued(); - - if bits == 0 { -- repair_ready_queue_after_drain(epoll_state, item.fd, &sub_state); -- epoll_empty_dequeue_entry(); -+ repair_ready_queue_after_drain(epoll_state, item.key, &sub_state); -+ continue; -+ } -+ -+ let bits = sub_state.validate_current_level_bits(bits); -+ if bits == 0 { -+ repair_ready_queue_after_drain(epoll_state, item.key, &sub_state); - continue; - } - -@@ -539,7 +777,7 @@ pub(crate) fn drain_ready_events( - .pending_bits - .fetch_or(undispatched_bits, Ordering::AcqRel); - } -- repair_ready_queue_after_drain(epoll_state, item.fd, &sub_state); -+ repair_ready_queue_after_drain(epoll_state, item.key, &sub_state); - - if ret.len() >= maxevents { - break; -@@ -548,59 +786,112 @@ pub(crate) fn drain_ready_events( - ret - } - -+/// Waits until at least one ready event can be drained. -+/// -+/// `tokio::Notify` requires the waiter to be enabled before the queue is -+/// checked. Otherwise a producer can enqueue readiness and call `notify_one` -+/// after an empty drain but before the `Notified` future is registered. -+pub(crate) async fn wait_for_ready_events( -+ epoll_state: &Arc, -+ maxevents: usize, -+) -> Vec<(EpollFd, EpollType)> { -+ loop { -+ let notified = epoll_state.notify.notified(); -+ tokio::pin!(notified); -+ notified.as_mut().enable(); -+ -+ let refreshed_ready_queue = epoll_state.refresh_level_readiness(); -+ let ret = drain_ready_events(epoll_state, maxevents); -+ if !ret.is_empty() { -+ return ret; -+ } -+ if refreshed_ready_queue { -+ continue; -+ } -+ if epoll_state.is_closed() { -+ return Vec::new(); -+ } -+ -+ notified.as_mut().await; -+ } -+} -+ - #[derive(Debug)] - struct EpollHandler { -- /// Watched fd associated with the subscription. -- fd: WasiFd, -+ /// Exact numeric-fd/open-file-description subscription identity. -+ key: EpollSubscriptionKey, - /// Parent epoll state for queueing and wakeups. -- epoll_state: Arc, -+ epoll_state: Weak, - /// Per-subscription state updated by interest callbacks. -- sub_state: Arc, -+ sub_state: Weak, - } - - impl EpollHandler { -- fn new(fd: WasiFd, epoll_state: Arc, sub_state: Arc) -> Box { -+ fn new( -+ key: EpollSubscriptionKey, -+ epoll_state: &Arc, -+ sub_state: &Arc, -+ ) -> Box { - Box::new(Self { -- fd, -- epoll_state, -- sub_state, -+ key, -+ epoll_state: Arc::downgrade(epoll_state), -+ sub_state: Arc::downgrade(sub_state), - }) - } - } - -+impl Drop for EpollHandler { -+ fn drop(&mut self) { -+ let Some(epoll_state) = self.epoll_state.upgrade() else { -+ return; -+ }; -+ let Some(sub_state) = self.sub_state.upgrade() else { -+ return; -+ }; -+ epoll_state.remove_if_same(self.key, &sub_state); -+ } -+} -+ - impl InterestHandler for EpollHandler { - /// Producer path: - /// set pending bits, enqueue once, and wake one waiter. - fn push_interest(&mut self, interest: InterestType) { -- EPOLL_ENQUEUE_ATTEMPTS.fetch_add(1, Ordering::Relaxed); -+ let Some(sub_state) = self.sub_state.upgrade() else { -+ return; -+ }; -+ let Some(epoll_state) = self.epoll_state.upgrade() else { -+ return; -+ }; -+ if !sub_state.is_active() { -+ return; -+ } - let bit = interest_to_pending_bit(interest); -- if !self.sub_state.set_pending(bit) { -- EPOLL_ENQUEUE_DEDUPE_HITS.fetch_add(1, Ordering::Relaxed); -+ if !sub_state.set_pending(bit) { - return; - } - -- if self.sub_state.mark_enqueued() { -- self.epoll_state -- .enqueue_ready(self.fd, self.sub_state.generation()); -- } else { -- EPOLL_ENQUEUE_DEDUPE_HITS.fetch_add(1, Ordering::Relaxed); -+ if sub_state.mark_enqueued() { -+ epoll_state.enqueue_ready(self.key, sub_state.registration_id()); - } - } - - /// Clears one readiness bit from this subscription only. - fn pop_interest(&mut self, interest: InterestType) -> bool { -+ let Some(sub_state) = self.sub_state.upgrade() else { -+ return false; -+ }; - let bit = interest_to_pending_bit(interest); -- let old = self -- .sub_state -- .pending_bits -- .fetch_and(!bit, Ordering::AcqRel); -+ let old = sub_state.pending_bits.fetch_and(!bit, Ordering::AcqRel); - (old & bit) != 0 - } - - /// Checks whether this subscription currently has a readiness bit set. - fn has_interest(&self, interest: InterestType) -> bool { -+ let Some(sub_state) = self.sub_state.upgrade() else { -+ return false; -+ }; - let bit = interest_to_pending_bit(interest); -- (self.sub_state.pending_bits() & bit) != 0 -+ (sub_state.pending_bits() & bit) != 0 - } - } - -@@ -608,7 +899,8 @@ impl InterestHandler for EpollHandler { - /// - /// `None` means the fd kind does not support handler attachment for epoll. - pub(crate) fn register_epoll_handler( -- state: &Arc, -+ fd_entry: &Fd, -+ key: EpollSubscriptionKey, - event: &EpollFd, - epoll_state: Arc, - sub_state: Arc, -@@ -637,57 +929,128 @@ pub(crate) fn register_epoll_handler( - }, - }; - -- let fd_guard = poll_fd_guard(state, peb.build(), event.fd(), s)?; -- let handler = EpollHandler::new(event.fd(), epoll_state.clone(), sub_state.clone()); -+ let requires_access = match s.type_ { -+ Eventtype::FdRead => Rights::FD_READ, -+ Eventtype::FdWrite => Rights::FD_WRITE, -+ _ => Rights::empty(), -+ }; -+ if !(fd_entry.inner.rights.contains(Rights::POLL_FD_READWRITE) -+ && fd_entry.inner.rights.contains(requires_access)) -+ { -+ return Err(Errno::Access); -+ } -+ let fd_guard = { -+ let guard = fd_entry.inode.read(); -+ InodeValFilePollGuard::new(event.fd(), peb.build(), s, &guard).ok_or(Errno::Badf)? -+ }; - - match &fd_guard.mode { - InodeValFilePollGuardMode::File(_) => { - // Intentionally ignored, epoll doesn't work with files - return Ok(None); - } -+ InodeValFilePollGuardMode::PipeTx { .. } => { -+ // The sending end of a pipe can't have an interest handler, since we -+ // only support "readable" interest on pipes; they're considered to -+ // always be writable. -+ prime_immediate_writable_if_applicable(key, event, &fd_guard, &epoll_state, &sub_state); -+ return Ok(Some(EpollJoinGuard::new(fd_guard, None))); -+ } -+ _ => {} -+ } -+ -+ // One source may feed multiple epolls and multiple dup-backed numeric fds. -+ // The OFD-owned fanout installs once on the source and gives this -+ // subscription an independently removable handler token. -+ let fanout = fd_entry.inner.ofd.interest_fanout(); -+ let handler_registration = fanout.register(EpollHandler::new(key, &epoll_state, &sub_state)); -+ match &fd_guard.mode { - InodeValFilePollGuardMode::Socket { inner, .. } => { - let mut inner = inner.protected.write().unwrap(); -- inner.set_handler(handler).map_err(net_error_into_io_err)?; -- drop(inner); -+ inner -+ .set_handler(Box::new(fanout.clone())) -+ .map_err(net_error_into_io_err)?; -+ } -+ InodeValFilePollGuardMode::EventNotifications(inner) => { -+ inner.set_interest_handler(Box::new(fanout.clone())); - } -- InodeValFilePollGuardMode::EventNotifications(inner) => inner.set_interest_handler(handler), - InodeValFilePollGuardMode::DuplexPipe { pipe } => { - let inner = pipe.write().unwrap(); -- inner.set_interest_handler(handler); -+ inner.set_interest_handler(Box::new(fanout.clone())); - } - InodeValFilePollGuardMode::PipeRx { rx } => { - let inner = rx.write().unwrap(); -- inner.set_interest_handler(handler); -+ inner.set_interest_handler(Box::new(fanout.clone())); - } -- InodeValFilePollGuardMode::PipeTx { .. } => { -- // The sending end of a pipe can't have an interest handler, since we -- // only support "readable" interest on pipes; they're considered to -- // always be writable. -- prime_immediate_writable_if_applicable(event, &fd_guard, &epoll_state, &sub_state); -- return Ok(None); -+ InodeValFilePollGuardMode::File(_) | InodeValFilePollGuardMode::PipeTx { .. } => { -+ unreachable!("non-handler fd kinds returned above") - } - } - -- prime_immediate_writable_if_applicable(event, &fd_guard, &epoll_state, &sub_state); -- -- Ok(Some(EpollJoinGuard::new(fd_guard))) --} -+ prime_immediate_writable_if_applicable(key, event, &fd_guard, &epoll_state, &sub_state); - --/// Increments stale-generation dequeue metric. --pub(crate) fn epoll_stale_generation_drop() { -- EPOLL_STALE_GENERATION_DROPS.fetch_add(1, Ordering::Relaxed); --} -- --/// Increments empty dequeue metric. --pub(crate) fn epoll_empty_dequeue_entry() { -- EPOLL_EMPTY_DEQUEUE_ENTRIES.fetch_add(1, Ordering::Relaxed); -+ Ok(Some(EpollJoinGuard::new( -+ fd_guard, -+ Some(handler_registration), -+ ))) - } - - #[cfg(test)] - mod tests { - use super::*; -- use std::sync::RwLock; -+ use crate::net::socket::{InodeSocket, InodeSocketKind}; -+ use std::{ -+ io::Write, -+ net::{Ipv4Addr, SocketAddr}, -+ sync::{Barrier, RwLock}, -+ task::{Context, Poll}, -+ thread, -+ }; - use virtual_fs::Pipe; -+ use virtual_mio::InterestHandler; -+ use virtual_net::{ -+ NetworkError, Result as NetResult, VirtualIoSource, VirtualTcpListener, VirtualTcpSocket, -+ }; -+ -+ #[derive(Debug)] -+ struct PendingTcpListener; -+ -+ impl VirtualIoSource for PendingTcpListener { -+ fn remove_handler(&mut self) {} -+ -+ fn poll_read_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { -+ Poll::Pending -+ } -+ -+ fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { -+ Poll::Pending -+ } -+ } -+ -+ impl VirtualTcpListener for PendingTcpListener { -+ fn try_accept(&mut self) -> NetResult<(Box, SocketAddr)> { -+ Err(NetworkError::WouldBlock) -+ } -+ -+ fn set_handler( -+ &mut self, -+ _handler: Box, -+ ) -> NetResult<()> { -+ Ok(()) -+ } -+ -+ fn addr_local(&self) -> NetResult { -+ Ok(SocketAddr::from((Ipv4Addr::LOCALHOST, 0))) -+ } -+ -+ fn set_ttl(&mut self, _ttl: u8) -> NetResult<()> { -+ Ok(()) -+ } -+ -+ fn ttl(&self) -> NetResult { -+ Ok(64) -+ } -+ } - - fn test_epoll_event_ctl(fd: WasiFd) -> EpollEventCtl { - EpollEventCtl { -@@ -699,6 +1062,10 @@ mod tests { - } - } - -+ fn test_key(fd: WasiFd) -> EpollSubscriptionKey { -+ EpollSubscriptionKey::new(fd, 10_000 + u64::from(fd)) -+ } -+ - fn test_epoll_handler(fd: WasiFd) -> (Arc, Arc, Box) { - let epoll_state = Arc::new(EpollState::new()); - let sub_state = Arc::new(EpollSubState::new( -@@ -714,7 +1081,7 @@ mod tests { - ), - 1, - )); -- let handler = EpollHandler::new(fd, epoll_state.clone(), sub_state.clone()); -+ let handler = EpollHandler::new(test_key(fd), &epoll_state, &sub_state); - (epoll_state, sub_state, handler) - } - -@@ -734,6 +1101,29 @@ mod tests { - )) - } - -+ fn test_pipe_tx_join(fd: WasiFd) -> EpollJoinGuard { -+ let (tx, _rx) = Pipe::new().split(); -+ EpollJoinGuard::new( -+ InodeValFilePollGuard { -+ fd, -+ peb: PollEventBuilder::new().build(), -+ subscription: Subscription { -+ userdata: 0, -+ type_: Eventtype::FdRead, -+ data: SubscriptionUnion { -+ fd_readwrite: SubscriptionFsReadwrite { -+ file_descriptor: fd, -+ }, -+ }, -+ }, -+ mode: InodeValFilePollGuardMode::PipeTx { -+ tx: Arc::new(RwLock::new(Box::new(tx))), -+ }, -+ }, -+ None, -+ ) -+ } -+ - #[test] - fn epoll_fd_from_event_ctl_uses_explicit_fd() { - let event = test_epoll_event_ctl(1234); -@@ -756,8 +1146,8 @@ mod tests { - EpollFd::new(EpollType::EPOLLIN, 0, 11, 0, 0), - 1, - )); -- let mut handler1 = EpollHandler::new(10, epoll_state.clone(), sub_state1.clone()); -- let mut handler2 = EpollHandler::new(11, epoll_state.clone(), sub_state2.clone()); -+ let mut handler1 = EpollHandler::new(test_key(10), &epoll_state, &sub_state1); -+ let mut handler2 = EpollHandler::new(test_key(11), &epoll_state, &sub_state2); - - handler1.push_interest(InterestType::Readable); - handler2.push_interest(InterestType::Readable); -@@ -805,6 +1195,22 @@ mod tests { - ); - } - -+ #[test] -+ fn epoll_handler_does_not_retain_closed_subscription_graph() { -+ let (epoll_state, sub_state, mut handler) = test_epoll_handler(8); -+ let epoll_state_weak = Arc::downgrade(&epoll_state); -+ let sub_state_weak = Arc::downgrade(&sub_state); -+ -+ drop(epoll_state); -+ drop(sub_state); -+ -+ assert!(epoll_state_weak.upgrade().is_none()); -+ assert!(sub_state_weak.upgrade().is_none()); -+ handler.push_interest(InterestType::Readable); -+ assert!(!handler.has_interest(InterestType::Readable)); -+ assert!(!handler.pop_interest(InterestType::Readable)); -+ } -+ - #[test] - fn epoll_type_to_pending_bit_has_stable_mapping() { - assert_eq!( -@@ -865,10 +1271,10 @@ mod tests { - sub_b.pending_bits.store(readable_bit, Ordering::Release); - sub_b.enqueued.store(true, Ordering::Release); - -- epoll_state.insert_subscription(10, sub_a); -- epoll_state.insert_subscription(11, sub_b); -- epoll_state.enqueue_ready(10, 1); -- epoll_state.enqueue_ready(11, 1); -+ epoll_state.insert_subscription(test_key(10), sub_a); -+ epoll_state.insert_subscription(test_key(11), sub_b); -+ epoll_state.enqueue_ready(test_key(10), 1); -+ epoll_state.enqueue_ready(test_key(11), 1); - - let events = drain_ready_events(&epoll_state, 8); - assert_eq!(events.len(), 2); -@@ -889,8 +1295,8 @@ mod tests { - sub.pending_bits - .store(READABLE_BIT | WRITABLE_BIT, Ordering::Release); - sub.enqueued.store(true, Ordering::Release); -- epoll_state.insert_subscription(90, sub.clone()); -- epoll_state.enqueue_ready(90, 1); -+ epoll_state.insert_subscription(test_key(90), sub.clone()); -+ epoll_state.enqueue_ready(test_key(90), 1); - - let first = drain_ready_events(&epoll_state, 1); - assert_eq!(first.len(), 1); -@@ -908,7 +1314,7 @@ mod tests { - } - - #[test] -- fn drain_ready_events_drops_stale_generation_items() { -+ fn drain_ready_events_drops_stale_registration_items() { - let epoll_state = Arc::new(EpollState::new()); - - let sub = test_sub_state(22, 2); -@@ -916,19 +1322,138 @@ mod tests { - sub.pending_bits.store(readable_bit, Ordering::Release); - sub.enqueued.store(true, Ordering::Release); - -- epoll_state.insert_subscription(22, sub.clone()); -- epoll_state.enqueue_ready(22, 1); -+ epoll_state.insert_subscription(test_key(22), sub.clone()); -+ epoll_state.enqueue_ready(test_key(22), 1); - - let events = drain_ready_events(&epoll_state, 8); - assert!( - events.is_empty(), -- "stale generation items must not emit events" -+ "stale registration items must not emit events" - ); - assert_eq!( - sub.pending_bits.load(Ordering::Acquire), - readable_bit, -- "stale dequeue must not clear pending bits for current generation" -+ "stale dequeue must not clear pending bits for current registration" -+ ); -+ } -+ -+ #[test] -+ fn refresh_level_readiness_requeues_current_pipe_readiness() { -+ let epoll_state = Arc::new(EpollState::new()); -+ let sub = Arc::new(EpollSubState::new( -+ EpollFd::new(EpollType::EPOLLIN, 0, 77, 0, 0), -+ 1, -+ )); -+ epoll_state.insert_subscription(test_key(77), sub.clone()); -+ -+ let (mut tx, rx) = Pipe::new().split(); -+ tx.write_all(b"ready").unwrap(); -+ sub.add_join(EpollJoinGuard::new( -+ InodeValFilePollGuard { -+ fd: 77, -+ peb: PollEventBuilder::new() -+ .add(PollEvent::PollIn) -+ .add(PollEvent::PollError) -+ .add(PollEvent::PollHangUp) -+ .build(), -+ subscription: Subscription { -+ userdata: 0, -+ type_: Eventtype::FdRead, -+ data: SubscriptionUnion { -+ fd_readwrite: SubscriptionFsReadwrite { -+ file_descriptor: 77, -+ }, -+ }, -+ }, -+ mode: InodeValFilePollGuardMode::PipeRx { -+ rx: Arc::new(RwLock::new(Box::new(rx))), -+ }, -+ }, -+ None, -+ )) -+ .unwrap(); -+ -+ epoll_state.refresh_level_readiness(); -+ -+ let events = drain_ready_events(&epoll_state, 8); -+ assert_eq!(events.len(), 1); -+ assert_eq!(events[0].0.fd(), 77); -+ assert_eq!(events[0].1, EpollType::EPOLLIN); -+ } -+ -+ #[test] -+ fn drain_ready_events_drops_stale_tcp_listener_readiness() { -+ let epoll_state = Arc::new(EpollState::new()); -+ let sub = Arc::new(EpollSubState::new( -+ EpollFd::new(EpollType::EPOLLIN, 0, 78, 0, 0), -+ 1, -+ )); -+ epoll_state.insert_subscription(test_key(78), sub.clone()); -+ -+ let listener = InodeSocket::new(InodeSocketKind::TcpListener { -+ socket: Box::new(PendingTcpListener), -+ accept_timeout: None, -+ }); -+ sub.add_join(EpollJoinGuard::new( -+ InodeValFilePollGuard { -+ fd: 78, -+ peb: PollEventBuilder::new() -+ .add(PollEvent::PollIn) -+ .add(PollEvent::PollError) -+ .add(PollEvent::PollHangUp) -+ .build(), -+ subscription: Subscription { -+ userdata: 0, -+ type_: Eventtype::FdRead, -+ data: SubscriptionUnion { -+ fd_readwrite: SubscriptionFsReadwrite { -+ file_descriptor: 78, -+ }, -+ }, -+ }, -+ mode: InodeValFilePollGuardMode::Socket { -+ inner: listener.inner.clone(), -+ }, -+ }, -+ None, -+ )) -+ .unwrap(); -+ sub.pending_bits.store(READABLE_BIT, Ordering::Release); -+ sub.enqueued.store(true, Ordering::Release); -+ epoll_state.enqueue_ready(test_key(78), 1); -+ -+ let events = drain_ready_events(&epoll_state, 8); -+ assert!( -+ events.is_empty(), -+ "stale producer readiness must not be dispatched after level revalidation" - ); -+ assert_eq!(sub.pending_bits(), 0); -+ assert!(!sub.enqueued.load(Ordering::Acquire)); -+ assert_eq!(epoll_state.ready.lock().unwrap().len(), 0); -+ } -+ -+ #[tokio::test] -+ async fn wait_for_ready_events_observes_async_producer() { -+ let (epoll_state, sub_state, mut handler) = test_epoll_handler(33); -+ epoll_state.insert_subscription(test_key(33), sub_state); -+ -+ let producer = tokio::spawn(async move { -+ tokio::task::yield_now().await; -+ handler.push_interest(InterestType::Readable); -+ handler -+ }); -+ -+ let events = tokio::time::timeout( -+ std::time::Duration::from_secs(1), -+ wait_for_ready_events(&epoll_state, 8), -+ ) -+ .await -+ .expect("waiter should wake when the producer enqueues readiness"); -+ -+ let _handler = producer.await.unwrap(); -+ assert_eq!(events.len(), 1); -+ assert_eq!(events[0].0.fd(), 33); -+ assert_eq!(events[0].1, EpollType::EPOLLIN); - } - - #[test] -@@ -939,11 +1464,11 @@ mod tests { - sub.pending_bits.store(writable_bit, Ordering::Release); - sub.enqueued.store(false, Ordering::Release); - -- repair_ready_queue_after_drain(&epoll_state, 44, &sub); -+ repair_ready_queue_after_drain(&epoll_state, test_key(44), &sub); - - assert!(sub.enqueued.load(Ordering::Acquire)); - let queued = epoll_state.ready.lock().unwrap().pop_front().unwrap(); -- assert_eq!(queued.fd, 44); -+ assert_eq!(queued.key.fd(), 44); - } - - #[test] -@@ -951,31 +1476,241 @@ mod tests { - let epoll_state = Arc::new(EpollState::new()); - let event = test_epoll_event_ctl(55); - let sub = Arc::new(EpollSubState::new(EpollFd::from_event_ctl(55, &event), 1)); -- epoll_state.insert_subscription(55, sub.clone()); -+ epoll_state.insert_subscription(test_key(55), sub.clone()); - - let (tx, _rx) = Pipe::new().split(); -- sub.add_join(EpollJoinGuard::new(InodeValFilePollGuard { -- fd: 55, -- peb: PollEventBuilder::new().build(), -- subscription: Subscription { -- userdata: 0, -- type_: Eventtype::FdRead, -- data: SubscriptionUnion { -- fd_readwrite: SubscriptionFsReadwrite { -- file_descriptor: 55, -+ sub.add_join(EpollJoinGuard::new( -+ InodeValFilePollGuard { -+ fd: 55, -+ peb: PollEventBuilder::new().build(), -+ subscription: Subscription { -+ userdata: 0, -+ type_: Eventtype::FdRead, -+ data: SubscriptionUnion { -+ fd_readwrite: SubscriptionFsReadwrite { -+ file_descriptor: 55, -+ }, - }, - }, -+ mode: InodeValFilePollGuardMode::PipeTx { -+ tx: Arc::new(RwLock::new(Box::new(tx))), -+ }, - }, -- mode: InodeValFilePollGuardMode::PipeTx { -- tx: Arc::new(RwLock::new(Box::new(tx))), -- }, -- })); -+ None, -+ )) -+ .unwrap(); - - let leaked_ref = sub.clone(); -- assert_eq!(leaked_ref.joins.lock().unwrap().len(), 1); -+ assert_eq!(leaked_ref.join_count(), 1); -+ -+ epoll_state.apply_del(test_key(55)).unwrap(); -+ -+ assert_eq!(leaked_ref.join_count(), 0); -+ } -+ -+ #[test] -+ fn add_join_racing_epoll_final_close_cannot_retain_guard() { -+ // Exercise both valid mutex orderings repeatedly: either ADD attaches -+ // first and close drains it, or close wins and ADD rejects the guard. -+ for iteration in 0..64 { -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(56, 50_000 + iteration); -+ let (_, subscription) = state.prepare_add(key, &test_epoll_event_ctl(56)).unwrap(); -+ let barrier = Arc::new(Barrier::new(3)); -+ -+ let add_subscription = subscription.clone(); -+ let add_barrier = barrier.clone(); -+ let add = thread::spawn(move || { -+ add_barrier.wait(); -+ drop(add_subscription.add_join(test_pipe_tx_join(56))); -+ }); -+ -+ let close_state = state.clone(); -+ let close_barrier = barrier.clone(); -+ let close = thread::spawn(move || { -+ close_barrier.wait(); -+ close_state.close(); -+ }); -+ -+ barrier.wait(); -+ add.join().unwrap(); -+ close.join().unwrap(); -+ -+ assert!(state.is_closed()); -+ assert!(!subscription.is_active()); -+ assert_eq!(subscription.join_count(), 0); -+ } -+ } -+ -+ #[test] -+ fn readiness_enqueue_racing_epoll_final_close_leaves_queue_drained() { -+ for iteration in 0..64 { -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(57, 60_000 + iteration); -+ let (_, subscription) = state.prepare_add(key, &test_epoll_event_ctl(57)).unwrap(); -+ let mut handler = EpollHandler::new(key, &state, &subscription); -+ let barrier = Arc::new(Barrier::new(3)); -+ -+ let push_barrier = barrier.clone(); -+ let push = thread::spawn(move || { -+ push_barrier.wait(); -+ handler.push_interest(InterestType::Readable); -+ handler -+ }); -+ -+ let close_state = state.clone(); -+ let close_barrier = barrier.clone(); -+ let close = thread::spawn(move || { -+ close_barrier.wait(); -+ close_state.close(); -+ }); -+ -+ barrier.wait(); -+ let _handler = push.join().unwrap(); -+ close.join().unwrap(); -+ -+ assert!(state.is_closed()); -+ assert!(state.ready.lock().unwrap().is_empty()); -+ assert_eq!(state.subscription_count(), 0); -+ assert!(!subscription.is_active()); -+ } -+ } -+ -+ #[test] -+ fn ofd_aware_keys_allow_dup_backed_old_watch_and_reused_numeric_fd() { -+ let state = Arc::new(EpollState::new()); -+ let old_key = EpollSubscriptionKey::new(5, 1001); -+ let new_key = EpollSubscriptionKey::new(5, 1002); -+ let mut old_event = test_epoll_event_ctl(5); -+ old_event.data1 = 1; -+ let mut new_event = test_epoll_event_ctl(5); -+ new_event.data1 = 2; -+ let (_, old_sub) = state.prepare_add(old_key, &old_event).unwrap(); -+ let (_, new_sub) = state.prepare_add(new_key, &new_event).unwrap(); -+ -+ old_sub.pending_bits.store(READABLE_BIT, Ordering::Release); -+ old_sub.enqueued.store(true, Ordering::Release); -+ new_sub.pending_bits.store(READABLE_BIT, Ordering::Release); -+ new_sub.enqueued.store(true, Ordering::Release); -+ state.enqueue_ready(old_key, old_sub.registration_id()); -+ state.enqueue_ready(new_key, new_sub.registration_id()); -+ -+ let events = drain_ready_events(&state, 8); -+ assert_eq!(events.len(), 2); -+ assert!(events.iter().all(|(event, _)| event.fd() == 5)); -+ let payloads = events -+ .iter() -+ .map(|(event, _)| event.data1()) -+ .collect::>(); -+ assert_eq!(payloads, std::collections::HashSet::from([1, 2])); -+ } - -- epoll_state.apply_del(55).unwrap(); -+ #[test] -+ fn del_add_same_ofd_key_cannot_replay_stale_ready_item() { -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(6, 2001); -+ let mut old_event = test_epoll_event_ctl(6); -+ old_event.data1 = 1; -+ let (_, old_sub) = state.prepare_add(key, &old_event).unwrap(); -+ old_sub.pending_bits.store(READABLE_BIT, Ordering::Release); -+ old_sub.enqueued.store(true, Ordering::Release); -+ state.enqueue_ready(key, old_sub.registration_id()); -+ state.apply_del(key).unwrap(); -+ -+ let mut new_event = test_epoll_event_ctl(6); -+ new_event.data1 = 2; -+ let (_, new_sub) = state.prepare_add(key, &new_event).unwrap(); -+ new_sub.pending_bits.store(READABLE_BIT, Ordering::Release); -+ new_sub.enqueued.store(true, Ordering::Release); -+ state.enqueue_ready(key, new_sub.registration_id()); -+ -+ let events = drain_ready_events(&state, 8); -+ assert_eq!(events.len(), 1); -+ assert_eq!(events[0].0.data1(), 2); -+ } -+ -+ #[test] -+ fn delayed_old_handler_drop_cannot_remove_replacement_subscription() { -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(7, 3001); -+ let (_, old_sub) = state.prepare_add(key, &test_epoll_event_ctl(7)).unwrap(); -+ let old_handler = EpollHandler::new(key, &state, &old_sub); -+ state.apply_del(key).unwrap(); -+ let (_, replacement) = state.prepare_add(key, &test_epoll_event_ctl(7)).unwrap(); - -- assert_eq!(leaked_ref.joins.lock().unwrap().len(), 0); -+ drop(old_handler); -+ -+ assert!(state.contains_exact_subscription(key, &replacement)); -+ assert!(replacement.is_active()); -+ } -+ -+ #[test] -+ fn failed_older_mod_cannot_rollback_over_later_subscription() { -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(71, 7001); -+ let (_, original) = state.prepare_add(key, &test_epoll_event_ctl(71)).unwrap(); -+ let (_, first_provisional, replaced_original) = -+ state.prepare_mod(key, &test_epoll_event_ctl(72)).unwrap(); -+ assert!(Arc::ptr_eq(&original, &replaced_original)); -+ -+ // Force the exact historical interleave: an older MOD is between map -+ // publication and source installation while a later MOD replaces it. -+ let later_published = Arc::new(Barrier::new(2)); -+ let later_state = state.clone(); -+ let later_barrier = later_published.clone(); -+ let first_for_later = first_provisional.clone(); -+ let later = thread::spawn(move || { -+ let (_, later_subscription, replaced_first) = later_state -+ .prepare_mod(key, &test_epoll_event_ctl(73)) -+ .unwrap(); -+ assert!(Arc::ptr_eq(&replaced_first, &first_for_later)); -+ replaced_first.deactivate_and_detach(); -+ later_barrier.wait(); -+ later_subscription -+ }); -+ -+ later_published.wait(); -+ let later_subscription = later.join().unwrap(); -+ -+ // The first operation now reports a source-install failure. Its -+ // rollback must be conditional on exact identity and restoration must -+ // not overwrite the later successful registration. -+ assert!(!state.rollback_registration(key, &first_provisional)); -+ assert!(matches!( -+ state.prepare_restore(key, replaced_original.fd_meta()), -+ Err(Errno::Exist) -+ )); -+ assert!(state.contains_exact_subscription(key, &later_subscription)); -+ assert!(later_subscription.is_active()); -+ } -+ -+ #[test] -+ fn fanout_delivers_to_two_epolls_and_removing_one_keeps_the_other() { -+ let first_state = Arc::new(EpollState::new()); -+ let second_state = Arc::new(EpollState::new()); -+ let first_key = EpollSubscriptionKey::new(8, 4001); -+ let second_key = EpollSubscriptionKey::new(9, 4001); -+ let (_, first_sub) = first_state -+ .prepare_add(first_key, &test_epoll_event_ctl(8)) -+ .unwrap(); -+ let (_, second_sub) = second_state -+ .prepare_add(second_key, &test_epoll_event_ctl(9)) -+ .unwrap(); -+ let mut fanout = virtual_mio::InterestHandlerFanout::default(); -+ let first_registration = -+ fanout.register(EpollHandler::new(first_key, &first_state, &first_sub)); -+ let _second_registration = -+ fanout.register(EpollHandler::new(second_key, &second_state, &second_sub)); -+ -+ fanout.push_interest(InterestType::Readable); -+ assert_eq!(drain_ready_events(&first_state, 8).len(), 1); -+ assert_eq!(drain_ready_events(&second_state, 8).len(), 1); -+ -+ drop(first_registration); -+ assert_eq!(first_state.subscription_count(), 0); -+ assert_eq!(fanout.handler_count(), 1); -+ fanout.push_interest(InterestType::Readable); -+ assert!(drain_ready_events(&first_state, 8).is_empty()); -+ assert_eq!(drain_ready_events(&second_state, 8).len(), 1); - } - } -diff --git a/lib/wasix/src/os/task/control_plane.rs b/lib/wasix/src/os/task/control_plane.rs -index 18e8ada..3468f5b 100644 ---- a/lib/wasix/src/os/task/control_plane.rs -+++ b/lib/wasix/src/os/task/control_plane.rs -@@ -7,8 +7,14 @@ use std::{ - time::Duration, - }; - --use crate::{WasiProcess, WasiProcessId}; -+#[cfg(test)] -+use std::sync::atomic::AtomicBool; -+ -+use crate::os::task::process::WasiChildPublicationGuard; -+use crate::os::task::thread::WasiMemoryLayout; -+use crate::{WasiProcess, WasiProcessId, WasiThreadHandle}; - use wasmer_types::ModuleHash; -+use wasmer_wasix_types::wasix::ThreadStartType; - - #[derive(Debug, Clone)] - pub struct WasiControlPlane { -@@ -73,6 +79,10 @@ struct State { - /// Total number of active tasks (threads) across all processes. - task_count: Arc, - -+ /// Deterministic failure injection for post-seal transaction tests. -+ #[cfg(test)] -+ fail_next_task_admission: AtomicBool, -+ - /// Mutable state. - mutable: RwLock, - } -@@ -92,6 +102,8 @@ impl WasiControlPlane { - state: Arc::new(State { - config, - task_count: Arc::new(AtomicUsize::new(0)), -+ #[cfg(test)] -+ fail_next_task_admission: AtomicBool::new(false), - mutable: RwLock::new(MutableState { - process_seed: 0, - processes: Default::default(), -@@ -105,8 +117,8 @@ impl WasiControlPlane { - } - - /// Get the current count of active tasks (threads). -- fn active_task_count(&self) -> usize { -- self.state.task_count.load(Ordering::SeqCst) -+ pub(crate) fn active_task_count(&self) -> usize { -+ self.state.task_count.load(Ordering::Acquire) - } - - /// Returns the configuration for this control plane -@@ -116,22 +128,73 @@ impl WasiControlPlane { - - /// Register a new task. - /// -- // Currently just increments the task counter. -+ // The CAS is the authoritative admission decision. An advisory process -+ // precheck cannot enforce a global limit when threads start concurrently. - pub(crate) fn register_task(&self) -> Result { -- let count = self.state.task_count.fetch_add(1, Ordering::SeqCst); -- if let Some(max) = self.state.config.max_task_count -- && count > max -+ #[cfg(test)] -+ if self -+ .state -+ .fail_next_task_admission -+ .swap(false, Ordering::AcqRel) - { -- self.state.task_count.fetch_sub(1, Ordering::SeqCst); -- return Err(ControlPlaneError::TaskLimitReached { max: count }); -+ return Err(ControlPlaneError::TaskLimitReached { -+ max: self.state.config.max_task_count.unwrap_or(usize::MAX), -+ }); -+ } -+ -+ let mut current = self.state.task_count.load(Ordering::Acquire); -+ loop { -+ if let Some(max) = self.state.config.max_task_count -+ && current >= max -+ { -+ return Err(ControlPlaneError::TaskLimitReached { max }); -+ } -+ let Some(next) = current.checked_add(1) else { -+ return Err(ControlPlaneError::TaskLimitReached { max: usize::MAX }); -+ }; -+ match self.state.task_count.compare_exchange_weak( -+ current, -+ next, -+ Ordering::AcqRel, -+ Ordering::Acquire, -+ ) { -+ Ok(_) => return Ok(TaskCountGuard(self.state.task_count.clone())), -+ Err(observed) => current = observed, -+ } - } -- Ok(TaskCountGuard(self.state.task_count.clone())) - } - -- /// Creates a new process -- // FIXME: De-register terminated processes! -- // Currently they just accumulate. -- pub fn new_process(&self, module_hash: ModuleHash) -> Result { -+ /// Creates and publishes a process together with its main thread. -+ /// -+ /// A process is never globally visible without a successfully admitted -+ /// main thread. Finished processes remain registered as zombies until a -+ /// successful join reaps them. -+ pub fn new_process_with_main_thread( -+ &self, -+ module_hash: ModuleHash, -+ layout: WasiMemoryLayout, -+ ) -> Result<(WasiProcess, WasiThreadHandle), ControlPlaneError> { -+ let (process, handle, mut registration) = -+ self.new_process_with_main_thread_guarded(module_hash, layout)?; -+ registration.commit(); -+ Ok((process, handle)) -+ } -+ -+ /// Reserves a unique PID and constructs a process off-map. Until the -+ /// returned guard is committed, registry lookups, signals, joins, and -+ /// process enumeration cannot observe or retain the tentative process. -+ pub(crate) fn new_process_guarded( -+ &self, -+ module_hash: ModuleHash, -+ ) -> Result<(WasiProcess, WasiProcessRegistrationGuard), ControlPlaneError> { -+ self.new_process_guarded_with_lifecycle(module_hash, None) -+ } -+ -+ fn new_process_guarded_with_lifecycle( -+ &self, -+ module_hash: ModuleHash, -+ child_parent: Option<&WasiProcess>, -+ ) -> Result<(WasiProcess, WasiProcessRegistrationGuard), ControlPlaneError> { - if let Some(max) = self.state.config.max_task_count - && self.active_task_count() >= max - { -@@ -140,15 +203,73 @@ impl WasiControlPlane { - return Err(ControlPlaneError::TaskLimitReached { max }); - } - -- // Create the process first to do all the allocations before locking. -- let mut proc = WasiProcess::new(WasiProcessId::from(0), module_hash, self.handle()); -+ let pid = self.state.mutable.write().unwrap().next_process_id()?; -+ let process = match child_parent { -+ Some(parent) => WasiProcess::new_pending_child( -+ pid, -+ module_hash, -+ self.handle(), -+ parent, -+ parent.tree_epoch(), -+ ), -+ None => WasiProcess::new(pid, module_hash, self.handle()), -+ }; -+ let guard = WasiProcessRegistrationGuard { -+ control_plane: self.clone(), -+ process: process.clone(), -+ parent: None, -+ parent_publication: None, -+ committed: false, -+ launch_complete: false, -+ }; -+ Ok((process, guard)) -+ } - -+ /// Atomically publishes a PID reserved by `new_process_guarded`. Reserved -+ /// IDs are monotonic and never reused, so occupancy indicates an internal -+ /// invariant violation rather than a recoverable runtime race. -+ fn publish_reserved_process(&self, process: &WasiProcess) { - let mut mutable = self.state.mutable.write().unwrap(); -+ match mutable.processes.entry(process.pid()) { -+ std::collections::hash_map::Entry::Vacant(entry) => { -+ entry.insert(process.clone()); -+ } -+ std::collections::hash_map::Entry::Occupied(_) => { -+ panic!("reserved process ID was published more than once"); -+ } -+ } -+ } - -- let pid = mutable.next_process_id()?; -- proc.set_pid(pid); -- mutable.processes.insert(pid, proc.clone()); -- Ok(proc) -+ /// Creates an off-map process and a registered main thread as one -+ /// failure-safe unit. Task admission failure drops the tentative object; -+ /// successful callers publish it by committing the returned guard. -+ pub(crate) fn new_process_with_main_thread_guarded( -+ &self, -+ module_hash: ModuleHash, -+ layout: WasiMemoryLayout, -+ ) -> Result<(WasiProcess, WasiThreadHandle, WasiProcessRegistrationGuard), ControlPlaneError> -+ { -+ let (process, guard) = self.new_process_guarded(module_hash)?; -+ let handle = process.new_thread(layout, ThreadStartType::MainThread)?; -+ Ok((process, handle, guard)) -+ } -+ -+ /// Reserves parent publication first, then creates the tentative child in -+ /// the parent's tree epoch with a pending guest-start gate. This is the -+ /// only supported constructor for a process that will be adopted. -+ pub(crate) fn new_child_process_with_main_thread_guarded( -+ &self, -+ parent: &WasiProcess, -+ module_hash: ModuleHash, -+ layout: WasiMemoryLayout, -+ ) -> Result<(WasiProcess, WasiThreadHandle, WasiProcessRegistrationGuard), ControlPlaneError> -+ { -+ let publication = parent.begin_child_publication()?; -+ let (process, guard) = -+ self.new_process_guarded_with_lifecycle(module_hash, Some(parent))?; -+ let handle = process.new_thread(layout, ThreadStartType::MainThread)?; -+ let guard = guard.with_parent(parent, publication); -+ Ok((process, handle, guard)) - } - - /// Generates a new process ID -@@ -167,6 +288,240 @@ impl WasiControlPlane { - .get(&pid) - .cloned() - } -+ -+ /// Removes a finished process from the global registry after it has been -+ /// joined. The identity check prevents a stale handle from removing a -+ /// future process if PID reuse is introduced. -+ pub(crate) fn reap_process(&self, process: &WasiProcess) -> bool { -+ if process.try_join().is_none() { -+ return false; -+ } -+ -+ let mut mutable = self.state.mutable.write().unwrap(); -+ let is_registered_process = mutable -+ .processes -+ .get(&process.pid()) -+ .is_some_and(|registered| Arc::ptr_eq(®istered.inner, &process.inner)); -+ if !is_registered_process { -+ return false; -+ } -+ mutable.processes.remove(&process.pid()); -+ true -+ } -+ -+ /// Retires a completed reusable execution epoch. An already-absent entry -+ /// is accepted (another legitimate join may have reaped it); a different -+ /// identity under the same PID is never removed. -+ pub(crate) fn retire_process_epoch( -+ &self, -+ process: &WasiProcess, -+ ) -> Result<(), ControlPlaneError> { -+ if process.try_join().is_none() { -+ return Err(ControlPlaneError::ProcessStillRunning { -+ pid: process.pid().raw(), -+ }); -+ } -+ -+ let mut mutable = self.state.mutable.write().unwrap(); -+ match mutable.processes.get(&process.pid()) { -+ None => Ok(()), -+ Some(registered) if registered.same_identity(process) => { -+ mutable.processes.remove(&process.pid()); -+ Ok(()) -+ } -+ Some(_) => Err(ControlPlaneError::ProcessIdentityChanged { -+ pid: process.pid().raw(), -+ }), -+ } -+ } -+ -+ /// Returns the number of process identities currently published in this -+ /// control plane. This is used by opt-in lifecycle diagnostics as well as -+ /// invariant tests; callers must not use it for admission decisions. -+ #[cfg(test)] -+ pub(crate) fn registered_process_count(&self) -> usize { -+ self.state.mutable.read().unwrap().processes.len() -+ } -+ -+ fn process_epoch_snapshot(&self, root: &WasiProcess) -> Vec { -+ let mut processes = vec![root.clone()]; -+ processes.extend( -+ self.state -+ .mutable -+ .read() -+ .unwrap() -+ .processes -+ .values() -+ .filter(|process| process.same_tree(root) && !process.same_identity(root)) -+ .cloned(), -+ ); -+ processes -+ } -+ -+ /// Wait for every published process in `root`'s epoch to publish terminal -+ /// status and release both execution and child-publication ownership. -+ /// -+ /// The final exclusive epoch pass linearizes against a child published -+ /// after the preceding registry snapshot. A process removed from the -+ /// registry is already join-claimed and quiescent, so it cannot hide live -+ /// guest work. Once this returns, terminal product evidence can take a -+ /// coherent process-tree snapshot without polling or timing assumptions. -+ pub(crate) async fn wait_for_process_tree_quiescence(&self, root: &WasiProcess) { -+ loop { -+ for process in self.process_epoch_snapshot(root) { -+ let _ = process.join().await; -+ } -+ -+ let epoch = root.tree_epoch(); -+ let _exclusive_epoch = epoch.write().unwrap(); -+ let stable = self.process_epoch_snapshot(root); -+ let all_quiescent = stable.iter().all(|process| { -+ if !process.finished.status().is_finished() { -+ return false; -+ } -+ let inner = process.inner.0.lock().unwrap(); -+ inner.execution_leases == 0 && inner.pending_child_publications == 0 -+ }); -+ if all_quiescent { -+ return; -+ } -+ } -+ } -+ -+ #[cfg(test)] -+ pub(crate) fn fail_next_task_admission(&self) { -+ self.state -+ .fail_next_task_admission -+ .store(true, Ordering::Release); -+ } -+} -+ -+/// Rollback token for a process registration that has not yet crossed its -+/// public success boundary. -+#[derive(Debug)] -+pub(crate) struct WasiProcessRegistrationGuard { -+ control_plane: WasiControlPlane, -+ process: WasiProcess, -+ parent: Option, -+ parent_publication: Option, -+ committed: bool, -+ launch_complete: bool, -+} -+ -+impl WasiProcessRegistrationGuard { -+ /// Records the potential parent so rollback also removes an adoption link -+ /// if a later construction step fails. -+ pub(crate) fn with_parent( -+ mut self, -+ parent: &WasiProcess, -+ publication: WasiChildPublicationGuard, -+ ) -> Self { -+ assert!( -+ self.process.same_tree(parent), -+ "child process must inherit its parent's tree epoch" -+ ); -+ self.parent = Some(parent.clone()); -+ self.parent_publication = Some(publication.bind_child(&self.process)); -+ self -+ } -+ -+ pub(crate) fn commit(&mut self) { -+ assert!( -+ self.parent.is_none() && self.parent_publication.is_none(), -+ "child process registration requires commit_child" -+ ); -+ assert!(!self.committed, "process registration committed twice"); -+ self.control_plane.publish_reserved_process(&self.process); -+ self.committed = true; -+ self.launch_complete = true; -+ } -+ -+ /// Atomically adopts and publishes a child while its exact parent-side -+ /// publication reservation prevents epoch retirement. No published child -+ /// can be observed without its parent link, and no permit can authorize a -+ /// different parent/child pair. -+ pub(crate) fn commit_child(&mut self) -> Result<(), ControlPlaneError> { -+ assert!(!self.committed, "process registration committed twice"); -+ let parent = self -+ .parent -+ .clone() -+ .ok_or(ControlPlaneError::InvalidChildPublication { -+ parent_pid: 0, -+ child_pid: self.process.pid().raw(), -+ })?; -+ let publication = -+ self.parent_publication -+ .take() -+ .ok_or(ControlPlaneError::InvalidChildPublication { -+ parent_pid: parent.pid().raw(), -+ child_pid: self.process.pid().raw(), -+ })?; -+ let control_plane = self.control_plane.clone(); -+ let process = self.process.clone(); -+ parent.adopt_child_process(process.clone(), publication, || { -+ control_plane.publish_reserved_process(&process); -+ })?; -+ self.committed = true; -+ process.commit_guest_start(); -+ Ok(()) -+ } -+ -+ /// Completes the child launch transaction after a task/command has been -+ /// accepted. Exit notification is deliberately delayed until this point, -+ /// so a launch rollback cannot emit a spurious SIGCHLD. -+ pub(crate) fn complete_child_launch( -+ &mut self, -+ tasks: &Arc, -+ ) { -+ assert!(self.committed, "child launch completed before publication"); -+ assert!(!self.launch_complete, "child launch completed twice"); -+ let parent = self -+ .parent -+ .clone() -+ .expect("child registration has no parent"); -+ parent.notify_on_child_exit(self.process.clone(), tasks); -+ self.launch_complete = true; -+ } -+ -+ /// Rolls back an already-published embryonic child whose launch failed. -+ /// The start gate was committed before instantiation, so termination and a -+ /// real lease-quiescence wait are required before exact topology/registry -+ /// removal. No polling or terminate-as-cancel shortcut is used. -+ pub(crate) fn rollback_child(&mut self, exit_code: wasmer_wasix_types::wasi::ExitCode) { -+ assert!(self.committed, "cannot roll back an unpublished child"); -+ assert!( -+ !self.launch_complete, -+ "cannot roll back a completed child launch" -+ ); -+ self.process.terminate(exit_code); -+ self.process.wait_for_execution_quiescence_blocking(); -+ if let Some(parent) = self.parent.as_ref() { -+ // A concurrent POSIX waiter may already have consumed the failed -+ // child's status and removed this exact link. -+ parent.remove_child_if_same(&self.process); -+ } -+ // Likewise, reap is idempotent with a concurrent exact-identity wait. -+ // PIDs are monotonic, so an absent entry cannot name a replacement. -+ self.control_plane.reap_process(&self.process); -+ self.committed = false; -+ self.launch_complete = true; -+ } -+} -+ -+impl Drop for WasiProcessRegistrationGuard { -+ fn drop(&mut self) { -+ if self.committed && !self.launch_complete { -+ self.rollback_child(wasmer_wasix_types::wasi::Errno::Canceled.into()); -+ return; -+ } -+ if self.committed || self.launch_complete { -+ return; -+ } -+ self.process.abort_guest_start(); -+ if let Some(parent) = self.parent.as_ref() { -+ parent.remove_child_if_same(&self.process); -+ } -+ } - } - - impl MutableState { -@@ -195,7 +550,11 @@ pub struct TaskCountGuard(Arc); - - impl Drop for TaskCountGuard { - fn drop(&mut self) { -- self.0.fetch_sub(1, Ordering::SeqCst); -+ self.0 -+ .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { -+ current.checked_sub(1) -+ }) -+ .expect("control-plane task counter underflow"); - } - } - -@@ -207,11 +566,47 @@ pub enum ControlPlaneError { - /// The maximum number of tasks. - max: usize, - }, -+ /// A live thread already owns the requested numeric thread ID. -+ #[error("Thread ID {tid} is already registered")] -+ DuplicateThreadId { tid: u32 }, -+ /// A reusable environment may only be reset after its current process exits. -+ #[error("Process {pid} is still running")] -+ ProcessStillRunning { pid: u32 }, -+ /// Terminal processes cannot accept a new thread or execution task. -+ #[error("Process {pid} has already finished")] -+ ProcessFinished { pid: u32 }, -+ /// Tentative children cannot instantiate guest code until their exact -+ /// parent adoption and registry publication transaction has committed. -+ #[error("Process {pid} has not been published for guest execution")] -+ ProcessNotPublished { pid: u32 }, -+ /// In-place process switching must begin inside an accepted TaskWasm so -+ /// its exact physical execution owner can authorize a later handoff. -+ #[error("Process {pid} thread {tid} has no accepted TaskWasm execution owner")] -+ ExecutionOwnerUnavailable { pid: u32, tid: u32 }, -+ /// The process registry no longer contains the expected process identity. -+ #[error("Process {pid} registration changed identity")] -+ ProcessIdentityChanged { pid: u32 }, -+ /// The process epoch has begun terminal retirement. -+ #[error("Process {pid} is retiring")] -+ ProcessRetiring { pid: u32 }, -+ /// Background thread handles/tasks still retain the process epoch. -+ #[error("Process {pid} still has {count} live non-main thread(s)")] -+ ProcessHasLiveThreads { pid: u32, count: usize }, -+ /// Child processes or in-flight child publications still retain the epoch. -+ #[error("Process {pid} still has {count} live child process(es)")] -+ ProcessHasLiveChildren { pid: u32, count: usize }, -+ /// A child publication token was used for a different process pair. -+ #[error("Invalid child publication for parent {parent_pid} and child {child_pid}")] -+ InvalidChildPublication { parent_pid: u32, child_pid: u32 }, -+ /// Fork must not snapshot address-space metadata while a fresh exec image -+ /// is taking ownership of inherited shared mappings. -+ #[error("shared-memory exec transition prevents a consistent fork snapshot")] -+ SharedMemoryForkUnavailable, - } - - #[cfg(test)] - mod tests { -- use wasmer_wasix_types::wasix::ThreadStartType; -+ use wasmer_wasix_types::{wasi::Errno, wasix::ThreadStartType}; - - use crate::os::task::thread::WasiMemoryLayout; - -@@ -226,16 +621,28 @@ mod tests { - enable_exponential_cpu_backoff: None, - }); - -- let p1 = p.new_process(ModuleHash::random()).unwrap(); -- let _t1 = p1 -- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -+ let (p1, _t1) = p -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) - .unwrap(); - let _t2 = p1 -- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) - .unwrap(); - - assert_eq!( -- p.new_process(ModuleHash::random()).unwrap_err(), -+ p1.new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) -+ .unwrap_err(), -+ ControlPlaneError::TaskLimitReached { max: 2 } -+ ); -+ -+ assert_eq!( -+ p.new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap_err(), - ControlPlaneError::TaskLimitReached { max: 2 } - ); - } -@@ -249,24 +656,1170 @@ mod tests { - enable_exponential_cpu_backoff: None, - }); - -- let p1 = p.new_process(ModuleHash::random()).unwrap(); -+ let (p1, _initial_main) = p -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); - - for _ in 0..10 { - let _thread = p1 -- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) - .unwrap(); - } - -- let _t1 = p1 -- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -- .unwrap(); - let _t2 = p1 -- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) - .unwrap(); - - assert_eq!( -- p.new_process(ModuleHash::random()).unwrap_err(), -+ p.new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap_err(), - ControlPlaneError::TaskLimitReached { max: 2 } - ); - } -+ -+ #[test] -+ fn finished_process_remains_until_join_reaps_it() { -+ let plane = WasiControlPlane::default(); -+ let (process, _main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ -+ assert!(!plane.reap_process(&process)); -+ assert!(plane.get_process(process.pid()).is_some()); -+ process.terminate(0u16.into()); -+ assert!(process.try_join().is_some()); -+ assert!(plane.get_process(process.pid()).is_some()); -+ -+ assert!(plane.reap_process(&process)); -+ assert!(plane.get_process(process.pid()).is_none()); -+ assert_eq!(plane.registered_process_count(), 0); -+ assert!(!plane.reap_process(&process)); -+ } -+ -+ #[test] -+ fn stale_process_identity_cannot_reap_registered_process() { -+ let plane = WasiControlPlane::default(); -+ let (registered, _registered_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let mut stale = WasiProcess::new(registered.pid(), ModuleHash::random(), plane.handle()); -+ stale.set_pid(registered.pid()); -+ let _main = stale -+ .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -+ .unwrap(); -+ stale.terminate(0u16.into()); -+ -+ assert!(!plane.reap_process(&stale)); -+ let current = plane.get_process(registered.pid()).unwrap(); -+ assert!(Arc::ptr_eq(¤t.inner, ®istered.inner)); -+ } -+ -+ #[test] -+ fn try_join_any_child_removes_parent_link_and_reaps_registry_entry() { -+ let plane = WasiControlPlane::default(); -+ let (mut parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, _main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let child_pid = child.pid(); -+ parent.lock().children.push(child.clone()); -+ -+ child.terminate(0u16.into()); -+ let joined = parent.try_join_any_child().unwrap().unwrap(); -+ -+ assert_eq!(joined.0, child_pid); -+ assert_eq!(joined.1, 0u16.into()); -+ assert!(parent.lock().children.is_empty()); -+ assert!(plane.get_process(child_pid).is_none()); -+ assert!(plane.get_process(parent.pid()).is_some()); -+ } -+ -+ #[test] -+ fn task_admission_honors_zero_and_one_as_exact_limits() { -+ let zero = WasiControlPlane::new(ControlPlaneConfig { -+ max_task_count: Some(0), -+ ..Default::default() -+ }); -+ assert_eq!( -+ zero.register_task().unwrap_err(), -+ ControlPlaneError::TaskLimitReached { max: 0 } -+ ); -+ assert_eq!(zero.active_task_count(), 0); -+ -+ let one = WasiControlPlane::new(ControlPlaneConfig { -+ max_task_count: Some(1), -+ ..Default::default() -+ }); -+ let (process, main) = one -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ assert_eq!(one.active_task_count(), 1); -+ assert_eq!( -+ process -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) -+ .unwrap_err(), -+ ControlPlaneError::TaskLimitReached { max: 1 } -+ ); -+ assert_eq!(one.active_task_count(), 1); -+ drop(main); -+ assert_eq!(one.active_task_count(), 0); -+ } -+ -+ #[test] -+ fn concurrent_task_reservations_never_oversubscribe_limit() { -+ use std::sync::{Barrier, mpsc}; -+ use std::thread; -+ -+ const LIMIT: usize = 4; -+ const CONTENDERS: usize = 24; -+ let plane = WasiControlPlane::new(ControlPlaneConfig { -+ max_task_count: Some(LIMIT), -+ ..Default::default() -+ }); -+ let start = Arc::new(Barrier::new(CONTENDERS + 1)); -+ let release = Arc::new(Barrier::new(LIMIT + 1)); -+ let (tx, rx) = mpsc::channel(); -+ let mut workers = Vec::new(); -+ -+ for _ in 0..CONTENDERS { -+ let plane = plane.clone(); -+ let start = start.clone(); -+ let release = release.clone(); -+ let tx = tx.clone(); -+ workers.push(thread::spawn(move || { -+ start.wait(); -+ match plane.register_task() { -+ Ok(guard) => { -+ tx.send(true).unwrap(); -+ release.wait(); -+ drop(guard); -+ } -+ Err(ControlPlaneError::TaskLimitReached { max: LIMIT }) => { -+ tx.send(false).unwrap(); -+ } -+ Err(other) => panic!("unexpected admission error: {other}"), -+ } -+ })); -+ } -+ drop(tx); -+ start.wait(); -+ let admitted = (0..CONTENDERS) -+ .map(|_| rx.recv().unwrap()) -+ .filter(|admitted| *admitted) -+ .count(); -+ assert_eq!(admitted, LIMIT); -+ assert_eq!(plane.active_task_count(), LIMIT); -+ release.wait(); -+ for worker in workers { -+ worker.join().unwrap(); -+ } -+ assert_eq!(plane.active_task_count(), 0); -+ } -+ -+ #[test] -+ fn duplicate_live_tid_is_rejected_without_count_or_slot_corruption() { -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let main_alias = main.clone(); -+ let observational_thread_clone = main.as_thread(); -+ let tid = main.id(); -+ -+ assert_eq!( -+ process -+ .new_thread_with_id( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ tid, -+ ) -+ .unwrap_err(), -+ ControlPlaneError::DuplicateThreadId { tid: tid.raw() } -+ ); -+ assert_eq!(process.active_threads(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ assert!(process.get_thread(&tid).unwrap().same_identity(&main)); -+ -+ drop(main); -+ assert_eq!(process.active_threads(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ assert_eq!( -+ process -+ .new_thread_with_id( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ tid, -+ ) -+ .unwrap_err(), -+ ControlPlaneError::DuplicateThreadId { tid: tid.raw() } -+ ); -+ drop(main_alias); -+ assert_eq!(process.active_threads(), 0); -+ assert_eq!(plane.active_task_count(), 0); -+ assert_eq!(observational_thread_clone.tid(), tid); -+ -+ assert_eq!( -+ process -+ .new_thread_with_id( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ tid, -+ ) -+ .unwrap_err(), -+ ControlPlaneError::ProcessFinished { -+ pid: process.pid().raw(), -+ } -+ ); -+ assert_eq!(process.active_threads(), 0); -+ assert_eq!(plane.active_task_count(), 0); -+ drop(observational_thread_clone); -+ } -+ -+ #[test] -+ fn unpublished_process_guard_rolls_back_registry_and_parent_link() { -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let baseline = plane.registered_process_count(); -+ let (child, child_main, guard) = plane -+ .new_child_process_with_main_thread_guarded( -+ &parent, -+ ModuleHash::random(), -+ WasiMemoryLayout::default(), -+ ) -+ .unwrap(); -+ parent.lock().children.push(child.clone()); -+ assert_eq!(plane.registered_process_count(), baseline); -+ assert!(plane.get_process(child.pid()).is_none()); -+ assert_eq!(parent.lock().children.len(), 1); -+ -+ drop(guard); -+ -+ assert_eq!(plane.registered_process_count(), baseline); -+ assert!(parent.lock().children.is_empty()); -+ assert!(plane.get_process(child.pid()).is_none()); -+ drop(child_main); -+ } -+ -+ #[test] -+ fn main_thread_admission_failure_cannot_leave_published_process() { -+ let plane = WasiControlPlane::new(ControlPlaneConfig { -+ max_task_count: Some(1), -+ ..Default::default() -+ }); -+ let (process, registration) = plane.new_process_guarded(ModuleHash::random()).unwrap(); -+ assert_eq!(plane.registered_process_count(), 0); -+ assert!(plane.get_process(process.pid()).is_none()); -+ -+ // Deterministically fill the only task slot after process publication, -+ // reproducing the race that an advisory precheck cannot prevent. -+ let competing_reservation = plane.register_task().unwrap(); -+ assert_eq!( -+ process -+ .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -+ .unwrap_err(), -+ ControlPlaneError::TaskLimitReached { max: 1 } -+ ); -+ drop(registration); -+ -+ assert_eq!(plane.registered_process_count(), 0); -+ assert!(plane.get_process(process.pid()).is_none()); -+ assert_eq!(plane.active_task_count(), 1); -+ drop(competing_reservation); -+ assert_eq!(plane.active_task_count(), 0); -+ } -+ -+ #[test] -+ fn tentative_process_is_invisible_and_abort_releases_last_object() { -+ use std::sync::Barrier; -+ use std::thread; -+ -+ let plane = WasiControlPlane::default(); -+ let (process, registration) = plane.new_process_guarded(ModuleHash::random()).unwrap(); -+ let pid = process.pid(); -+ let weak_process = Arc::downgrade(&process.inner); -+ let phases = Arc::new(Barrier::new(2)); -+ let observer_plane = plane.clone(); -+ let observer_phases = phases.clone(); -+ let observer = thread::spawn(move || { -+ observer_phases.wait(); -+ let visible_before_abort = observer_plane.get_process(pid).is_some(); -+ observer_phases.wait(); -+ observer_phases.wait(); -+ let visible_after_abort = observer_plane.get_process(pid).is_some(); -+ (visible_before_abort, visible_after_abort) -+ }); -+ -+ phases.wait(); -+ phases.wait(); -+ assert!(plane.get_process(pid).is_none()); -+ assert_eq!(plane.registered_process_count(), 0); -+ drop(registration); -+ drop(process); -+ assert!(weak_process.upgrade().is_none()); -+ phases.wait(); -+ -+ assert_eq!(observer.join().unwrap(), (false, false)); -+ assert_eq!(plane.registered_process_count(), 0); -+ } -+ -+ #[test] -+ fn process_wait_status_has_exactly_one_concurrent_consumer() { -+ use std::sync::Barrier; -+ use std::thread; -+ -+ let plane = WasiControlPlane::default(); -+ let (process, _main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ process.terminate(0u16.into()); -+ let start = Arc::new(Barrier::new(3)); -+ let mut waiters = Vec::new(); -+ for _ in 0..2 { -+ let process = process.clone(); -+ let start = start.clone(); -+ waiters.push(thread::spawn(move || { -+ start.wait(); -+ process.try_claim_join().is_some() -+ })); -+ } -+ start.wait(); -+ let winners = waiters -+ .into_iter() -+ .map(|waiter| waiter.join().unwrap()) -+ .filter(|won| *won) -+ .count(); -+ assert_eq!(winners, 1); -+ assert!( -+ process.try_join().is_some(), -+ "observation remains non-consuming" -+ ); -+ assert!(process.try_claim_join().is_none()); -+ } -+ -+ #[test] -+ fn concurrent_parent_waiters_return_one_child_status() { -+ use std::sync::Barrier; -+ use std::thread; -+ -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, _child_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let child_pid = child.pid(); -+ parent.lock().children.push(child.clone()); -+ child.terminate(0u16.into()); -+ -+ let start = Arc::new(Barrier::new(3)); -+ let mut waiters = Vec::new(); -+ for _ in 0..2 { -+ let mut parent = parent.clone(); -+ let start = start.clone(); -+ waiters.push(thread::spawn(move || { -+ start.wait(); -+ parent.try_join_any_child() -+ })); -+ } -+ start.wait(); -+ let results: Vec<_> = waiters -+ .into_iter() -+ .map(|waiter| waiter.join().unwrap()) -+ .collect(); -+ let joined: Vec<_> = results -+ .iter() -+ .filter_map(|result| result.as_ref().ok().and_then(|status| *status)) -+ .collect(); -+ -+ assert_eq!(joined, vec![(child_pid, 0u16.into())]); -+ assert!(parent.lock().children.is_empty()); -+ assert!(plane.get_process(child_pid).is_none()); -+ } -+ -+ #[test] -+ fn public_process_construction_never_publishes_without_a_main_thread() { -+ let plane = WasiControlPlane::new(ControlPlaneConfig { -+ max_task_count: Some(0), -+ ..Default::default() -+ }); -+ -+ assert_eq!( -+ plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default(),) -+ .unwrap_err(), -+ ControlPlaneError::TaskLimitReached { max: 0 } -+ ); -+ assert_eq!(plane.registered_process_count(), 0); -+ assert_eq!(plane.active_task_count(), 0); -+ } -+ -+ #[test] -+ fn retirement_seals_a_finished_process_against_new_threads() { -+ use std::sync::Barrier; -+ use std::thread; -+ -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ process.terminate(0u16.into()); -+ -+ let start = Arc::new(Barrier::new(3)); -+ let thread_process = process.clone(); -+ let thread_start = start.clone(); -+ let thread_racer = thread::spawn(move || { -+ thread_start.wait(); -+ thread_process.new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) -+ }); -+ let retirement_process = process.clone(); -+ let retirement_start = start.clone(); -+ let retirement_racer = thread::spawn(move || { -+ retirement_start.wait(); -+ retirement_process.begin_epoch_retirement() -+ }); -+ start.wait(); -+ -+ let thread_result = thread_racer.join().unwrap(); -+ let retirement_result = retirement_racer.join().unwrap(); -+ match (thread_result, retirement_result) { -+ ( -+ Err( -+ ControlPlaneError::ProcessFinished { .. } -+ | ControlPlaneError::ProcessRetiring { .. }, -+ ), -+ Ok(children), -+ ) => { -+ assert!(children.is_empty()); -+ assert!(process.lock().retiring); -+ } -+ (thread_result, retirement_result) => panic!( -+ "finished-process retirement admitted a new thread: thread={thread_result:?}, retirement={retirement_result:?}" -+ ), -+ } -+ -+ assert_eq!(process.active_threads(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ process.retire_epoch_task_registrations(); -+ plane.retire_process_epoch(&process).unwrap(); -+ drop(main); -+ assert_eq!(plane.active_task_count(), 0); -+ assert_eq!(plane.registered_process_count(), 0); -+ } -+ -+ #[test] -+ fn retirement_and_child_publication_linearize_without_leaking_permit() { -+ use std::sync::Barrier; -+ use std::thread; -+ -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ process.terminate(0u16.into()); -+ -+ let start = Arc::new(Barrier::new(3)); -+ let publication_process = process.clone(); -+ let publication_start = start.clone(); -+ let publication_racer = thread::spawn(move || { -+ publication_start.wait(); -+ publication_process.begin_child_publication() -+ }); -+ let retirement_process = process.clone(); -+ let retirement_start = start.clone(); -+ let retirement_racer = thread::spawn(move || { -+ retirement_start.wait(); -+ retirement_process.begin_epoch_retirement() -+ }); -+ start.wait(); -+ -+ let publication_result = publication_racer.join().unwrap(); -+ let retirement_result = retirement_racer.join().unwrap(); -+ match (publication_result, retirement_result) { -+ ( -+ Err( -+ ControlPlaneError::ProcessFinished { .. } -+ | ControlPlaneError::ProcessRetiring { .. }, -+ ), -+ Ok(children), -+ ) => { -+ assert!(children.is_empty()); -+ assert!(process.lock().retiring); -+ assert_eq!(process.lock().pending_child_publications, 0); -+ } -+ (publication_result, retirement_result) => panic!( -+ "race did not produce exactly one winner: publication={publication_result:?}, retirement={retirement_result:?}" -+ ), -+ } -+ -+ process.retire_epoch_task_registrations(); -+ plane.retire_process_epoch(&process).unwrap(); -+ drop(main); -+ assert_eq!(plane.active_task_count(), 0); -+ assert_eq!(plane.registered_process_count(), 0); -+ } -+ -+ #[test] -+ fn child_adoption_rejects_a_permit_for_a_different_parent_without_panic() { -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (other_parent, _other_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, child_handle, registration) = plane -+ .new_process_with_main_thread_guarded(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let publication = parent.begin_child_publication().unwrap().bind_child(&child); -+ -+ let error = other_parent -+ .adopt_child_process(child.clone(), publication, || { -+ panic!("invalid adoption must not publish the child") -+ }) -+ .unwrap_err(); -+ assert_eq!( -+ error, -+ ControlPlaneError::InvalidChildPublication { -+ parent_pid: other_parent.pid().raw(), -+ child_pid: child.pid().raw(), -+ } -+ ); -+ assert_eq!(parent.lock().pending_child_publications, 0); -+ assert!(other_parent.lock().children.is_empty()); -+ assert!(plane.get_process(child.pid()).is_none()); -+ drop(registration); -+ drop(child_handle); -+ } -+ -+ #[test] -+ fn child_execution_admission_requires_exact_publication() { -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, child_main, mut registration) = plane -+ .new_child_process_with_main_thread_guarded( -+ &parent, -+ ModuleHash::random(), -+ WasiMemoryLayout::default(), -+ ) -+ .unwrap(); -+ -+ assert_eq!( -+ child.acquire_execution_lease().unwrap_err(), -+ ControlPlaneError::ProcessNotPublished { -+ pid: child.pid().raw(), -+ } -+ ); -+ assert!(plane.get_process(child.pid()).is_none()); -+ assert!(parent.lock().children.is_empty()); -+ -+ registration.commit_child().unwrap(); -+ let lease = child.acquire_execution_lease().unwrap(); -+ assert_eq!(child.ppid(), parent.pid()); -+ assert!(child.has_exact_parent(&parent)); -+ assert!( -+ plane -+ .get_process(child.pid()) -+ .is_some_and(|registered| registered.same_identity(&child)) -+ ); -+ assert!( -+ parent -+ .lock() -+ .children -+ .iter() -+ .any(|candidate| candidate.same_identity(&child)) -+ ); -+ -+ child.terminate(Errno::Canceled.into()); -+ assert!( -+ child.try_join().is_none(), -+ "terminal status is not quiescence" -+ ); -+ assert!(!plane.reap_process(&child)); -+ drop(lease); -+ registration.rollback_child(Errno::Canceled.into()); -+ assert!(plane.get_process(child.pid()).is_none()); -+ assert!(parent.lock().children.is_empty()); -+ drop(child_main); -+ } -+ -+ #[test] -+ fn failed_launch_rollback_waits_for_real_execution_quiescence() { -+ use std::{sync::mpsc, thread, time::Duration}; -+ -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, child_main, mut registration) = plane -+ .new_child_process_with_main_thread_guarded( -+ &parent, -+ ModuleHash::random(), -+ WasiMemoryLayout::default(), -+ ) -+ .unwrap(); -+ registration.commit_child().unwrap(); -+ let lease = child.acquire_execution_lease().unwrap(); -+ -+ let (entered_tx, entered_rx) = mpsc::channel(); -+ let (done_tx, done_rx) = mpsc::channel(); -+ let rollback = thread::spawn(move || { -+ entered_tx.send(()).unwrap(); -+ registration.rollback_child(Errno::Canceled.into()); -+ done_tx.send(()).unwrap(); -+ }); -+ -+ entered_rx.recv().unwrap(); -+ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ assert!(plane.get_process(child.pid()).is_some()); -+ assert!( -+ parent -+ .lock() -+ .children -+ .iter() -+ .any(|candidate| candidate.same_identity(&child)) -+ ); -+ -+ drop(lease); -+ done_rx.recv_timeout(Duration::from_secs(1)).unwrap(); -+ rollback.join().unwrap(); -+ assert!(plane.get_process(child.pid()).is_none()); -+ assert!(parent.lock().children.is_empty()); -+ drop(child_main); -+ } -+ -+ #[test] -+ fn process_tree_barrier_waits_for_late_descendant_execution_quiescence() { -+ use std::{sync::mpsc, thread, time::Duration}; -+ -+ let plane = WasiControlPlane::default(); -+ let (root, _root_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let root_lease = root.acquire_execution_lease().unwrap(); -+ let (child, _child_main, mut child_registration) = plane -+ .new_child_process_with_main_thread_guarded( -+ &root, -+ ModuleHash::random(), -+ WasiMemoryLayout::default(), -+ ) -+ .unwrap(); -+ child_registration.commit_child().unwrap(); -+ let child_lease = child.acquire_execution_lease().unwrap(); -+ -+ root.terminate(Errno::Success.into()); -+ drop(root_lease); -+ -+ let waiting_plane = plane.clone(); -+ let waiting_root = root.clone(); -+ let (done_tx, done_rx) = mpsc::channel(); -+ let waiter = thread::spawn(move || { -+ futures::executor::block_on( -+ waiting_plane.wait_for_process_tree_quiescence(&waiting_root), -+ ); -+ done_tx.send(()).unwrap(); -+ }); -+ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ -+ // Publish a grandchild after the barrier's first registry snapshot. -+ // The terminal child cannot hide it: the exclusive epoch recheck must -+ // discover the new identity and wait for its execution lease too. -+ let (grandchild, _grandchild_main, mut grandchild_registration) = plane -+ .new_child_process_with_main_thread_guarded( -+ &child, -+ ModuleHash::random(), -+ WasiMemoryLayout::default(), -+ ) -+ .unwrap(); -+ grandchild_registration.commit_child().unwrap(); -+ let grandchild_lease = grandchild.acquire_execution_lease().unwrap(); -+ -+ child.terminate(Errno::Success.into()); -+ drop(child_lease); -+ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ -+ grandchild.terminate(Errno::Success.into()); -+ drop(grandchild_lease); -+ done_rx.recv_timeout(Duration::from_secs(1)).unwrap(); -+ waiter.join().unwrap(); -+ -+ grandchild_registration.rollback_child(Errno::Success.into()); -+ child_registration.rollback_child(Errno::Success.into()); -+ } -+ -+ #[test] -+ fn process_join_waits_for_pending_child_publication_ownership() { -+ use std::{sync::mpsc, thread, time::Duration}; -+ -+ let plane = WasiControlPlane::default(); -+ let (process, _main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let publication = process.begin_child_publication().unwrap(); -+ process.terminate(Errno::Success.into()); -+ assert!(process.try_join().is_none()); -+ -+ let waiting_process = process.clone(); -+ let (done_tx, done_rx) = mpsc::channel(); -+ let waiter = thread::spawn(move || { -+ futures::executor::block_on(waiting_process.join()).unwrap(); -+ done_tx.send(()).unwrap(); -+ }); -+ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ -+ drop(publication); -+ assert!(process.try_join().is_some()); -+ done_rx.recv_timeout(Duration::from_secs(1)).unwrap(); -+ waiter.join().unwrap(); -+ assert_eq!(process.lock().pending_child_publications, 0); -+ } -+ -+ #[test] -+ fn execution_guard_publishes_terminal_before_quiescence() { -+ use std::{sync::mpsc, thread, time::Duration}; -+ -+ let plane = WasiControlPlane::default(); -+ let (process, _main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let guard = process.acquire_execution_guard().unwrap(); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ let observer = process.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let waiter = thread::spawn(move || { -+ observer.wait_for_execution_quiescence_blocking(); -+ observed_tx.send(observer.try_join()).unwrap(); -+ }); -+ -+ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ guard.finish(Ok(7u16.into())); -+ -+ let observed = observed_rx.recv_timeout(Duration::from_secs(1)).unwrap(); -+ assert_eq!( -+ observed.expect("process never became joinable").unwrap(), -+ 7u16.into() -+ ); -+ waiter.join().unwrap(); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn panicking_monitor_manager_terminates_and_reaps_before_quiescence() { -+ use std::{future::Future, pin::Pin, sync::mpsc, thread, time::Duration}; -+ -+ use crate::{ -+ WasiThreadError, -+ os::task::{OwnedTaskStatus, signal::SignalHandlerAbi}, -+ runtime::task_manager::{TaskWasm, VirtualTaskManager}, -+ }; -+ use wasmer::FromToNativeWasmType; -+ use wasmer_wasix_types::wasi::Signal; -+ -+ #[derive(Debug)] -+ struct FinishingSignalHandler { -+ sender: mpsc::Sender, -+ } -+ -+ impl SignalHandlerAbi for FinishingSignalHandler { -+ fn signal( -+ &self, -+ signal: u8, -+ ) -> Result<(), crate::os::task::signal::SignalDeliveryError> { -+ self.sender -+ .send(signal) -+ .map_err(|_| crate::os::task::signal::SignalDeliveryError) -+ } -+ } -+ -+ #[derive(Debug)] -+ struct PanickingMonitorTaskManager; -+ -+ impl VirtualTaskManager for PanickingMonitorTaskManager { -+ fn sleep_now( -+ &self, -+ _time: Duration, -+ ) -> Pin + Send + Sync + 'static>> { -+ Box::pin(async {}) -+ } -+ -+ fn task_shared( -+ &self, -+ task: Box futures::future::BoxFuture<'static, ()> + Send + 'static>, -+ ) -> Result<(), WasiThreadError> { -+ // Model a custom manager that constructs, then cancels, the -+ // admitted watcher before panicking out of task admission. -+ drop(task()); -+ panic!("synthetic process-monitor admission panic") -+ } -+ -+ fn task_wasm(&self, _task: TaskWasm) -> Result<(), WasiThreadError> { -+ unreachable!("process monitoring does not schedule Wasm") -+ } -+ -+ fn task_dedicated( -+ &self, -+ _task: Box, -+ ) -> Result<(), WasiThreadError> { -+ unreachable!("process monitoring does not use a dedicated task") -+ } -+ -+ fn thread_parallelism(&self) -> Result { -+ Ok(1) -+ } -+ } -+ -+ let plane = WasiControlPlane::default(); -+ let (process, _main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let execution = process.acquire_execution_guard().unwrap(); -+ -+ let (signal_tx, signal_rx) = mpsc::channel(); -+ let mut command = OwnedTaskStatus::default(); -+ command.set_signal_handler(Arc::new(FinishingSignalHandler { sender: signal_tx })); -+ let command = Arc::new(command); -+ let command_handle = command.handle(); -+ let command_finisher = command.clone(); -+ let finisher = thread::spawn(move || { -+ assert_eq!( -+ signal_rx.recv_timeout(Duration::from_secs(1)).unwrap(), -+ Signal::Sigkill.to_native() as u8 -+ ); -+ command_finisher.set_finished(Ok(9u16.into())); -+ }); -+ -+ let observed_process = process.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = thread::spawn(move || { -+ observed_process.wait_for_execution_quiescence_blocking(); -+ let _ = observed_tx.send(observed_process.try_join()); -+ }); -+ -+ let tasks: Arc = Arc::new(PanickingMonitorTaskManager); -+ assert!(matches!( -+ execution.monitor(command_handle, &tasks), -+ Err(WasiThreadError::InvalidWasmContext) -+ )); -+ -+ finisher.join().unwrap(); -+ let status = observed_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .unwrap() -+ .expect("monitor released quiescence before publishing terminal status"); -+ assert_eq!(status.unwrap(), Errno::Canceled.into()); -+ assert_eq!( -+ command -+ .status() -+ .into_finished() -+ .expect("authoritative command handle was not reaped") -+ .unwrap(), -+ 9u16.into() -+ ); -+ assert_eq!(process.lock().execution_leases, 0); -+ observer.join().unwrap(); -+ } -+ -+ #[test] -+ fn abandoned_host_execution_fails_closed_only_after_last_guard_clone() { -+ let plane = WasiControlPlane::default(); -+ let (process, _main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let guard = process.acquire_execution_guard().unwrap(); -+ let successor = guard.clone(); -+ -+ drop(guard); -+ assert!(process.try_join().is_none()); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ drop(successor); -+ assert_eq!(process.lock().execution_leases, 0); -+ assert_eq!( -+ process -+ .try_join() -+ .expect("abandoned execution never became joinable") -+ .unwrap(), -+ Errno::Canceled.into() -+ ); -+ } -+ -+ #[test] -+ fn supplemental_parent_guard_requires_an_accepted_successor() { -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let thread = main.as_thread(); -+ let owner = Arc::new(()); -+ let owner_weak = Arc::downgrade(&owner); -+ let task_wasm_lease = process.acquire_execution_lease().unwrap(); -+ let parent_guard = process -+ .acquire_supplemental_execution_guard(thread.clone(), owner_weak.clone()) -+ .unwrap(); -+ assert_eq!(process.lock().execution_leases, 2); -+ -+ assert!(parent_guard.try_handoff_to_current_task_wasm(&process, &thread, &owner_weak)); -+ assert_eq!(process.lock().execution_leases, 1); -+ assert!(process.finished.status().into_finished().is_none()); -+ -+ drop(task_wasm_lease); -+ assert_eq!(process.lock().execution_leases, 0); -+ assert!(process.finished.status().into_finished().is_none()); -+ } -+ -+ #[test] -+ fn repeated_parent_switch_guards_remain_bounded_without_a_task_successor() { -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let thread = main.as_thread(); -+ let owner = Arc::new(()); -+ let owner_weak = Arc::downgrade(&owner); -+ let task_wasm_lease = process.acquire_execution_lease().unwrap(); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ for _ in 0..2_000 { -+ let supplemental = process -+ .acquire_supplemental_execution_guard(thread.clone(), owner_weak.clone()) -+ .unwrap(); -+ assert!(supplemental.try_handoff_to_current_task_wasm(&process, &thread, &owner_weak)); -+ assert_eq!(process.lock().execution_leases, 1); -+ } -+ drop(task_wasm_lease); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn concurrent_guard_clones_have_one_linearizable_handoff_winner() { -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let thread = main.as_thread(); -+ let owner = Arc::new(()); -+ let owner_weak = Arc::downgrade(&owner); -+ let successor = process.acquire_execution_lease().unwrap(); -+ let guard = process -+ .acquire_supplemental_execution_guard(thread.clone(), owner_weak.clone()) -+ .unwrap(); -+ let barrier = Arc::new(std::sync::Barrier::new(3)); -+ let (result_tx, result_rx) = std::sync::mpsc::channel(); -+ let mut workers = Vec::new(); -+ -+ for candidate in [guard.clone(), guard] { -+ let process = process.clone(); -+ let thread = thread.clone(); -+ let owner_weak = owner_weak.clone(); -+ let barrier = barrier.clone(); -+ let result_tx = result_tx.clone(); -+ workers.push(std::thread::spawn(move || { -+ barrier.wait(); -+ result_tx -+ .send(candidate.try_handoff_to_current_task_wasm( -+ &process, -+ &thread, -+ &owner_weak, -+ )) -+ .unwrap(); -+ })); -+ } -+ barrier.wait(); -+ let outcomes = [result_rx.recv().unwrap(), result_rx.recv().unwrap()]; -+ assert_eq!(outcomes.into_iter().filter(|won| *won).count(), 1); -+ for worker in workers { -+ worker.join().unwrap(); -+ } -+ assert_eq!(process.lock().execution_leases, 1); -+ drop(successor); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn unrelated_task_or_thread_cannot_authorize_parent_guard_handoff() { -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let main_thread = main.as_thread(); -+ let other_thread = process -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) -+ .unwrap(); -+ let owner = Arc::new(()); -+ let wrong_owner = Arc::new(()); -+ let owner_weak = Arc::downgrade(&owner); -+ let guard = process -+ .acquire_supplemental_execution_guard(main_thread.clone(), owner_weak.clone()) -+ .unwrap(); -+ -+ assert!(!guard.try_handoff_to_current_task_wasm( -+ &process, -+ &main_thread, -+ &Arc::downgrade(&wrong_owner) -+ )); -+ assert!(!guard.try_handoff_to_accepted_task_wasm(&process, &other_thread)); -+ assert_eq!(process.lock().execution_leases, 1); -+ assert!(guard.try_handoff_to_current_task_wasm(&process, &main_thread, &owner_weak)); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn committed_vfork_parent_guard_fails_closed_before_quiescence() { -+ use std::{sync::mpsc, thread, time::Duration}; -+ -+ let plane = WasiControlPlane::default(); -+ let (parent, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let owner = Arc::new(()); -+ let original_execution = parent.acquire_execution_lease().unwrap(); -+ let parent_vfork = parent -+ .acquire_supplemental_execution_guard(main.as_thread(), Arc::downgrade(&owner)) -+ .unwrap(); -+ parent_vfork.arm_fail_closed(); -+ drop(original_execution); -+ -+ let observer_process = parent.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = thread::spawn(move || { -+ observer_process.wait_for_execution_quiescence_blocking(); -+ let _ = observed_tx.send(observer_process.try_join()); -+ }); -+ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ -+ drop(parent_vfork); -+ let status = observed_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .unwrap() -+ .expect("committed vfork parent became quiescent before terminal status"); -+ assert_eq!(status.unwrap(), Errno::Canceled.into()); -+ assert_eq!(parent.lock().execution_leases, 0); -+ observer.join().unwrap(); -+ } -+ -+ #[test] -+ fn abandoned_non_main_vfork_owner_terminalizes_whole_process_before_quiescence() { -+ use std::{sync::mpsc, thread, time::Duration}; -+ -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let worker = process -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) -+ .unwrap(); -+ let owner = Arc::new(()); -+ let original_execution = process.acquire_execution_lease().unwrap(); -+ let vfork_owner = process -+ .acquire_supplemental_execution_guard(worker.as_thread(), Arc::downgrade(&owner)) -+ .unwrap(); -+ vfork_owner.arm_fail_closed(); -+ drop(original_execution); -+ -+ let observer_process = process.clone(); -+ let observer_main = main.as_thread(); -+ let observer_worker = worker.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = thread::spawn(move || { -+ observer_process.wait_for_execution_quiescence_blocking(); -+ observed_tx -+ .send((observer_main.try_join(), observer_worker.try_join())) -+ .unwrap(); -+ }); -+ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ -+ drop(vfork_owner); -+ let (main_status, worker_status) = -+ observed_rx.recv_timeout(Duration::from_secs(1)).unwrap(); -+ assert_eq!(main_status.unwrap().unwrap(), Errno::Canceled.into()); -+ assert_eq!(worker_status.unwrap().unwrap(), Errno::Canceled.into()); -+ assert_eq!(process.lock().execution_leases, 0); -+ observer.join().unwrap(); -+ } -+ -+ #[test] -+ fn vfork_parent_and_child_ownership_coexist_across_deep_sleep_handoffs() { -+ let plane = WasiControlPlane::default(); -+ let (parent, parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, child_main, mut registration) = plane -+ .new_child_process_with_main_thread_guarded( -+ &parent, -+ ModuleHash::random(), -+ WasiMemoryLayout::default(), -+ ) -+ .unwrap(); -+ registration.commit_child().unwrap(); -+ -+ // The currently executing TaskWasm and the supplemental vfork guard -+ // both own the suspended parent while the in-place child has its own -+ // fail-closed guard. -+ let parent_owner = Arc::new(()); -+ let parent_task = parent.acquire_execution_lease().unwrap(); -+ let parent_vfork = parent -+ .acquire_supplemental_execution_guard( -+ parent_main.as_thread(), -+ Arc::downgrade(&parent_owner), -+ ) -+ .unwrap(); -+ let child_vfork = child.acquire_execution_guard().unwrap(); -+ assert_eq!(parent.lock().execution_leases, 2); -+ assert_eq!(child.lock().execution_leases, 1); -+ -+ // A deep-sleep child successor acquires before the old callback gives -+ // up the parent TaskWasm lease. -+ let child_successor = child.acquire_execution_lease().unwrap(); -+ assert_eq!(child.lock().execution_leases, 2); -+ drop(parent_task); -+ assert_eq!(parent.lock().execution_leases, 1); -+ -+ // Restoring the parent terminalizes the child in-place ownership, but -+ // the accepted child successor remains until its callback returns. -+ child_vfork.finish(Ok(Errno::Success.into())); -+ assert_eq!(child.lock().execution_leases, 1); -+ assert!(child.try_join().is_none()); -+ -+ // Parent resume acquires its successor before the supplemental guard -+ // follows the restored environment to terminal cleanup. -+ let parent_successor = parent.acquire_execution_lease().unwrap(); -+ assert_eq!(parent.lock().execution_leases, 2); -+ drop(child_successor); -+ assert!(child.try_join().is_some()); -+ -+ parent.terminate(Errno::Success.into()); -+ drop(parent_vfork); -+ assert_eq!(parent.lock().execution_leases, 1); -+ assert!(parent.try_join().is_none()); -+ drop(parent_successor); -+ assert!(parent.try_join().is_some()); -+ -+ registration.rollback_child(Errno::Success.into()); -+ drop(child_main); -+ } - } -diff --git a/lib/wasix/src/os/task/mod.rs b/lib/wasix/src/os/task/mod.rs -index 755645f..e49fbc2 100644 ---- a/lib/wasix/src/os/task/mod.rs -+++ b/lib/wasix/src/os/task/mod.rs -@@ -9,6 +9,16 @@ pub mod thread; - - #[allow(unused_imports)] - pub(crate) use process::WasiProcessInner; -+pub(crate) use task_join_handle::terminate_and_reap_abandoned_task; -+#[cfg(all(feature = "ctrlc", unix))] - pub use task_join_handle::{ -- OwnedTaskStatus, TaskJoinHandle, TaskStatus, TaskTerminatedError, VirtualTaskHandle, -+ HostLifecycleSupervisor, UnixHostLifecycleError, UnixHostLifecycleSupervisor, -+}; -+#[cfg(all(feature = "ctrlc", windows))] -+pub use task_join_handle::{ -+ HostLifecycleSupervisor, WindowsHostLifecycleError, WindowsHostLifecycleSupervisor, -+}; -+pub use task_join_handle::{ -+ OwnedTaskStatus, TaskJoinHandle, TaskSignalController, TaskSignalError, TaskStatus, -+ TaskTerminatedError, VirtualTaskHandle, - }; -diff --git a/lib/wasix/src/os/task/process.rs b/lib/wasix/src/os/task/process.rs -index f1d0f12..f83b6b5 100644 ---- a/lib/wasix/src/os/task/process.rs -+++ b/lib/wasix/src/os/task/process.rs -@@ -1,18 +1,18 @@ --use crate::{WasiEnv, WasiRuntimeError, journal::SnapshotTrigger}; -+use crate::{WasiEnv, WasiRuntimeError, journal::SnapshotTrigger, runtime::VirtualTaskManager}; - #[cfg(feature = "journal")] - use crate::{WasiResult, journal::JournalEffector, syscalls::do_checkpoint_from_outside, unwind}; - use serde::{Deserialize, Serialize}; --#[cfg(feature = "journal")] --use std::collections::HashSet; - use std::{ -- collections::HashMap, -+ collections::{HashMap, HashSet}, - convert::TryInto, -+ future::Future, - ops::Range, -+ pin::Pin, - sync::{ - Arc, Condvar, Mutex, MutexGuard, RwLock, Weak, -- atomic::{AtomicU32, Ordering}, -+ atomic::{AtomicBool, AtomicU32, Ordering}, - }, -- task::Waker, -+ task::{Context, Poll, Waker}, - time::Duration, - }; - use tracing::trace; -@@ -34,8 +34,11 @@ use super::{ - backoff::WasiProcessCpuBackoff, - control_plane::{ControlPlaneError, WasiControlPlaneHandle}, - signal::{SignalDeliveryError, SignalHandlerAbi}, -- task_join_handle::OwnedTaskStatus, -- thread::WasiMemoryLayout, -+ task_join_handle::{ -+ OwnedTaskStatus, TaskAbandonCallback, TaskCompletionCallback, TaskJoinHandle, -+ terminate_and_reap_abandoned_task, -+ }, -+ thread::{WasiMemoryLayout, WasiThreadError}, - }; - - /// Represents the ID of a sub-process -@@ -86,6 +89,121 @@ impl std::fmt::Debug for WasiProcessId { - - pub type LockableWasiProcessInner = Arc<(Mutex, Condvar)>; - -+/// Serializes process-tree admission and topology changes against reusable -+/// epoch retirement. Every descendant inherits the same gate. -+pub(crate) type WasiProcessTreeEpoch = Arc>; -+ -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+enum WasiProcessStartState { -+ Pending, -+ Committed, -+ Aborted, -+} -+ -+#[derive(Debug)] -+struct WasiProcessStartInner { -+ state: WasiProcessStartState, -+} -+ -+/// A one-shot publication barrier for a tentative child. Task admission is -+/// rejected while pending, because Wasmer instantiation itself may execute a -+/// guest start function. Adoption and registry publication commit this gate -+/// before any TaskWasm can be constructed. -+#[derive(Debug, Clone)] -+pub(crate) struct WasiProcessStartGate { -+ inner: Arc>, -+} -+ -+impl WasiProcessStartGate { -+ fn new(state: WasiProcessStartState) -> Self { -+ Self { -+ inner: Arc::new(Mutex::new(WasiProcessStartInner { state })), -+ } -+ } -+ -+ fn pending() -> Self { -+ Self::new(WasiProcessStartState::Pending) -+ } -+ -+ fn committed() -> Self { -+ Self::new(WasiProcessStartState::Committed) -+ } -+ -+ fn transition(&self, next: WasiProcessStartState) { -+ let mut inner = self.inner.lock().unwrap(); -+ if inner.state != WasiProcessStartState::Pending { -+ return; -+ } -+ inner.state = next; -+ } -+ -+ pub(crate) fn commit(&self) { -+ self.transition(WasiProcessStartState::Committed); -+ } -+ -+ pub(crate) fn abort(&self) { -+ self.transition(WasiProcessStartState::Aborted); -+ } -+ -+ pub(crate) fn is_committed(&self) -> bool { -+ self.inner.lock().unwrap().state == WasiProcessStartState::Committed -+ } -+} -+ -+/// Keeps a process non-reapable while accepted Wasm work can still execute. -+/// It deliberately outlives terminal status and is released only after the -+/// final callback (or cancellation/drop) has completed. -+#[derive(Debug)] -+pub struct WasiProcessExecutionLease { -+ inner: Weak<(Mutex, Condvar)>, -+} -+ -+struct WasiProcessQuiescenceFuture { -+ inner: LockableWasiProcessInner, -+} -+ -+impl Future for WasiProcessQuiescenceFuture { -+ type Output = (); -+ -+ fn poll(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll { -+ let mut inner = self.inner.0.lock().unwrap(); -+ if inner.execution_leases == 0 && inner.pending_child_publications == 0 { -+ return Poll::Ready(()); -+ } -+ if !inner -+ .quiescence_wakers -+ .iter() -+ .any(|waker| waker.will_wake(cx.waker())) -+ { -+ inner.quiescence_wakers.push(cx.waker().clone()); -+ } -+ Poll::Pending -+ } -+} -+ -+impl Drop for WasiProcessExecutionLease { -+ fn drop(&mut self) { -+ let Some(inner) = self.inner.upgrade() else { -+ return; -+ }; -+ let mut state = inner.0.lock().unwrap(); -+ state.execution_leases = state -+ .execution_leases -+ .checked_sub(1) -+ .expect("process execution lease underflow"); -+ notify_process_quiescence_if_ready(&mut state, &inner.1); -+ } -+} -+ -+fn notify_process_quiescence_if_ready(state: &mut WasiProcessInner, condvar: &Condvar) { -+ if state.execution_leases == 0 && state.pending_child_publications == 0 { -+ for waker in state.quiescence_wakers.drain(..) { -+ waker.wake(); -+ } -+ condvar.notify_all(); -+ } -+} -+ - /// Represents a process running within the compute state - /// TODO: fields should be private and only accessed via methods. - #[derive(Debug, Clone)] -@@ -95,16 +213,24 @@ pub struct WasiProcess { - /// Hash of the module that this process is using - pub(crate) module_hash: ModuleHash, - /// List of all the children spawned from this thread -- pub(crate) parent: Option>>, -+ pub(crate) parent: Option, Condvar)>>, - /// The inner protected region of the process with a conditional - /// variable that is used for coordination such as snapshots. - pub(crate) inner: LockableWasiProcessInner, -+ /// Shared by this process and every descendant. The lock order is tree -+ /// epoch, process inner, then the control-plane registry. -+ pub(crate) tree_epoch: WasiProcessTreeEpoch, -+ /// Child publication barrier inherited by every TaskWasm for this process. -+ pub(crate) start_gate: WasiProcessStartGate, - /// Reference back to the compute engine - // TODO: remove this reference, access should happen via separate state instead - // (we don't want cyclical references) - pub(crate) compute: WasiControlPlaneHandle, - /// Reference to the exit code for the main thread - pub(crate) finished: Arc, -+ /// POSIX wait status is consumable once, even when multiple parent waiters -+ /// race through clones of the same process handle. -+ pub(crate) join_claimed: Arc, - /// Number of threads waiting for children to exit - pub(crate) waiting: Arc, - /// Number of tokens that are currently active and thus -@@ -113,6 +239,231 @@ pub struct WasiProcess { - pub(crate) cpu_run_tokens: Arc, - } - -+/// Execution ownership for host commands and vfork continuations that do not -+/// naturally live inside one `TaskWasm`. Clones share one lease. Completing -+/// the associated task publishes terminal status before releasing that lease; -+/// abandoning the guard fails closed with `Canceled`. -+#[derive(Clone, Debug)] -+pub struct WasiProcessExecutionGuard { -+ inner: Arc, -+} -+ -+#[derive(Debug)] -+struct WasiProcessExecutionGuardInner { -+ process: WasiProcess, -+ lease: Mutex>, -+ fail_closed: AtomicBool, -+ handoff_owner: Option, -+} -+ -+#[derive(Debug)] -+struct WasiProcessExecutionHandoffOwner { -+ thread: WasiThread, -+ task_wasm_owner: Weak<()>, -+} -+ -+impl WasiProcessExecutionGuardInner { -+ fn finish(&self, status: Result>) { -+ let lease = self.lease.lock().unwrap().take(); -+ if lease.is_none() { -+ return; -+ } -+ self.process.finished.set_finished(status); -+ drop(lease); -+ } -+} -+ -+impl Drop for WasiProcessExecutionGuardInner { -+ fn drop(&mut self) { -+ let lease = self.lease.get_mut().unwrap().take(); -+ if lease.is_none() { -+ return; -+ } -+ if self.fail_closed.load(Ordering::Acquire) { -+ if let Some(owner) = &self.handoff_owner { -+ owner.thread.set_status_finished(Ok(Errno::Canceled.into())); -+ } -+ // Losing an armed vfork owner is process-terminal. Mark every -+ // registered thread before releasing the final quiescence lease; -+ // the exact recorded thread above also covers a concurrent -+ // registry removal. -+ self.process.terminate(Errno::Canceled.into()); -+ self.process -+ .finished -+ .set_finished(Ok(Errno::Canceled.into())); -+ } -+ drop(lease); -+ } -+} -+ -+impl WasiProcessExecutionGuard { -+ fn new( -+ process: WasiProcess, -+ lease: WasiProcessExecutionLease, -+ fail_closed: bool, -+ handoff_owner: Option, -+ ) -> Self { -+ Self { -+ inner: Arc::new(WasiProcessExecutionGuardInner { -+ process, -+ lease: Mutex::new(Some(lease)), -+ fail_closed: AtomicBool::new(fail_closed), -+ handoff_owner, -+ }), -+ } -+ } -+ -+ pub(crate) fn finish(&self, status: Result>) { -+ self.inner.finish(status); -+ } -+ -+ /// Converts a setup-only supplemental guard into authoritative execution -+ /// ownership once an in-place vfork commits. Before commit, rollback can -+ /// hand it back to the original TaskWasm; after commit, abandoning the -+ /// stored/restored parent environment must fail closed. -+ pub(crate) fn arm_fail_closed(&self) { -+ self.inner.fail_closed.store(true, Ordering::Release); -+ } -+ -+ pub(crate) fn matches_process_thread( -+ &self, -+ process: &WasiProcess, -+ thread: &WasiThread, -+ ) -> bool { -+ self.inner.process.same_identity(process) -+ && self -+ .inner -+ .handoff_owner -+ .as_ref() -+ .is_some_and(|owner| owner.thread.same_identity(thread)) -+ } -+ -+ fn release_handoff_lease(&self) -> bool { -+ let mut lease = self.inner.lease.lock().unwrap(); -+ let Some(current) = lease.take() else { -+ return false; -+ }; -+ drop(lease); -+ drop(current); -+ true -+ } -+ -+ /// Releases a supplemental owner only when the currently executing -+ /// physical TaskWasm is exactly the owner recorded before the process -+ /// switch. Process-wide lease counts cannot authorize this transition. -+ pub(crate) fn try_handoff_to_current_task_wasm( -+ &self, -+ process: &WasiProcess, -+ thread: &WasiThread, -+ current_owner: &Weak<()>, -+ ) -> bool { -+ let Some(expected) = self.inner.handoff_owner.as_ref() else { -+ return false; -+ }; -+ if !self.matches_process_thread(process, thread) -+ || !Weak::ptr_eq(&expected.task_wasm_owner, current_owner) -+ || current_owner.upgrade().is_none() -+ { -+ return false; -+ } -+ self.release_handoff_lease() -+ } -+ -+ /// An accepted successor may adopt deferred ownership for the exact same -+ /// guest thread. The caller holds the successor guard while invoking this -+ /// method, so releasing the predecessor is successor-before-predecessor. -+ pub(crate) fn try_handoff_to_accepted_task_wasm( -+ &self, -+ process: &WasiProcess, -+ thread: &WasiThread, -+ ) -> bool { -+ self.matches_process_thread(process, thread) && self.release_handoff_lease() -+ } -+ -+ /// Retains execution ownership until the command's authoritative task -+ /// status becomes terminal. The task-manager acceptance boundary is the -+ /// handoff: rejection, panic, or cancellation transfers the command and -+ /// its execution guard to an independent reaper. -+ pub(crate) fn monitor( -+ self, -+ handle: TaskJoinHandle, -+ tasks: &Arc, -+ ) -> Result<(), WasiThreadError> { -+ let watcher = ProcessMonitorLifecycleGuard::new(handle, self); -+ let admission = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ tasks.task_shared(Box::new(move || { -+ Box::pin(async move { watcher.wait_finished().await }) -+ })) -+ })); -+ match admission { -+ Ok(result) => result, -+ Err(_) => { -+ tracing::error!("task manager panicked while admitting a process monitor"); -+ Err(WasiThreadError::InvalidWasmContext) -+ } -+ } -+ } -+} -+ -+/// Owns both sides of an admitted command-monitor relationship. -+/// -+/// A task manager may reject or panic while consuming the closure, or cancel -+/// the returned future later. Dropping a `TaskJoinHandle` is only a detach, so -+/// every abnormal path must transfer the exact handle and execution lease to a -+/// stable host reaper. -+struct ProcessMonitorLifecycleGuard { -+ task: Option, -+ execution: Option, -+} -+ -+impl ProcessMonitorLifecycleGuard { -+ fn new(task: TaskJoinHandle, execution: WasiProcessExecutionGuard) -> Self { -+ Self { -+ task: Some(task), -+ execution: Some(execution), -+ } -+ } -+ -+ async fn wait_finished(mut self) { -+ let Some(task) = self.task.as_mut() else { -+ return; -+ }; -+ let status = task.wait_finished().await; -+ if let Some(execution) = self.execution.take() { -+ execution.finish(status); -+ } -+ // Normal completion observed the authoritative status and published -+ // process terminal state before releasing the lease. Disarm so the hot -+ // path never creates a reaper thread. -+ self.task.take(); -+ } -+} -+ -+impl Drop for ProcessMonitorLifecycleGuard { -+ fn drop(&mut self) { -+ let Some(task) = self.task.take() else { -+ return; -+ }; -+ let process = self -+ .execution -+ .as_ref() -+ .map(|execution| execution.inner.process.clone()); -+ let abandon = process.map(|process| { -+ Box::new(move || process.terminate(Errno::Canceled.into())) as TaskAbandonCallback -+ }); -+ let completion = self.execution.take().map(|execution| { -+ Box::new(move |status| execution.finish(status)) as TaskCompletionCallback -+ }); -+ terminate_and_reap_abandoned_task( -+ task, -+ "wasmer-process-monitor-reaper", -+ "process monitor", -+ abandon, -+ completion, -+ ); -+ } -+} -+ - /// Represents a freeze of all threads to perform some action - /// on the total state-machine. This is normally done for - /// things like snapshots which require the memory to remain -@@ -165,6 +516,16 @@ pub struct WasiProcessInner { - pub signal_intervals: HashMap, - /// List of all the children spawned from this thread - pub children: Vec, -+ /// Prevents new threads/children from entering an execution epoch once a -+ /// reusable environment has begun terminal retirement. -+ pub(crate) retiring: bool, -+ /// Fork/spawn operations that have reserved the right to publish a child -+ /// but have not yet completed parent adoption. -+ pub(crate) pending_child_publications: u32, -+ /// Accepted Wasm tasks which have not reached their final callback/drop. -+ pub(crate) execution_leases: u32, -+ /// Joiners waiting for terminal execution quiescence. -+ pub(crate) quiescence_wakers: Vec, - /// Represents a checkpoint which blocks all the threads - /// and then executes some maintenance action - pub checkpoint: WasiProcessCheckpoint, -@@ -242,9 +603,12 @@ impl WasiProcessInner { - let thread_layout = ctx.data().thread.memory_layout().clone(); - unwind::(ctx, move |mut ctx, memory_stack, rewind_stack| { - // Grab all the globals and serialize them -- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) -- .serialize() -- .unwrap(); -+ let snapshot = -+ match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { -+ Ok(snapshot) => snapshot, -+ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), -+ }; -+ let store_data = snapshot.serialize().unwrap(); - let memory_stack = memory_stack.freeze(); - let rewind_stack = rewind_stack.freeze(); - let store_data = Bytes::from(store_data); -@@ -414,8 +778,84 @@ impl Drop for WasiProcessWait { - } - } - -+/// Holds one parent-side child publication reservation across process -+/// construction, registry commit, and exact parent adoption. -+#[derive(Debug)] -+pub(crate) struct WasiChildPublicationGuard { -+ parent: WasiProcess, -+ child_identity: Option, -+ active: bool, -+} -+ -+impl Drop for WasiChildPublicationGuard { -+ fn drop(&mut self) { -+ if !self.active { -+ return; -+ } -+ let _epoch = self.parent.tree_epoch.read().unwrap(); -+ let mut inner = self.parent.inner.0.lock().unwrap(); -+ inner.pending_child_publications = inner -+ .pending_child_publications -+ .checked_sub(1) -+ .expect("pending child publication count underflow"); -+ notify_process_quiescence_if_ready(&mut inner, &self.parent.inner.1); -+ } -+} -+ -+impl WasiChildPublicationGuard { -+ pub(crate) fn bind_child(mut self, child: &WasiProcess) -> Self { -+ assert!( -+ self.child_identity.is_none(), -+ "child publication bound twice" -+ ); -+ self.child_identity = Some(Arc::as_ptr(&child.inner) as usize); -+ self -+ } -+ -+ fn matches(&self, parent: &WasiProcess, child: &WasiProcess) -> bool { -+ self.active -+ && self.parent.same_identity(parent) -+ && self.child_identity == Some(Arc::as_ptr(&child.inner) as usize) -+ } -+} -+ - impl WasiProcess { - pub fn new(pid: WasiProcessId, module_hash: ModuleHash, plane: WasiControlPlaneHandle) -> Self { -+ Self::new_with_lifecycle( -+ pid, -+ module_hash, -+ plane, -+ None, -+ Arc::new(RwLock::new(())), -+ WasiProcessStartGate::committed(), -+ ) -+ } -+ -+ pub(crate) fn new_pending_child( -+ pid: WasiProcessId, -+ module_hash: ModuleHash, -+ plane: WasiControlPlaneHandle, -+ parent: &WasiProcess, -+ tree_epoch: WasiProcessTreeEpoch, -+ ) -> Self { -+ Self::new_with_lifecycle( -+ pid, -+ module_hash, -+ plane, -+ Some(Arc::downgrade(&parent.inner)), -+ tree_epoch, -+ WasiProcessStartGate::pending(), -+ ) -+ } -+ -+ fn new_with_lifecycle( -+ pid: WasiProcessId, -+ module_hash: ModuleHash, -+ plane: WasiControlPlaneHandle, -+ parent: Option, Condvar)>>, -+ tree_epoch: WasiProcessTreeEpoch, -+ start_gate: WasiProcessStartGate, -+ ) -> Self { - let max_cpu_backoff_time = plane - .upgrade() - .and_then(|p| p.config().enable_exponential_cpu_backoff) -@@ -430,6 +870,10 @@ impl WasiProcess { - thread_count: Default::default(), - signal_intervals: Default::default(), - children: Default::default(), -+ retiring: false, -+ pending_child_publications: 0, -+ execution_leases: 0, -+ quiescence_wakers: Default::default(), - checkpoint: WasiProcessCheckpoint::Execute, - wakers: Default::default(), - waiting: waiting.clone(), -@@ -460,18 +904,22 @@ impl WasiProcess { - WasiProcess { - pid, - module_hash, -- parent: None, -+ parent, - compute: plane, - inner: inner.clone(), -+ tree_epoch, -+ start_gate, - finished: Arc::new( - OwnedTaskStatus::new(TaskStatus::Pending) - .with_signal_handler(Arc::new(SignalHandler(inner))), - ), -+ join_claimed: Arc::new(AtomicBool::new(false)), - waiting, - cpu_run_tokens: Arc::new(AtomicU32::new(0)), - } - } - -+ #[cfg(test)] - pub(super) fn set_pid(&mut self, pid: WasiProcessId) { - self.pid = pid; - } -@@ -481,12 +929,126 @@ impl WasiProcess { - self.pid - } - -+ pub(crate) fn same_identity(&self, other: &Self) -> bool { -+ Arc::ptr_eq(&self.inner, &other.inner) -+ } -+ -+ pub(crate) fn same_tree(&self, other: &Self) -> bool { -+ Arc::ptr_eq(&self.tree_epoch, &other.tree_epoch) -+ } -+ -+ pub(crate) fn has_exact_parent(&self, parent: &Self) -> bool { -+ self.parent -+ .as_ref() -+ .and_then(Weak::upgrade) -+ .is_some_and(|candidate| Arc::ptr_eq(&candidate, &parent.inner)) -+ } -+ -+ pub(crate) fn tree_epoch(&self) -> WasiProcessTreeEpoch { -+ self.tree_epoch.clone() -+ } -+ -+ pub(crate) fn start_gate(&self) -> WasiProcessStartGate { -+ self.start_gate.clone() -+ } -+ -+ pub(crate) fn guest_start_is_committed(&self) -> bool { -+ self.start_gate.is_committed() -+ } -+ -+ pub(crate) fn commit_guest_start(&self) { -+ self.start_gate.commit(); -+ } -+ -+ pub(crate) fn abort_guest_start(&self) { -+ self.start_gate.abort(); -+ } -+ -+ pub(crate) fn acquire_execution_lease( -+ &self, -+ ) -> Result { -+ if !self.guest_start_is_committed() { -+ return Err(ControlPlaneError::ProcessNotPublished { -+ pid: self.pid().raw(), -+ }); -+ } -+ let _epoch = self.tree_epoch.read().unwrap(); -+ let mut inner = self.inner.0.lock().unwrap(); -+ if inner.retiring { -+ return Err(ControlPlaneError::ProcessRetiring { -+ pid: self.pid().raw(), -+ }); -+ } -+ if self.finished.status().into_finished().is_some() { -+ return Err(ControlPlaneError::ProcessFinished { -+ pid: self.pid().raw(), -+ }); -+ } -+ inner.execution_leases = inner -+ .execution_leases -+ .checked_add(1) -+ .expect("process execution lease count exhausted"); -+ Ok(WasiProcessExecutionLease { -+ inner: Arc::downgrade(&self.inner), -+ }) -+ } -+ -+ pub(crate) fn acquire_execution_guard( -+ &self, -+ ) -> Result { -+ let lease = self.acquire_execution_lease()?; -+ Ok(WasiProcessExecutionGuard::new( -+ self.clone(), -+ lease, -+ true, -+ None, -+ )) -+ } -+ -+ pub(crate) fn acquire_supplemental_execution_guard( -+ &self, -+ thread: WasiThread, -+ task_wasm_owner: Weak<()>, -+ ) -> Result { -+ let owns_thread = self -+ .inner -+ .0 -+ .lock() -+ .unwrap() -+ .threads -+ .values() -+ .any(|candidate| candidate.same_identity(&thread)); -+ if !owns_thread || task_wasm_owner.upgrade().is_none() { -+ return Err(ControlPlaneError::ExecutionOwnerUnavailable { -+ pid: self.pid().raw(), -+ tid: thread.tid().raw(), -+ }); -+ } -+ let lease = self.acquire_execution_lease()?; -+ Ok(WasiProcessExecutionGuard::new( -+ self.clone(), -+ lease, -+ false, -+ Some(WasiProcessExecutionHandoffOwner { -+ thread, -+ task_wasm_owner, -+ }), -+ )) -+ } -+ -+ pub(crate) fn wait_for_execution_quiescence_blocking(&self) { -+ let mut inner = self.inner.0.lock().unwrap(); -+ while inner.execution_leases != 0 { -+ inner = self.inner.1.wait(inner).unwrap(); -+ } -+ } -+ - /// Gets the process ID of the parent process - pub fn ppid(&self) -> WasiProcessId { - self.parent - .iter() - .filter_map(|parent| parent.upgrade()) -- .map(|parent| parent.read().unwrap().pid) -+ .map(|parent| parent.0.lock().unwrap().pid) - .next() - .unwrap_or(WasiProcessId(0)) - } -@@ -528,12 +1090,30 @@ impl WasiProcess { - tid: WasiThreadId, - ) -> Result { - let control_plane = self.compute.must_upgrade(); -- let task_count_guard = control_plane.register_task()?; -- - let is_main = matches!(start, ThreadStartType::MainThread); - -- // The wait finished should be the process version if its the main thread -+ // Numeric TID ownership and insertion linearize under the shared tree -+ // admission gate followed by the process lock. -+ // Task admission itself is an atomic reservation and rolls back if any -+ // construction below returns before the handle is published. -+ let _epoch = self.tree_epoch.read().unwrap(); - let mut inner = self.inner.0.lock().unwrap(); -+ if inner.retiring { -+ return Err(ControlPlaneError::ProcessRetiring { -+ pid: self.pid().raw(), -+ }); -+ } -+ if self.finished.status().into_finished().is_some() { -+ return Err(ControlPlaneError::ProcessFinished { -+ pid: self.pid().raw(), -+ }); -+ } -+ if inner.threads.contains_key(&tid) { -+ return Err(ControlPlaneError::DuplicateThreadId { tid: tid.raw() }); -+ } -+ let task_count_guard = control_plane.register_task()?; -+ -+ // The wait finished should be the process version if its the main thread - let finished = if is_main { - self.finished.clone() - } else { -@@ -551,7 +1131,10 @@ impl WasiProcess { - start, - ); - inner.threads.insert(tid, ctrl.clone()); -- inner.thread_count += 1; -+ inner.thread_count = inner -+ .thread_count -+ .checked_add(1) -+ .expect("process thread count exhausted"); - - Ok(WasiThreadHandle::new(ctrl, &self.inner)) - } -@@ -597,6 +1180,248 @@ impl WasiProcess { - signal_process_internal(&self.inner, signal); - } - -+ /// Adds and publishes the exact child authorized by `publication` while -+ /// holding the parent epoch lock. Retirement therefore linearizes either -+ /// before the reservation or after both the parent link and registry entry. -+ pub(crate) fn adopt_child_process( -+ &self, -+ child: WasiProcess, -+ mut publication: WasiChildPublicationGuard, -+ publish: impl FnOnce(), -+ ) -> Result<(), ControlPlaneError> { -+ if !publication.matches(self, &child) { -+ return Err(ControlPlaneError::InvalidChildPublication { -+ parent_pid: self.pid().raw(), -+ child_pid: child.pid().raw(), -+ }); -+ } -+ if !self.same_tree(&child) || !child.has_exact_parent(self) { -+ return Err(ControlPlaneError::InvalidChildPublication { -+ parent_pid: self.pid().raw(), -+ child_pid: child.pid().raw(), -+ }); -+ } -+ -+ let _epoch = self.tree_epoch.read().unwrap(); -+ let mut inner = self.inner.0.lock().unwrap(); -+ if inner.retiring { -+ drop(inner); -+ return Err(ControlPlaneError::ProcessRetiring { -+ pid: self.pid().raw(), -+ }); -+ } -+ if self.finished.status().into_finished().is_some() { -+ drop(inner); -+ return Err(ControlPlaneError::ProcessFinished { -+ pid: self.pid().raw(), -+ }); -+ } -+ if inner.pending_child_publications == 0 -+ || inner -+ .children -+ .iter() -+ .any(|candidate| candidate.same_identity(&child)) -+ { -+ drop(inner); -+ return Err(ControlPlaneError::InvalidChildPublication { -+ parent_pid: self.pid().raw(), -+ child_pid: child.pid().raw(), -+ }); -+ } -+ -+ inner.pending_child_publications -= 1; -+ publication.active = false; -+ inner.children.push(child); -+ publish(); -+ notify_process_quiescence_if_ready(&mut inner, &self.inner.1); -+ Ok(()) -+ } -+ -+ /// Arranges POSIX-style child-exit notification after the child -+ /// publication transaction has become externally visible. -+ pub(crate) fn notify_on_child_exit( -+ &self, -+ child: WasiProcess, -+ tasks: &Arc, -+ ) { -+ let parent = self.clone(); -+ let parent_pid = parent.pid(); -+ let child_pid = child.pid(); -+ if let Err(err) = tasks.task_shared(Box::new(move || { -+ Box::pin(async move { -+ let _ = child.join().await; -+ tracing::trace!(%parent_pid, %child_pid, "signaling child exit"); -+ parent.signal_process(Signal::Sigchld); -+ }) -+ })) { -+ tracing::warn!( -+ %parent_pid, -+ %child_pid, -+ "failed to schedule child-exit signal delivery: {err}" -+ ); -+ } -+ } -+ -+ /// Reserves an in-flight child publication so epoch retirement cannot -+ /// pass quiescence while fork/spawn is between construction and adoption. -+ pub(crate) fn begin_child_publication( -+ &self, -+ ) -> Result { -+ let _epoch = self.tree_epoch.read().unwrap(); -+ let mut inner = self.inner.0.lock().unwrap(); -+ if inner.retiring { -+ return Err(ControlPlaneError::ProcessRetiring { -+ pid: self.pid().raw(), -+ }); -+ } -+ if self.finished.status().into_finished().is_some() { -+ return Err(ControlPlaneError::ProcessFinished { -+ pid: self.pid().raw(), -+ }); -+ } -+ inner.pending_child_publications = inner -+ .pending_child_publications -+ .checked_add(1) -+ .expect("pending child publication count exhausted"); -+ Ok(WasiChildPublicationGuard { -+ parent: self.clone(), -+ child_identity: None, -+ active: true, -+ }) -+ } -+ -+ /// Verifies and seals a complete process tree without changing any node on -+ /// rejection. The exclusive tree gate makes the validation and sealing -+ /// passes one atomic admission epoch: descendants cannot add work or -+ /// topology between the two passes. -+ /// -+ /// The returned descendants are in post-order. Every returned process and -+ /// `self` is already sealed when this method succeeds. -+ pub(crate) fn begin_epoch_retirement(&self) -> Result, ControlPlaneError> { -+ let _epoch = self.tree_epoch.write().unwrap(); -+ self.validate_retirement_node()?; -+ -+ let children = self.inner.0.lock().unwrap().children.clone(); -+ let mut visited = HashSet::from([Arc::as_ptr(&self.inner) as usize]); -+ let mut descendants = Vec::new(); -+ let mut live_children = 0; -+ for child in children { -+ if child.collect_retirement_candidates(&self.tree_epoch, &mut visited, &mut descendants) -+ { -+ descendants.push(child); -+ } else { -+ live_children += 1; -+ } -+ } -+ if live_children != 0 { -+ return Err(ControlPlaneError::ProcessHasLiveChildren { -+ pid: self.pid().raw(), -+ count: live_children, -+ }); -+ } -+ -+ // Pass two begins only after every node passed validation. No -+ // admission or topology mutation can interleave while `_epoch` lives. -+ for process in descendants.iter().chain(std::iter::once(self)) { -+ let mut inner = process.inner.0.lock().unwrap(); -+ debug_assert!(!inner.retiring); -+ inner.retiring = true; -+ } -+ Ok(descendants) -+ } -+ -+ fn validate_retirement_node(&self) -> Result<(), ControlPlaneError> { -+ if self.finished.status().into_finished().is_none() { -+ return Err(ControlPlaneError::ProcessStillRunning { -+ pid: self.pid().raw(), -+ }); -+ } -+ let inner = self.inner.0.lock().unwrap(); -+ if inner.retiring { -+ return Err(ControlPlaneError::ProcessRetiring { -+ pid: self.pid().raw(), -+ }); -+ } -+ if inner.execution_leases != 0 { -+ return Err(ControlPlaneError::ProcessStillRunning { -+ pid: self.pid().raw(), -+ }); -+ } -+ let live_non_main_threads = inner -+ .threads -+ .values() -+ .filter(|thread| !thread.is_main()) -+ .count(); -+ if live_non_main_threads != 0 { -+ return Err(ControlPlaneError::ProcessHasLiveThreads { -+ pid: self.pid().raw(), -+ count: live_non_main_threads, -+ }); -+ } -+ if inner.pending_child_publications != 0 { -+ return Err(ControlPlaneError::ProcessHasLiveChildren { -+ pid: self.pid().raw(), -+ count: inner.pending_child_publications as usize, -+ }); -+ } -+ Ok(()) -+ } -+ -+ fn collect_retirement_candidates( -+ &self, -+ expected_tree: &WasiProcessTreeEpoch, -+ visited: &mut HashSet, -+ descendants: &mut Vec, -+ ) -> bool { -+ let identity = Arc::as_ptr(&self.inner) as usize; -+ if !Arc::ptr_eq(&self.tree_epoch, expected_tree) || !visited.insert(identity) { -+ // Process children are a tree. A cycle is malformed retained state, -+ // and must block retirement rather than deadlock recursive locks. -+ return false; -+ } -+ if self.validate_retirement_node().is_err() { -+ return false; -+ } -+ let children = self.inner.0.lock().unwrap().children.clone(); -+ for child in children { -+ if !child.collect_retirement_candidates(expected_tree, visited, descendants) { -+ return false; -+ } -+ descendants.push(child); -+ } -+ true -+ } -+ -+ /// Releases task-count reservations for a sealed, quiescent epoch. The -+ /// thread objects may remain referenced by stale handles, but can no -+ /// longer consume embedded-runtime admission or execute new work. -+ pub(crate) fn retire_epoch_task_registrations(&self) { -+ let inner = self.inner.0.lock().unwrap(); -+ assert!(inner.retiring, "task retirement requires a sealed epoch"); -+ for thread in inner.threads.values() { -+ thread.retire_task_registration(); -+ } -+ } -+ -+ /// Removes only the exact child identity. PID equality alone is not safe -+ /// once stale handles and eventual PID reuse are considered. -+ pub(crate) fn remove_child_if_same(&self, child: &WasiProcess) -> bool { -+ let _epoch = self.tree_epoch.read().unwrap(); -+ let mut inner = self.inner.0.lock().unwrap(); -+ let before = inner.children.len(); -+ inner -+ .children -+ .retain(|candidate| !candidate.same_identity(child)); -+ inner.children.len() != before -+ } -+ -+ pub(crate) fn clear_retired_children(&self) { -+ let _epoch = self.tree_epoch.read().unwrap(); -+ let mut inner = self.inner.0.lock().unwrap(); -+ assert!(inner.retiring, "child cleanup requires a sealed epoch"); -+ inner.children.clear(); -+ } -+ - /// Takes a snapshot of the process and disables journaling returning - /// a future that can be waited on for the snapshot to complete - /// -@@ -765,12 +1590,78 @@ impl WasiProcess { - /// Waits until the process is finished. - pub async fn join(&self) -> Result> { - let _guard = WasiProcessWait::new(self); -- self.finished.await_termination().await -+ let status = self.finished.await_termination().await; -+ WasiProcessQuiescenceFuture { -+ inner: self.inner.clone(), -+ } -+ .await; -+ status - } - - /// Attempts to join on the process - pub fn try_join(&self) -> Option>> { -- self.finished.status().into_finished() -+ let status = self.finished.status().into_finished()?; -+ let inner = self.inner.0.lock().unwrap(); -+ (inner.execution_leases == 0 && inner.pending_child_publications == 0).then_some(status) -+ } -+ -+ /// Attempts to observe and atomically consume this process's wait status. -+ pub(crate) fn try_claim_join(&self) -> Option>> { -+ let status = self.try_join()?; -+ self.join_claimed -+ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) -+ .ok() -+ .map(|_| status) -+ } -+ -+ /// Waits for termination and atomically consumes the status. A competing -+ /// waiter that already claimed it causes `None` rather than a duplicate -+ /// successful wait result. -+ pub(crate) async fn claim_join(&self) -> Option>> { -+ let status = self.join().await; -+ self.join_claimed -+ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) -+ .ok() -+ .map(|_| status) -+ } -+ -+ /// Removes this process from the control-plane registry after its exit -+ /// status has been consumed by a join path. Process exit alone deliberately -+ /// does not call this so zombie/wait semantics remain intact. -+ pub(crate) fn reap(&self) -> bool { -+ self.compute -+ .upgrade() -+ .is_some_and(|control_plane| control_plane.reap_process(self)) -+ } -+ -+ /// Attempts to join any finished child without blocking. -+ pub fn try_join_any_child(&mut self) -> Result, Errno> { -+ let children: Vec<_> = { -+ let inner = self.inner.0.lock().unwrap(); -+ inner.children.clone() -+ }; -+ if children.is_empty() { -+ return Err(Errno::Child); -+ } -+ -+ for child in children { -+ let process = self -+ .compute -+ .must_upgrade() -+ .get_process(child.pid) -+ .unwrap_or_else(|| child.clone()); -+ let Some(status) = process.try_claim_join() else { -+ continue; -+ }; -+ -+ self.remove_child_if_same(&child); -+ let code = status -+ .unwrap_or_else(|e| e.as_exit_code().unwrap_or_else(|| Errno::Canceled.into())); -+ process.reap(); -+ return Ok(Some((child.pid, code))); -+ } -+ -+ Ok(None) - } - - /// Waits for all the children to be finished -@@ -785,17 +1676,26 @@ impl WasiProcess { - } - let mut waits = Vec::new(); - for child in children { -- if let Some(process) = self.compute.must_upgrade().get_process(child.pid) { -- let inner = self.inner.clone(); -- waits.push(async move { -- let join = process.join().await; -- let mut inner = inner.0.lock().unwrap(); -- inner.children.retain(|a| a.pid != child.pid); -- join -- }) -- } -+ let process = self -+ .compute -+ .must_upgrade() -+ .get_process(child.pid) -+ .unwrap_or_else(|| child.clone()); -+ let parent = self.clone(); -+ waits.push(async move { -+ let join = process.claim_join().await; -+ parent.remove_child_if_same(&child); -+ if join.is_some() { -+ process.reap(); -+ } -+ join -+ }) - } -- futures::future::join_all(waits).await.into_iter().next() -+ futures::future::join_all(waits) -+ .await -+ .into_iter() -+ .flatten() -+ .next() - } - - /// Waits for any of the children to finished -@@ -811,24 +1711,33 @@ impl WasiProcess { - - let mut waits = Vec::new(); - for child in children { -- if let Some(process) = self.compute.must_upgrade().get_process(child.pid) { -- let inner = self.inner.clone(); -- waits.push(async move { -- let join = process.join().await; -- let mut inner = inner.0.lock().unwrap(); -- inner.children.retain(|a| a.pid != child.pid); -- (child, join) -- }) -- } -+ let process = self -+ .compute -+ .must_upgrade() -+ .get_process(child.pid) -+ .unwrap_or_else(|| child.clone()); -+ let parent = self.clone(); -+ waits.push(async move { -+ let join = process.claim_join().await; -+ parent.remove_child_if_same(&child); -+ (child, process, join) -+ }) -+ } -+ let mut waits: Vec<_> = waits.into_iter().map(Box::pin).collect(); -+ while !waits.is_empty() { -+ let ((child, process, claimed), _, remaining) = -+ futures::future::select_all(waits).await; -+ waits = remaining; -+ let Some(res) = claimed else { -+ continue; -+ }; -+ process.reap(); -+ let code = -+ res.unwrap_or_else(|e| e.as_exit_code().unwrap_or_else(|| Errno::Canceled.into())); -+ return Ok(Some((child.pid, code))); - } -- let (child, res) = futures::future::select_all(waits.into_iter().map(Box::pin)) -- .await -- .0; -- -- let code = -- res.unwrap_or_else(|e| e.as_exit_code().unwrap_or_else(|| Errno::Canceled.into())); - -- Ok(Some((child.pid, code))) -+ Ok(None) - } - - /// Terminate the process and all its threads -diff --git a/lib/wasix/src/os/task/task_join_handle.rs b/lib/wasix/src/os/task/task_join_handle.rs -index d503817..c37169a 100644 ---- a/lib/wasix/src/os/task/task_join_handle.rs -+++ b/lib/wasix/src/os/task/task_join_handle.rs -@@ -1,14 +1,48 @@ - use std::{ - pin::Pin, -- sync::Arc, -+ sync::{Arc, Mutex}, - task::{Context, Poll}, - }; - -+#[cfg(all(feature = "ctrlc", windows))] -+use std::{ -+ ffi::c_void, -+ io, -+ sync::atomic::{AtomicBool, AtomicPtr, AtomicU32, AtomicUsize, Ordering}, -+ thread::JoinHandle, -+}; -+#[cfg(all(feature = "ctrlc", unix))] -+use std::{ -+ io, -+ mem::MaybeUninit, -+ os::fd::RawFd, -+ sync::atomic::{AtomicBool, AtomicI32, AtomicU32, AtomicUsize, Ordering}, -+ thread::JoinHandle, -+}; -+ -+#[cfg(all(feature = "ctrlc", windows))] -+use windows_sys::{ -+ Win32::{ -+ Foundation::{CloseHandle, HANDLE, WAIT_OBJECT_0}, -+ System::{ -+ Console::{ -+ CTRL_BREAK_EVENT, CTRL_C_EVENT, CTRL_CLOSE_EVENT, CTRL_LOGOFF_EVENT, -+ CTRL_SHUTDOWN_EVENT, SetConsoleCtrlHandler, -+ }, -+ Threading::{CreateEventW, INFINITE, SetEvent, WaitForSingleObject}, -+ }, -+ }, -+ core::BOOL, -+}; -+ - use wasmer_wasix_types::wasi::{Errno, ExitCode}; - -+use wasmer::FromToNativeWasmType; -+use wasmer_wasix_types::wasi::Signal; -+ - use crate::WasiRuntimeError; - --use super::signal::{DynSignalHandlerAbi, default_signal_handler}; -+use super::signal::{DynSignalHandlerAbi, SignalDeliveryError, default_signal_handler}; - - #[derive(Clone, Debug)] - pub enum TaskStatus { -@@ -54,6 +88,44 @@ impl TaskStatus { - #[error("Task already terminated")] - pub struct TaskTerminatedError; - -+/// A platform-neutral, cloneable controller for delivering a WASI signal to -+/// one task. It does not install or mutate host operating-system signal -+/// dispositions. -+#[derive(Clone, Debug)] -+pub struct TaskSignalController { -+ signal_handler: Arc, -+ watch: tokio::sync::watch::Receiver, -+} -+ -+#[derive(thiserror::Error, Debug)] -+pub enum TaskSignalError { -+ #[error("task already terminated")] -+ Terminated, -+ #[error(transparent)] -+ Delivery(#[from] SignalDeliveryError), -+} -+ -+impl TaskSignalController { -+ /// Retrieve the current task status without changing it. -+ pub fn status(&self) -> TaskStatus { -+ self.watch.borrow().clone() -+ } -+ -+ /// Deliver a WASI signal directly to this task. -+ /// -+ /// The status check prevents knowingly targeting a completed task. Task -+ /// completion can still race signal delivery, exactly as it can for a -+ /// native process. -+ pub fn send_signal(&self, signal: Signal) -> Result<(), TaskSignalError> { -+ if self.status().is_finished() { -+ return Err(TaskSignalError::Terminated); -+ } -+ self.signal_handler -+ .signal(signal.to_native() as u8) -+ .map_err(TaskSignalError::from) -+ } -+} -+ - pub trait VirtualTaskHandle: std::fmt::Debug + Send + Sync + 'static { - fn status(&self) -> TaskStatus; - -@@ -176,7 +248,6 @@ impl Default for OwnedTaskStatus { - /// A handle that allows awaiting the termination of a task, and retrieving its exit code. - #[derive(Clone, Debug)] - pub struct TaskJoinHandle { -- #[allow(unused)] - signal_handler: Arc, - watch: tokio::sync::watch::Receiver, - } -@@ -187,22 +258,18 @@ impl TaskJoinHandle { - self.watch.borrow().clone() - } - -- #[cfg(feature = "ctrlc")] -- pub fn install_ctrlc_handler(&self) { -- use wasmer::FromToNativeWasmType; -- use wasmer_wasix_types::wasi::Signal; -- -- let signal_handler = self.signal_handler.clone(); -+ /// Create a platform-neutral controller that can deliver WASI signals to -+ /// this task without taking ownership of host OS signal policy. -+ pub fn signal_controller(&self) -> TaskSignalController { -+ TaskSignalController { -+ signal_handler: self.signal_handler.clone(), -+ watch: self.watch.clone(), -+ } -+ } - -- tokio::spawn(async move { -- // Loop sending ctrl-c presses as signals to the signal handler -- while tokio::signal::ctrl_c().await.is_ok() { -- if let Err(err) = signal_handler.signal(Signal::Sigint.to_native() as u8) { -- tracing::error!("failed to process signal - {}", err); -- std::process::exit(1); -- } -- } -- }); -+ /// Deliver a WASI signal directly to this task. -+ pub fn send_signal(&self, signal: Signal) -> Result<(), TaskSignalError> { -+ self.signal_controller().send_signal(signal) - } - - /// Wait until the task finishes. -@@ -221,3 +288,1230 @@ impl TaskJoinHandle { - } - } - } -+ -+pub(crate) type TaskCompletion = Result>; -+pub(crate) type TaskAbandonCallback = Box; -+pub(crate) type TaskCompletionCallback = Box; -+ -+type AbandonedTask = ( -+ TaskJoinHandle, -+ Option, -+ Option, -+); -+ -+fn take_abandoned_task(slot: &Mutex>) -> Option { -+ match slot.lock() { -+ Ok(mut slot) => slot.take(), -+ Err(poisoned) => poisoned.into_inner().take(), -+ } -+} -+ -+fn terminate_and_wait_for_abandoned_task( -+ mut task: TaskJoinHandle, -+ abandon: Option, -+ completion: Option, -+ task_description: &'static str, -+) { -+ if let Some(abandon) = abandon -+ && std::panic::catch_unwind(std::panic::AssertUnwindSafe(abandon)).is_err() -+ { -+ tracing::error!( -+ task = task_description, -+ "abandoned task termination callback panicked" -+ ); -+ } -+ -+ let signal = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ task.send_signal(Signal::Sigkill) -+ })); -+ match signal { -+ Ok(Ok(())) => {} -+ Ok(Err(error)) => { -+ tracing::debug!(%error, task = task_description, "abandoned task rejected SIGKILL") -+ } -+ Err(_) => { -+ tracing::error!( -+ task = task_description, -+ "abandoned task signal handler panicked" -+ ) -+ } -+ } -+ -+ let status = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ virtual_mio::block_on(task.wait_finished()) -+ })); -+ let Ok(status) = status else { -+ // Dropping the completion callback is intentional. Lifecycle callbacks -+ // own their fail-closed process guard, so even a broken task-status -+ // implementation cannot unwind this independent reaper or release a -+ // live process lease without first publishing terminal status. -+ tracing::error!( -+ task = task_description, -+ "abandoned task reaper panicked while waiting" -+ ); -+ return; -+ }; -+ -+ match &status { -+ Ok(code) => { -+ tracing::debug!( -+ exit_code = code.raw(), -+ task = task_description, -+ "abandoned task reaped" -+ ) -+ } -+ Err(error) => { -+ tracing::debug!(%error, task = task_description, "abandoned task reaped with error") -+ } -+ } -+ -+ if let Some(completion) = completion -+ && std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| completion(status))).is_err() -+ { -+ tracing::error!( -+ task = task_description, -+ "abandoned task completion callback panicked" -+ ); -+ } -+} -+ -+/// Transfers an abandoned accepted task to a stable host reaper. -+/// -+/// `Builder::spawn` consumes its closure even on failure. The shared slot lets -+/// the caller recover both the authoritative task handle and its terminal -+/// callback for a synchronous fallback, so neither cancellation nor unwinding -+/// can silently detach accepted guest work. -+pub(crate) fn terminate_and_reap_abandoned_task( -+ task: TaskJoinHandle, -+ reaper_name: &'static str, -+ task_description: &'static str, -+ abandon: Option, -+ completion: Option, -+) { -+ let abandoned = Arc::new(Mutex::new(Some((task, abandon, completion)))); -+ let reaper_task = abandoned.clone(); -+ let spawn = std::thread::Builder::new() -+ .name(reaper_name.to_string()) -+ .spawn(move || { -+ if let Some((task, abandon, completion)) = take_abandoned_task(&reaper_task) { -+ terminate_and_wait_for_abandoned_task(task, abandon, completion, task_description); -+ } -+ }); -+ if let Err(error) = spawn { -+ tracing::error!(%error, task = task_description, "failed to spawn task reaper; waiting synchronously"); -+ if let Some((task, abandon, completion)) = take_abandoned_task(&abandoned) { -+ terminate_and_wait_for_abandoned_task(task, abandon, completion, task_description); -+ } -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+const UNIX_HOST_SIGNALS: [(libc::c_int, u32, Signal, &str); 4] = [ -+ (libc::SIGINT, 1 << 0, Signal::Sigint, "SIGINT"), -+ (libc::SIGTERM, 1 << 1, Signal::Sigterm, "SIGTERM"), -+ (libc::SIGQUIT, 1 << 2, Signal::Sigquit, "SIGQUIT"), -+ (libc::SIGHUP, 1 << 3, Signal::Sighup, "SIGHUP"), -+]; -+ -+#[cfg(all(feature = "ctrlc", unix))] -+static UNIX_HOST_SIGNAL_OWNER_ACTIVE: AtomicBool = AtomicBool::new(false); -+#[cfg(all(feature = "ctrlc", unix))] -+static UNIX_HOST_SIGNAL_WRITE_FD: AtomicI32 = AtomicI32::new(-1); -+#[cfg(all(feature = "ctrlc", unix))] -+static UNIX_HOST_SIGNAL_PENDING: AtomicU32 = AtomicU32::new(0); -+#[cfg(all(feature = "ctrlc", unix))] -+static UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT: AtomicUsize = AtomicUsize::new(0); -+ -+/// Failure to install or bind the explicit Unix host lifecycle supervisor. -+#[cfg(all(feature = "ctrlc", unix))] -+#[derive(thiserror::Error, Debug)] -+pub enum UnixHostLifecycleError { -+ #[error( -+ "Unix host lifecycle supervision is unsupported on this target; use TaskSignalController" -+ )] -+ UnsupportedPlatform, -+ #[error("another Unix host lifecycle supervisor already owns this process")] -+ AlreadyOwned, -+ #[error("this Unix host lifecycle supervisor is already bound to a task")] -+ AlreadyBound, -+ #[error("{operation} failed: {source}")] -+ Io { -+ operation: &'static str, -+ #[source] -+ source: io::Error, -+ }, -+ #[error("{operation} failed for signal {signal}: {source}")] -+ Sigaction { -+ operation: &'static str, -+ signal: libc::c_int, -+ #[source] -+ source: io::Error, -+ }, -+ #[error("buffered {signal:?} could not be delivered when the task was bound: {source}")] -+ BufferedSignalDelivery { -+ signal: Signal, -+ #[source] -+ source: TaskSignalError, -+ }, -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+#[derive(Default)] -+struct UnixHostSignalRoute { -+ target: Option, -+ buffered: u32, -+ bound: bool, -+ stopping: bool, -+} -+ -+/// Exclusive process-scoped Unix lifecycle signal supervision. -+/// -+/// This is an explicit CLI/supervisor adapter, not an embedded-library -+/// default. It captures all of SIGINT, SIGTERM, SIGQUIT, and SIGHUP before a -+/// guest is scheduled, buffers signals until exactly one root task is bound, -+/// and restores the exact previous `sigaction` values when the final owner is -+/// dropped. Use [`TaskSignalController`] directly in embedded applications. -+/// -+/// The low-level errno adapters cover Linux, Android, Apple, the supported BSD -+/// families, and Solaris/illumos. Installation fails without changing -+/// dispositions on any other Unix target; its process owner can still route -+/// signals through [`TaskSignalController`] directly. -+#[cfg(all(feature = "ctrlc", unix))] -+pub struct UnixHostLifecycleSupervisor { -+ resources: UnixHostSignalResources, -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+pub type HostLifecycleSupervisor = UnixHostLifecycleSupervisor; -+ -+#[cfg(all(feature = "ctrlc", unix))] -+impl std::fmt::Debug for UnixHostLifecycleSupervisor { -+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { -+ f.debug_struct("UnixHostLifecycleSupervisor") -+ .field("read_fd", &self.resources.read_fd) -+ .field("write_fd", &self.resources.write_fd) -+ .finish_non_exhaustive() -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+struct UnixHostSignalResources { -+ read_fd: RawFd, -+ write_fd: RawFd, -+ previous_actions: Vec<(libc::c_int, libc::sigaction)>, -+ route: Arc>, -+ forwarder: Option>, -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+impl UnixHostLifecycleSupervisor { -+ /// Whether the process-global adapter is qualified for this target. -+ pub const fn is_supported() -> bool { -+ cfg!(any( -+ target_os = "linux", -+ target_os = "android", -+ target_os = "macos", -+ target_os = "ios", -+ target_os = "tvos", -+ target_os = "watchos", -+ target_os = "visionos", -+ target_os = "freebsd", -+ target_os = "dragonfly", -+ target_os = "netbsd", -+ target_os = "openbsd", -+ target_os = "solaris", -+ target_os = "illumos", -+ )) -+ } -+ -+ /// Install the exclusive process signal owner and its dedicated forwarder. -+ /// All four dispositions are installed transactionally or the operation -+ /// restores every disposition already changed and fails. -+ pub fn install() -> Result { -+ Self::install_signals(&UNIX_HOST_SIGNALS.map(|entry| entry.0)) -+ } -+ -+ fn install_signals(signals: &[libc::c_int]) -> Result { -+ if !Self::is_supported() { -+ return Err(UnixHostLifecycleError::UnsupportedPlatform); -+ } -+ UNIX_HOST_SIGNAL_OWNER_ACTIVE -+ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) -+ .map_err(|_| UnixHostLifecycleError::AlreadyOwned)?; -+ -+ let (read_fd, write_fd) = match create_unix_signal_pipe() { -+ Ok(fds) => fds, -+ Err(source) => { -+ UNIX_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); -+ return Err(UnixHostLifecycleError::Io { -+ operation: "create host signal self-pipe", -+ source, -+ }); -+ } -+ }; -+ let route = Arc::new(Mutex::new(UnixHostSignalRoute::default())); -+ let thread_route = route.clone(); -+ let forwarder = match std::thread::Builder::new() -+ .name("wasmer-host-signals".to_string()) -+ .spawn(move || unix_host_signal_forwarder(read_fd, thread_route)) -+ { -+ Ok(forwarder) => forwarder, -+ Err(source) => { -+ unsafe { -+ libc::close(read_fd); -+ libc::close(write_fd); -+ } -+ UNIX_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); -+ return Err(UnixHostLifecycleError::Io { -+ operation: "spawn host signal forwarder", -+ source, -+ }); -+ } -+ }; -+ -+ let mut resources = UnixHostSignalResources { -+ read_fd, -+ write_fd, -+ previous_actions: Vec::with_capacity(signals.len()), -+ route, -+ forwarder: Some(forwarder), -+ }; -+ UNIX_HOST_SIGNAL_PENDING.store(0, Ordering::Release); -+ UNIX_HOST_SIGNAL_WRITE_FD.store(write_fd, Ordering::SeqCst); -+ -+ let action = unix_host_signal_action()?; -+ for &signal in signals { -+ let mut previous = MaybeUninit::::uninit(); -+ let rc = unsafe { libc::sigaction(signal, &action, previous.as_mut_ptr()) }; -+ if rc != 0 { -+ return Err(UnixHostLifecycleError::Sigaction { -+ operation: "install host lifecycle handler", -+ signal, -+ source: io::Error::last_os_error(), -+ }); -+ } -+ resources -+ .previous_actions -+ .push((signal, unsafe { previous.assume_init() })); -+ } -+ -+ Ok(Self { resources }) -+ } -+ -+ /// Bind the sole root guest task. Signals captured before this call are -+ /// retained and delivered to this task, even if the dedicated reader has -+ /// not yet observed its self-pipe wakeup when this call returns. -+ pub fn bind_task(&self, task: &TaskJoinHandle) -> Result<(), UnixHostLifecycleError> { -+ let controller = task.signal_controller(); -+ let buffered = { -+ let mut route = lock_unix_signal_route(&self.resources.route); -+ if route.bound { -+ return Err(UnixHostLifecycleError::AlreadyBound); -+ } -+ route.bound = true; -+ route.target = Some(controller.clone()); -+ std::mem::take(&mut route.buffered) -+ }; -+ deliver_unix_signal_mask(buffered, &controller).map_err(|(signal, source)| { -+ UnixHostLifecycleError::BufferedSignalDelivery { signal, source } -+ }) -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+impl Drop for UnixHostSignalResources { -+ fn drop(&mut self) { -+ let mut restoration_failed = false; -+ for (signal, previous) in self.previous_actions.iter().rev() { -+ if unsafe { libc::sigaction(*signal, previous, std::ptr::null_mut()) } != 0 { -+ restoration_failed = true; -+ tracing::error!( -+ host_signal = *signal, -+ error = %io::Error::last_os_error(), -+ "failed to restore previous host signal disposition" -+ ); -+ } -+ } -+ self.previous_actions.clear(); -+ -+ if restoration_failed { -+ // Never close or recycle a descriptor that an unrestored signal -+ // disposition can still reach. Keeping the reader, descriptors, -+ // route, and exclusivity token alive is the only fail-closed state -+ // available from Drop; the detached reader's Arc retains `route`. -+ // This path means the kernel rejected restoration of a disposition -+ // that it previously accepted, so leaking these process-scoped -+ // resources is intentional and safer than a stale-handler UAF. -+ tracing::error!( -+ "host signal restoration was incomplete; retaining the supervisor for process lifetime" -+ ); -+ return; -+ } -+ -+ // The SeqCst protocol closes the stale-handler/reused-fd window: -+ // -+ // * every handler increments before loading the descriptor; -+ // * a handler that loaded the old descriptor is therefore included in -+ // the count observed below and must finish before close; -+ // * a handler whose increment is ordered after the zero observation is -+ // also ordered after this -1 store and cannot load the old value. -+ UNIX_HOST_SIGNAL_WRITE_FD.store(-1, Ordering::SeqCst); -+ while UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT.load(Ordering::SeqCst) != 0 { -+ std::thread::yield_now(); -+ } -+ -+ // No handler can add work after this point. Ask the sole normal bitset -+ // consumer to drain and exit, keeping the target bound until it has -+ // joined. A full pipe needs no extra byte: it already contains a wakeup. -+ { -+ let mut route = lock_unix_signal_route(&self.route); -+ route.stopping = true; -+ } -+ wake_unix_signal_forwarder(self.write_fd); -+ if let Some(forwarder) = self.forwarder.take() -+ && forwarder.join().is_err() -+ { -+ tracing::error!("host signal forwarder panicked during shutdown"); -+ } -+ // Recover pending work if the reader exited because of an unexpected -+ // pipe failure. Under normal operation it has already drained to zero. -+ route_unix_signal_mask( -+ UNIX_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), -+ &self.route, -+ ); -+ { -+ let mut route = lock_unix_signal_route(&self.route); -+ route.target = None; -+ route.buffered = 0; -+ } -+ -+ unsafe { -+ libc::close(self.read_fd); -+ libc::close(self.write_fd); -+ } -+ UNIX_HOST_SIGNAL_PENDING.store(0, Ordering::Release); -+ UNIX_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn unix_host_signal_action() -> Result { -+ let mut action = unsafe { std::mem::zeroed::() }; -+ action.sa_sigaction = unix_host_lifecycle_signal_handler as *const () as usize; -+ action.sa_flags = libc::SA_RESTART; -+ if unsafe { libc::sigemptyset(&mut action.sa_mask) } != 0 { -+ return Err(UnixHostLifecycleError::Io { -+ operation: "initialize host signal mask", -+ source: io::Error::last_os_error(), -+ }); -+ } -+ for (signal, _, _, _) in UNIX_HOST_SIGNALS { -+ if unsafe { libc::sigaddset(&mut action.sa_mask, signal) } != 0 { -+ return Err(UnixHostLifecycleError::Sigaction { -+ operation: "block nested host lifecycle signal", -+ signal, -+ source: io::Error::last_os_error(), -+ }); -+ } -+ } -+ Ok(action) -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+extern "C" fn unix_host_lifecycle_signal_handler(signal: libc::c_int) { -+ UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_add(1, Ordering::SeqCst); -+ let errno = unsafe { unix_errno_location() }; -+ let saved_errno = if errno.is_null() { -+ 0 -+ } else { -+ unsafe { errno.read() } -+ }; -+ -+ if let Some(bit) = unix_host_signal_bit(signal) { -+ let write_fd = UNIX_HOST_SIGNAL_WRITE_FD.load(Ordering::SeqCst); -+ if write_fd >= 0 { -+ UNIX_HOST_SIGNAL_PENDING.fetch_or(bit, Ordering::Release); -+ let wake = [1u8]; -+ loop { -+ let written = unsafe { -+ libc::write(write_fd, wake.as_ptr().cast::(), wake.len()) -+ }; -+ if written >= 0 { -+ break; -+ } -+ let current_errno = if errno.is_null() { -+ 0 -+ } else { -+ unsafe { errno.read() } -+ }; -+ if current_errno != libc::EINTR { -+ break; -+ } -+ } -+ } -+ } -+ -+ if !errno.is_null() { -+ unsafe { errno.write(saved_errno) }; -+ } -+ UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_sub(1, Ordering::SeqCst); -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn unix_host_signal_bit(signal: libc::c_int) -> Option { -+ UNIX_HOST_SIGNALS -+ .iter() -+ .find_map(|(candidate, bit, _, _)| (*candidate == signal).then_some(*bit)) -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn deliver_unix_signal_mask( -+ mask: u32, -+ controller: &TaskSignalController, -+) -> Result<(), (Signal, TaskSignalError)> { -+ for (_, bit, signal, _) in UNIX_HOST_SIGNALS { -+ if mask & bit != 0 { -+ controller -+ .send_signal(signal) -+ .map_err(|source| (signal, source))?; -+ } -+ } -+ Ok(()) -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn route_unix_signal_mask(mask: u32, route: &Arc>) { -+ if mask == 0 { -+ return; -+ } -+ let target = { -+ let mut route = lock_unix_signal_route(route); -+ match &route.target { -+ Some(target) => Some(target.clone()), -+ None => { -+ route.buffered |= mask; -+ None -+ } -+ } -+ }; -+ if let Some(target) = target -+ && let Err((signal, error)) = deliver_unix_signal_mask(mask, &target) -+ { -+ tracing::warn!(?signal, %error, "failed to forward host lifecycle signal"); -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn unix_host_signal_forwarder(read_fd: RawFd, route: Arc>) { -+ let mut wakeups = [0u8; 64]; -+ loop { -+ let read = unsafe { -+ libc::read( -+ read_fd, -+ wakeups.as_mut_ptr().cast::(), -+ wakeups.len(), -+ ) -+ }; -+ if read < 0 { -+ let error = io::Error::last_os_error(); -+ if error.raw_os_error() == Some(libc::EINTR) { -+ continue; -+ } -+ tracing::error!(%error, "host signal self-pipe read failed"); -+ return; -+ } -+ if read == 0 { -+ return; -+ } -+ route_unix_signal_mask(UNIX_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), &route); -+ if lock_unix_signal_route(&route).stopping { -+ // The writer is quiescent before `stopping` is set. One final swap -+ // closes the small interval between the previous swap and the -+ // stopping observation without competing with another consumer. -+ route_unix_signal_mask(UNIX_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), &route); -+ return; -+ } -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn lock_unix_signal_route( -+ route: &Mutex, -+) -> std::sync::MutexGuard<'_, UnixHostSignalRoute> { -+ route -+ .lock() -+ .unwrap_or_else(|poisoned| poisoned.into_inner()) -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn wake_unix_signal_forwarder(write_fd: RawFd) { -+ let wake = [1u8]; -+ unsafe { -+ libc::write(write_fd, wake.as_ptr().cast::(), wake.len()); -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn create_unix_signal_pipe() -> io::Result<(RawFd, RawFd)> { -+ let mut fds = [-1; 2]; -+ #[cfg(any(target_os = "linux", target_os = "android"))] -+ let result = unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }; -+ #[cfg(not(any(target_os = "linux", target_os = "android")))] -+ let result = unsafe { libc::pipe(fds.as_mut_ptr()) }; -+ if result != 0 { -+ return Err(io::Error::last_os_error()); -+ } -+ -+ let configured = configure_unix_signal_pipe(fds); -+ if let Err(error) = configured { -+ unsafe { -+ libc::close(fds[0]); -+ libc::close(fds[1]); -+ } -+ return Err(error); -+ } -+ Ok((fds[0], fds[1])) -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn configure_unix_signal_pipe(fds: [RawFd; 2]) -> io::Result<()> { -+ #[cfg(not(any(target_os = "linux", target_os = "android")))] -+ { -+ set_unix_fd_flag(fds[0], libc::F_GETFD, libc::F_SETFD, libc::FD_CLOEXEC)?; -+ set_unix_fd_flag(fds[1], libc::F_GETFD, libc::F_SETFD, libc::FD_CLOEXEC)?; -+ } -+ set_unix_fd_flag(fds[1], libc::F_GETFL, libc::F_SETFL, libc::O_NONBLOCK) -+} -+ -+#[cfg(all(feature = "ctrlc", unix))] -+fn set_unix_fd_flag( -+ fd: RawFd, -+ get: libc::c_int, -+ set: libc::c_int, -+ flag: libc::c_int, -+) -> io::Result<()> { -+ let current = unsafe { libc::fcntl(fd, get) }; -+ if current < 0 { -+ return Err(io::Error::last_os_error()); -+ } -+ if unsafe { libc::fcntl(fd, set, current | flag) } < 0 { -+ return Err(io::Error::last_os_error()); -+ } -+ Ok(()) -+} -+ -+#[cfg(all( -+ feature = "ctrlc", -+ unix, -+ any(target_os = "linux", target_os = "dragonfly") -+))] -+unsafe fn unix_errno_location() -> *mut libc::c_int { -+ unsafe { libc::__errno_location() } -+} -+ -+#[cfg(all(feature = "ctrlc", unix, target_os = "android"))] -+unsafe fn unix_errno_location() -> *mut libc::c_int { -+ unsafe { libc::__errno() } -+} -+ -+#[cfg(all( -+ feature = "ctrlc", -+ unix, -+ any( -+ target_os = "macos", -+ target_os = "ios", -+ target_os = "tvos", -+ target_os = "watchos", -+ target_os = "visionos", -+ target_os = "freebsd", -+ ) -+))] -+unsafe fn unix_errno_location() -> *mut libc::c_int { -+ unsafe { libc::__error() } -+} -+ -+#[cfg(all( -+ feature = "ctrlc", -+ unix, -+ any(target_os = "netbsd", target_os = "openbsd") -+))] -+unsafe fn unix_errno_location() -> *mut libc::c_int { -+ unsafe { libc::__errno() } -+} -+ -+#[cfg(all( -+ feature = "ctrlc", -+ unix, -+ any(target_os = "solaris", target_os = "illumos") -+))] -+unsafe fn unix_errno_location() -> *mut libc::c_int { -+ unsafe { libc::___errno() } -+} -+ -+#[cfg(all( -+ feature = "ctrlc", -+ unix, -+ not(any( -+ target_os = "linux", -+ target_os = "android", -+ target_os = "macos", -+ target_os = "ios", -+ target_os = "tvos", -+ target_os = "watchos", -+ target_os = "visionos", -+ target_os = "freebsd", -+ target_os = "dragonfly", -+ target_os = "netbsd", -+ target_os = "openbsd", -+ target_os = "solaris", -+ target_os = "illumos", -+ )) -+))] -+unsafe fn unix_errno_location() -> *mut libc::c_int { -+ std::ptr::null_mut() -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+const WINDOWS_GUEST_SIGNALS: [(u32, Signal); 4] = [ -+ (1 << 0, Signal::Sigint), -+ (1 << 1, Signal::Sigterm), -+ (1 << 2, Signal::Sigquit), -+ (1 << 3, Signal::Sighup), -+]; -+ -+#[cfg(all(feature = "ctrlc", windows))] -+static WINDOWS_HOST_SIGNAL_OWNER_ACTIVE: AtomicBool = AtomicBool::new(false); -+#[cfg(all(feature = "ctrlc", windows))] -+static WINDOWS_HOST_SIGNAL_EVENT: AtomicPtr = AtomicPtr::new(std::ptr::null_mut()); -+#[cfg(all(feature = "ctrlc", windows))] -+static WINDOWS_HOST_SIGNAL_PENDING: AtomicU32 = AtomicU32::new(0); -+#[cfg(all(feature = "ctrlc", windows))] -+static WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT: AtomicUsize = AtomicUsize::new(0); -+ -+/// Failure to install or bind the explicit Windows console lifecycle owner. -+#[cfg(all(feature = "ctrlc", windows))] -+#[derive(thiserror::Error, Debug)] -+pub enum WindowsHostLifecycleError { -+ #[error("another Windows host lifecycle supervisor already owns this process")] -+ AlreadyOwned, -+ #[error("this Windows host lifecycle supervisor is already bound to a task")] -+ AlreadyBound, -+ #[error("{operation} failed: {source}")] -+ Io { -+ operation: &'static str, -+ #[source] -+ source: io::Error, -+ }, -+ #[error("buffered {signal:?} could not be delivered when the task was bound: {source}")] -+ BufferedSignalDelivery { -+ signal: Signal, -+ #[source] -+ source: TaskSignalError, -+ }, -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+#[derive(Default)] -+struct WindowsHostSignalRoute { -+ target: Option, -+ buffered: u32, -+ bound: bool, -+ stopping: bool, -+} -+ -+/// Exclusive process-scoped Windows console lifecycle supervision. -+/// -+/// This adapter is installed only by an explicit process owner. It preserves -+/// the previous Windows handler stack by adding and later removing exactly its -+/// own callback. Console Ctrl+C maps to WASI `SIGINT`, Ctrl+Break to `SIGQUIT`, -+/// close/shutdown to `SIGTERM`, and logoff to `SIGHUP`. -+#[cfg(all(feature = "ctrlc", windows))] -+pub struct WindowsHostLifecycleSupervisor { -+ resources: WindowsHostSignalResources, -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+pub type HostLifecycleSupervisor = WindowsHostLifecycleSupervisor; -+ -+#[cfg(all(feature = "ctrlc", windows))] -+impl std::fmt::Debug for WindowsHostLifecycleSupervisor { -+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { -+ f.debug_struct("WindowsHostLifecycleSupervisor") -+ .field("event", &self.resources.event) -+ .finish_non_exhaustive() -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+struct WindowsHostSignalResources { -+ event: usize, -+ handler_installed: bool, -+ route: Arc>, -+ forwarder: Option>, -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+impl WindowsHostLifecycleSupervisor { -+ /// Install one exclusive console-control owner and its dedicated forwarder. -+ pub fn install() -> Result { -+ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE -+ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) -+ .map_err(|_| WindowsHostLifecycleError::AlreadyOwned)?; -+ -+ let event = unsafe { CreateEventW(std::ptr::null(), 0, 0, std::ptr::null()) }; -+ if event.is_null() { -+ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); -+ return Err(WindowsHostLifecycleError::Io { -+ operation: "create host lifecycle event", -+ source: io::Error::last_os_error(), -+ }); -+ } -+ -+ let route = Arc::new(Mutex::new(WindowsHostSignalRoute::default())); -+ let thread_route = route.clone(); -+ let event_address = event as usize; -+ let forwarder = match std::thread::Builder::new() -+ .name("wasmer-host-signals".to_string()) -+ .spawn(move || windows_host_signal_forwarder(event_address as HANDLE, thread_route)) -+ { -+ Ok(forwarder) => forwarder, -+ Err(source) => { -+ unsafe { CloseHandle(event) }; -+ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); -+ return Err(WindowsHostLifecycleError::Io { -+ operation: "spawn host signal forwarder", -+ source, -+ }); -+ } -+ }; -+ -+ let mut resources = WindowsHostSignalResources { -+ event: event_address, -+ handler_installed: false, -+ route, -+ forwarder: Some(forwarder), -+ }; -+ WINDOWS_HOST_SIGNAL_PENDING.store(0, Ordering::Release); -+ WINDOWS_HOST_SIGNAL_EVENT.store(event, Ordering::SeqCst); -+ -+ if unsafe { SetConsoleCtrlHandler(Some(windows_host_lifecycle_handler), 1) } == 0 { -+ return Err(WindowsHostLifecycleError::Io { -+ operation: "install Windows console-control handler", -+ source: io::Error::last_os_error(), -+ }); -+ } -+ resources.handler_installed = true; -+ Ok(Self { resources }) -+ } -+ -+ /// Bind the sole root guest task. Console events received earlier remain -+ /// buffered until the root is available. -+ pub fn bind_task(&self, task: &TaskJoinHandle) -> Result<(), WindowsHostLifecycleError> { -+ let controller = task.signal_controller(); -+ let buffered = { -+ let mut route = lock_windows_signal_route(&self.resources.route); -+ if route.bound { -+ return Err(WindowsHostLifecycleError::AlreadyBound); -+ } -+ route.bound = true; -+ route.target = Some(controller.clone()); -+ std::mem::take(&mut route.buffered) -+ }; -+ deliver_windows_signal_mask(buffered, &controller).map_err(|(signal, source)| { -+ WindowsHostLifecycleError::BufferedSignalDelivery { signal, source } -+ }) -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+impl Drop for WindowsHostSignalResources { -+ fn drop(&mut self) { -+ if self.handler_installed -+ && unsafe { SetConsoleCtrlHandler(Some(windows_host_lifecycle_handler), 0) } == 0 -+ { -+ // As on Unix, retaining the callback target and exclusive owner is -+ // safer than allowing a stale callback to reach a recycled handle. -+ tracing::error!( -+ error = %io::Error::last_os_error(), -+ "failed to remove Windows console-control handler; retaining supervisor for process lifetime" -+ ); -+ return; -+ } -+ -+ WINDOWS_HOST_SIGNAL_EVENT.store(std::ptr::null_mut(), Ordering::SeqCst); -+ while WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT.load(Ordering::SeqCst) != 0 { -+ std::thread::yield_now(); -+ } -+ -+ { -+ let mut route = lock_windows_signal_route(&self.route); -+ route.stopping = true; -+ } -+ wake_windows_signal_forwarder(self.event as HANDLE); -+ if let Some(forwarder) = self.forwarder.take() -+ && forwarder.join().is_err() -+ { -+ tracing::error!("Windows host signal forwarder panicked during shutdown"); -+ } -+ route_windows_signal_mask( -+ WINDOWS_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), -+ &self.route, -+ ); -+ { -+ let mut route = lock_windows_signal_route(&self.route); -+ route.target = None; -+ route.buffered = 0; -+ } -+ -+ unsafe { CloseHandle(self.event as HANDLE) }; -+ WINDOWS_HOST_SIGNAL_PENDING.store(0, Ordering::Release); -+ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+unsafe extern "system" fn windows_host_lifecycle_handler(control: u32) -> BOOL { -+ WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_add(1, Ordering::SeqCst); -+ let handled = if let Some(bit) = windows_host_signal_bit(control) { -+ let event = WINDOWS_HOST_SIGNAL_EVENT.load(Ordering::SeqCst); -+ if event.is_null() { -+ false -+ } else { -+ WINDOWS_HOST_SIGNAL_PENDING.fetch_or(bit, Ordering::Release); -+ // Only claim the console event when its forwarder was actually -+ // woken. The pending bit intentionally remains set on failure so -+ // a later successful wake can still drain it. -+ (unsafe { SetEvent(event) }) != 0 -+ } -+ } else { -+ false -+ }; -+ WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_sub(1, Ordering::SeqCst); -+ if handled { 1 } else { 0 } -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+fn windows_host_signal_bit(control: u32) -> Option { -+ match control { -+ CTRL_C_EVENT => Some(1 << 0), -+ CTRL_CLOSE_EVENT | CTRL_SHUTDOWN_EVENT => Some(1 << 1), -+ CTRL_BREAK_EVENT => Some(1 << 2), -+ CTRL_LOGOFF_EVENT => Some(1 << 3), -+ _ => None, -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+fn deliver_windows_signal_mask( -+ mask: u32, -+ controller: &TaskSignalController, -+) -> Result<(), (Signal, TaskSignalError)> { -+ for (bit, signal) in WINDOWS_GUEST_SIGNALS { -+ if mask & bit != 0 { -+ controller -+ .send_signal(signal) -+ .map_err(|source| (signal, source))?; -+ } -+ } -+ Ok(()) -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+fn route_windows_signal_mask(mask: u32, route: &Arc>) { -+ if mask == 0 { -+ return; -+ } -+ let target = { -+ let mut route = lock_windows_signal_route(route); -+ match &route.target { -+ Some(target) => Some(target.clone()), -+ None => { -+ route.buffered |= mask; -+ None -+ } -+ } -+ }; -+ if let Some(target) = target -+ && let Err((signal, error)) = deliver_windows_signal_mask(mask, &target) -+ { -+ tracing::warn!(?signal, %error, "failed to forward Windows host lifecycle event"); -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+fn windows_host_signal_forwarder(event: HANDLE, route: Arc>) { -+ loop { -+ let wait = unsafe { WaitForSingleObject(event, INFINITE) }; -+ if wait != WAIT_OBJECT_0 { -+ tracing::error!( -+ error = %io::Error::last_os_error(), -+ wait, -+ "Windows host lifecycle event wait failed" -+ ); -+ return; -+ } -+ route_windows_signal_mask( -+ WINDOWS_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), -+ &route, -+ ); -+ if lock_windows_signal_route(&route).stopping { -+ route_windows_signal_mask( -+ WINDOWS_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), -+ &route, -+ ); -+ return; -+ } -+ } -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+fn wake_windows_signal_forwarder(event: HANDLE) { -+ unsafe { SetEvent(event) }; -+} -+ -+#[cfg(all(feature = "ctrlc", windows))] -+fn lock_windows_signal_route( -+ route: &Mutex, -+) -> std::sync::MutexGuard<'_, WindowsHostSignalRoute> { -+ route -+ .lock() -+ .unwrap_or_else(|poisoned| poisoned.into_inner()) -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use crate::os::task::signal::SignalHandlerAbi; -+ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] -+ use std::time::Instant; -+ use std::{sync::mpsc, time::Duration}; -+ -+ #[derive(Debug)] -+ struct RecordingSignalHandler { -+ sender: mpsc::Sender, -+ } -+ -+ impl SignalHandlerAbi for RecordingSignalHandler { -+ fn signal(&self, signal: u8) -> Result<(), SignalDeliveryError> { -+ self.sender.send(signal).map_err(|_| SignalDeliveryError) -+ } -+ } -+ -+ fn status_and_signals() -> (OwnedTaskStatus, mpsc::Receiver, TaskJoinHandle) { -+ let (sender, receiver) = mpsc::channel(); -+ let mut status = OwnedTaskStatus::default(); -+ status.set_signal_handler(Arc::new(RecordingSignalHandler { sender })); -+ let handle = status.handle(); -+ (status, receiver, handle) -+ } -+ -+ #[test] -+ fn direct_signal_controller_is_platform_neutral_and_rejects_finished_tasks() { -+ let (status, recorded, handle) = status_and_signals(); -+ let controller = handle.signal_controller(); -+ for signal in [ -+ Signal::Sigint, -+ Signal::Sigterm, -+ Signal::Sigquit, -+ Signal::Sighup, -+ ] { -+ controller.send_signal(signal).unwrap(); -+ assert_eq!( -+ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), -+ signal.to_native() as u8 -+ ); -+ } -+ status.set_finished(Ok(Errno::Success.into())); -+ assert!(matches!( -+ controller.send_signal(Signal::Sigterm), -+ Err(TaskSignalError::Terminated) -+ )); -+ } -+ -+ #[cfg(all(feature = "ctrlc", windows))] -+ #[test] -+ fn windows_console_events_have_explicit_wasi_mappings() { -+ assert_eq!(windows_host_signal_bit(CTRL_C_EVENT), Some(1 << 0)); -+ assert_eq!(windows_host_signal_bit(CTRL_CLOSE_EVENT), Some(1 << 1)); -+ assert_eq!(windows_host_signal_bit(CTRL_SHUTDOWN_EVENT), Some(1 << 1)); -+ assert_eq!(windows_host_signal_bit(CTRL_BREAK_EVENT), Some(1 << 2)); -+ assert_eq!(windows_host_signal_bit(CTRL_LOGOFF_EVENT), Some(1 << 3)); -+ assert_eq!(windows_host_signal_bit(u32::MAX), None); -+ } -+ -+ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] -+ static PREVIOUS_HUP_COUNT: AtomicUsize = AtomicUsize::new(0); -+ -+ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] -+ extern "C" fn previous_hup_handler(_: libc::c_int) { -+ PREVIOUS_HUP_COUNT.fetch_add(1, Ordering::SeqCst); -+ } -+ -+ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] -+ fn wait_until(description: &str, predicate: impl Fn() -> bool) { -+ let deadline = Instant::now() + Duration::from_secs(2); -+ while !predicate() { -+ assert!( -+ Instant::now() < deadline, -+ "timed out waiting for {description}" -+ ); -+ std::thread::sleep(Duration::from_millis(1)); -+ } -+ } -+ -+ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] -+ fn run_unix_supervisor_child() { -+ PREVIOUS_HUP_COUNT.store(0, Ordering::SeqCst); -+ let mut previous_hup = MaybeUninit::::uninit(); -+ let mut custom_hup = unsafe { std::mem::zeroed::() }; -+ custom_hup.sa_sigaction = previous_hup_handler as *const () as usize; -+ custom_hup.sa_flags = libc::SA_RESTART; -+ assert_eq!(unsafe { libc::sigemptyset(&mut custom_hup.sa_mask) }, 0); -+ assert_eq!( -+ unsafe { libc::sigaction(libc::SIGHUP, &custom_hup, previous_hup.as_mut_ptr()) }, -+ 0 -+ ); -+ let previous_hup = unsafe { previous_hup.assume_init() }; -+ let mut expected_custom_hup = MaybeUninit::::uninit(); -+ assert_eq!( -+ unsafe { -+ libc::sigaction( -+ libc::SIGHUP, -+ std::ptr::null(), -+ expected_custom_hup.as_mut_ptr(), -+ ) -+ }, -+ 0 -+ ); -+ let expected_custom_hup = unsafe { expected_custom_hup.assume_init() }; -+ -+ // Force a failure after one successful sigaction and prove the setup -+ // guard restores the first disposition and releases exclusivity. -+ assert!(matches!( -+ UnixHostLifecycleSupervisor::install_signals(&[libc::SIGHUP, -1]), -+ Err(UnixHostLifecycleError::Sigaction { signal: -1, .. }) -+ )); -+ assert!(!UNIX_HOST_SIGNAL_OWNER_ACTIVE.load(Ordering::Acquire)); -+ assert_eq!(unsafe { libc::kill(libc::getpid(), libc::SIGHUP) }, 0); -+ wait_until("partial-install rollback handler", || { -+ PREVIOUS_HUP_COUNT.load(Ordering::SeqCst) == 1 -+ }); -+ -+ let supervisor = UnixHostLifecycleSupervisor::install().unwrap(); -+ assert!(matches!( -+ UnixHostLifecycleSupervisor::install(), -+ Err(UnixHostLifecycleError::AlreadyOwned) -+ )); -+ let (status, recorded, handle) = status_and_signals(); -+ -+ // Capture SIGTERM before the root task exists, and require it to be in -+ // the supervisor's pre-bind buffer before binding. -+ assert_eq!(unsafe { libc::kill(libc::getpid(), libc::SIGTERM) }, 0); -+ wait_until("pre-bind SIGTERM buffering", || { -+ lock_unix_signal_route(&supervisor.resources.route).buffered & (1 << 1) != 0 -+ }); -+ supervisor.bind_task(&handle).unwrap(); -+ assert!(matches!( -+ supervisor.bind_task(&handle), -+ Err(UnixHostLifecycleError::AlreadyBound) -+ )); -+ assert_eq!( -+ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), -+ Signal::Sigterm.to_native() as u8 -+ ); -+ -+ // The host callback must not perturb errno in interrupted code. -+ unsafe { unix_errno_location().write(libc::EDOM) }; -+ unix_host_lifecycle_signal_handler(libc::SIGINT); -+ assert_eq!(unsafe { unix_errno_location().read() }, libc::EDOM); -+ assert_eq!( -+ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), -+ Signal::Sigint.to_native() as u8 -+ ); -+ -+ for (host, guest) in [ -+ (libc::SIGQUIT, Signal::Sigquit), -+ (libc::SIGHUP, Signal::Sighup), -+ ] { -+ assert_eq!(unsafe { libc::kill(libc::getpid(), host) }, 0); -+ assert_eq!( -+ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), -+ guest.to_native() as u8 -+ ); -+ } -+ assert_eq!(PREVIOUS_HUP_COUNT.load(Ordering::SeqCst), 1); -+ -+ drop(supervisor); -+ assert!(!UNIX_HOST_SIGNAL_OWNER_ACTIVE.load(Ordering::Acquire)); -+ assert_eq!(unsafe { libc::kill(libc::getpid(), libc::SIGHUP) }, 0); -+ wait_until("restored previous SIGHUP disposition", || { -+ PREVIOUS_HUP_COUNT.load(Ordering::SeqCst) == 2 -+ }); -+ -+ // Repeated ownership cycles must not leak the self-pipe descriptors or -+ // leave a forwarder behind. `/proc/self/fd` counts its own directory -+ // handle in both measurements, so equality remains deterministic. -+ let descriptors_before = std::fs::read_dir("/proc/self/fd").unwrap().count(); -+ for _ in 0..16 { -+ drop(UnixHostLifecycleSupervisor::install().unwrap()); -+ } -+ let descriptors_after = std::fs::read_dir("/proc/self/fd").unwrap().count(); -+ assert_eq!(descriptors_after, descriptors_before); -+ -+ let mut restored_custom_hup = MaybeUninit::::uninit(); -+ assert_eq!( -+ unsafe { -+ libc::sigaction( -+ libc::SIGHUP, -+ std::ptr::null(), -+ restored_custom_hup.as_mut_ptr(), -+ ) -+ }, -+ 0 -+ ); -+ let restored_custom_hup = unsafe { restored_custom_hup.assume_init() }; -+ assert_eq!( -+ restored_custom_hup.sa_sigaction, -+ expected_custom_hup.sa_sigaction -+ ); -+ assert_eq!(restored_custom_hup.sa_flags, expected_custom_hup.sa_flags); -+ for signal in [libc::SIGINT, libc::SIGTERM, libc::SIGQUIT, libc::SIGHUP] { -+ assert_eq!( -+ unsafe { libc::sigismember(&restored_custom_hup.sa_mask, signal) }, -+ unsafe { libc::sigismember(&expected_custom_hup.sa_mask, signal) } -+ ); -+ } -+ -+ assert_eq!( -+ unsafe { libc::sigaction(libc::SIGHUP, &previous_hup, std::ptr::null_mut()) }, -+ 0 -+ ); -+ drop(status); -+ } -+ -+ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] -+ #[test] -+ fn unix_supervisor_real_signals_restore_and_route_exclusively() { -+ const CHILD_ENV: &str = "WASMER_SIGNAL_SUPERVISOR_TEST_CHILD"; -+ if std::env::var_os(CHILD_ENV).is_some() { -+ run_unix_supervisor_child(); -+ return; -+ } -+ -+ let output = std::process::Command::new(std::env::current_exe().unwrap()) -+ .arg("os::task::task_join_handle::tests::unix_supervisor_real_signals_restore_and_route_exclusively") -+ .arg("--exact") -+ .arg("--nocapture") -+ .arg("--test-threads=1") -+ .env(CHILD_ENV, "1") -+ .output() -+ .unwrap(); -+ assert!( -+ output.status.success(), -+ "signal supervisor child failed with {}\nstdout:\n{}\nstderr:\n{}", -+ output.status, -+ String::from_utf8_lossy(&output.stdout), -+ String::from_utf8_lossy(&output.stderr), -+ ); -+ } -+} -diff --git a/lib/wasix/src/os/task/thread.rs b/lib/wasix/src/os/task/thread.rs -index cb7df5f..d079b08 100644 ---- a/lib/wasix/src/os/task/thread.rs -+++ b/lib/wasix/src/os/task/thread.rs -@@ -1,5 +1,5 @@ - use super::{ -- control_plane::TaskCountGuard, -+ control_plane::{ControlPlaneError, TaskCountGuard}, - task_join_handle::{OwnedTaskStatus, TaskJoinHandle}, - }; - use crate::{ -@@ -247,7 +247,7 @@ struct WasiThreadState { - - // Registers the task termination with the ControlPlane on drop. - // Never accessed, since it's a drop guard. -- _task_count_guard: TaskCountGuard, -+ task_count_guard: Mutex>, - } - - static NO_MORE_BYTES: [u8; 0] = [0u8; 0]; -@@ -273,7 +273,7 @@ impl WasiThread { - #[cfg(feature = "journal")] - check_pointing: AtomicBool::new(false), - deep_sleeping: AtomicBool::new(false), -- _task_count_guard: guard, -+ task_count_guard: Mutex::new(Some(guard)), - }), - layout, - start, -@@ -291,6 +291,17 @@ impl WasiThread { - self.state.id - } - -+ pub(crate) fn same_identity(&self, other: &Self) -> bool { -+ Arc::ptr_eq(&self.state, &other.state) -+ } -+ -+ /// Releases this thread's control-plane task reservation exactly once. -+ /// Reusable environments use this when retiring an old execution epoch -+ /// whose thread value may still be referenced by an instance handle. -+ pub(crate) fn retire_task_registration(&self) -> bool { -+ self.state.task_count_guard.lock().unwrap().take().is_some() -+ } -+ - /// Returns true if this thread is the main thread - pub fn is_main(&self) -> bool { - self.state.is_main -@@ -594,13 +605,28 @@ impl WasiThreadHandle { - - impl Drop for WasiThreadHandleProtected { - fn drop(&mut self) { -+ // The handle, not observational WasiThread clones, owns execution -+ // admission. Releasing the last handle must return the task slot even -+ // when an environment snapshot still references the thread state. -+ self.thread.retire_task_registration(); - let id = self.thread.tid(); - if let Some(inner) = Weak::upgrade(&self.inner) { - let mut inner = inner.0.lock().unwrap(); -- if let Some(ctrl) = inner.threads.remove(&id) { -+ let owns_slot = inner -+ .threads -+ .get(&id) -+ .is_some_and(|current| current.same_identity(&self.thread)); -+ if owns_slot { -+ let ctrl = inner -+ .threads -+ .remove(&id) -+ .expect("thread identity was checked under the same lock"); - ctrl.set_status_finished(Ok(Errno::Success.into())); -+ inner.thread_count = inner -+ .thread_count -+ .checked_sub(1) -+ .expect("process thread count underflow"); - } -- inner.thread_count -= 1; - } - } - } -@@ -635,6 +661,10 @@ pub enum WasiThreadError { - /// This will happen if WASM is running in a thread has not been created by the spawn_wasm call - #[error("WASM context is invalid")] - InvalidWasmContext, -+ #[error("Process lifecycle rejected the task - {0}")] -+ ProcessLifecycle(ControlPlaneError), -+ #[error("Failed to capture WASIX store snapshot - {0}")] -+ StoreSnapshotCaptureFailed(crate::StoreSnapshotCaptureError), - } - - impl From for Errno { -@@ -649,6 +679,8 @@ impl From for Errno { - WasiThreadError::InstanceCreateFailed(_) => Errno::Noexec, - WasiThreadError::InitFailed(_) => Errno::Noexec, - WasiThreadError::InvalidWasmContext => Errno::Noexec, -+ WasiThreadError::ProcessLifecycle(_) => Errno::Perm, -+ WasiThreadError::StoreSnapshotCaptureFailed(_) => Errno::Noexec, - } - } - } -diff --git a/lib/wasix/src/runners/wasi.rs b/lib/wasix/src/runners/wasi.rs -index c416457..f17df68 100644 ---- a/lib/wasix/src/runners/wasi.rs -+++ b/lib/wasix/src/runners/wasi.rs -@@ -1,6 +1,6 @@ - //! WebC container support for running WASI modules - --use std::{path::PathBuf, sync::Arc}; -+use std::{collections::HashMap, path::PathBuf, sync::Arc}; - - use anyhow::{Context, Error}; - use tracing::Instrument; -@@ -9,25 +9,134 @@ use wasmer::{Engine, Module}; - use wasmer_types::ModuleHash; - use webc::metadata::{Command, annotations::Wasi}; - -+#[cfg(feature = "journal")] -+use crate::journal::{DynJournal, DynReadableJournal, SnapshotTrigger}; -+#[cfg(all(feature = "ctrlc", any(unix, windows)))] -+use crate::os::task::HostLifecycleSupervisor; - use crate::{ - Runtime, WasiEnvBuilder, WasiError, WasiRuntimeError, - bin_factory::BinaryPackage, - capabilities::Capabilities, -- journal::{DynJournal, DynReadableJournal, SnapshotTrigger}, -+ os::task::{TaskJoinHandle, terminate_and_reap_abandoned_task}, - runners::{MappedDirectory, MountedDirectory, wasi_common::CommonWasiOptions}, - runtime::task_manager::VirtualTaskManagerExt, -+ state::{PreinitializedMemoryImageHandle, PreinitializedMemoryImageMode}, - }; -+use wasmer_wasix_types::wasi::{ExitCode, Signal}; - - use super::wasi_common::{ - ExistingMountConflictBehavior, MAPPED_CURRENT_DIR_DEFAULT_PATH, MappedCommand, - }; - -+/// Owns a spawned root until its authoritative completion has been observed. -+/// -+/// The admitted watcher future can still be canceled or panic after guest -+/// spawn. A plain `TaskJoinHandle` drop does not terminate anything, so this -+/// guard transfers every abnormal path to an independent, non-Tokio reaper. -+struct RootTaskLifecycleGuard { -+ task: Option, -+} -+ -+impl RootTaskLifecycleGuard { -+ fn new(task: TaskJoinHandle) -> Self { -+ Self { task: Some(task) } -+ } -+ -+ fn task(&self) -> &TaskJoinHandle { -+ self.task -+ .as_ref() -+ .expect("armed root lifecycle guard has no task") -+ } -+ -+ async fn wait_finished(mut self) -> Result> { -+ let result = self -+ .task -+ .as_mut() -+ .expect("armed root lifecycle guard has no task") -+ .wait_finished() -+ .await; -+ // Authoritative terminal status was observed; disarm the hot path so -+ // normal completion never creates a reaper thread. -+ self.task.take(); -+ result -+ } -+ -+ async fn terminate_and_reap(mut self, bind_error: Error) -> Error { -+ let result = terminate_root_after_lifecycle_bind_failure( -+ self.task -+ .as_mut() -+ .expect("armed root lifecycle guard has no task"), -+ bind_error, -+ ) -+ .await; -+ self.task.take(); -+ result -+ } -+} -+ -+impl Drop for RootTaskLifecycleGuard { -+ fn drop(&mut self) { -+ if let Some(task) = self.task.take() { -+ terminate_and_reap_abandoned_task(task, "wasmer-root-reaper", "root task", None, None); -+ } -+ } -+} -+ -+async fn terminate_root_after_lifecycle_bind_failure( -+ task: &mut TaskJoinHandle, -+ bind_error: Error, -+) -> Error { -+ let signal_error = task.send_signal(Signal::Sigkill).err(); -+ let completion = task.wait_finished().await; -+ let cleanup = match (signal_error, completion) { -+ (None, Ok(code)) => format!( -+ "root task terminated and reaped with exit code {}", -+ code.raw() -+ ), -+ (None, Err(error)) => format!("root task terminated and reaped with error: {error}"), -+ (Some(signal_error), Ok(code)) => format!( -+ "root task finished and was reaped with exit code {} after SIGKILL delivery failed: {signal_error}", -+ code.raw() -+ ), -+ (Some(signal_error), Err(error)) => format!( -+ "root task finished and was reaped with error {error} after SIGKILL delivery failed: {signal_error}" -+ ), -+ }; -+ bind_error.context(cleanup) -+} -+ -+/// Spawn, optionally bind, and join one root inside a single admitted future. -+/// -+/// Keeping `spawn_root` inside this async function is deliberate: constructing -+/// the future has no guest-side effects. A task manager that rejects (or -+/// panics during) `spawn_and_block_on` admission drops this future without -+/// creating a root that has no waiter. -+async fn spawn_and_wait_for_root( -+ spawn_root: Spawn, -+ bind_root: Bind, -+) -> Result>, Error> -+where -+ Spawn: FnOnce() -> Result, -+ Bind: FnOnce(&TaskJoinHandle) -> Result<(), Error>, -+{ -+ let task_handle = spawn_root()?; -+ let root = RootTaskLifecycleGuard::new(task_handle); -+ if let Err(bind_error) = bind_root(root.task()) { -+ return Err(root.terminate_and_reap(bind_error).await); -+ } -+ Ok(root.wait_finished().await) -+} -+ - #[derive(Debug, Default, Clone)] - pub struct WasiRunner { - wasi: CommonWasiOptions, - stdin: Option, - stdout: Option, - stderr: Option, -+ sealed_modules: HashMap)>, -+ preinitialized_memory_image: Option, -+ #[cfg(all(feature = "ctrlc", any(unix, windows)))] -+ host_lifecycle_supervisor: Option>, - } - - pub enum PackageOrHash<'a> { -@@ -184,6 +293,48 @@ impl WasiRunner { - self - } - -+ /// Registers an immutable guest executable for every process spawned by -+ /// this runner. A non-empty registry closes executable lookup: paths not -+ /// present in it are denied before consulting the guest filesystem. -+ pub fn with_sealed_module( -+ &mut self, -+ guest_path: impl Into, -+ module_hash: ModuleHash, -+ preinitialized_memory_image: Option, -+ ) -> &mut Self { -+ self.sealed_modules.insert( -+ guest_path.into(), -+ (module_hash, preinitialized_memory_image), -+ ); -+ self -+ } -+ -+ /// Applies or captures a preinitialized memory image for only the initial -+ /// module executed by this runner. Exec aliases carry their own sealed -+ /// image through the immutable executable registry. -+ pub fn with_preinitialized_memory_image( -+ &mut self, -+ image: PreinitializedMemoryImageMode, -+ ) -> &mut Self { -+ self.preinitialized_memory_image = Some(image); -+ self -+ } -+ -+ /// Routes this runner's single root task through an already-installed, -+ /// process-scoped host lifecycle supervisor. -+ /// -+ /// `WasiRunner` never installs host signal dispositions itself. Embedders -+ /// therefore retain host policy by default, while a process owner such as -+ /// the CLI can opt in before any guest is scheduled. -+ #[cfg(all(feature = "ctrlc", any(unix, windows)))] -+ pub fn with_host_lifecycle_supervisor( -+ &mut self, -+ supervisor: Arc, -+ ) -> &mut Self { -+ self.host_lifecycle_supervisor = Some(supervisor); -+ self -+ } -+ - #[cfg(feature = "journal")] - pub fn with_snapshot_trigger(&mut self, on: SnapshotTrigger) -> &mut Self { - self.wasi.snapshot_on.push(on); -@@ -372,9 +523,8 @@ impl WasiRunner { - None, - )?; - -- #[cfg(feature = "ctrlc")] -- { -- builder = builder.attach_ctrl_c(); -+ for (guest_path, (module_hash, image)) in &runner.sealed_modules { -+ builder.add_sealed_module(guest_path.clone(), *module_hash, image.clone()); - } - - #[cfg(feature = "journal")] -@@ -412,15 +562,50 @@ impl WasiRunner { - let env = builder.build()?; - let runtime = env.runtime.clone(); - let tasks = runtime.task_manager().clone(); -+ let root_process = env.process.clone(); -+ let control_plane = env.control_plane.clone(); -+ -+ #[cfg(all(feature = "ctrlc", any(unix, windows)))] -+ let host_lifecycle_supervisor = runner.host_lifecycle_supervisor.clone(); -+ let root_task = async move { -+ let result = spawn_and_wait_for_root( -+ move || { -+ crate::bin_factory::spawn_exec_module_with_preinitialized_memory_image( -+ module, -+ env, -+ &runtime, -+ runner.preinitialized_memory_image, -+ ) -+ .context("Spawn failed") -+ }, -+ move |task_handle| { -+ #[cfg(not(all(feature = "ctrlc", any(unix, windows))))] -+ let _ = task_handle; -+ -+ #[cfg(all(feature = "ctrlc", any(unix, windows)))] -+ if let Some(supervisor) = &host_lifecycle_supervisor -+ && let Err(error) = supervisor.bind_task(task_handle) -+ { -+ return Err(Error::new(error).context( -+ "Unable to bind the root task to host lifecycle supervision", -+ )); -+ } -+ -+ Ok(()) -+ }, -+ ) -+ .await?; -+ control_plane -+ .wait_for_process_tree_quiescence(&root_process) -+ .await; -+ Ok::<_, Error>(result) -+ } -+ .in_current_span(); - -- let mut task_handle = -- crate::bin_factory::spawn_exec_module(module, env, &runtime).context("Spawn failed")?; -- -- #[cfg(feature = "ctrlc")] -- task_handle.install_ctrlc_handler(); -- let task_handle = async move { task_handle.wait_finished().await }.in_current_span(); -- -- let result = tasks.spawn_and_block_on(task_handle)?; -+ // Spawn, binding, fail-closed cleanup, and the normal join all run -+ // inside the same admitted future. No guest effect precedes watcher -+ // admission, and bind failure never needs a second admission. -+ let result = tasks.spawn_and_block_on(root_task)??; - let exit_code = result - .map_err(|err| { - // We do our best to recover the error -@@ -515,6 +700,8 @@ impl WasiRunner { - let command_name = command_name.to_string(); - let tasks = runtime.task_manager().clone(); - let pkg = pkg.clone(); -+ #[cfg(all(feature = "ctrlc", any(unix, windows)))] -+ let host_lifecycle_supervisor = self.host_lifecycle_supervisor.clone(); - - // Wrapping the call to `spawn_and_block_on` in a call to `spawn_await` could help to prevent deadlocks - // because then blocking in here won't block the tokio runtime -@@ -522,16 +709,21 @@ impl WasiRunner { - // See run_wasm above for a possible fix - let exit_code = tasks.spawn_and_block_on( - async move { -- let mut task_handle = -- crate::bin_factory::spawn_exec(pkg, &command_name, env, &runtime) -- .await -- .context("Spawn failed")?; -- -- #[cfg(feature = "ctrlc")] -- task_handle.install_ctrlc_handler(); -+ let task_handle = crate::bin_factory::spawn_exec(pkg, &command_name, env, &runtime) -+ .await -+ .context("Spawn failed")?; -+ let root = RootTaskLifecycleGuard::new(task_handle); -+ -+ #[cfg(all(feature = "ctrlc", any(unix, windows)))] -+ if let Some(supervisor) = &host_lifecycle_supervisor -+ && let Err(error) = supervisor.bind_task(root.task()) -+ { -+ let bind_error = Error::new(error) -+ .context("Unable to bind the root task to host lifecycle supervision"); -+ return Err(root.terminate_and_reap(bind_error).await); -+ } - -- task_handle -- .wait_finished() -+ root.wait_finished() - .await - .map_err(|err| { - // We do our best to recover the error -@@ -594,6 +786,9 @@ fn wasi_runtime_error_to_owned(err: &WasiRuntimeError) -> WasiRuntimeError { - WasiRuntimeError::Wasi(WasiError::DlSymbolResolutionFailed(symbol)) => { - WasiRuntimeError::Wasi(WasiError::DlSymbolResolutionFailed(symbol.clone())) - } -+ WasiRuntimeError::Wasi(WasiError::StoreSnapshot(err)) => { -+ WasiRuntimeError::Wasi(WasiError::StoreSnapshot(err.clone())) -+ } - WasiRuntimeError::ControlPlane(a) => WasiRuntimeError::ControlPlane(a.clone()), - WasiRuntimeError::Runtime(a) => WasiRuntimeError::Runtime(a.clone()), - WasiRuntimeError::Thread(a) => WasiRuntimeError::Thread(a.clone()), -@@ -601,10 +796,180 @@ fn wasi_runtime_error_to_owned(err: &WasiRuntimeError) -> WasiRuntimeError { - } - } - -+#[cfg(all(test, feature = "ctrlc", any(unix, windows)))] -+mod host_lifecycle_tests { -+ use std::{ -+ future::Future, -+ sync::{Arc, mpsc}, -+ task::{Context as TaskContext, Poll}, -+ time::Duration, -+ }; -+ -+ use wasmer::FromToNativeWasmType; -+ -+ use super::*; -+ use crate::os::task::{ -+ OwnedTaskStatus, TaskStatus, -+ signal::{SignalDeliveryError, SignalHandlerAbi}, -+ }; -+ -+ #[derive(Debug)] -+ struct FinishingSignalHandler { -+ sender: mpsc::Sender, -+ } -+ -+ impl SignalHandlerAbi for FinishingSignalHandler { -+ fn signal(&self, signal: u8) -> Result<(), SignalDeliveryError> { -+ self.sender.send(signal).map_err(|_| SignalDeliveryError) -+ } -+ } -+ -+ fn finishing_root() -> (Arc, mpsc::Receiver, TaskJoinHandle) { -+ let (sender, receiver) = mpsc::channel(); -+ let mut owned = OwnedTaskStatus::default(); -+ owned.set_signal_handler(Arc::new(FinishingSignalHandler { sender })); -+ let owned = Arc::new(owned); -+ let task = owned.handle(); -+ (owned, receiver, task) -+ } -+ -+ fn finish_root_after_kill( -+ owned: Arc, -+ receiver: mpsc::Receiver, -+ ) -> std::thread::JoinHandle<()> { -+ std::thread::spawn(move || { -+ assert_eq!( -+ receiver.recv_timeout(Duration::from_secs(1)).unwrap(), -+ Signal::Sigkill.to_native() as u8 -+ ); -+ owned.set_finished(Ok(0u16.into())); -+ }) -+ } -+ -+ #[test] -+ fn wasi_runner_never_owns_host_lifecycle_by_default() { -+ assert!(WasiRunner::new().host_lifecycle_supervisor.is_none()); -+ } -+ -+ #[test] -+ fn lifecycle_bind_failure_kills_and_reaps_spawned_root() { -+ let (owned, receiver, mut task) = finishing_root(); -+ let finisher = finish_root_after_kill(owned.clone(), receiver); -+ -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let error = runtime.block_on(terminate_root_after_lifecycle_bind_failure( -+ &mut task, -+ anyhow::anyhow!("synthetic bind failure"), -+ )); -+ -+ finisher.join().unwrap(); -+ assert!(matches!(owned.status(), TaskStatus::Finished(_))); -+ assert!(error.to_string().contains("terminated and reaped")); -+ assert!(format!("{error:#}").contains("synthetic bind failure")); -+ } -+ -+ #[test] -+ fn bind_panic_terminates_and_reaps_spawned_root() { -+ let (owned, receiver, task) = finishing_root(); -+ let finisher = finish_root_after_kill(owned.clone(), receiver); -+ let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ virtual_mio::block_on(spawn_and_wait_for_root( -+ move || Ok(task), -+ |_| -> Result<(), Error> { panic!("synthetic bind panic") }, -+ )) -+ })); -+ -+ assert!(result.is_err()); -+ finisher.join().unwrap(); -+ assert!(matches!(owned.status(), TaskStatus::Finished(_))); -+ } -+ -+ #[test] -+ fn dropping_admitted_watcher_terminates_and_reaps_spawned_root() { -+ let (owned, receiver, task) = finishing_root(); -+ let finisher = finish_root_after_kill(owned.clone(), receiver); -+ let mut watcher = Box::pin(spawn_and_wait_for_root(move || Ok(task), |_| Ok(()))); -+ let waker = futures::task::noop_waker(); -+ let mut context = TaskContext::from_waker(&waker); -+ assert!(matches!(watcher.as_mut().poll(&mut context), Poll::Pending)); -+ -+ drop(watcher); -+ finisher.join().unwrap(); -+ assert!(matches!(owned.status(), TaskStatus::Finished(_))); -+ } -+} -+ - #[cfg(test)] - mod tests { - use super::*; - -+ #[derive(Debug)] -+ struct RejectingSharedTaskManager; -+ -+ impl crate::runtime::task_manager::VirtualTaskManager for RejectingSharedTaskManager { -+ fn sleep_now( -+ &self, -+ _time: std::time::Duration, -+ ) -> std::pin::Pin + Send + Sync + 'static>> -+ { -+ Box::pin(async {}) -+ } -+ -+ fn task_shared( -+ &self, -+ _task: Box futures::future::BoxFuture<'static, ()> + Send + 'static>, -+ ) -> Result<(), crate::WasiThreadError> { -+ Err(crate::WasiThreadError::Unsupported) -+ } -+ -+ fn task_wasm( -+ &self, -+ _task: crate::runtime::task_manager::TaskWasm, -+ ) -> Result<(), crate::WasiThreadError> { -+ unreachable!("root spawn must not reach task_wasm before shared-task admission") -+ } -+ -+ fn task_dedicated( -+ &self, -+ _task: Box, -+ ) -> Result<(), crate::WasiThreadError> { -+ unreachable!("root admission does not use a dedicated task") -+ } -+ -+ fn thread_parallelism(&self) -> Result { -+ Ok(1) -+ } -+ } -+ -+ #[test] -+ fn direct_root_spawn_has_no_effect_before_watcher_admission() { -+ use std::sync::atomic::{AtomicBool, Ordering}; -+ -+ let spawned = Arc::new(AtomicBool::new(false)); -+ let spawn_observer = spawned.clone(); -+ let root_task = spawn_and_wait_for_root( -+ move || { -+ spawn_observer.store(true, Ordering::SeqCst); -+ Err(anyhow::anyhow!("root spawn should not be polled")) -+ }, -+ |_| Ok(()), -+ ); -+ assert!(!spawned.load(Ordering::SeqCst)); -+ -+ // VirtualTaskManagerExt currently panics on a rejected task_shared -+ // admission. The admitted future must be dropped unpolled in that -+ // path, so its root-spawn closure must remain untouched. -+ let manager = Arc::new(RejectingSharedTaskManager); -+ let admission = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ let _ = manager.spawn_and_block_on(root_task); -+ })); -+ assert!(admission.is_err()); -+ assert!(!spawned.load(Ordering::SeqCst)); -+ } -+ - #[test] - fn send_and_sync() { - fn assert_send() {} -diff --git a/lib/wasix/src/runners/wasi_common.rs b/lib/wasix/src/runners/wasi_common.rs -index afb8669..a52c008 100644 ---- a/lib/wasix/src/runners/wasi_common.rs -+++ b/lib/wasix/src/runners/wasi_common.rs -@@ -5,6 +5,7 @@ use std::{ - }; - - use anyhow::{Context, Error}; -+#[cfg(feature = "host-fs")] - use tokio::runtime::Handle; - use virtual_fs::{ - ArcFileSystem, ExactMountConflictMode, FileSystem, MountFileSystem, OverlayFileSystem, -@@ -12,12 +13,13 @@ use virtual_fs::{ - }; - use webc::metadata::annotations::Wasi as WasiAnnotation; - -+#[cfg(feature = "journal")] -+use crate::journal::{DynJournal, DynReadableJournal, SnapshotTrigger}; - use crate::{ - WasiEnvBuilder, - bin_factory::{BinaryPackage, BinaryPackageMounts}, - capabilities::Capabilities, - fs::WasiFsRoot, -- journal::{DynJournal, DynReadableJournal, SnapshotTrigger}, - }; - - pub const MAPPED_CURRENT_DIR_DEFAULT_PATH: &str = "/home"; -@@ -48,10 +50,15 @@ pub(crate) struct CommonWasiOptions { - pub(crate) is_home_mapped: bool, - pub(crate) injected_packages: Vec, - pub(crate) capabilities: Capabilities, -+ #[cfg(feature = "journal")] - pub(crate) read_only_journals: Vec>, -+ #[cfg(feature = "journal")] - pub(crate) writable_journals: Vec>, -+ #[cfg(feature = "journal")] - pub(crate) snapshot_on: Vec, -+ #[cfg(feature = "journal")] - pub(crate) snapshot_interval: Option, -+ #[cfg(feature = "journal")] - pub(crate) stop_running_after_snapshot: bool, - pub(crate) skip_stdio_during_bootstrap: bool, - pub(crate) current_dir: Option, -diff --git a/lib/wasix/src/runtime/mod.rs b/lib/wasix/src/runtime/mod.rs -index 2023d22..e810d39 100644 ---- a/lib/wasix/src/runtime/mod.rs -+++ b/lib/wasix/src/runtime/mod.rs -@@ -1,6 +1,7 @@ - pub mod module_cache; - pub mod package_loader; - pub mod resolver; -+pub mod sealed_loader_audit; - pub mod task_manager; - - use self::module_cache::CacheError; -@@ -147,6 +148,16 @@ impl<'a> ModuleInput<'a> { - } - } - -+/// Process resource limits exposed by a WASIX runtime. -+/// -+/// Limits are runtime policy: embedders may choose conservative values that -+/// reflect host-side constraints that are not directly visible in guest memory. -+#[derive(Debug, Clone, Copy, Default)] -+pub struct ResourceLimits { -+ /// Effective process stack limit in bytes. -+ pub stack: Option, -+} -+ - /// Runtime components used when running WebAssembly programs. - /// - /// Think of this as the "System" in "WebAssembly Systems Interface". -@@ -161,6 +172,11 @@ where - /// Retrieve the active [`VirtualTaskManager`]. - fn task_manager(&self) -> &Arc; - -+ /// Process resource limits reported to WASIX programs. -+ fn resource_limits(&self) -> ResourceLimits { -+ ResourceLimits::default() -+ } -+ - /// A package loader. - fn package_loader(&self) -> Arc { - Arc::new(UnsupportedPackageLoader) -@@ -377,6 +393,9 @@ pub async fn load_module( - - match result { - Ok(module) => return Ok(module), -+ Err(error) if module_cache.is_authoritative() => { -+ return Err(crate::SpawnError::CacheError(error)); -+ } - Err(CacheError::NotFound) => {} - Err(other) => { - tracing::warn!( -@@ -459,6 +478,7 @@ pub struct PluggableRuntime { - pub source: Arc, - pub engine: Engine, - pub module_cache: Arc, -+ pub resource_limits: ResourceLimits, - pub tty: Option>, - #[cfg(feature = "journal")] - pub read_only_journals: Vec>, -@@ -500,6 +520,7 @@ impl PluggableRuntime { - source: Arc::new(source), - package_loader: Arc::new(loader), - module_cache: Arc::new(module_cache::in_memory()), -+ resource_limits: ResourceLimits::default(), - #[cfg(feature = "journal")] - read_only_journals: Vec::new(), - #[cfg(feature = "journal")] -@@ -522,6 +543,11 @@ impl PluggableRuntime { - self - } - -+ pub fn set_resource_limits(&mut self, resource_limits: ResourceLimits) -> &mut Self { -+ self.resource_limits = resource_limits; -+ self -+ } -+ - pub fn set_tty(&mut self, tty: Arc) -> &mut Self { - self.tty = Some(tty); - self -@@ -627,6 +653,10 @@ impl Runtime for PluggableRuntime { - &self.rt - } - -+ fn resource_limits(&self) -> ResourceLimits { -+ self.resource_limits -+ } -+ - fn tty(&self) -> Option<&(dyn TtyBridge + Send + Sync)> { - self.tty.as_deref() - } -@@ -688,6 +718,7 @@ pub struct OverriddenRuntime { - source: Option>, - engine: Option, - module_cache: Option>, -+ resource_limits: Option, - tty: Option>, - additional_imports: Vec, - instance_callbacks: Vec, -@@ -708,6 +739,7 @@ impl OverriddenRuntime { - source: None, - engine: None, - module_cache: None, -+ resource_limits: None, - tty: None, - additional_imports: Vec::new(), - instance_callbacks: Vec::new(), -@@ -756,6 +788,11 @@ impl OverriddenRuntime { - self - } - -+ pub fn with_resource_limits(mut self, resource_limits: ResourceLimits) -> Self { -+ self.resource_limits.replace(resource_limits); -+ self -+ } -+ - pub fn with_tty(mut self, tty: Arc) -> Self { - self.tty.replace(tty); - self -@@ -820,6 +857,11 @@ impl Runtime for OverriddenRuntime { - } - } - -+ fn resource_limits(&self) -> ResourceLimits { -+ self.resource_limits -+ .unwrap_or_else(|| self.inner.resource_limits()) -+ } -+ - fn source(&self) -> Arc { - if let Some(source) = self.source.clone() { - source -@@ -930,3 +972,58 @@ impl Runtime for OverriddenRuntime { - } - } - } -+ -+#[cfg(all(test, not(target_arch = "wasm32")))] -+mod authoritative_cache_tests { -+ use super::*; -+ -+ #[derive(Debug)] -+ struct AuthoritativeMissCache; -+ -+ #[async_trait::async_trait] -+ impl ModuleCache for AuthoritativeMissCache { -+ fn is_authoritative(&self) -> bool { -+ true -+ } -+ -+ async fn load(&self, _key: ModuleHash, _engine: &Engine) -> Result { -+ Err(CacheError::NotFound) -+ } -+ -+ async fn contains(&self, _key: ModuleHash, _engine: &Engine) -> Result { -+ Ok(false) -+ } -+ -+ async fn save( -+ &self, -+ _key: ModuleHash, -+ _engine: &Engine, -+ _module: &Module, -+ ) -> Result<(), CacheError> { -+ Err(CacheError::other(std::io::Error::new( -+ std::io::ErrorKind::PermissionDenied, -+ "authoritative cache is immutable", -+ ))) -+ } -+ } -+ -+ #[tokio::test] -+ async fn authoritative_miss_is_terminal_without_compilation_fallback() { -+ let engine = Engine::default(); -+ let cache: Arc = Arc::new(AuthoritativeMissCache); -+ assert!(cache.is_authoritative()); -+ -+ let error = load_module( -+ &engine, -+ cache.as_ref(), -+ ModuleInput::Bytes(Cow::Borrowed(b"\0asm\x01\0\0\0")), -+ None, -+ ) -+ .await -+ .unwrap_err(); -+ assert!(matches!( -+ error, -+ crate::SpawnError::CacheError(CacheError::NotFound) -+ )); -+ } -+} -diff --git a/lib/wasix/src/runtime/module_cache/filesystem.rs b/lib/wasix/src/runtime/module_cache/filesystem.rs -index 675c1bf..fd657ca 100644 ---- a/lib/wasix/src/runtime/module_cache/filesystem.rs -+++ b/lib/wasix/src/runtime/module_cache/filesystem.rs -@@ -42,10 +42,11 @@ impl FileSystemCache { - /// A tokio reactor must be available - #[tracing::instrument(level = "debug", skip_all, fields(? path))] - async fn tokio_load(path: PathBuf, engine: Engine) -> Result { -- let bytes = read_file(&path).await?; -- let deserialized = tokio::task::spawn_blocking(move || deserialize(&bytes, &engine)) -- .await -- .unwrap(); -+ let deserialize_path = path.clone(); -+ let deserialized = -+ tokio::task::spawn_blocking(move || deserialize_file(&deserialize_path, &engine)) -+ .await -+ .unwrap(); - match deserialized { - Ok(m) => { - tracing::debug!("Cache hit!"); -@@ -174,8 +175,8 @@ impl ModuleCache for FileSystemCache { - } - } - --async fn read_file(path: &Path) -> Result, CacheError> { -- match tokio::fs::read(path).await { -+fn read_file(path: &Path) -> Result, CacheError> { -+ match std::fs::read(path) { - Ok(bytes) => Ok(bytes), - Err(e) if e.kind() == std::io::ErrorKind::NotFound => Err(CacheError::NotFound), - Err(error) => Err(CacheError::FileRead { -@@ -185,7 +186,7 @@ async fn read_file(path: &Path) -> Result, CacheError> { - } - } - --fn deserialize(bytes: &[u8], engine: &Engine) -> Result { -+fn deserialize_file(path: &Path, engine: &Engine) -> Result { - // We used to compress our compiled modules using LZW encoding in the past. - // This was removed because it has a negative impact on startup times for - // "wasmer run", so all new compiled modules should be saved directly to -@@ -201,18 +202,22 @@ fn deserialize(bytes: &[u8], engine: &Engine) -> Result { - // - ModuleCache::save(): 2.4s, 72MB binary - // - ModuleCache::load(): 822ms - -- match unsafe { Module::deserialize(engine, bytes) } { -+ match unsafe { Module::deserialize_from_file(engine, path) } { - // The happy case - Ok(m) => Ok(m), - Err(wasmer::DeserializeError::Incompatible(_)) => { -+ let bytes = read_file(path)?; - let bytes = weezl::decode::Decoder::new(weezl::BitOrder::Msb, 8) -- .decode(bytes) -+ .decode(&bytes) - .map_err(CacheError::other)?; - - let m = unsafe { Module::deserialize(engine, bytes)? }; - - Ok(m) - } -+ Err(wasmer::DeserializeError::Io(e)) if e.kind() == std::io::ErrorKind::NotFound => { -+ Err(CacheError::NotFound) -+ } - Err(e) => Err(CacheError::Deserialize(e)), - } - } -diff --git a/lib/wasix/src/runtime/module_cache/types.rs b/lib/wasix/src/runtime/module_cache/types.rs -index 3c4510c..0c1abd0 100644 ---- a/lib/wasix/src/runtime/module_cache/types.rs -+++ b/lib/wasix/src/runtime/module_cache/types.rs -@@ -23,6 +23,16 @@ use crate::runtime::module_cache::{FallbackCache, progress::ModuleLoadProgressRe - /// - #[async_trait::async_trait] - pub trait ModuleCache: Debug { -+ /// Whether this cache is the authoritative module closure. -+ /// -+ /// An authoritative cache is a policy boundary rather than an optimization: -+ /// a miss or load error must be returned to the caller without attempting to -+ /// compile the supplied WebAssembly bytes. This is used by sealed, -+ /// precompiled-only runtimes. -+ fn is_authoritative(&self) -> bool { -+ false -+ } -+ - /// Load a module based on its hash. - async fn load(&self, key: ModuleHash, engine: &Engine) -> Result; - -@@ -95,6 +105,10 @@ where - D: Deref + Debug + Send + Sync, - C: ModuleCache + Send + Sync + ?Sized, - { -+ fn is_authoritative(&self) -> bool { -+ (**self).is_authoritative() -+ } -+ - async fn load(&self, key: ModuleHash, engine: &Engine) -> Result { - (**self).load(key, engine).await - } -diff --git a/lib/wasix/src/runtime/sealed_loader_audit.rs b/lib/wasix/src/runtime/sealed_loader_audit.rs -new file mode 100644 -index 0000000..d974629 ---- /dev/null -+++ b/lib/wasix/src/runtime/sealed_loader_audit.rs -@@ -0,0 +1,310 @@ -+//! Non-faulting host-page-cache evidence for sealed loader inputs. -+//! -+//! These helpers describe advisory calls and Linux `mincore(2)` observations; -+//! they never participate in artifact admission or integrity decisions. -+ -+use std::fs::File; -+ -+/// Outcome of one or more best-effort file-cache advisory calls. -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub struct FileAdviceAudit { -+ pub supported: bool, -+ pub calls: u64, -+ pub successes: u64, -+ /// The first nonzero error returned by an advisory call. -+ pub first_errno: Option, -+} -+ -+impl FileAdviceAudit { -+ pub const fn unsupported() -> Self { -+ Self { -+ supported: false, -+ calls: 0, -+ successes: 0, -+ first_errno: None, -+ } -+ } -+ -+ pub const fn not_applicable() -> Self { -+ Self { -+ supported: true, -+ calls: 0, -+ successes: 0, -+ first_errno: None, -+ } -+ } -+} -+ -+/// A point-in-time observation of file-backed page-cache residency. -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub struct FileResidencyAudit { -+ pub state: FileResidencyState, -+ pub page_size: Option, -+ pub total_pages: Option, -+ pub resident_pages: Option, -+ pub resident_bytes: Option, -+ pub errno: Option, -+} -+ -+impl FileResidencyAudit { -+ pub const fn unsupported() -> Self { -+ Self { -+ state: FileResidencyState::Unsupported, -+ page_size: None, -+ total_pages: None, -+ resident_pages: None, -+ resident_bytes: None, -+ errno: None, -+ } -+ } -+ -+ pub const fn not_applicable() -> Self { -+ Self { -+ state: FileResidencyState::NotApplicable, -+ page_size: None, -+ total_pages: None, -+ resident_pages: None, -+ resident_bytes: None, -+ errno: None, -+ } -+ } -+ -+ #[cfg(target_os = "linux")] -+ const fn error(errno: i32) -> Self { -+ Self { -+ state: FileResidencyState::Error, -+ page_size: None, -+ total_pages: None, -+ resident_pages: None, -+ resident_bytes: None, -+ errno: Some(errno), -+ } -+ } -+} -+ -+/// Portable state attached to a residency checkpoint. -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub enum FileResidencyState { -+ Measured, -+ Unsupported, -+ NotApplicable, -+ Error, -+} -+ -+impl FileResidencyState { -+ pub const fn as_str(self) -> &'static str { -+ match self { -+ Self::Measured => "measured", -+ Self::Unsupported => "unsupported-platform", -+ Self::NotApplicable => "not-applicable", -+ Self::Error => "error", -+ } -+ } -+} -+ -+/// Apply sequential and no-reuse hints before a one-shot verified read. -+#[cfg(target_os = "linux")] -+pub fn advise_file_for_one_shot_read(file: &File) -> FileAdviceAudit { -+ use std::os::fd::AsRawFd; -+ -+ let results = [ -+ // SAFETY: `file` owns this live descriptor and no memory is accessed. -+ unsafe { libc::posix_fadvise(file.as_raw_fd(), 0, 0, libc::POSIX_FADV_SEQUENTIAL) }, -+ // SAFETY: as above; this is a best-effort page-replacement hint. -+ unsafe { libc::posix_fadvise(file.as_raw_fd(), 0, 0, libc::POSIX_FADV_NOREUSE) }, -+ ]; -+ FileAdviceAudit { -+ supported: true, -+ calls: results.len() as u64, -+ successes: results.iter().filter(|&&result| result == 0).count() as u64, -+ first_errno: results.into_iter().find(|&result| result != 0), -+ } -+} -+ -+#[cfg(not(target_os = "linux"))] -+pub fn advise_file_for_one_shot_read(_file: &File) -> FileAdviceAudit { -+ FileAdviceAudit::unsupported() -+} -+ -+/// Ask the kernel to release clean source pages after verified activation. -+#[cfg(target_os = "linux")] -+pub fn advise_file_away(file: &File) -> FileAdviceAudit { -+ use std::os::fd::AsRawFd; -+ -+ // SAFETY: `file` owns this live descriptor. DONTNEED neither changes file -+ // contents nor forms part of the loader's correctness contract. -+ let result = unsafe { libc::posix_fadvise(file.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) }; -+ FileAdviceAudit { -+ supported: true, -+ calls: 1, -+ successes: u64::from(result == 0), -+ first_errno: (result != 0).then_some(result), -+ } -+} -+ -+#[cfg(not(target_os = "linux"))] -+pub fn advise_file_away(_file: &File) -> FileAdviceAudit { -+ FileAdviceAudit::unsupported() -+} -+ -+/// Observe file-backed residency without reading or faulting payload bytes. -+/// -+/// Linux maps the descriptor `PROT_NONE`, asks `mincore(2)` for the existing -+/// page-cache vector, and immediately unmaps it. The mapping itself cannot be -+/// dereferenced and therefore cannot make a cold artifact resident. -+#[cfg(target_os = "linux")] -+pub fn file_residency(file: &File, logical_bytes: u64) -> FileResidencyAudit { -+ use std::os::fd::AsRawFd; -+ -+ fn errno_or(fallback: i32) -> i32 { -+ std::io::Error::last_os_error() -+ .raw_os_error() -+ .filter(|errno| *errno != 0) -+ .unwrap_or(fallback) -+ } -+ -+ let metadata = match file.metadata() { -+ Ok(metadata) => metadata, -+ Err(error) => { -+ return FileResidencyAudit::error(error.raw_os_error().unwrap_or(libc::EIO)); -+ } -+ }; -+ if metadata.len() != logical_bytes { -+ return FileResidencyAudit::error(libc::ESTALE); -+ } -+ -+ let page_size = unsafe { libc::sysconf(libc::_SC_PAGESIZE) }; -+ if page_size <= 0 { -+ return FileResidencyAudit::error(errno_or(libc::EINVAL)); -+ } -+ let page_size = page_size as u64; -+ if logical_bytes == 0 { -+ return FileResidencyAudit { -+ state: FileResidencyState::Measured, -+ page_size: Some(page_size), -+ total_pages: Some(0), -+ resident_pages: Some(0), -+ resident_bytes: Some(0), -+ errno: None, -+ }; -+ } -+ -+ let Ok(mapping_len) = usize::try_from(logical_bytes) else { -+ return FileResidencyAudit::error(libc::EOVERFLOW); -+ }; -+ let total_pages = logical_bytes.div_ceil(page_size); -+ let Ok(vector_len) = usize::try_from(total_pages) else { -+ return FileResidencyAudit::error(libc::EOVERFLOW); -+ }; -+ let mut vector = Vec::new(); -+ if vector.try_reserve_exact(vector_len).is_err() { -+ return FileResidencyAudit::error(libc::ENOMEM); -+ } -+ vector.resize(vector_len, 0_u8); -+ -+ // SAFETY: this creates a new non-dereferenceable mapping over the live -+ // descriptor. No loader mapping or executable allocation is modified. -+ let address = unsafe { -+ libc::mmap( -+ std::ptr::null_mut(), -+ mapping_len, -+ libc::PROT_NONE, -+ libc::MAP_SHARED, -+ file.as_raw_fd(), -+ 0, -+ ) -+ }; -+ if address == libc::MAP_FAILED { -+ return FileResidencyAudit::error(errno_or(libc::EIO)); -+ } -+ -+ // SAFETY: `address` names the full live mapping and `vector` has exactly -+ // one byte for every native page in that range. -+ let mincore_result = unsafe { libc::mincore(address, mapping_len, vector.as_mut_ptr()) }; -+ let mincore_errno = (mincore_result != 0).then(|| errno_or(libc::EIO)); -+ // SAFETY: this releases exactly the mapping created above. -+ let unmap_result = unsafe { libc::munmap(address, mapping_len) }; -+ let unmap_errno = (unmap_result != 0).then(|| errno_or(libc::EIO)); -+ if let Some(errno) = mincore_errno.or(unmap_errno) { -+ return FileResidencyAudit::error(errno); -+ } -+ -+ let resident_pages = vector.iter().filter(|entry| **entry & 1 != 0).count() as u64; -+ let resident_bytes = vector -+ .iter() -+ .enumerate() -+ .filter(|(_, entry)| **entry & 1 != 0) -+ .map(|(index, _)| { -+ let offset = index as u64 * page_size; -+ (logical_bytes - offset).min(page_size) -+ }) -+ .sum(); -+ FileResidencyAudit { -+ state: FileResidencyState::Measured, -+ page_size: Some(page_size), -+ total_pages: Some(total_pages), -+ resident_pages: Some(resident_pages), -+ resident_bytes: Some(resident_bytes), -+ errno: None, -+ } -+} -+ -+#[cfg(not(target_os = "linux"))] -+pub fn file_residency(_file: &File, _logical_bytes: u64) -> FileResidencyAudit { -+ FileResidencyAudit::unsupported() -+} -+ -+#[cfg(all(test, target_os = "linux"))] -+mod tests { -+ use super::*; -+ use std::io::{Read, Seek, SeekFrom, Write}; -+ -+ #[test] -+ fn mincore_checkpoint_does_not_move_or_read_the_descriptor() { -+ let mut file = tempfile::tempfile().unwrap(); -+ let bytes = vec![0x5a; 8193]; -+ file.write_all(&bytes).unwrap(); -+ file.seek(SeekFrom::Start(7)).unwrap(); -+ -+ let audit = file_residency(&file, bytes.len() as u64); -+ assert_eq!(audit.state, FileResidencyState::Measured); -+ assert_eq!(audit.total_pages, Some(3)); -+ assert_eq!(audit.resident_pages, Some(3)); -+ assert_eq!(audit.resident_bytes, Some(bytes.len() as u64)); -+ -+ let mut byte = [0_u8; 1]; -+ file.read_exact(&mut byte).unwrap(); -+ assert_eq!(byte[0], 0x5a); -+ assert_eq!(file.stream_position().unwrap(), 8); -+ } -+ -+ #[test] -+ fn mincore_checkpoint_rejects_a_changed_file_size() { -+ let mut file = tempfile::tempfile().unwrap(); -+ file.write_all(b"stable-size").unwrap(); -+ -+ let audit = file_residency(&file, 12); -+ assert_eq!(audit.state, FileResidencyState::Error); -+ assert_eq!(audit.errno, Some(libc::ESTALE)); -+ assert_eq!(audit.resident_pages, None); -+ assert_eq!(audit.resident_bytes, None); -+ } -+ -+ #[test] -+ fn advisory_audits_preserve_call_and_errno_shape() { -+ let mut file = tempfile::tempfile().unwrap(); -+ file.write_all(b"advice").unwrap(); -+ let read = advise_file_for_one_shot_read(&file); -+ assert!(read.supported); -+ assert_eq!(read.calls, 2); -+ assert_eq!(read.successes + u64::from(read.first_errno.is_some()), 2); -+ -+ let eviction = advise_file_away(&file); -+ assert!(eviction.supported); -+ assert_eq!(eviction.calls, 1); -+ assert_eq!( -+ eviction.successes, -+ u64::from(eviction.first_errno.is_none()) -+ ); -+ } -+} -diff --git a/lib/wasix/src/runtime/task_manager/mod.rs b/lib/wasix/src/runtime/task_manager/mod.rs -index 2a6457e..2418f25 100644 ---- a/lib/wasix/src/runtime/task_manager/mod.rs -+++ b/lib/wasix/src/runtime/task_manager/mod.rs -@@ -4,7 +4,7 @@ pub mod tokio; - - use std::ops::Deref; - use std::task::{Context, Poll}; --use std::{pin::Pin, time::Duration}; -+use std::{pin::Pin, sync::Arc, time::Duration}; - - use bytes::Bytes; - use derive_more::Debug; -@@ -17,7 +17,13 @@ use wasmer_wasix_types::wasi::{Errno, ExitCode}; - - use crate::syscalls::AsyncifyFuture; - use crate::{StoreSnapshot, WasiEnv, WasiFunctionEnv, WasiThread, capture_store_snapshot}; --use crate::{os::task::thread::WasiThreadError, state::Linker}; -+use crate::{ -+ os::task::{ -+ process::{WasiProcessExecutionLease, WasiProcessStartGate}, -+ thread::WasiThreadError, -+ }, -+ state::{Linker, PreinitializedMemoryImageMode}, -+}; - - pub use virtual_mio::waker::*; - -@@ -49,15 +55,157 @@ pub enum SpawnMemoryTypeOrStore { - StoreAndMemory(wasmer::Store, Option), - } - --pub type WasmResumeTask = dyn FnOnce(WasiFunctionEnv, Store, Bytes) + Send + 'static; -+pub type WasmResumeTask = dyn FnOnce(WasiFunctionEnv, Store, Bytes, Option) -+ + Send -+ + 'static; - - pub type WasmResumeTrigger = dyn FnOnce() -> Pin> + Send + 'static>> - + Send - + Sync; - -+/// Fail-closed ownership for a `TaskWasm` that has not entered its accepted -+/// run callback yet. -+/// -+/// Memory construction, Wasmer instantiation, async trigger polling, and the -+/// blocking worker queue all happen before the callback owns execution. If -+/// any of those stages drops the task, this guard terminalizes the exact WASI -+/// thread before releasing the process-quiescence lease. The only successful -+/// disarm boundary is [`Self::accept_callback`], invoked from inside the -+/// worker callback immediately before calling `TaskWasmRun`. -+#[derive(Debug)] -+#[must_use = "dropping pending TaskWasm execution terminalizes its thread"] -+pub struct TaskWasmExecutionGuard { -+ thread: WasiThread, -+ lease: Option, -+ owner_identity: Arc<()>, -+} -+ -+impl TaskWasmExecutionGuard { -+ fn new(thread: WasiThread, lease: WasiProcessExecutionLease) -> Self { -+ Self { -+ thread, -+ lease: Some(lease), -+ owner_identity: Arc::new(()), -+ } -+ } -+ -+ /// Transfers responsibility to an accepted `TaskWasmRun` callback. -+ /// -+ /// Task-manager implementations must call this only from within the -+ /// worker closure, after every fallible pre-run stage has completed and -+ /// immediately before invoking the callback. The callback must publish a -+ /// terminal thread status or acquire an accepted successor lease before -+ /// the returned guard is dropped. -+ fn into_accepted_callback(mut self) -> TaskWasmAcceptedExecutionGuard { -+ TaskWasmAcceptedExecutionGuard { -+ thread: self.thread.clone(), -+ lease: self.lease.take(), -+ owner_identity: Arc::clone(&self.owner_identity), -+ } -+ } -+ -+ fn bind_pending_owner(&self, env: &mut WasiEnv) { -+ env.bind_pending_task_wasm_execution(&self.thread, &self.owner_identity); -+ } -+ -+ /// Accepts this callback only if [`TaskWasm::new`] bound this exact guard's -+ /// unforgeable identity to the physical instance in `env`. -+ /// -+ /// Every [`VirtualTaskManager`] implementation must invoke this from the -+ /// accepted worker closure after all fallible setup and immediately before -+ /// constructing [`TaskWasmRunProperties`]. The exact-token assertion and -+ /// private raw conversion make alternate task managers follow the same -+ /// vfork ownership contract as the built-in Tokio implementation. -+ pub fn accept_callback(self, env: &mut WasiEnv) -> TaskWasmAcceptedExecutionGuard { -+ let accepted = self.into_accepted_callback(); -+ env.accept_task_wasm_execution(&accepted); -+ accepted -+ } -+} -+ -+impl Drop for TaskWasmExecutionGuard { -+ fn drop(&mut self) { -+ let Some(lease) = self.lease.take() else { -+ return; -+ }; -+ if self.thread.try_join().is_none() { -+ self.thread -+ .set_status_finished(Err(crate::RuntimeError::new( -+ "TaskWasm execution canceled before callback acceptance", -+ ) -+ .into())); -+ } -+ // Exact-thread terminal status must happen-before process quiescence. -+ drop(lease); -+ } -+} -+ -+/// Fail-closed ownership after a `TaskWasm` callback has been admitted. -+/// -+/// Arbitrary callback code may panic or abandon its properties. Keeping the -+/// exact thread beside the process lease makes that path safe by construction: -+/// dropping an armed guard publishes terminal thread status before releasing -+/// process quiescence. A deep-sleep continuation can disarm it only through -+/// [`Self::admit_successor`], which performs successor admission itself and -+/// releases the predecessor lease only after admission returns successfully. -+#[derive(Debug)] -+#[must_use = "dropping accepted TaskWasm execution terminalizes its thread"] -+pub struct TaskWasmAcceptedExecutionGuard { -+ thread: WasiThread, -+ lease: Option, -+ owner_identity: Arc<()>, -+} -+ -+impl TaskWasmAcceptedExecutionGuard { -+ pub(crate) fn thread(&self) -> &WasiThread { -+ &self.thread -+ } -+ -+ pub(crate) fn owner_identity(&self) -> &Arc<()> { -+ &self.owner_identity -+ } -+ -+ fn admit_successor(mut self, admission: impl FnOnce() -> Result) -> Result { -+ let accepted = admission()?; -+ // `admission` constructed and transferred the successor TaskWasm, -+ // whose pending guard acquired its lease in TaskWasm::new. Taking the -+ // predecessor lease only after Ok makes successor-before-predecessor -+ // ordering structural, including unwinding inside `admission`. -+ let lease = self -+ .lease -+ .take() -+ .expect("accepted TaskWasm execution handed off twice"); -+ drop(lease); -+ Ok(accepted) -+ } -+} -+ -+impl Drop for TaskWasmAcceptedExecutionGuard { -+ fn drop(&mut self) { -+ let Some(lease) = self.lease.take() else { -+ return; -+ }; -+ if self.thread.try_join().is_none() { -+ self.thread -+ .set_status_finished(Err(crate::RuntimeError::new( -+ "TaskWasm callback ended without terminal status or an accepted successor", -+ ) -+ .into())); -+ } -+ // Exact-thread terminal status must happen-before process quiescence. -+ drop(lease); -+ } -+} -+ - /// The properties passed to the task - #[derive(derive_more::Debug)] - pub struct TaskWasmRunProperties { -+ /// Accepted execution ownership. It is deliberately the first field so -+ /// abandoning the complete properties value terminalizes before dropping -+ /// the environment's owning handles. Callbacks that move it out retain -+ /// the same fail-closed guarantee through the guard's own destructor. -+ #[debug(ignore)] -+ pub execution_guard: Option, - pub ctx: WasiFunctionEnv, - pub store: Store, - /// The result of the asynchronous trigger serialized into bytes using the bincode serializer -@@ -94,6 +242,15 @@ pub type TaskWasmRecycle = dyn FnOnce(TaskWasmRecycleProperties) + Send + 'stati - - /// Represents a WASM task that will be executed on a dedicated thread - pub struct TaskWasm<'a> { -+ /// Held fail-closed from construction through every pre-callback stage. -+ /// This is deliberately the first field so abandoning the whole task -+ /// terminalizes before its environment can drop owning thread handles. -+ /// Task managers disarm it only from inside the accepted run callback. -+ pub execution_lease: -+ Result, -+ /// Immutable proof that child publication committed before construction; -+ /// Wasmer instantiation itself may execute guest start functions. -+ pub(crate) start_gate: WasiProcessStartGate, - pub run: Box, - pub recycle: Option>, - pub env: WasiEnv, -@@ -103,17 +260,36 @@ pub struct TaskWasm<'a> { - pub trigger: Option>, - pub update_layout: bool, - pub call_initialize: bool, -+ pub preinitialized_memory_image: Option, - pub pre_run: Option>, - } - - impl<'a> TaskWasm<'a> { - pub fn new( - run: Box, -- env: WasiEnv, -+ mut env: WasiEnv, - module: Module, - update_layout: bool, - call_initialize: bool, - ) -> Self { -+ let execution_lease = if env.process.guest_start_is_committed() { -+ env.process -+ .acquire_execution_lease() -+ .map(|lease| TaskWasmExecutionGuard::new(env.thread.clone(), lease)) -+ } else { -+ Err( -+ crate::os::task::control_plane::ControlPlaneError::ProcessNotPublished { -+ pid: env.pid().raw(), -+ }, -+ ) -+ }; -+ if let Ok(execution) = &execution_lease { -+ // Bind structurally in the constructor: every successfully -+ // constructed TaskWasm is ready for guest imports that may run -+ // during instantiation, independent of task-manager behavior. -+ execution.bind_pending_owner(&mut env); -+ } -+ let start_gate = env.process.start_gate(); - let shared_memory = module.imports().memories().next().map(|a| *a.ty()); - Self { - run, -@@ -127,11 +303,25 @@ impl<'a> TaskWasm<'a> { - trigger: None, - update_layout, - call_initialize, -+ preinitialized_memory_image: None, - recycle: None, - pre_run: None, -+ execution_lease, -+ start_gate, - } - } - -+ /// Must be called by task-manager implementations before allocating or -+ /// instantiating guest state. Keeping lifecycle failure inside TaskWasm -+ /// preserves the long-standing infallible constructor while making task -+ /// acceptance fail closed. -+ pub fn validate_lifecycle(&self) -> Result<(), WasiThreadError> { -+ self.execution_lease -+ .as_ref() -+ .map(|_| ()) -+ .map_err(|err| WasiThreadError::ProcessLifecycle(err.clone())) -+ } -+ - pub fn with_memory(mut self, spawn_type: SpawnType<'a>) -> Self { - self.spawn_type = spawn_type; - self -@@ -163,6 +353,14 @@ impl<'a> TaskWasm<'a> { - self.pre_run.replace(pre_run); - self - } -+ -+ pub fn with_preinitialized_memory_image( -+ mut self, -+ image: PreinitializedMemoryImageMode, -+ ) -> Self { -+ self.preinitialized_memory_image = Some(image); -+ self -+ } - } - - /// A task executor backed by a thread pool. -@@ -199,7 +397,7 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { - // browser otherwise creation will fail. - let _ = ty.maximum.get_or_insert(wasmer_types::Pages::max_value()); - -- let mem = Memory::new(&mut store, ty).map_err(|err| { -+ let mem = { Memory::new(&mut store, ty) }.map_err(|err| { - tracing::error!( - error = &err as &dyn std::error::Error, - memory_type=?ty, -@@ -210,7 +408,7 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { - Ok(Some(mem)) - } - SpawnType::ShareMemory(mem, old_store) => { -- let mem = mem.share_in_store(&old_store, store).map_err(|err| { -+ let mem = { mem.share_in_store(&old_store, store) }.map_err(|err| { - tracing::warn!( - error = &err as &dyn std::error::Error, - "could not clone memory", -@@ -220,7 +418,7 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { - Ok(Some(mem)) - } - SpawnType::CopyMemory(mem, old_store) => { -- let mem = mem.copy_to_store(&old_store, store).map_err(|err| { -+ let mem = { mem.copy_to_store(&old_store, store) }.map_err(|err| { - tracing::warn!( - error = &err as &dyn std::error::Error, - "could not copy memory", -@@ -259,6 +457,12 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { - /// - /// This is primarily used inside the context of a syscall and allows - /// the transfer of things like [`wasmer::Module`] across threads. -+ /// Implementations must call [`TaskWasm::validate_lifecycle`] before any -+ /// guest memory allocation, instantiation, initialization, or execution. -+ /// From inside the accepted worker closure, after all other fallible setup, -+ /// they must convert the pending guard with -+ /// [`TaskWasmExecutionGuard::accept_callback`] against the callback's -+ /// concrete [`WasiEnv`] immediately before invoking [`TaskWasmRun`]. - fn task_wasm(&self, task: TaskWasm) -> Result<(), WasiThreadError>; - - /// Run a blocking operation on the thread pool. -@@ -361,6 +565,7 @@ impl dyn VirtualTaskManager { - ctx: WasiFunctionEnv, - mut store: Store, - trigger: Pin>, -+ current_execution: Option, - ) -> Result<(), WasiThreadError> { - // This poller will process any signals when the main working function is idle - struct AsyncifyPollerOwned { -@@ -384,60 +589,70 @@ impl dyn VirtualTaskManager { - } - } - -- let snapshot = capture_store_snapshot(&mut store.as_store_mut()); -- let env = ctx.data(&store); -- let env_inner = env.inner(); -- let handles = env_inner -- .static_module_instance_handles() -- .ok_or(WasiThreadError::Unsupported)?; -- let module = handles.module_clone(); -- let memory = handles.memory_clone(); -+ let snapshot = capture_store_snapshot(&mut store.as_store_mut()) -+ .map_err(WasiThreadError::StoreSnapshotCaptureFailed)?; -+ let env = ctx.data_mut(&mut store); -+ let (module, memory) = { -+ let env_inner = env.inner(); -+ let handles = env_inner -+ .static_module_instance_handles() -+ .ok_or(WasiThreadError::Unsupported)?; -+ (handles.module_clone(), handles.memory_clone()) -+ }; - let thread = env.thread.clone(); -- let env = env.clone(); -+ let observer_env = env.clone(); -+ let continuation_env = env.take_for_same_thread_continuation(); - - let thread_inner = thread.clone(); -- self.task_wasm( -- TaskWasm::new( -- Box::new(move |props| { -- let result = props -- .trigger_result -- .expect("If there is no result then its likely the trigger did not run"); -- let result = match result { -- Ok(r) => r, -- Err(exit_code) => { -- thread.set_status_finished(Ok(exit_code)); -- return; -- } -- }; -- task(props.ctx, props.store, result) -- }), -- env.clone(), -- module, -- false, -- false, -+ let admit_successor = || { -+ self.task_wasm( -+ TaskWasm::new( -+ Box::new(move |mut props| { -+ let result = props.trigger_result.expect( -+ "If there is no result then its likely the trigger did not run", -+ ); -+ let result = match result { -+ Ok(r) => r, -+ Err(exit_code) => { -+ thread.set_status_finished(Ok(exit_code)); -+ return; -+ } -+ }; -+ task(props.ctx, props.store, result, props.execution_guard.take()) -+ }), -+ continuation_env, -+ module, -+ false, -+ false, -+ ) -+ .with_memory(SpawnType::ShareMemory(memory, store.as_store_ref())) -+ .with_globals(snapshot) -+ .with_trigger(Box::new(move || { -+ Box::pin(async move { -+ let mut poller = AsyncifyPollerOwned { -+ thread: thread_inner, -+ trigger, -+ }; -+ let res = Pin::new(&mut poller).await; -+ let res = match res { -+ Ok(res) => res, -+ Err(exit_code) => { -+ observer_env.thread.set_status_finished(Ok(exit_code)); -+ return Err(exit_code); -+ } -+ }; -+ -+ tracing::trace!("deep sleep woken - res.len={}", res.len()); -+ Ok(res) -+ }) -+ })), - ) -- .with_memory(SpawnType::ShareMemory(memory, store.as_store_ref())) -- .with_globals(snapshot) -- .with_trigger(Box::new(move || { -- Box::pin(async move { -- let mut poller = AsyncifyPollerOwned { -- thread: thread_inner, -- trigger, -- }; -- let res = Pin::new(&mut poller).await; -- let res = match res { -- Ok(res) => res, -- Err(exit_code) => { -- env.thread.set_status_finished(Ok(exit_code)); -- return Err(exit_code); -- } -- }; -- -- tracing::trace!("deep sleep woken - res.len={}", res.len()); -- Ok(res) -- }) -- })), -- ) -+ }; -+ -+ match current_execution { -+ Some(execution) => execution.admit_successor(admit_successor), -+ None => admit_successor(), -+ } - } - } - -@@ -504,3 +719,156 @@ where - Box::new(receiver.map_err(|e| Box::new(e).into())) - } - } -+ -+#[cfg(test)] -+mod lifecycle_tests { -+ use std::{sync::mpsc, thread, time::Duration}; -+ -+ use wasmer_types::ModuleHash; -+ use wasmer_wasix_types::wasi::Errno; -+ -+ use crate::{WasiControlPlane, os::task::thread::WasiMemoryLayout}; -+ -+ use super::{TaskWasmAcceptedExecutionGuard, TaskWasmExecutionGuard}; -+ -+ fn pending_execution() -> ( -+ crate::WasiProcess, -+ crate::WasiThreadHandle, -+ TaskWasmExecutionGuard, -+ ) { -+ let plane = WasiControlPlane::default(); -+ let (process, handle) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let lease = process.acquire_execution_lease().unwrap(); -+ let guard = TaskWasmExecutionGuard::new(handle.as_thread(), lease); -+ (process, handle, guard) -+ } -+ -+ fn accepted_execution() -> ( -+ crate::WasiProcess, -+ crate::WasiThreadHandle, -+ TaskWasmAcceptedExecutionGuard, -+ ) { -+ let (process, handle, pending) = pending_execution(); -+ (process, handle, pending.into_accepted_callback()) -+ } -+ -+ #[test] -+ fn canceled_task_wasm_terminalizes_exact_thread_before_quiescence() { -+ let (process, handle, guard) = pending_execution(); -+ let observer_process = process.clone(); -+ let observer_thread = handle.as_thread(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = thread::spawn(move || { -+ observer_process.wait_for_execution_quiescence_blocking(); -+ observed_tx.send(observer_thread.try_join()).unwrap(); -+ }); -+ -+ drop(guard); -+ -+ let status = observed_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .unwrap() -+ .expect("thread was not terminal when process became quiescent"); -+ assert!(status.is_err()); -+ assert_eq!(process.lock().execution_leases, 0); -+ observer.join().unwrap(); -+ drop(handle); -+ } -+ -+ #[test] -+ fn closed_worker_queue_drops_pending_execution_fail_closed() { -+ let (process, handle, guard) = pending_execution(); -+ let queued_callback: Box = Box::new(move || { -+ let _accepted = guard.into_accepted_callback(); -+ }); -+ -+ // A closed queue owns and drops its callback without invoking it. -+ drop(queued_callback); -+ -+ assert!(handle.try_join().is_some_and(|status| status.is_err())); -+ assert_eq!(process.lock().execution_leases, 0); -+ drop(handle); -+ } -+ -+ #[test] -+ fn accepted_callback_conversion_is_the_only_successful_disarm() { -+ let (process, handle, guard) = pending_execution(); -+ let lease = guard.into_accepted_callback(); -+ -+ assert!(handle.try_join().is_none()); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ handle.set_status_finished(Ok(Errno::Success.into())); -+ drop(lease); -+ assert_eq!(process.lock().execution_leases, 0); -+ assert_eq!(handle.try_join().unwrap().unwrap(), Errno::Success.into()); -+ } -+ -+ #[test] -+ fn accepted_callback_panic_terminalizes_before_quiescence() { -+ let (process, handle, accepted) = accepted_execution(); -+ let observer_process = process.clone(); -+ let observer_thread = handle.as_thread(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = thread::spawn(move || { -+ observer_process.wait_for_execution_quiescence_blocking(); -+ let _ = observed_tx.send(observer_thread.try_join()); -+ }); -+ -+ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ let _ = accepted.admit_successor(|| -> Result<(), ()> { -+ panic!("synthetic accepted-callback panic") -+ }); -+ })); -+ -+ assert!(panic.is_err()); -+ let status = observed_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .unwrap() -+ .expect("accepted callback released quiescence before terminal status"); -+ assert!(status.is_err()); -+ assert_eq!(process.lock().execution_leases, 0); -+ observer.join().unwrap(); -+ drop(handle); -+ } -+ -+ #[test] -+ fn rejected_successor_keeps_predecessor_until_fail_closed_drop() { -+ let (process, handle, accepted) = accepted_execution(); -+ -+ let result = accepted.admit_successor(|| { -+ assert_eq!(process.lock().execution_leases, 1); -+ Err::<(), _>("synthetic admission rejection") -+ }); -+ -+ assert_eq!(result.unwrap_err(), "synthetic admission rejection"); -+ assert!(handle.try_join().is_some_and(|status| status.is_err())); -+ assert_eq!(process.lock().execution_leases, 0); -+ drop(handle); -+ } -+ -+ #[test] -+ fn deep_sleep_handoff_keeps_thread_live_until_successor_guard_finishes() { -+ let (process, handle, accepted) = accepted_execution(); -+ let mut successor = None; -+ -+ accepted -+ .admit_successor(|| { -+ assert_eq!(process.lock().execution_leases, 1); -+ let lease = process.acquire_execution_lease().unwrap(); -+ successor = Some(TaskWasmExecutionGuard::new(handle.as_thread(), lease)); -+ assert_eq!(process.lock().execution_leases, 2); -+ Ok::<_, ()>(()) -+ }) -+ .unwrap(); -+ -+ assert!(handle.try_join().is_none()); -+ assert_eq!(process.lock().execution_leases, 1); -+ drop(successor); -+ assert!(handle.try_join().is_some_and(|status| status.is_err())); -+ assert_eq!(process.lock().execution_leases, 0); -+ drop(handle); -+ } -+} -diff --git a/lib/wasix/src/runtime/task_manager/tokio.rs b/lib/wasix/src/runtime/task_manager/tokio.rs -index e390436..36c1785 100644 ---- a/lib/wasix/src/runtime/task_manager/tokio.rs -+++ b/lib/wasix/src/runtime/task_manager/tokio.rs -@@ -1,5 +1,5 @@ - use std::sync::Mutex; --use std::{num::NonZeroUsize, pin::Pin, sync::Arc, time::Duration}; -+use std::{pin::Pin, sync::Arc, time::Duration}; - - use futures::{Future, future::BoxFuture}; - use tokio::runtime::{Handle, Runtime}; -@@ -74,6 +74,33 @@ impl std::fmt::Debug for ThreadPool { - pub struct TokioTaskManager { - rt: RuntimeOrHandle, - pool: Arc, -+ config: TokioTaskManagerConfig, -+} -+ -+/// Host worker policy for blocking WASIX tasks. -+/// -+/// A small persistent core avoids retaining one host thread and allocator arena -+/// for every guest task ever observed. The pool can still grow to `max_threads` -+/// while guests are active, and excess workers retire after `idle_timeout`. -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+pub struct TokioTaskManagerConfig { -+ pub core_threads: usize, -+ pub max_threads: usize, -+ pub idle_timeout: Duration, -+} -+ -+impl Default for TokioTaskManagerConfig { -+ fn default() -> Self { -+ let concurrency = std::thread::available_parallelism() -+ .map(usize::from) -+ .unwrap_or(1); -+ -+ Self { -+ core_threads: 1, -+ max_threads: 200usize.max(concurrency.saturating_mul(100)), -+ idle_timeout: Duration::from_secs(1), -+ } -+ } - } - - impl TokioTaskManager { -@@ -81,20 +108,33 @@ impl TokioTaskManager { - where - I: Into, - { -- let concurrency = std::thread::available_parallelism() -- .unwrap_or(NonZeroUsize::new(1).unwrap()) -- .get(); -- let max_threads = 200usize.max(concurrency * 100); -+ Self::new_with_config(rt, TokioTaskManagerConfig::default()) -+ } -+ -+ pub fn new_with_config(rt: I, config: TokioTaskManagerConfig) -> Self -+ where -+ I: Into, -+ { -+ assert!( -+ config.core_threads > 0, -+ "core_threads must be greater than 0" -+ ); -+ assert!( -+ config.max_threads >= config.core_threads, -+ "max_threads must be greater than or equal to core_threads" -+ ); - - Self { - rt: rt.into(), - pool: Arc::new(ThreadPool { - inner: rusty_pool::Builder::new() - .name("TokioTaskManager Thread Pool".to_string()) -- .core_size(max_threads) -- .max_size(max_threads) -+ .core_size(config.core_threads) -+ .max_size(config.max_threads) -+ .keep_alive(config.idle_timeout) - .build(), - }), -+ config, - } - } - -@@ -105,6 +145,10 @@ impl TokioTaskManager { - pub fn pool_handle(&self) -> Arc { - self.pool.clone() - } -+ -+ pub fn config(&self) -> TokioTaskManagerConfig { -+ self.config -+ } - } - - impl Default for TokioTaskManager { -@@ -140,10 +184,19 @@ impl VirtualTaskManager for TokioTaskManager { - - /// See [`VirtualTaskManager::task_wasm`]. - fn task_wasm(&self, task: TaskWasm) -> Result<(), WasiThreadError> { -+ // This must precede memory creation and instance construction: Wasmer -+ // may execute a module start function while instantiating. -+ task.validate_lifecycle()?; - let run = task.run; - let recycle = task.recycle; - let env = task.env; -+ let lifecycle_thread = env.thread.clone(); - let pre_run = task.pre_run; -+ let preinitialized_memory_image = task.preinitialized_memory_image; -+ let execution_guard = task -+ .execution_lease -+ .expect("TaskWasm lifecycle was validated above"); -+ let start_gate = task.start_gate; - - let make_memory: SpawnMemoryTypeOrStore = match &task.spawn_type { - SpawnType::CreateMemory | SpawnType::NewLinkerInstanceGroup(..) => { -@@ -152,7 +205,15 @@ impl VirtualTaskManager for TokioTaskManager { - SpawnType::CreateMemoryOfType(t) => SpawnMemoryTypeOrStore::Type(*t), - SpawnType::ShareMemory(_, _) | SpawnType::CopyMemory(_, _) => { - let mut store = env.runtime().new_store(); -- let memory = self.build_memory(&mut store.as_store_mut(), &task.spawn_type)?; -+ let memory = self -+ .build_memory(&mut store.as_store_mut(), &task.spawn_type) -+ .map_err(|err| { -+ lifecycle_thread.set_status_finished(Err(crate::RuntimeError::new( -+ format!("Wasm task memory construction failed: {err}"), -+ ) -+ .into())); -+ err -+ })?; - SpawnMemoryTypeOrStore::StoreAndMemory(store, memory) - } - }; -@@ -161,9 +222,8 @@ impl VirtualTaskManager for TokioTaskManager { - // See the comment below for why we can't do it there yet. - // - // For now block_in_place at least ensures that we don't block the async runtime -- let ret = tokio::task::block_in_place(move || { -- if let SpawnType::NewLinkerInstanceGroup(linker, func_env, mut store) = task.spawn_type -- { -+ let ret = tokio::task::block_in_place(move || match task.spawn_type { -+ SpawnType::NewLinkerInstanceGroup(linker, func_env, mut store) => { - WasiFunctionEnv::new_with_store( - task.module, - env, -@@ -171,21 +231,37 @@ impl VirtualTaskManager for TokioTaskManager { - make_memory, - task.update_layout, - task.call_initialize, -+ preinitialized_memory_image, - Some((linker, &mut func_env.into_mut(&mut store))), - ) -- } else { -- WasiFunctionEnv::new_with_store( -- task.module, -- env, -- task.globals, -- make_memory, -- task.update_layout, -- task.call_initialize, -- None, -- ) - } -+ _ => WasiFunctionEnv::new_with_store( -+ task.module, -+ env, -+ task.globals, -+ make_memory, -+ task.update_layout, -+ task.call_initialize, -+ preinitialized_memory_image, -+ None, -+ ), - }); - -+ // `new_with_store` can fail after a task has been accepted but before -+ // its run callback (and its normal run guard) exists. Publish terminal -+ // status while the execution lease is still held instead of relying -+ // on environment/handle field drop order. -+ let (mut ctx, mut store) = match ret { -+ Ok(context) => context, -+ Err(err) => { -+ lifecycle_thread.set_status_finished(Err(crate::RuntimeError::new(format!( -+ "Wasm task instantiation failed: {err}" -+ )) -+ .into())); -+ return Err(err); -+ } -+ }; -+ - if let Some(trigger) = task.trigger { - tracing::trace!("spawning task_wasm trigger in async pool"); - // In principle, we'd need to create this in the `pool.execute` function below, that is -@@ -217,11 +293,11 @@ impl VirtualTaskManager for TokioTaskManager { - // pool's one), and let it fail for runtimes that don't support entities created in a - // thread that's not the one in which execution happens in; this until we can clone - // stores. -- let (mut ctx, mut store) = ret?; -- - let mut trigger = trigger(); - let pool = self.pool.clone(); - self.rt.handle().spawn(async move { -+ debug_assert!(start_gate.is_committed()); -+ - // We wait for either the trigger or for a snapshot to take place - let result = loop { - let env = ctx.data(&store); -@@ -264,12 +340,17 @@ impl VirtualTaskManager for TokioTaskManager { - - // Build the task that will go on the callback - pool.execute(move || { -- // Invoke the callback -+ // This is the sole successful disarm boundary: all -+ // fallible setup and queueing stages are behind us, and -+ // the accepted callback now owns terminal/successor state. -+ let execution_guard = -+ execution_guard.accept_callback(ctx.data_mut(&mut store)); - run(TaskWasmRunProperties { - ctx, - store, - trigger_result: Some(result), - recycle, -+ execution_guard: Some(execution_guard), - }); - }); - }); -@@ -281,32 +362,26 @@ impl VirtualTaskManager for TokioTaskManager { - // Run the callback on a dedicated thread - self.pool.execute(move || { - tracing::trace!("task_wasm started in blocking thread"); -- let (mut ctx, mut store) = match ret { -- Ok(x) => { -- sx.send(Ok(())).unwrap(); -- x -- } -- Err(c) => { -- sx.send(Err(c)).unwrap(); -- return; -- } -- }; -+ let _ = sx.send(()); -+ -+ debug_assert!(start_gate.is_committed()); - - if let Some(pre_run) = pre_run { - block_on(pre_run(&mut ctx, &mut store)); - } - -- // Invoke the callback -+ // Disarm only after the queue and pre-run stages completed. -+ let execution_guard = execution_guard.accept_callback(ctx.data_mut(&mut store)); - run(TaskWasmRunProperties { - ctx, - store, - trigger_result: None, - recycle, -+ execution_guard: Some(execution_guard), - }); - }); - -- rx.recv() -- .map_err(|_| WasiThreadError::InvalidWasmContext)??; -+ rx.recv().map_err(|_| WasiThreadError::InvalidWasmContext)?; - } - Ok(()) - } -@@ -361,3 +436,490 @@ impl Drop for SleepNow { - } - } - } -+ -+#[cfg(test)] -+mod tests { -+ use std::sync::atomic::{AtomicBool, Ordering}; -+ use std::sync::{Arc, Barrier, Mutex, mpsc}; -+ use std::time::{Duration, Instant}; -+ -+ use wasmer::{AsStoreRef, Engine, Function, Imports, Memory, MemoryType, Module, Store}; -+ use wasmer_wasix_types::wasi::Errno; -+ -+ use crate::{ -+ PluggableRuntime, WasiControlPlane, WasiEnv, WasiProcess, WasiThreadError, -+ runtime::task_manager::{SpawnType, TaskWasm}, -+ }; -+ -+ use super::{TokioTaskManager, TokioTaskManagerConfig}; -+ -+ fn task_wasm_test_env( -+ runtime: &tokio::runtime::Runtime, -+ engine: &Engine, -+ name: &str, -+ ) -> WasiEnv { -+ let mut wasix_runtime = -+ PluggableRuntime::new(Arc::new(TokioTaskManager::new(runtime.handle().clone()))); -+ wasix_runtime.set_engine(engine.clone()); -+ WasiEnv::builder(name) -+ .runtime(Arc::new(wasix_runtime)) -+ .build() -+ .unwrap() -+ } -+ -+ #[test] -+ fn non_core_workers_retire_after_the_idle_timeout() { -+ const TASKS: usize = 8; -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let config = TokioTaskManagerConfig { -+ core_threads: 1, -+ max_threads: TASKS, -+ idle_timeout: Duration::from_millis(20), -+ }; -+ let manager = TokioTaskManager::new_with_config(runtime.handle().clone(), config); -+ let pool = manager.pool_handle(); -+ let barrier = Arc::new(Barrier::new(TASKS + 1)); -+ -+ for _ in 0..TASKS { -+ let barrier = Arc::clone(&barrier); -+ pool.execute(move || { -+ barrier.wait(); -+ }); -+ } -+ -+ barrier.wait(); -+ pool.join(); -+ assert_eq!(manager.config(), config); -+ assert_eq!(pool.get_current_worker_count(), TASKS); -+ -+ let deadline = Instant::now() + Duration::from_secs(2); -+ while pool.get_current_worker_count() != config.core_threads && Instant::now() < deadline { -+ std::thread::sleep(Duration::from_millis(10)); -+ } -+ -+ assert_eq!(pool.get_current_worker_count(), config.core_threads); -+ } -+ -+ #[test] -+ fn memory_construction_failure_terminalizes_before_releasing_lease() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ let env = task_wasm_test_env(&runtime, &engine, "memory-failure-test"); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let tasks = env.tasks().clone(); -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (memory (export "memory") 1) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let mut old_store = Store::new(engine.clone()); -+ let non_shared = Memory::new(&mut old_store, MemoryType::new(1, Some(1), false)).unwrap(); -+ let run_called = Arc::new(AtomicBool::new(false)); -+ let run_called_inner = Arc::clone(&run_called); -+ -+ let result = tasks.task_wasm( -+ TaskWasm::new( -+ Box::new(move |_| run_called_inner.store(true, Ordering::Release)), -+ env, -+ module, -+ false, -+ false, -+ ) -+ .with_memory(SpawnType::ShareMemory(non_shared, old_store.as_store_ref())), -+ ); -+ -+ assert!(matches!( -+ result, -+ Err(WasiThreadError::MemoryCreateFailed(_)) -+ )); -+ assert!(!run_called.load(Ordering::Acquire)); -+ assert!(thread.try_join().is_some_and(|status| status.is_err())); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn instantiation_failure_terminalizes_before_releasing_lease() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ let env = task_wasm_test_env(&runtime, &engine, "instantiation-failure-test"); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let tasks = env.tasks().clone(); -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (import "wasi_snapshot_preview1" "proc_exit" (func $proc_exit (param i32))) -+ (import "missing" "function" (func $missing)) -+ (memory (export "memory") 1) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let run_called = Arc::new(AtomicBool::new(false)); -+ let run_called_inner = Arc::clone(&run_called); -+ -+ let result = tasks.task_wasm(TaskWasm::new( -+ Box::new(move |_| run_called_inner.store(true, Ordering::Release)), -+ env, -+ module, -+ false, -+ false, -+ )); -+ -+ assert!(result.is_err(), "missing import unexpectedly instantiated"); -+ assert!(!run_called.load(Ordering::Acquire)); -+ assert!(thread.try_join().is_some(), "thread remained nonterminal"); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn panicking_custom_task_wasm_callback_terminalizes_before_quiescence() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ let env = task_wasm_test_env(&runtime, &engine, "callback-panic-test"); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let tasks = env.tasks().clone(); -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (memory (export "memory") 1) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let (entered_tx, entered_rx) = mpsc::channel(); -+ -+ tasks -+ .task_wasm(TaskWasm::new( -+ Box::new(move |_props| { -+ let _ = entered_tx.send(()); -+ panic!("synthetic custom TaskWasmRun panic"); -+ }), -+ env, -+ module, -+ false, -+ false, -+ )) -+ .unwrap(); -+ entered_rx.recv_timeout(Duration::from_secs(2)).unwrap(); -+ -+ let observer_process = process.clone(); -+ let observer_thread = thread.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = std::thread::spawn(move || { -+ observer_process.wait_for_execution_quiescence_blocking(); -+ let _ = observed_tx.send(observer_thread.try_join()); -+ }); -+ let status = observed_rx -+ .recv_timeout(Duration::from_secs(2)) -+ .unwrap() -+ .expect("custom callback released quiescence before terminal status"); -+ -+ assert!(status.is_err()); -+ assert_eq!(process.lock().execution_leases, 0); -+ observer.join().unwrap(); -+ } -+ -+ #[test] -+ fn accepted_restored_process_adopts_its_single_deferred_parent_guard() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ let mut env = task_wasm_test_env(&runtime, &engine, "restored-parent-handoff-test"); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let tasks = env.tasks().clone(); -+ let prior_owner = Arc::new(()); -+ let deferred = process -+ .acquire_supplemental_execution_guard(thread.clone(), Arc::downgrade(&prior_owner)) -+ .unwrap(); -+ deferred.arm_fail_closed(); -+ env.deferred_parent_execution = Some(deferred); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (memory (export "memory") 1) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let callback_process = process.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ tasks -+ .task_wasm(TaskWasm::new( -+ Box::new(move |props| { -+ let env = props.ctx.data(&props.store); -+ observed_tx -+ .send(( -+ env.deferred_parent_execution.is_none(), -+ callback_process.lock().execution_leases, -+ )) -+ .unwrap(); -+ env.thread.set_status_finished(Ok(Errno::Success.into())); -+ }), -+ env, -+ module, -+ false, -+ false, -+ )) -+ .unwrap(); -+ -+ assert_eq!( -+ observed_rx.recv_timeout(Duration::from_secs(2)).unwrap(), -+ (true, 1) -+ ); -+ process.wait_for_execution_quiescence_blocking(); -+ assert_eq!(process.lock().execution_leases, 0); -+ assert!(thread.try_join().is_some()); -+ } -+ -+ #[test] -+ fn task_wasm_constructor_binds_owner_before_manager_instantiation() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ let env = task_wasm_test_env(&runtime, &engine, "constructor-owner-bind-test"); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (memory (export "memory") 1) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let mut task = TaskWasm::new(Box::new(|_| unreachable!()), env, module, false, false); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ let supplemental = task.env.acquire_parent_execution_guard().unwrap(); -+ assert_eq!(process.lock().execution_leases, 2); -+ task.env.restore_parent_execution_guard(supplemental); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ drop(task); -+ process.wait_for_execution_quiescence_blocking(); -+ assert!(thread.try_join().is_some_and(|status| status.is_err())); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn pending_task_drop_with_deferred_owner_is_fail_closed_and_bounded() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ let mut env = task_wasm_test_env(&runtime, &engine, "deferred-queue-drop-test"); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let predecessor = Arc::new(()); -+ env.bind_pending_task_wasm_execution(&thread, &predecessor); -+ let deferred = env.acquire_parent_execution_guard().unwrap(); -+ deferred.arm_fail_closed(); -+ env.deferred_parent_execution = Some(deferred); -+ drop(predecessor); -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (memory (export "memory") 1) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let task = TaskWasm::new(Box::new(|_| unreachable!()), env, module, false, false); -+ assert_eq!(process.lock().execution_leases, 2); -+ -+ drop(task); -+ -+ process.wait_for_execution_quiescence_blocking(); -+ assert!(thread.try_join().is_some()); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn foreign_pending_token_cannot_accept_same_guest_thread() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ let env = task_wasm_test_env(&runtime, &engine, "foreign-owner-bind-test"); -+ let process = env.process.clone(); -+ let thread = env.thread.clone(); -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (memory (export "memory") 1) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ let task_a = TaskWasm::new( -+ Box::new(|_| unreachable!()), -+ env.clone(), -+ module.clone(), -+ false, -+ false, -+ ); -+ let mut task_b = TaskWasm::new(Box::new(|_| unreachable!()), env, module, false, false); -+ assert_eq!(process.lock().execution_leases, 2); -+ let pending_a = task_a.execution_lease.unwrap(); -+ -+ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { -+ let _ = pending_a.accept_callback(&mut task_b.env); -+ })); -+ assert!(panic.is_err()); -+ assert!(thread.try_join().is_some_and(|status| status.is_err())); -+ assert_eq!(process.lock().execution_leases, 1); -+ -+ let observer_process = process.clone(); -+ let observer_thread = thread.clone(); -+ let (observed_tx, observed_rx) = mpsc::channel(); -+ let observer = std::thread::spawn(move || { -+ observer_process.wait_for_execution_quiescence_blocking(); -+ observed_tx.send(observer_thread.try_join()).unwrap(); -+ }); -+ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); -+ drop(task_b); -+ assert!( -+ observed_rx -+ .recv_timeout(Duration::from_secs(1)) -+ .unwrap() -+ .is_some_and(|status| status.is_err()) -+ ); -+ observer.join().unwrap(); -+ assert_eq!(process.lock().execution_leases, 0); -+ } -+ -+ #[test] -+ fn module_start_observes_child_parent_after_exact_publication() { -+ let runtime = tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ let _runtime_guard = runtime.enter(); -+ let engine = Engine::default(); -+ -+ type ExpectedPublication = (WasiProcess, WasiProcess, WasiControlPlane); -+ let expected_publication = Arc::new(Mutex::new(None::)); -+ let (start_observed_tx, start_observed_rx) = mpsc::channel(); -+ let mut wasix_runtime = -+ PluggableRuntime::new(Arc::new(TokioTaskManager::new(runtime.handle().clone()))); -+ wasix_runtime.set_engine(engine.clone()); -+ let expected_for_import = Arc::clone(&expected_publication); -+ wasix_runtime.with_additional_imports(move |_module, store| { -+ let expected_for_start = Arc::clone(&expected_for_import); -+ let start_observed_tx = start_observed_tx.clone(); -+ let observe_publication = Function::new_typed(store, move || { -+ let expected = expected_for_start.lock().unwrap(); -+ let (parent, child, plane) = expected -+ .as_ref() -+ .expect("module start ran before the test installed its expectation"); -+ let registered = plane -+ .get_process(child.pid()) -+ .is_some_and(|process| process.same_identity(child)); -+ let adopted = parent -+ .lock() -+ .children -+ .iter() -+ .any(|process| process.same_identity(child)); -+ start_observed_tx -+ .send((child.ppid().raw(), registered, adopted)) -+ .unwrap(); -+ }); -+ let mut imports = Imports::new(); -+ imports.define("test", "observe_publication", observe_publication); -+ Ok(imports) -+ }); -+ let parent_env = WasiEnv::builder("publication-before-start-test") -+ .runtime(Arc::new(wasix_runtime)) -+ .build() -+ .unwrap(); -+ let parent = parent_env.process.clone(); -+ let plane = parent_env.control_plane.clone(); -+ let tasks = parent_env.tasks().clone(); -+ let (child_env, child_handle, mut registration) = parent_env.fork_guarded().unwrap(); -+ let child = child_env.process.clone(); -+ -+ let module = Module::new( -+ &engine, -+ br#"(module -+ (import "test" "observe_publication" (func $observe_publication)) -+ (memory (export "memory") 1) -+ (func $constructor call $observe_publication) -+ (start $constructor) -+ (func (export "_start")))"#, -+ ) -+ .unwrap(); -+ -+ assert!(matches!( -+ TaskWasm::new( -+ Box::new(|_| unreachable!()), -+ child_env.clone(), -+ module.clone(), -+ false, -+ false -+ ) -+ .validate_lifecycle(), -+ Err(crate::WasiThreadError::ProcessLifecycle( -+ crate::os::task::control_plane::ControlPlaneError::ProcessNotPublished { .. } -+ )) -+ )); -+ -+ registration.commit_child().unwrap(); -+ *expected_publication.lock().unwrap() = Some((parent.clone(), child, plane)); -+ let (run_tx, run_rx) = mpsc::channel(); -+ let expected_parent_pid = parent.pid().raw(); -+ let run = move |props: crate::runtime::task_manager::TaskWasmRunProperties| { -+ props -+ .ctx -+ .data(&props.store) -+ .thread -+ .set_status_finished(Ok(Errno::Success.into())); -+ run_tx.send(()).unwrap(); -+ }; -+ -+ tasks -+ .task_wasm(TaskWasm::new( -+ Box::new(run), -+ child_env, -+ module, -+ false, -+ false, -+ )) -+ .unwrap(); -+ registration.complete_child_launch(&tasks); -+ let (observed_parent, registered, adopted) = start_observed_rx -+ .recv_timeout(Duration::from_secs(2)) -+ .unwrap(); -+ assert_eq!(observed_parent, expected_parent_pid); -+ assert!(registered); -+ assert!(adopted); -+ run_rx.recv_timeout(Duration::from_secs(2)).unwrap(); -+ drop(child_handle); -+ } -+} -diff --git a/lib/wasix/src/state/builder.rs b/lib/wasix/src/state/builder.rs -index b66c0e1..9dc9baf 100644 ---- a/lib/wasix/src/state/builder.rs -+++ b/lib/wasix/src/state/builder.rs -@@ -23,7 +23,7 @@ use crate::{ - fs::{WasiFs, WasiFsRoot, WasiInodes}, - os::command::VirtualCommand, - os::task::control_plane::{ControlPlaneConfig, ControlPlaneError, WasiControlPlane}, -- state::WasiState, -+ state::{PreinitializedMemoryImageHandle, WasiState}, - syscalls::types::{__WASI_STDERR_FILENO, __WASI_STDIN_FILENO, __WASI_STDOUT_FILENO}, - }; - use wasmer_types::ModuleHash; -@@ -79,6 +79,11 @@ pub struct WasiEnvBuilder { - - pub(super) module_hash: Option, - -+ /// Exact immutable guest executable aliases. Modules are resolved lazily -+ /// from the runtime's authoritative cache by hash. -+ pub(super) sealed_modules: -+ HashMap)>, -+ - /// List of host commands to map into the WASI instance. - pub(super) map_commands: HashMap, - /// Indicates if internal builtin commands should be disabled. -@@ -127,6 +132,7 @@ impl std::fmt::Debug for WasiEnvBuilder { - .field("builtin_commands_count", &self.builtin_commands.len()) - .field("engine_override_exists", &self.engine.is_some()) - .field("runtime_override_exists", &self.runtime.is_some()) -+ .field("sealed_module_count", &self.sealed_modules.len()) - .finish() - } - } -@@ -394,6 +400,23 @@ impl WasiEnvBuilder { - self - } - -+ /// Registers an immutable guest executable for this process tree. -+ /// -+ /// Once any path is registered, exact matches are resolved by hash and all -+ /// other executable paths are denied before guest-filesystem lookup. -+ pub fn add_sealed_module( -+ &mut self, -+ guest_path: impl Into, -+ module_hash: ModuleHash, -+ preinitialized_memory_image: Option, -+ ) -> &mut Self { -+ self.sealed_modules.insert( -+ guest_path.into(), -+ (module_hash, preinitialized_memory_image), -+ ); -+ self -+ } -+ - /// Adds a container this module inherits from. - /// - /// This will make all of the container's files and commands available to the -@@ -1004,6 +1027,10 @@ impl WasiEnvBuilder { - clock_offset: Default::default(), - envs: std::sync::Mutex::new(conv_env_vars(self.envs)), - signals: std::sync::Mutex::new(self.signals.iter().map(|s| (s.sig, s.disp)).collect()), -+ shared_futex_registries: Default::default(), -+ shared_memory_mappings: Default::default(), -+ shared_memory_exec_phase: Default::default(), -+ shared_memory_exec_reservations: Default::default(), - }; - - let runtime = self.runtime.unwrap_or_else(|| { -@@ -1046,7 +1073,8 @@ impl WasiEnvBuilder { - let disable_default_builtins = self.disable_default_builtins; - let builtin_commands = self.builtin_commands; - -- let mut bin_factory = BinFactory::new(runtime.clone()); -+ let mut bin_factory = -+ BinFactory::new_with_sealed_modules(runtime.clone(), self.sealed_modules); - if disable_default_builtins { - bin_factory.clear_builtin_commands(); - } -@@ -1148,7 +1176,7 @@ impl WasiEnvBuilder { - .map(|ty| wasmer::Memory::new(store, ty)) - .transpose() - .map_err(WasiThreadError::MemoryCreateFailed)?; -- Ok(env.instantiate(module, store, memory, true, call_init, None)?) -+ Ok(env.instantiate(module, store, memory, true, call_init, None, false, None)?) - } - } - -diff --git a/lib/wasix/src/state/env.rs b/lib/wasix/src/state/env.rs -index 783291f..915aa13 100644 ---- a/lib/wasix/src/state/env.rs -+++ b/lib/wasix/src/state/env.rs -@@ -1,5 +1,6 @@ - #[cfg(feature = "journal")] - use crate::journal::{DynJournal, JournalEffector, SnapshotTrigger}; -+use crate::runtime::task_manager::TaskWasmAcceptedExecutionGuard; - use crate::{ - Runtime, VirtualTaskManager, WasiControlPlane, WasiEnvBuilder, WasiError, WasiFunctionEnv, - WasiResult, WasiRuntimeError, WasiStateCreationError, WasiThreadError, WasiVFork, -@@ -8,8 +9,8 @@ use crate::{ - fs::{WasiFsRoot, WasiInodes}, - import_object_for_all_wasi_versions, - os::task::{ -- control_plane::ControlPlaneError, -- process::{WasiProcess, WasiProcessId}, -+ control_plane::{ControlPlaneError, WasiProcessRegistrationGuard}, -+ process::{WasiProcess, WasiProcessExecutionGuard, WasiProcessId}, - thread::{WasiMemoryLayout, WasiThread, WasiThreadHandle, WasiThreadId}, - }, - syscalls::platform_clock_time_get, -@@ -21,7 +22,7 @@ use std::{ - ops::Deref, - path::{Path, PathBuf}, - str, -- sync::Arc, -+ sync::{Arc, Weak}, - time::Duration, - }; - use virtual_fs::{FileSystem, FsError, VirtualFile}; -@@ -41,7 +42,10 @@ use wasmer_wasix_types::{ - use webc::metadata::annotations::Wasi; - - pub use super::handles::*; --use super::{Linker, WasiState, context_switching::ContextSwitchingEnvironment, conv_env_vars}; -+use super::{ -+ Linker, PreinitializedMemoryImageMode, WasiState, -+ context_switching::ContextSwitchingEnvironment, conv_env_vars, -+}; - - async fn write_readonly_buffer_to_fs( - fs: &WasiFsRoot, -@@ -127,6 +131,10 @@ impl WasiEnvInit { - args: std::sync::Mutex::new(self.state.args.lock().unwrap().clone()), - envs: std::sync::Mutex::new(self.state.envs.lock().unwrap().deref().clone()), - signals: std::sync::Mutex::new(self.state.signals.lock().unwrap().deref().clone()), -+ shared_futex_registries: Default::default(), -+ shared_memory_mappings: Default::default(), -+ shared_memory_exec_phase: Default::default(), -+ shared_memory_exec_reservations: Default::default(), - preopen: self.state.preopen.clone(), - }, - runtime: self.runtime.clone(), -@@ -171,6 +179,14 @@ pub struct WasiEnv { - /// List of the handles that are owned by this context - /// (this can be used to ensure that threads own themselves or others) - pub owned_handles: Vec, -+ /// The sole deferred execution owner retained across an in-place vfork -+ /// process switch. Nested vfork is unsupported, so an `Option` encodes the -+ /// real cardinality and makes per-switch accumulation impossible. -+ pub(crate) deferred_parent_execution: Option, -+ /// Weak identity of the accepted TaskWasm that owns the physical module -+ /// instance currently stored in `inner`. It is deliberately not cloned; -+ /// `swap_inner` moves it with that physical instance across vfork. -+ current_task_wasm_owner: Weak<()>, - /// Implementation of the WASI runtime. - pub runtime: Arc, - -@@ -231,6 +247,10 @@ impl Clone for WasiEnv { - bin_factory: self.bin_factory.clone(), - inner: Default::default(), - owned_handles: self.owned_handles.clone(), -+ // Generic clones are non-owning. Only an explicitly declared -+ // same-thread continuation may alias a deferred vfork owner. -+ deferred_parent_execution: None, -+ current_task_wasm_owner: Weak::new(), - runtime: self.runtime.clone(), - capabilities: self.capabilities.clone(), - enable_deep_sleep: self.enable_deep_sleep, -@@ -244,7 +264,136 @@ impl Clone for WasiEnv { - } - } - -+fn retire_finished_process_tree( -+ control_plane: &WasiControlPlane, -+ process: &WasiProcess, -+) -> Result<(), ControlPlaneError> { -+ let descendants = process.begin_epoch_retirement()?; -+ for descendant in &descendants { -+ control_plane.retire_process_epoch(descendant)?; -+ descendant.retire_epoch_task_registrations(); -+ } -+ control_plane.retire_process_epoch(process)?; -+ process.retire_epoch_task_registrations(); -+ for descendant in &descendants { -+ descendant.clear_retired_children(); -+ } -+ process.clear_retired_children(); -+ Ok(()) -+} -+ - impl WasiEnv { -+ /// Binds a pending TaskWasm's unforgeable identity before Wasmer can run -+ /// guest start functions during instantiation. The pending guard remains -+ /// fail-closed; this only establishes which physical execution may request -+ /// an in-place vfork ownership handoff. -+ pub(crate) fn bind_pending_task_wasm_execution( -+ &mut self, -+ thread: &WasiThread, -+ owner_identity: &Arc<()>, -+ ) { -+ assert!( -+ thread.same_identity(&self.thread), -+ "pending TaskWasm thread does not match its environment" -+ ); -+ let owner = Arc::downgrade(owner_identity); -+ assert!( -+ self.current_task_wasm_owner.upgrade().is_none() -+ || Weak::ptr_eq(&self.current_task_wasm_owner, &owner), -+ "environment is already bound to a different TaskWasm" -+ ); -+ self.current_task_wasm_owner = owner; -+ } -+ -+ /// Transfers bounded process-switch ownership to an accepted TaskWasm for -+ /// the exact restored guest thread. The accepted successor already holds -+ /// its own lease, so predecessor release remains successor-before- -+ /// predecessor without consulting process-wide lease counts. -+ pub(crate) fn accept_task_wasm_execution(&mut self, accepted: &TaskWasmAcceptedExecutionGuard) { -+ assert!( -+ accepted.thread().same_identity(&self.thread), -+ "accepted TaskWasm thread does not match its environment" -+ ); -+ let accepted_owner = Arc::downgrade(accepted.owner_identity()); -+ assert!( -+ Weak::ptr_eq(&self.current_task_wasm_owner, &accepted_owner) -+ && self.current_task_wasm_owner.upgrade().is_some(), -+ "accepted TaskWasm does not match the pending owner bound before instantiation" -+ ); -+ let Some(guard) = self.deferred_parent_execution.take() else { -+ return; -+ }; -+ assert!( -+ guard.try_handoff_to_accepted_task_wasm(&self.process, accepted.thread()), -+ "deferred parent execution does not match its accepted successor" -+ ); -+ } -+ -+ pub(crate) fn clear_task_wasm_execution(&mut self) { -+ self.current_task_wasm_owner = Weak::new(); -+ self.deferred_parent_execution.take(); -+ } -+ -+ /// Restores a supplemental owner after switching back to its process. The -+ /// common path hands it directly to the still-accepted TaskWasm. If no -+ /// such successor exists, one fail-closed owner is retained until a later -+ /// accepted callback can adopt it. -+ pub(crate) fn restore_parent_execution_guard(&mut self, guard: WasiProcessExecutionGuard) { -+ assert!( -+ guard.matches_process_thread(&self.process, &self.thread), -+ "restored parent execution does not match its environment" -+ ); -+ if guard.try_handoff_to_current_task_wasm( -+ &self.process, -+ &self.thread, -+ &self.current_task_wasm_owner, -+ ) { -+ return; -+ } -+ assert!( -+ self.deferred_parent_execution.is_none(), -+ "nested deferred parent execution ownership is unsupported" -+ ); -+ self.deferred_parent_execution = Some(guard); -+ } -+ -+ pub(crate) fn acquire_parent_execution_guard( -+ &self, -+ ) -> Result { -+ self.process.acquire_supplemental_execution_guard( -+ self.thread.clone(), -+ self.current_task_wasm_owner.clone(), -+ ) -+ } -+ -+ /// Clones process-visible WASI state for a newly spawned guest thread -+ /// without copying execution ownership that belongs to the calling thread. -+ pub(crate) fn clone_for_thread_spawn( -+ &self, -+ thread: WasiThread, -+ layout: WasiMemoryLayout, -+ ) -> Self { -+ assert!( -+ self.vfork.is_none(), -+ "a vfork child cannot spawn a guest thread" -+ ); -+ let mut env = self.clone(); -+ env.deferred_parent_execution.take(); -+ env.current_task_wasm_owner = Weak::new(); -+ env.thread = thread; -+ env.layout = layout; -+ env -+ } -+ -+ /// Clones an environment for a successor TaskWasm of the exact same guest -+ /// thread. This is the only clone path allowed to carry the one deferred -+ /// parent execution owner across deep sleep or exec continuation. -+ pub(crate) fn take_for_same_thread_continuation(&mut self) -> Self { -+ let mut env = self.clone(); -+ env.deferred_parent_execution = self.deferred_parent_execution.take(); -+ env -+ } -+ - /// Construct a new [`WasiEnvBuilder`] that allows customizing an environment. - pub fn builder(program_name: impl Into) -> WasiEnvBuilder { - WasiEnvBuilder::new(program_name) -@@ -252,13 +401,32 @@ impl WasiEnv { - - /// Forking the WasiState is used when either fork or vfork is called - pub fn fork(&self) -> Result<(Self, WasiThreadHandle), ControlPlaneError> { -- let process = self.control_plane.new_process(self.process.module_hash)?; -- let handle = process.new_thread(self.layout.clone(), ThreadStartType::MainThread)?; -+ let (env, handle, mut registration) = self.fork_guarded()?; -+ registration.commit_child()?; -+ registration.complete_child_launch(self.tasks()); -+ Ok((env, handle)) -+ } -+ -+ /// Internal fork transaction whose process registration is committed only -+ /// when the syscall crosses its externally observable success boundary. -+ pub(crate) fn fork_guarded( -+ &self, -+ ) -> Result<(Self, WasiThreadHandle, WasiProcessRegistrationGuard), ControlPlaneError> { -+ let (process, handle, registration) = self -+ .control_plane -+ .new_child_process_with_main_thread_guarded( -+ &self.process, -+ self.process.module_hash, -+ self.layout.clone(), -+ )?; - - let thread = handle.as_thread(); - thread.copy_stack_from(&self.thread); - -- let state = Arc::new(self.state.fork()); -+ let state = self.state.fork_with(|_| {}).map_err(|errno| { -+ tracing::warn!(%errno, "could not fork process state"); -+ ControlPlaneError::SharedMemoryForkUnavailable -+ })?; - - let bin_factory = self.bin_factory.clone(); - -@@ -272,7 +440,11 @@ impl WasiEnv { - bin_factory, - state, - inner: Default::default(), -- owned_handles: Vec::new(), -+ // The child owns its main-thread lifetime. Parent environments -+ // must never accumulate handles for already-joined children. -+ owned_handles: vec![handle.clone()], -+ deferred_parent_execution: None, -+ current_task_wasm_owner: Weak::new(), - runtime: self.runtime.clone(), - capabilities: self.capabilities.clone(), - enable_deep_sleep: self.enable_deep_sleep, -@@ -283,7 +455,7 @@ impl WasiEnv { - disable_fs_cleanup: self.disable_fs_cleanup, - context_switching_environment: None, - }; -- Ok((new_env, handle)) -+ Ok((new_env, handle, registration)) - } - - pub fn pid(&self) -> WasiProcessId { -@@ -297,22 +469,35 @@ impl WasiEnv { - /// Returns true if this WASM process will need and try to use - /// asyncify while its running which normally means. - pub fn will_use_asyncify(&self) -> bool { -- self.inner() -- .static_module_instance_handles() -- .map(|handles| self.enable_deep_sleep || handles.has_stack_checkpoint) -- .unwrap_or(false) -+ let handles = self.inner().main_module_instance_handles(); -+ self.enable_deep_sleep || handles.has_stack_checkpoint - } - - /// Re-initializes this environment so that it can be executed again - pub fn reinit(&mut self) -> Result<(), WasiStateCreationError> { -+ // Verify the whole old process tree before mutating registry, handle, -+ // descriptor, or task-count state. Rejection therefore leaves the -+ // reusable environment exactly as it was. Success seals the epoch; -+ // any later filesystem/setup error is terminal for that old epoch and -+ // a second reinit attempt is rejected as ProcessRetiring. -+ retire_finished_process_tree(&self.control_plane, &self.process)?; -+ self.owned_handles.clear(); -+ self.deferred_parent_execution.take(); -+ self.current_task_wasm_owner = Weak::new(); -+ - // If the cleanup logic is enabled then we need to rebuild the - // file descriptors which would have been destroyed when the - // main thread exited - if !self.disable_fs_cleanup { - // First we clear any open files as the descriptors would - // otherwise clash -- if let Ok(mut map) = self.state.fs.fd_map.write() { -- map.clear(); -+ let removed = if let Ok(mut map) = self.state.fs.fd_map.write() { -+ map.drain_deferred() -+ } else { -+ Vec::new() -+ }; -+ for fd in removed { -+ fd.release_descriptor(); - } - self.state.fs.preopen_fds.write().unwrap().clear(); - *self.state.fs.current_dir.lock().unwrap() = "/".to_string(); -@@ -331,21 +516,17 @@ impl WasiEnv { - .map_err(WasiStateCreationError::WasiFsSetupError)?; - } - -- // The process and thread state need to be reset -- self.process = WasiProcess::new( -- self.process.pid, -- self.process.module_hash, -- self.process.compute.clone(), -- ); -- self.thread = WasiThread::new( -- self.thread.pid(), -- self.thread.tid(), -- self.thread.is_main(), -- self.process.finished.clone(), -- self.process.compute.must_upgrade().register_task()?, -- self.thread.memory_layout().clone(), -- self.thread.thread_start_type(), -- ); -+ // Allocate a fresh PID and register the replacement main thread through -+ // the same guarded process/thread transaction used by spawn and fork. -+ let module_hash = self.process.module_hash; -+ let (process, handle, mut registration) = self -+ .control_plane -+ .new_process_with_main_thread_guarded(module_hash, self.layout.clone())?; -+ let thread = handle.as_thread(); -+ registration.commit(); -+ self.process = process; -+ self.thread = thread; -+ self.owned_handles.push(handle); - - Ok(()) - } -@@ -377,13 +558,12 @@ impl WasiEnv { - self.try_inner() - .map(|handles| { - handles -- .static_module_instance_handles() -- .map(|handles| { -- handles.asyncify_get_state.is_some() -- && handles.asyncify_start_rewind.is_some() -- && handles.asyncify_start_unwind.is_some() -- }) -- .unwrap_or(false) -+ .main_module_instance_handles() -+ .supports_asyncify_stack_rewind() -+ && handles -+ .main_module_instance_handles() -+ .asyncify_get_state -+ .is_some() - }) - .unwrap_or(false) - } -@@ -398,10 +578,11 @@ impl WasiEnv { - init: WasiEnvInit, - module_hash: ModuleHash, - ) -> Result { -- let process = if let Some(p) = init.process { -- p -+ let (process, process_registration) = if let Some(p) = init.process { -+ (p, None) - } else { -- init.control_plane.new_process(module_hash)? -+ let (process, registration) = init.control_plane.new_process_guarded(module_hash)?; -+ (process, Some(registration)) - }; - - #[cfg(feature = "journal")] -@@ -418,6 +599,7 @@ impl WasiEnv { - process.new_thread(layout.clone(), ThreadStartType::MainThread)? - }; - -+ let state = Arc::new(init.state); - let mut env = Self { - control_plane: init.control_plane, - process, -@@ -425,9 +607,11 @@ impl WasiEnv { - layout, - vfork: None, - poll_seed: 0, -- state: Arc::new(init.state), -+ state, - inner: Default::default(), - owned_handles: Vec::new(), -+ deferred_parent_execution: None, -+ current_task_wasm_owner: Weak::new(), - #[cfg(feature = "journal")] - enable_journal: init.runtime.active_journal().is_some(), - #[cfg(not(feature = "journal"))] -@@ -455,6 +639,10 @@ impl WasiEnv { - #[cfg(feature = "sys")] - env.map_commands(init.mapped_commands.clone())?; - -+ if let Some(mut registration) = process_registration { -+ registration.commit(); -+ } -+ - Ok(env) - } - -@@ -467,15 +655,40 @@ impl WasiEnv { - memory: Option, - update_layout: bool, - call_initialize: bool, -- parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut)>, -+ preinitialized_memory_image: Option, -+ fresh_zeroed_memory: bool, -+ parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut<'_, WasiEnv>)>, - ) -> Result<(Instance, WasiFunctionEnv), WasiThreadError> { - let pid = self.process.pid(); - - let mut store = store.as_store_mut(); - let engine = self.runtime().engine(); -- let mut func_env = WasiFunctionEnv::new(&mut store, self); -+ let mut func_env = { WasiFunctionEnv::new(&mut store, self) }; - - let is_dl = super::linker::is_dynamically_linked(&module); -+ if !is_dl -+ && func_env -+ .data(&store) -+ .state -+ .has_shared_memory_exec_reservations() -+ { -+ return Err(WasiThreadError::LinkError(Arc::new( -+ super::linker::LinkError::SharedMemoryExecReservation( -+ "fresh exec images with inherited mappings require an imported linear memory that can be reserved before module start" -+ .to_string(), -+ ), -+ ))); -+ } -+ if preinitialized_memory_image.is_some() && (!is_dl || parent_linker_and_ctx.is_some()) { -+ let reason = if !is_dl { -+ "preinitialized memory images require a fresh dynamically linked main module" -+ } else { -+ "preinitialized memory images cannot replace shared instance-group memory" -+ }; -+ return Err(WasiThreadError::LinkError(Arc::new( -+ super::linker::LinkError::PreinitializedMemoryImage(reason.to_string()), -+ ))); -+ } - if is_dl { - let linker = match parent_linker_and_ctx { - Some((linker, ctx)) => linker.create_instance_group(ctx, &mut store, &mut func_env), -@@ -502,15 +715,19 @@ impl WasiEnv { - }; - - // TODO: make stack size configurable -- Linker::new( -- engine, -- &module, -- &mut store, -- memory, -- &mut func_env, -- 8 * 1024 * 1024, -- &ld_library_path, -- ) -+ { -+ Linker::new( -+ engine, -+ &module, -+ &mut store, -+ memory, -+ &mut func_env, -+ 8 * 1024 * 1024, -+ &ld_library_path, -+ preinitialized_memory_image, -+ fresh_zeroed_memory, -+ ) -+ } - } - }; - -@@ -539,8 +756,7 @@ impl WasiEnv { - import_object.define("env", "memory", memory); - } - let runtime = func_env.data(&store).runtime.clone(); -- let additional_imports = runtime -- .additional_imports(&module, &mut store) -+ let additional_imports = { runtime.additional_imports(&module, &mut store) } - .map_err(|err| WasiThreadError::AdditionalImportCreationFailed(Arc::new(err)))?; - - for ((namespace, name), value) in &additional_imports { -@@ -564,7 +780,7 @@ impl WasiEnv { - }); - - // Construct the instance. -- let instance = match Instance::new(&mut store, &module, &import_object) { -+ let instance = match { Instance::new(&mut store, &module, &import_object) } { - Ok(a) => a, - Err(err) => { - tracing::error!( -@@ -579,16 +795,14 @@ impl WasiEnv { - } - }; - -- runtime -- .configure_new_instance(&module, &mut store, &instance, imported_memory.as_ref()) -- .map_err(|err| WasiThreadError::AdditionalImportCreationFailed(Arc::new(err)))?; -+ { -+ runtime.configure_new_instance(&module, &mut store, &instance, imported_memory.as_ref()) -+ } -+ .map_err(|err| WasiThreadError::AdditionalImportCreationFailed(Arc::new(err)))?; - - let handles = match imported_memory { -- Some(memory) => WasiModuleTreeHandles::Static(WasiModuleInstanceHandles::new( -- memory, -- &store, -- instance.clone(), -- None, -+ Some(memory) => Ok(WasiModuleTreeHandles::Static( -+ WasiModuleInstanceHandles::new(memory, &store, instance.clone(), None), - )), - None => { - let exported_memory = instance -@@ -605,23 +819,22 @@ impl WasiEnv { - .ok_or(WasiThreadError::ExportError(ExportError::Missing( - "No imported or exported memory found".to_owned(), - )))?; -- WasiModuleTreeHandles::Static(WasiModuleInstanceHandles::new( -- exported_memory, -- &store, -- instance.clone(), -- None, -+ Ok(WasiModuleTreeHandles::Static( -+ WasiModuleInstanceHandles::new(exported_memory, &store, instance.clone(), None), - )) - } -- }; -+ }?; - - // Initialize the WASI environment -- if let Err(err) = func_env.initialize_handles_and_layout( -- &mut store, -- instance.clone(), -- handles, -- None, -- update_layout, -- ) { -+ if let Err(err) = { -+ func_env.initialize_handles_and_layout( -+ &mut store, -+ instance.clone(), -+ handles, -+ None, -+ update_layout, -+ ) -+ } { - tracing::error!( - %pid, - error = &err as &dyn std::error::Error, -@@ -635,7 +848,7 @@ impl WasiEnv { - - // If this module exports an _initialize function, run that first. - if call_initialize && let Ok(initialize) = instance.exports.get_function("_initialize") { -- let initialize_result = initialize.call(&mut store, &[]); -+ let initialize_result = { initialize.call(&mut store, &[]) }; - if let Err(err) = initialize_result { - func_env - .data(&store) -@@ -796,11 +1009,13 @@ impl WasiEnv { - } - } - -+ let mut processed_guest_signal = false; - for signal in signals { - // Skip over Sigwakeup, which is host-side-only - if matches!(signal, Signal::Sigwakeup) { - continue; - } -+ processed_guest_signal = true; - - tracing::trace!( - pid=%ctx.data().pid(), -@@ -836,7 +1051,7 @@ impl WasiEnv { - "signal processed", - ); - } -- Ok(true) -+ Ok(processed_guest_signal) - } else { - tracing::trace!("no signal handler"); - Ok(false) -@@ -915,13 +1130,10 @@ impl WasiEnv { - #[doc(hidden)] - pub(crate) fn swap_inner(&mut self, other: &mut Self) { - std::mem::swap(&mut self.inner, &mut other.inner); -- } -- -- /// Helper function to ensure the module isn't dynamically linked, needed since -- /// we only support a subset of WASIX functionality for dynamically linked modules. -- /// Specifically, anything that requires asyncify is not supported right now. -- pub(crate) fn ensure_static_module(&self) -> Result<(), ()> { -- self.inner.get().unwrap().ensure_static_module() -+ std::mem::swap( -+ &mut self.current_task_wasm_owner, -+ &mut other.current_task_wasm_owner, -+ ); - } - - /// Tries to clone the instance from this environment, but only if it's a static -@@ -1259,6 +1471,8 @@ impl WasiEnv { - pub fn on_exit(&self, process_exit_code: Option) -> BoxFuture<'static, ()> { - const CLEANUP_TIMEOUT: Duration = Duration::from_secs(10); - -+ if process_exit_code.is_some() {} -+ - // If snap-shooting is enabled then we should record an event that the thread has exited. - #[cfg(feature = "journal")] - if self.should_journal() && self.has_active_journal() { -@@ -1348,3 +1562,578 @@ impl WasiEnv { - } - } - } -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use wasmer::Engine; -+ -+ fn enter_tokio_runtime() -> Option { -+ #[cfg(not(target_arch = "wasm32"))] -+ { -+ Some( -+ tokio::runtime::Builder::new_multi_thread() -+ .enable_all() -+ .build() -+ .unwrap(), -+ ) -+ } -+ -+ #[cfg(target_arch = "wasm32")] -+ { -+ None -+ } -+ } -+ -+ #[test] -+ fn repeated_reinit_uses_fresh_registered_epoch_and_plateaus_counts() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-reinit-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let mut previous_pid = env.pid(); -+ -+ for _ in 0..8 { -+ env.process.terminate(Errno::Success.into()); -+ env.reinit().unwrap(); -+ -+ assert_ne!(env.pid(), previous_pid); -+ previous_pid = env.pid(); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ assert_eq!(env.owned_handles.len(), 1); -+ assert_eq!(env.process.active_threads(), 1); -+ assert_eq!(env.process.all_threads(), vec![env.tid()]); -+ let registered = plane.get_process(env.pid()).unwrap(); -+ assert!(registered.same_identity(&env.process)); -+ } -+ -+ env.process.terminate(Errno::Success.into()); -+ plane.retire_process_epoch(&env.process).unwrap(); -+ env.thread.retire_task_registration(); -+ env.owned_handles.clear(); -+ assert_eq!(plane.registered_process_count(), 0); -+ assert_eq!(plane.active_task_count(), 0); -+ } -+ -+ #[test] -+ fn repeated_child_construction_keeps_main_handle_owned_by_child() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let env = WasiEnv::builder("child-handle-plateau-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let baseline_handles = env.owned_handles.len(); -+ let baseline_tasks = plane.active_task_count(); -+ let baseline_processes = plane.registered_process_count(); -+ -+ for _ in 0..32 { -+ let (child_env, construction_handle, registration) = env.fork_guarded().unwrap(); -+ assert_eq!(env.owned_handles.len(), baseline_handles); -+ assert_eq!(child_env.owned_handles.len(), 1); -+ assert_eq!(plane.active_task_count(), baseline_tasks + 1); -+ assert_eq!(plane.registered_process_count(), baseline_processes); -+ -+ // Aborting an unpublished launch drops the child-owned clone; the -+ // parent never becomes a lifetime owner for the child main task. -+ drop(registration); -+ drop(construction_handle); -+ drop(child_env); -+ assert_eq!(env.owned_handles.len(), baseline_handles); -+ assert_eq!(plane.active_task_count(), baseline_tasks); -+ assert_eq!(plane.registered_process_count(), baseline_processes); -+ assert_eq!(env.process.lock().pending_child_publications, 0); -+ } -+ } -+ -+ #[test] -+ fn reinit_rejects_a_running_epoch_without_mutation() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-running-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let pid = env.pid(); -+ let error = env.reinit().unwrap_err(); -+ -+ assert!(matches!( -+ error, -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessStillRunning { -+ pid: running_pid -+ }) if running_pid == pid.raw() -+ )); -+ assert_eq!(env.pid(), pid); -+ assert_eq!(env.control_plane.registered_process_count(), 1); -+ assert_eq!(env.control_plane.active_task_count(), 1); -+ assert!(!env.process.lock().retiring); -+ } -+ -+ #[test] -+ fn reinit_requires_terminal_status_and_execution_quiescence() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-execution-lease-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let old_pid = env.pid(); -+ let lease = env.process.acquire_execution_lease().unwrap(); -+ env.process.terminate(Errno::Success.into()); -+ -+ assert_eq!( -+ env.reinit().unwrap_err(), -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessStillRunning { -+ pid: old_pid.raw(), -+ }) -+ ); -+ assert_eq!(env.pid(), old_pid); -+ assert!(!env.process.lock().retiring); -+ assert!(plane.get_process(old_pid).is_some()); -+ -+ drop(lease); -+ env.reinit().unwrap(); -+ assert_ne!(env.pid(), old_pid); -+ assert!(plane.get_process(old_pid).is_none()); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ } -+ -+ #[test] -+ fn reinit_rejects_live_background_thread_without_mutating_epoch() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-live-thread-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let pid = env.pid(); -+ let background = env -+ .process -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) -+ .unwrap(); -+ env.process.terminate(Errno::Success.into()); -+ -+ let error = env.reinit().unwrap_err(); -+ assert_eq!( -+ error, -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveThreads { -+ pid: pid.raw(), -+ count: 1, -+ }) -+ ); -+ assert_eq!(env.pid(), pid); -+ assert_eq!(env.process.active_threads(), 2); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 2); -+ assert!(!env.process.lock().retiring); -+ -+ drop(background); -+ env.reinit().unwrap(); -+ assert_ne!(env.pid(), pid); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ } -+ -+ #[test] -+ fn reinit_rejects_live_child_then_reaps_finished_child_exactly() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-live-child-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let parent_pid = env.pid(); -+ let (child, child_main, mut child_registration) = plane -+ .new_child_process_with_main_thread_guarded( -+ &env.process, -+ ModuleHash::random(), -+ WasiMemoryLayout::default(), -+ ) -+ .unwrap(); -+ child_registration.commit_child().unwrap(); -+ child_registration.complete_child_launch(env.tasks()); -+ let child_pid = child.pid(); -+ env.process.terminate(Errno::Success.into()); -+ -+ let error = env.reinit().unwrap_err(); -+ assert_eq!( -+ error, -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveChildren { -+ pid: parent_pid.raw(), -+ count: 1, -+ }) -+ ); -+ assert_eq!(env.pid(), parent_pid); -+ assert_eq!(env.process.lock().children.len(), 1); -+ assert_eq!(plane.registered_process_count(), 2); -+ assert_eq!(plane.active_task_count(), 2); -+ assert!(!env.process.lock().retiring); -+ -+ child.terminate(Errno::Success.into()); -+ drop(child_main); -+ env.reinit().unwrap(); -+ assert_ne!(env.pid(), parent_pid); -+ assert!(plane.get_process(child_pid).is_none()); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ } -+ -+ #[test] -+ fn reinit_cannot_cross_inflight_child_publication() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-pending-child-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let parent_pid = env.pid(); -+ let (child_env, child_handle, child_registration) = env.fork_guarded().unwrap(); -+ env.process.terminate(Errno::Success.into()); -+ -+ let error = env.reinit().unwrap_err(); -+ assert_eq!( -+ error, -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveChildren { -+ pid: parent_pid.raw(), -+ count: 1, -+ }) -+ ); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 2); -+ assert!(!env.process.lock().retiring); -+ -+ drop(child_registration); -+ drop(child_handle); -+ drop(child_env); -+ env.reinit().unwrap(); -+ assert_ne!(env.pid(), parent_pid); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ } -+ -+ #[test] -+ fn sealed_epoch_is_terminal_after_later_reset_failure() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-terminal-reset-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let pid = env.pid(); -+ env.process.terminate(Errno::Success.into()); -+ assert!(env.process.begin_epoch_retirement().unwrap().is_empty()); -+ -+ let error = env.reinit().unwrap_err(); -+ assert_eq!( -+ error, -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessRetiring { -+ pid: pid.raw(), -+ }) -+ ); -+ assert_eq!(env.pid(), pid); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ -+ plane.retire_process_epoch(&env.process).unwrap(); -+ env.process.retire_epoch_task_registrations(); -+ env.owned_handles.clear(); -+ } -+ -+ #[test] -+ fn public_fork_publishes_and_adopts_before_releasing_parent_permit() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("public-fork-adoption-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let (child_env, child_handle) = env.fork().unwrap(); -+ let child_pid = child_env.pid(); -+ -+ assert!( -+ env.process -+ .lock() -+ .children -+ .iter() -+ .any(|child| child.same_identity(&child_env.process)) -+ ); -+ assert!( -+ plane -+ .get_process(child_pid) -+ .unwrap() -+ .same_identity(&child_env.process) -+ ); -+ assert_eq!(plane.registered_process_count(), 2); -+ assert_eq!(plane.active_task_count(), 2); -+ -+ child_env.process.terminate(Errno::Success.into()); -+ drop(child_handle); -+ env.process.terminate(Errno::Success.into()); -+ env.reinit().unwrap(); -+ assert!(plane.get_process(child_pid).is_none()); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ } -+ -+ #[test] -+ fn reinit_recursively_retires_finished_grandchildren() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-grandchild-retirement-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let (child_env, child_handle) = env.fork().unwrap(); -+ let child_pid = child_env.pid(); -+ let (grandchild_env, grandchild_handle) = child_env.fork().unwrap(); -+ let grandchild_pid = grandchild_env.pid(); -+ -+ grandchild_env.process.terminate(Errno::Success.into()); -+ drop(grandchild_handle); -+ child_env.process.terminate(Errno::Success.into()); -+ drop(child_handle); -+ env.process.terminate(Errno::Success.into()); -+ -+ env.reinit().unwrap(); -+ assert!(plane.get_process(child_pid).is_none()); -+ assert!(plane.get_process(grandchild_pid).is_none()); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ assert!(env.process.lock().children.is_empty()); -+ } -+ -+ #[test] -+ fn descendant_validation_failure_seals_no_process_in_tree() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-tree-validation-transaction-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let (child_env, child_handle) = env.fork().unwrap(); -+ let (grandchild_env, grandchild_handle) = child_env.fork().unwrap(); -+ -+ child_env.process.terminate(Errno::Success.into()); -+ env.process.terminate(Errno::Success.into()); -+ assert!(matches!( -+ env.reinit().unwrap_err(), -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveChildren { -+ count: 1, -+ .. -+ }) -+ )); -+ assert!(!env.process.lock().retiring); -+ assert!(!child_env.process.lock().retiring); -+ assert!(!grandchild_env.process.lock().retiring); -+ -+ grandchild_env.process.terminate(Errno::Success.into()); -+ drop(grandchild_handle); -+ drop(child_handle); -+ env.reinit().unwrap(); -+ assert_eq!(env.control_plane.registered_process_count(), 1); -+ assert_eq!(env.control_plane.active_task_count(), 1); -+ } -+ -+ #[test] -+ fn stale_environment_cannot_fork_or_start_threads_after_reinit() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-stale-epoch-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let stale = env.clone(); -+ let stale_pid = stale.pid(); -+ let plane = env.control_plane.clone(); -+ -+ env.process.terminate(Errno::Success.into()); -+ env.reinit().unwrap(); -+ assert_ne!(env.pid(), stale_pid); -+ -+ assert_eq!( -+ stale.fork().unwrap_err(), -+ ControlPlaneError::ProcessRetiring { -+ pid: stale_pid.raw(), -+ } -+ ); -+ assert_eq!( -+ stale -+ .process -+ .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) -+ .unwrap_err(), -+ ControlPlaneError::ProcessRetiring { -+ pid: stale_pid.raw(), -+ } -+ ); -+ assert_eq!(plane.registered_process_count(), 1); -+ assert_eq!(plane.active_task_count(), 1); -+ assert!(plane.get_process(stale_pid).is_none()); -+ assert!(stale.process.lock().children.is_empty()); -+ assert_eq!(stale.process.lock().pending_child_publications, 0); -+ } -+ -+ #[test] -+ fn retargeted_thread_clone_drops_calling_thread_execution_ownership() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("thread-clone-ownership-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let owner = Arc::new(()); -+ let main = env.thread.clone(); -+ env.bind_pending_task_wasm_execution(&main, &owner); -+ let deferred = env.acquire_parent_execution_guard().unwrap(); -+ env.deferred_parent_execution = Some(deferred); -+ assert_eq!(env.process.lock().execution_leases, 1); -+ -+ let spawned = env -+ .process -+ .new_thread( -+ WasiMemoryLayout::default(), -+ ThreadStartType::ThreadSpawn { start_ptr: 0 }, -+ ) -+ .unwrap(); -+ let spawned_thread = spawned.as_thread(); -+ let cloned = -+ env.clone_for_thread_spawn(spawned_thread.clone(), WasiMemoryLayout::default()); -+ assert!(cloned.thread.same_identity(&spawned_thread)); -+ assert!(cloned.deferred_parent_execution.is_none()); -+ assert!(cloned.current_task_wasm_owner.upgrade().is_none()); -+ assert!(cloned.vfork.is_none()); -+ assert!(env.deferred_parent_execution.is_some()); -+ assert_eq!(env.process.lock().execution_leases, 1); -+ -+ let deferred = env.deferred_parent_execution.take().unwrap(); -+ env.restore_parent_execution_guard(deferred); -+ assert_eq!(env.process.lock().execution_leases, 0); -+ drop(spawned); -+ } -+ -+ #[test] -+ fn swap_inner_uses_exact_physical_owner_for_immediate_and_deferred_restore() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ -+ // Ordinary vfork restore returns to the same physical TaskWasm owner, -+ // so the supplemental parent lease is consumed immediately. -+ { -+ let mut current = WasiEnv::builder("ordinary-vfork-owner-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let parent_process = current.process.clone(); -+ let parent_thread = current.thread.clone(); -+ let owner = Arc::new(()); -+ current.bind_pending_task_wasm_execution(&parent_thread, &owner); -+ let supplemental = current.acquire_parent_execution_guard().unwrap(); -+ let (mut stored, _child_handle, mut registration) = current.fork_guarded().unwrap(); -+ registration.commit_child().unwrap(); -+ registration.complete_child_launch(current.tasks()); -+ -+ stored.swap_inner(&mut current); -+ std::mem::swap(&mut current, &mut stored); -+ assert!(!current.process.same_identity(&stored.process)); -+ stored.swap_inner(&mut current); -+ std::mem::swap(&mut current, &mut stored); -+ assert!(current.process.same_identity(&parent_process)); -+ current.restore_parent_execution_guard(supplemental); -+ assert!(current.deferred_parent_execution.is_none()); -+ assert_eq!(parent_process.lock().execution_leases, 0); -+ } -+ -+ // A child successor has a different physical owner. Restoration must -+ // retain exactly one parent lease and transfer it to a later exact -+ // parent continuation before releasing it. -+ { -+ let mut current = WasiEnv::builder("deferred-vfork-owner-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let parent_process = current.process.clone(); -+ let parent_thread = current.thread.clone(); -+ let predecessor = Arc::new(()); -+ current.bind_pending_task_wasm_execution(&parent_thread, &predecessor); -+ let supplemental = current.acquire_parent_execution_guard().unwrap(); -+ let (mut stored, _child_handle, mut registration) = current.fork_guarded().unwrap(); -+ registration.commit_child().unwrap(); -+ registration.complete_child_launch(current.tasks()); -+ -+ stored.swap_inner(&mut current); -+ std::mem::swap(&mut current, &mut stored); -+ drop(predecessor); -+ let child_successor = Arc::new(()); -+ let child_thread = current.thread.clone(); -+ current.bind_pending_task_wasm_execution(&child_thread, &child_successor); -+ -+ stored.swap_inner(&mut current); -+ std::mem::swap(&mut current, &mut stored); -+ current.restore_parent_execution_guard(supplemental); -+ assert!(current.deferred_parent_execution.is_some()); -+ assert_eq!(parent_process.lock().execution_leases, 1); -+ -+ let mut parent_successor = current.take_for_same_thread_continuation(); -+ assert!(current.deferred_parent_execution.is_none()); -+ let successor_identity = Arc::new(()); -+ let successor_lease = parent_process.acquire_execution_lease().unwrap(); -+ parent_successor.bind_pending_task_wasm_execution(&parent_thread, &successor_identity); -+ let deferred = parent_successor.deferred_parent_execution.take().unwrap(); -+ assert!(deferred.try_handoff_to_accepted_task_wasm(&parent_process, &parent_thread)); -+ assert_eq!(parent_process.lock().execution_leases, 1); -+ drop(successor_lease); -+ assert_eq!(parent_process.lock().execution_leases, 0); -+ drop(child_successor); -+ } -+ } -+ -+ #[test] -+ fn post_seal_main_thread_admission_failure_is_terminal_and_leak_free() { -+ let runtime = enter_tokio_runtime(); -+ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); -+ let mut env = WasiEnv::builder("dcgi-post-seal-admission-test") -+ .engine(Engine::default()) -+ .build() -+ .unwrap(); -+ let plane = env.control_plane.clone(); -+ let old_pid = env.pid(); -+ env.process.terminate(Errno::Success.into()); -+ plane.fail_next_task_admission(); -+ -+ assert_eq!( -+ env.reinit().unwrap_err(), -+ WasiStateCreationError::ControlPlane(ControlPlaneError::TaskLimitReached { -+ max: usize::MAX, -+ }) -+ ); -+ assert_eq!(env.pid(), old_pid); -+ assert!(env.process.lock().retiring); -+ assert!(plane.get_process(old_pid).is_none()); -+ assert_eq!(plane.registered_process_count(), 0); -+ assert_eq!(plane.active_task_count(), 0); -+ -+ assert_eq!( -+ env.reinit().unwrap_err(), -+ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessRetiring { -+ pid: old_pid.raw(), -+ }) -+ ); -+ assert_eq!(plane.registered_process_count(), 0); -+ assert_eq!(plane.active_task_count(), 0); -+ } -+} -diff --git a/lib/wasix/src/state/func_env.rs b/lib/wasix/src/state/func_env.rs -index 914edc2..e5c514c 100644 ---- a/lib/wasix/src/state/func_env.rs -+++ b/lib/wasix/src/state/func_env.rs -@@ -17,7 +17,7 @@ use crate::{ - utils::{get_wasi_version, get_wasi_versions, store::restore_store_snapshot}, - }; - --use super::Linker; -+use super::{Linker, PreinitializedMemoryImageMode}; - - /// The default stack size for WASIX - the number itself is the default that compilers - /// have used in the past when compiling WASM apps. -@@ -46,12 +46,17 @@ impl WasiFunctionEnv { - spawn_type: SpawnMemoryTypeOrStore, - update_layout: bool, - call_initialize: bool, -- parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut)>, -+ preinitialized_memory_image: Option, -+ parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut<'_, WasiEnv>)>, - ) -> Result<(Self, Store), WasiThreadError> { - // Create a new store and put the memory object in it - // (but only if it has imported memory) -- let (memory, store): (Option, Option) = match spawn_type { -- SpawnMemoryTypeOrStore::New => (None, None), -+ let (memory, store, fresh_zeroed_memory): ( -+ Option, -+ Option, -+ bool, -+ ) = match spawn_type { -+ SpawnMemoryTypeOrStore::New => (None, None, true), - SpawnMemoryTypeOrStore::Type(mut ty) => { - ty.shared = true; - -@@ -61,7 +66,7 @@ impl WasiFunctionEnv { - // browser otherwise creation will fail. - let _ = ty.maximum.get_or_insert(wasmer_types::Pages::max_value()); - -- let mem = Memory::new(&mut store, ty).map_err(|err| { -+ let mem = { Memory::new(&mut store, ty) }.map_err(|err| { - tracing::error!( - error = &err as &dyn std::error::Error, - memory_type=?ty, -@@ -69,27 +74,36 @@ impl WasiFunctionEnv { - ); - WasiThreadError::MemoryCreateFailed(err) - })?; -- (Some(mem), Some(store)) -+ (Some(mem), Some(store), true) - } -- SpawnMemoryTypeOrStore::StoreAndMemory(s, m) => (m, Some(s)), -+ SpawnMemoryTypeOrStore::StoreAndMemory(s, m) => (m, Some(s), false), - }; - -- let mut store = store.unwrap_or_else(|| env.runtime().new_store()); -+ let mut store = match store { -+ Some(store) => store, -+ None => env.runtime().new_store(), -+ }; - -- let (_, ctx) = env.instantiate( -- module, -- &mut store, -- memory, -- update_layout, -- call_initialize, -- parent_linker_and_ctx, -- )?; -+ let (_, ctx) = { -+ env.instantiate( -+ module, -+ &mut store, -+ memory, -+ update_layout, -+ call_initialize, -+ preinitialized_memory_image, -+ fresh_zeroed_memory, -+ parent_linker_and_ctx, -+ ) -+ }?; - - // FIXME: shouldn't this happen _before_ instantiating, so the startup code in the instance - // has access to the globals? - // Set all the globals - if let Some(snapshot) = store_snapshot { -- restore_store_snapshot(&mut store, &snapshot); -+ restore_store_snapshot(&mut store, &snapshot).map_err(|err| { -+ WasiThreadError::InitFailed(std::sync::Arc::new(anyhow::Error::new(err))) -+ })?; - } - - Ok((ctx, store)) -diff --git a/lib/wasix/src/state/handles/mod.rs b/lib/wasix/src/state/handles/mod.rs -index 4fa1030..6de50be 100644 ---- a/lib/wasix/src/state/handles/mod.rs -+++ b/lib/wasix/src/state/handles/mod.rs -@@ -47,9 +47,6 @@ pub struct WasiModuleInstanceHandles { - /// Points to the start of the TLS area - pub(crate) tls_base: Option, - -- /// Main function that will be invoked (name = "_start") -- pub(crate) start: Option>, -- - /// Function thats invoked to initialize the WASM module (name = "_initialize") - // TODO: review allow... - #[allow(dead_code)] -@@ -140,7 +137,6 @@ impl WasiModuleInstanceHandles { - stack_low: instance.exports.get_global("__stack_low").cloned().ok(), - stack_high: instance.exports.get_global("__stack_high").cloned().ok(), - tls_base: instance.exports.get_global("__tls_base").cloned().ok(), -- start: instance.exports.get_typed_function(store, "_start").ok(), - initialize: instance - .exports - .get_typed_function(store, "_initialize") -@@ -187,6 +183,13 @@ impl WasiModuleInstanceHandles { - self.instance.module().clone() - } - -+ pub(crate) fn supports_asyncify_stack_rewind(&self) -> bool { -+ self.asyncify_start_unwind.is_some() -+ && self.asyncify_stop_unwind.is_some() -+ && self.asyncify_start_rewind.is_some() -+ && self.asyncify_stop_rewind.is_some() -+ } -+ - /// Providers safe access to the memory - /// (it must be initialized before it can be used) - pub fn memory_view<'a>(&'a self, store: &'a (impl AsStoreRef + ?Sized)) -> MemoryView<'a> { -@@ -245,6 +248,68 @@ impl From for Errno { - } - } - -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use wasmer::{Instance, Module, Store, imports}; -+ -+ fn supports_asyncify_stack_rewind(wat: &str) -> bool { -+ let mut store = Store::default(); -+ let module = Module::new(&store, wat).unwrap(); -+ let instance = Instance::new(&mut store, &module, &imports! {}).unwrap(); -+ let memory = instance.exports.get_memory("memory").unwrap().clone(); -+ let handles = WasiModuleInstanceHandles::new(memory, &store, instance, None); -+ -+ handles.supports_asyncify_stack_rewind() -+ } -+ -+ #[test] -+ fn stack_rewind_is_unsupported_without_exports() { -+ let supported = supports_asyncify_stack_rewind( -+ r#" -+ (module -+ (memory (export "memory") 1) -+ ) -+ "#, -+ ); -+ -+ assert!(!supported); -+ } -+ -+ #[test] -+ fn stack_rewind_requires_complete_asyncify_exports() { -+ let supported = supports_asyncify_stack_rewind( -+ r#" -+ (module -+ (memory (export "memory") 1) -+ (func (export "asyncify_start_unwind") (param i32)) -+ (func (export "asyncify_stop_unwind")) -+ (func (export "asyncify_start_rewind") (param i32)) -+ ) -+ "#, -+ ); -+ -+ assert!(!supported); -+ } -+ -+ #[test] -+ fn stack_rewind_supports_complete_asyncify_exports() { -+ let supported = supports_asyncify_stack_rewind( -+ r#" -+ (module -+ (memory (export "memory") 1) -+ (func (export "asyncify_start_unwind") (param i32)) -+ (func (export "asyncify_stop_unwind")) -+ (func (export "asyncify_start_rewind") (param i32)) -+ (func (export "asyncify_stop_rewind")) -+ ) -+ "#, -+ ); -+ -+ assert!(supported); -+ } -+} -+ - impl WasiModuleTreeHandles { - /// Can be used to get the `WasiModuleInstanceHandles` of the main module. - /// If access to the side modules' instance handles is required, one must go -@@ -290,16 +355,6 @@ impl WasiModuleTreeHandles { - } - } - -- /// Helper function to ensure the module isn't dynamically linked, needed since -- /// we only support a subset of WASIX functionality for dynamically linked modules. -- /// Specifically, anything that requires asyncify is not supported right now. -- pub(crate) fn ensure_static_module(&self) -> Result<(), ()> { -- match self { -- WasiModuleTreeHandles::Static(_) => Ok(()), -- _ => Err(()), -- } -- } -- - /// Providers safe access to the memory - /// (it must be initialized before it can be used) - pub fn memory_view<'a>(&'a self, store: &'a (impl AsStoreRef + ?Sized)) -> MemoryView<'a> { -diff --git a/lib/wasix/src/state/linker.rs b/lib/wasix/src/state/linker.rs -index 0abec18..4090d25 100644 ---- a/lib/wasix/src/state/linker.rs -+++ b/lib/wasix/src/state/linker.rs -@@ -266,22 +266,21 @@ use std::{ - ops::{Deref, DerefMut}, - path::{Path, PathBuf}, - sync::{ -- Arc, Barrier, Mutex, MutexGuard, RwLock, RwLockWriteGuard, TryLockError, -+ Arc, Barrier, Condvar, Mutex, MutexGuard, RwLock, RwLockWriteGuard, TryLockError, - atomic::{AtomicBool, Ordering}, - }, - }; - --use bus::Bus; - use derive_more::Debug; - use shared_buffer::OwnedBuffer; - use tracing::trace; - use virtual_fs::{AsyncReadExt, FileSystem, FsError}; - use virtual_mio::block_on; - use wasmer::{ -- AsStoreMut, AsStoreRef, Engine, ExportError, Exportable, Extern, ExternType, Function, -- FunctionEnv, FunctionEnvMut, FunctionType, Global, GlobalType, ImportType, Imports, Instance, -- InstantiationError, Memory, MemoryError, Module, RuntimeError, StoreMut, Table, Tag, Type, -- Value, WASM_PAGE_SIZE, WasmTypeList, -+ AsStoreMut, Engine, ExportError, Extern, ExternType, Function, FunctionEnv, FunctionEnvMut, -+ FunctionType, Global, GlobalType, ImportType, Imports, Instance, InstantiationError, Memory, -+ MemoryError, Module, RuntimeError, StoreMut, Table, Tag, Type, Value, WASM_PAGE_SIZE, -+ WasmTypeList, - }; - use wasmer_wasix_types::wasix::WasiMemoryLayout; - -@@ -291,7 +290,15 @@ use crate::{ - runtime::module_cache::HashedModuleData, - }; - --use super::{WasiModuleInstanceHandles, WasiState}; -+use super::{PreinitializedMemoryImageMode, WasiModuleInstanceHandles, WasiState}; -+ -+/// Linear witness that ordinary WebAssembly start returned normally for the -+/// newly instantiated main module. The private field can be constructed only -+/// in this linker module, and the non-`Copy` value is consumed by image -+/// application at the immediately following boundary. -+pub(crate) struct OrdinaryModuleStartCompleted { -+ _private: (), -+} - - // Module handle 1 is always the main module. Side modules get handles starting from the next one after the main module. - pub static MAIN_MODULE_HANDLE: ModuleHandle = ModuleHandle(1); -@@ -323,6 +330,224 @@ impl Display for ModuleHandle { - - const DEFAULT_RUNTIME_PATH: [&str; 3] = ["/lib", "/usr/lib", "/usr/local/lib"]; - -+/// Exports retained eagerly by WASIX's dynamic linker. -+/// -+/// These are the exact fixed entry points consumed while constructing runtime -+/// handles or running module initialization. Every dynamic symbol remains -+/// available through store-aware lookup against the module's complete export -+/// metadata; it is intentionally not copied into the per-instance export map -+/// until the linker actually asks for it. -+const DYNAMIC_INSTANCE_INITIAL_EXPORTS: &[&str] = &[ -+ "__indirect_function_table", -+ "__stack_pointer", -+ "__data_end", -+ "__stack_low", -+ "__stack_high", -+ "__tls_base", -+ "_start", -+ "_initialize", -+ "wasi_thread_start", -+ "__wasm_signal", -+ "asyncify_start_unwind", -+ "asyncify_stop_unwind", -+ "asyncify_start_rewind", -+ "asyncify_stop_rewind", -+ "asyncify_get_state", -+ "__wasm_apply_data_relocs", -+ "__wasm_apply_tls_relocs", -+ "__wasm_call_ctors", -+ "__wasix_init_tls", -+]; -+ -+/// A dependency-free, single-slot broadcast channel for linker rendezvous. -+/// -+/// Dynamic-link operations cannot overlap: the linker write lock serializes the -+/// producer and the two barriers serialize every participating instance group. -+/// A general-purpose ring-buffer bus is therefore unnecessary here. In -+/// particular, `bus::Bus` creates a permanent helper thread for each channel, -+/// even though these channels are almost always idle. -+/// -+/// The slot remains occupied until every receiver that existed at broadcast -+/// time has either received the value or disconnected. Receivers added while a -+/// value is pending only observe future broadcasts. All state transitions and -+/// condition-variable waits use the same mutex, so broadcasts and sender -+/// closure cannot be lost between checking the state and going to sleep. -+struct SingleSlotBroadcast { -+ inner: Arc>, -+} -+ -+struct SingleSlotReceiver { -+ inner: Arc>, -+ reader_id: u64, -+} -+ -+struct SingleSlotBroadcastInner { -+ state: Mutex>, -+ changed: Condvar, -+} -+ -+struct SingleSlotBroadcastState { -+ slot: Option, -+ readers: HashMap, -+ pending_readers: usize, -+ next_reader_id: u64, -+ closed: bool, -+ #[cfg(test)] -+ waiting_readers: usize, -+} -+ -+impl SingleSlotBroadcastInner { -+ fn lock(&self) -> MutexGuard<'_, SingleSlotBroadcastState> { -+ self.state -+ .lock() -+ .unwrap_or_else(std::sync::PoisonError::into_inner) -+ } -+} -+ -+impl SingleSlotBroadcast { -+ fn new() -> Self { -+ Self { -+ inner: Arc::new(SingleSlotBroadcastInner { -+ state: Mutex::new(SingleSlotBroadcastState { -+ slot: None, -+ readers: HashMap::new(), -+ pending_readers: 0, -+ next_reader_id: 0, -+ closed: false, -+ #[cfg(test)] -+ waiting_readers: 0, -+ }), -+ changed: Condvar::new(), -+ }), -+ } -+ } -+ -+ fn add_rx(&mut self) -> SingleSlotReceiver { -+ let mut state = self.inner.lock(); -+ let reader_id = state.next_reader_id; -+ state.next_reader_id = state -+ .next_reader_id -+ .checked_add(1) -+ .expect("single-slot broadcast reader ID space exhausted"); -+ assert!( -+ state.readers.insert(reader_id, false).is_none(), -+ "single-slot broadcast reader ID reused" -+ ); -+ drop(state); -+ -+ SingleSlotReceiver { -+ inner: Arc::clone(&self.inner), -+ reader_id, -+ } -+ } -+ -+ fn rx_count(&self) -> usize { -+ self.inner.lock().readers.len() -+ } -+ -+ /// Broadcast without blocking. On failure, no receiver observes `value`. -+ fn try_broadcast(&mut self, value: T) -> Result<(), T> { -+ let mut state = self.inner.lock(); -+ debug_assert!(!state.closed, "live sender marked closed"); -+ -+ if state.slot.is_some() { -+ return Err(value); -+ } -+ -+ if state.readers.is_empty() { -+ return Ok(()); -+ } -+ -+ state.pending_readers = state.readers.len(); -+ for pending in state.readers.values_mut() { -+ *pending = true; -+ } -+ state.slot = Some(value); -+ drop(state); -+ self.inner.changed.notify_all(); -+ Ok(()) -+ } -+} -+ -+impl Drop for SingleSlotBroadcast { -+ fn drop(&mut self) { -+ let mut state = self.inner.lock(); -+ state.closed = true; -+ drop(state); -+ self.inner.changed.notify_all(); -+ } -+} -+ -+impl SingleSlotReceiver { -+ fn recv(&mut self) -> Result { -+ let mut state = self.inner.lock(); -+ -+ loop { -+ let Some(is_pending) = state.readers.get(&self.reader_id).copied() else { -+ return Err(std::sync::mpsc::RecvError); -+ }; -+ -+ if is_pending { -+ if state.pending_readers == 1 { -+ state.readers.insert(self.reader_id, false); -+ state.pending_readers = 0; -+ return Ok(state -+ .slot -+ .take() -+ .expect("pending broadcast has no stored value")); -+ } -+ -+ let value = state -+ .slot -+ .as_ref() -+ .expect("pending broadcast has no stored value") -+ .clone(); -+ state.readers.insert(self.reader_id, false); -+ state.pending_readers -= 1; -+ return Ok(value); -+ } -+ -+ // Values already in the slot at sender shutdown remain readable by -+ // their original receivers. Only report disconnect after this -+ // receiver has drained everything it was entitled to receive. -+ if state.closed { -+ return Err(std::sync::mpsc::RecvError); -+ } -+ -+ #[cfg(test)] -+ { -+ state.waiting_readers += 1; -+ } -+ state = self -+ .inner -+ .changed -+ .wait(state) -+ .unwrap_or_else(std::sync::PoisonError::into_inner); -+ #[cfg(test)] -+ { -+ state.waiting_readers -= 1; -+ } -+ } -+ } -+} -+ -+impl Drop for SingleSlotReceiver { -+ fn drop(&mut self) { -+ let mut state = self.inner.lock(); -+ let Some(was_pending) = state.readers.remove(&self.reader_id) else { -+ return; -+ }; -+ -+ if was_pending { -+ state.pending_readers -= 1; -+ if state.pending_readers == 0 { -+ state.slot = None; -+ } -+ } -+ } -+} -+ -+#[derive(Clone)] - struct AllocatedPage { - // The base_ptr is mutable, and will move forward as memory is allocated from the page. - base_ptr: u32, -@@ -336,6 +561,7 @@ struct AllocatedPage { - // out, since each module may request a specific amount of memory to be allocated - // for it before starting it up. - // TODO: Only supports Memory32, should implement proper Memory64 support -+#[derive(Clone)] - struct MemoryAllocator { - allocated_pages: Vec, - } -@@ -525,6 +751,12 @@ pub enum LinkError { - #[error("Bad __tls_base export, expected a global of type I32 or I64")] - BadTlsBaseExport, - -+ #[error("Preinitialized memory image rejected: {0}")] -+ PreinitializedMemoryImage(String), -+ -+ #[error("Shared-memory exec reservation rejected: {0}")] -+ SharedMemoryExecReservation(String), -+ - #[error( - "TLS symbol {0} cannot be resolved from module {1} because it does not export its __tls_base" - )] -@@ -737,7 +969,7 @@ pub enum SymbolResolutionKey { - }, - } - --#[derive(Debug)] -+#[derive(Debug, Clone)] - pub enum SymbolResolutionResult { - // The symbol was resolved to a global address. We don't resolve again because - // the value of globals and the memory_base for each module and all of its instances -@@ -788,6 +1020,7 @@ enum DlOperation { - }, - } - -+#[derive(Clone)] - struct DlModule { - module: Module, - dylink_info: DylinkInfo, -@@ -816,10 +1049,10 @@ struct InstanceGroupState { - - // Once the dl_operation_pending flag is set, a barrier is created and broadcast - // by the instigating group, which others must use to rendezvous with it. -- recv_pending_operation_barrier: bus::BusReader>, -+ recv_pending_operation_barrier: SingleSlotReceiver>, - // The corresponding sender is stored in the shared linker state, and is used - // by the instigating instance group to broadcast the results. -- recv_pending_operation: bus::BusReader, -+ recv_pending_operation: SingleSlotReceiver, - } - - // There is only one LinkerState for all instance groups -@@ -854,8 +1087,8 @@ struct LinkerState { - - symbol_resolution_records: HashMap, - -- send_pending_operation_barrier: bus::Bus>, -- send_pending_operation: bus::Bus, -+ send_pending_operation_barrier: SingleSlotBroadcast>, -+ send_pending_operation: SingleSlotBroadcast, - } - - /// The linker is responsible for loading and linking dynamic modules at runtime, -@@ -964,12 +1197,15 @@ impl Linker { - func_env: &mut WasiFunctionEnv, - stack_size: u64, - ld_library_path: &[&Path], -+ preinitialized_memory_image: Option, -+ fresh_zeroed_memory: bool, - ) -> Result<(Self, LinkedMainModule), LinkError> { -- let dylink_section = parse_dylink0_section(main_module)?; -+ let dylink_section = { parse_dylink0_section(main_module) }?; - - trace!(?dylink_section, "Loading main module"); - -- let mut imports = import_object_for_all_wasi_versions(main_module, store, &func_env.env); -+ let mut imports = -+ { import_object_for_all_wasi_versions(main_module, store, &func_env.env) }; - - let function_table_type = main_module - .imports() -@@ -993,8 +1229,9 @@ impl Linker { - minimum_size = ?function_table_type.minimum, - "Creating indirect function table" - ); -- let indirect_function_table = Table::new(store, function_table_type, Value::FuncRef(None)) -- .map_err(LinkError::TableAllocationError)?; -+ let indirect_function_table = -+ { Table::new(store, function_table_type, Value::FuncRef(None)) } -+ .map_err(LinkError::TableAllocationError)?; - - let expected_table_length = - dylink_section.mem_info.table_size + MAIN_MODULE_TABLE_BASE as u32; -@@ -1003,8 +1240,7 @@ impl Linker { - let current_size = indirect_function_table.size(store); - let delta = expected_table_length - current_size; - trace!(?current_size, ?delta, "Growing indirect function table"); -- indirect_function_table -- .grow(store, delta, Value::FuncRef(None)) -+ { indirect_function_table.grow(store, delta, Value::FuncRef(None)) } - .map_err(LinkError::TableAllocationError)?; - } - -@@ -1050,15 +1286,60 @@ impl Linker { - - let stack_high = stack_low + stack_size; - -- // Allocate memory for the stack. This does not need to go through the memory allocator -- // because it's always placed directly after the main module's data -- memory.grow_at_least(store, stack_high)?; -+ let exec_reservations = func_env.data(store).state.shared_memory_exec_reservations(); -+ let mut exec_reservation_end = None; -+ for reservation in &exec_reservations { -+ let end = reservation -+ .start -+ .checked_add(reservation.len) -+ .ok_or_else(|| { -+ LinkError::SharedMemoryExecReservation( -+ "reservation range overflowed the guest address space".to_string(), -+ ) -+ })?; -+ if reservation.len == 0 || reservation.start < stack_high { -+ return Err(LinkError::SharedMemoryExecReservation(format!( -+ "reservation [{:#x}, {end:#x}) overlaps main data/stack ending at {stack_high:#x}", -+ reservation.start -+ ))); -+ } -+ exec_reservation_end = Some(exec_reservation_end.map_or(end, |old: u64| old.max(end))); -+ } -+ let initial_memory_end = exec_reservation_end.map_or(stack_high, |end| end.max(stack_high)); -+ -+ // Growing before Instance::new is the actual address reservation. -+ // Patched allocators claim only exact positive-sbrk return intervals, -+ // so this runtime-owned growth can never become malloc-owned space even -+ // when module start code allocates before PostgreSQL reattaches files. -+ // Growth reserves virtual address space; it does not fault every -+ // intervening page into RSS. -+ { -+ reserve_exec_memory_window( -+ &memory, -+ store, -+ initial_memory_end, -+ !exec_reservations.is_empty(), -+ ) -+ }?; -+ if !exec_reservations.is_empty() { -+ func_env -+ .data(store) -+ .state -+ .mark_shared_memory_exec_reservations_installed() -+ .map_err(|errno| { -+ LinkError::SharedMemoryExecReservation(format!( -+ "could not install reserved address window after memory growth: {errno}" -+ )) -+ })?; -+ } - - trace!( - memory_pages = ?memory.grow(store, 0).unwrap(), - memory_base, - stack_low, - stack_high, -+ exec_reservation_end, -+ initial_memory_end, - "Memory layout" - ); - -@@ -1069,14 +1350,15 @@ impl Linker { - "__stack_pointer".to_string(), - ))?; - -- let stack_pointer = define_integer_global_import(store, &stack_pointer_import, stack_high)?; -+ let stack_pointer = -+ { define_integer_global_import(store, &stack_pointer_import, stack_high) }?; - - let c_longjmp = Tag::new(store, vec![Type::I32]); - let cpp_exception = Tag::new(store, vec![Type::I32]); - -- let mut barrier_tx = Bus::new(1); -+ let mut barrier_tx = SingleSlotBroadcast::new(); - let barrier_rx = barrier_tx.add_rx(); -- let mut operation_tx = Bus::new(1); -+ let mut operation_tx = SingleSlotBroadcast::new(); - let operation_rx = operation_tx.add_rx(); - - let mut instance_group = InstanceGroupState { -@@ -1085,7 +1367,7 @@ impl Linker { - // `__tls_base` global export from the instance after instantiation. - main_instance_tls_base: None, - side_instances: HashMap::new(), -- stack_pointer, -+ stack_pointer: stack_pointer.clone(), - memory: memory.clone(), - indirect_function_table: indirect_function_table.clone(), - c_longjmp, -@@ -1122,35 +1404,71 @@ impl Linker { - ]; - - trace!("Resolving main module's symbols"); -- linker_state.resolve_symbols( -- &instance_group, -- store, -- main_module, -- MAIN_MODULE_HANDLE, -- &mut link_state, -- &well_known_imports, -- )?; -+ { -+ linker_state.resolve_symbols( -+ &instance_group, -+ store, -+ main_module, -+ MAIN_MODULE_HANDLE, -+ &mut link_state, -+ &well_known_imports, -+ ) -+ }?; - - trace!("Populating main module's imports object"); -- instance_group.populate_imports_from_link_state( -- MAIN_MODULE_HANDLE, -- &mut linker_state, -- &mut link_state, -- store, -- main_module, -- &mut imports, -- &func_env.env, -- &well_known_imports, -- )?; -+ { -+ instance_group.populate_imports_from_link_state( -+ MAIN_MODULE_HANDLE, -+ &mut linker_state, -+ &mut link_state, -+ store, -+ main_module, -+ &mut imports, -+ &func_env.env, -+ &well_known_imports, -+ ) -+ }?; - - // TODO: figure out which way is faster (stubs in main or stubs in sides), - // use that ordering. My *guess* is that, since main exports all the libc - // functions and those are called frequently by basically any code, then giving - // stubs to main will be faster, but we need numbers before we decide this. -- let main_instance = Instance::new(store, main_module, &imports)?; -+ let main_instance = { -+ Instance::new_with_export_names( -+ store, -+ main_module, -+ &imports, -+ DYNAMIC_INSTANCE_INITIAL_EXPORTS, -+ ) -+ }?; -+ let ordinary_start_completed = OrdinaryModuleStartCompleted { _private: () }; -+ if let Some(image) = preinitialized_memory_image { -+ match image { -+ PreinitializedMemoryImageMode::Apply(image) => image.apply( -+ ordinary_start_completed, -+ main_module, -+ &memory, -+ store, -+ memory_type, -+ &linker_state.main_module_dylink_info, -+ memory_base, -+ stack_low, -+ fresh_zeroed_memory, -+ ), -+ PreinitializedMemoryImageMode::Capture(capture) => capture.capture( -+ main_module, -+ &memory, -+ store, -+ memory_type, -+ &linker_state.main_module_dylink_info, -+ memory_base, -+ stack_low, -+ ), -+ }?; -+ } - instance_group.main_instance = Some(main_instance.clone()); - -- let tls_base = get_tls_base_export(&main_instance, store)?; -+ let tls_base = { get_tls_base_export(&main_instance, store) }?; - instance_group.main_instance_tls_base = tls_base; - - let runtime_path = linker_state.main_module_dylink_info.runtime_path.clone(); -@@ -1160,23 +1478,25 @@ impl Linker { - // guard.resolve_imports. - trace!(name = needed, "Loading module needed by main"); - let wasi_env = func_env.data(store); -- linker_state.load_module_tree( -- DlModuleSpec::FileSystem { -- module_spec: Path::new(needed.as_str()), -- ld_library_path, -- }, -- &mut link_state, -- &wasi_env.runtime, -- &wasi_env.state, -- runtime_path.as_ref(), -- // HACK: The main module doesn't have to exist in the virtual FS at all; e.g. -- // if one runs `wasmer ../module.wasm --volume .`, we won't have access to the -- // main module's folder within the virtual FS. This is why we're picking PWD -- // as the $ORIGIN of the main module, which should at least be slightly -- // sensible. The `main.wasm` file name will be stripped and only the `./` -- // will be taken into account by `locate_module`. -- Some(Path::new("./main.wasm")), -- )?; -+ { -+ linker_state.load_module_tree( -+ DlModuleSpec::FileSystem { -+ module_spec: Path::new(needed.as_str()), -+ ld_library_path, -+ }, -+ &mut link_state, -+ &wasi_env.runtime, -+ &wasi_env.state, -+ runtime_path.as_ref(), -+ // HACK: The main module doesn't have to exist in the virtual FS at all; e.g. -+ // if one runs `wasmer ../module.wasm --volume .`, we won't have access to the -+ // main module's folder within the virtual FS. This is why we're picking PWD -+ // as the $ORIGIN of the main module, which should at least be slightly -+ // sensible. The `main.wasm` file name will be stripped and only the `./` -+ // will be taken into account by `locate_module`. -+ Some(Path::new("./main.wasm")), -+ ) -+ }?; - } - - for module_handle in link_state -@@ -1186,13 +1506,15 @@ impl Linker { - .collect::>() - { - trace!(?module_handle, "Instantiating module"); -- instance_group.instantiate_side_module_from_link_state( -- &mut linker_state, -- store, -- &func_env.env, -- &mut link_state, -- module_handle, -- )?; -+ { -+ instance_group.instantiate_side_module_from_link_state( -+ &mut linker_state, -+ store, -+ &func_env.env, -+ &mut link_state, -+ module_handle, -+ ) -+ }?; - } - - let linker = Self { -@@ -1208,25 +1530,28 @@ impl Linker { - guard_size: 0, - tls_base, - }; -+ let mut main_module_instance_handles = WasiModuleInstanceHandles::new( -+ memory.clone(), -+ store, -+ main_instance.clone(), -+ Some(indirect_function_table.clone()), -+ ); -+ main_module_instance_handles.stack_pointer = Some(stack_pointer); - let module_handles = WasiModuleTreeHandles::Dynamic { - linker: linker.clone(), -- main_module_instance_handles: WasiModuleInstanceHandles::new( -- memory.clone(), -- store, -- main_instance.clone(), -- Some(indirect_function_table.clone()), -- ), -+ main_module_instance_handles, - }; - -- func_env -- .initialize_handles_and_layout( -+ { -+ func_env.initialize_handles_and_layout( - store, - main_instance.clone(), - module_handles, - Some(stack_layout), - true, - ) -- .map_err(LinkError::MainModuleHandleInitFailed)?; -+ } -+ .map_err(LinkError::MainModuleHandleInitFailed)?; - - { - trace!(?link_state, "Finalizing linking of main module"); -@@ -1235,23 +1560,33 @@ impl Linker { - let mut linker_state = linker.linker_state.write().unwrap(); - - let group_state = group_guard.as_mut().unwrap(); -- group_state.finalize_pending_globals( -- &mut linker_state, -- store, -- &link_state.unresolved_globals, -- )?; -+ { -+ group_state.finalize_pending_globals( -+ &mut linker_state, -+ store, -+ &link_state.unresolved_globals, -+ ) -+ }?; - - // The main module isn't added to the link state's list of new modules, so we need to - // call its initialization functions separately - trace!("Calling data relocator function for main module"); -- call_initialization_function::<()>(&main_instance, store, "__wasm_apply_data_relocs")?; -- call_initialization_function::<()>(&main_instance, store, "__wasm_apply_tls_relocs")?; -+ { -+ call_initialization_function::<()>( -+ &main_instance, -+ store, -+ "__wasm_apply_data_relocs", -+ ) -+ }?; -+ { -+ call_initialization_function::<()>(&main_instance, store, "__wasm_apply_tls_relocs") -+ }?; - -- linker.initialize_new_modules(group_guard, store, link_state)?; -+ { linker.initialize_new_modules(group_guard, store, link_state) }?; - } - - trace!("Calling main module's _initialize function"); -- call_initialization_function::<()>(&main_instance, store, "_initialize")?; -+ { call_initialization_function::<()>(&main_instance, store, "_initialize") }?; - - trace!("Link complete"); - -@@ -1289,11 +1624,14 @@ impl Linker { - let parent_store = parent_ctx.as_store_mut(); - - let main_module = linker_state.main_module.clone(); -- let memory = parent_group_state -- .memory -- .share_in_store(&parent_store, store)?; -+ let memory = { -+ parent_group_state -+ .memory -+ .share_in_store(&parent_store, store) -+ }?; - -- let mut imports = import_object_for_all_wasi_versions(&main_module, store, &func_env.env); -+ let mut imports = -+ { import_object_for_all_wasi_versions(&main_module, store, &func_env.env) }; - - let indirect_function_table_type = - parent_group_state.indirect_function_table.ty(&parent_store); -@@ -1303,7 +1641,7 @@ impl Linker { - ); - - let indirect_function_table = -- Table::new(store, indirect_function_table_type, Value::FuncRef(None)) -+ { Table::new(store, indirect_function_table_type, Value::FuncRef(None)) } - .map_err(LinkError::TableAllocationError)?; - - let expected_table_length = parent_group_state -@@ -1314,8 +1652,7 @@ impl Linker { - let current_size = indirect_function_table.size(store); - let delta = expected_table_length - current_size; - trace!(?current_size, ?delta, "Growing indirect function table"); -- indirect_function_table -- .grow(store, delta, Value::FuncRef(None)) -+ { indirect_function_table.grow(store, delta, Value::FuncRef(None)) } - .map_err(LinkError::TableAllocationError)?; - } - -@@ -1329,13 +1666,7 @@ impl Linker { - // FIXME: this needs to become a parameter if we ever decouple the linker from WASIX - let (stack_low, stack_high, tls_base) = { - let layout = &func_env.env.as_ref(store).layout; -- ( -- layout.stack_lower, -- layout.stack_upper, -- layout.tls_base.expect( -- "tls_base must be set in memory layout of new instance group's main instance", -- ), -- ) -+ (layout.stack_lower, layout.stack_upper, layout.tls_base) - }; - - trace!(stack_low, stack_high, "Memory layout"); -@@ -1359,9 +1690,9 @@ impl Linker { - - let mut instance_group = InstanceGroupState { - main_instance: None, -- main_instance_tls_base: Some(tls_base), -+ main_instance_tls_base: tls_base, - side_instances: HashMap::new(), -- stack_pointer, -+ stack_pointer: stack_pointer.clone(), - memory: memory.clone(), - indirect_function_table: indirect_function_table.clone(), - c_longjmp, -@@ -1381,37 +1712,48 @@ impl Linker { - ]; - - trace!("Populating imports object for new instance group's main instance"); -- instance_group.populate_imports_from_linker( -- MAIN_MODULE_HANDLE, -- &linker_state, -- store, -- &main_module, -- &mut imports, -- &func_env.env, -- &well_known_imports, -- &mut pending_resolutions, -- )?; -+ { -+ instance_group.populate_imports_from_linker( -+ MAIN_MODULE_HANDLE, -+ &linker_state, -+ store, -+ &main_module, -+ &mut imports, -+ &func_env.env, -+ &well_known_imports, -+ &mut pending_resolutions, -+ ) -+ }?; - -- let main_instance = Instance::new(store, &main_module, &imports)?; -+ let main_instance = { -+ Instance::new_with_export_names( -+ store, -+ &main_module, -+ &imports, -+ DYNAMIC_INSTANCE_INITIAL_EXPORTS, -+ ) -+ }?; - - instance_group.main_instance = Some(main_instance.clone()); - - for side in &linker_state.side_modules { - trace!(module_handle = ?side.0, "Instantiating existing side module"); -- instance_group.instantiate_side_module_from_linker( -- &linker_state, -- store, -- &func_env.env, -- *side.0, -- &mut pending_resolutions, -- )?; -+ { -+ instance_group.instantiate_side_module_from_linker( -+ &linker_state, -+ store, -+ &func_env.env, -+ *side.0, -+ &mut pending_resolutions, -+ ) -+ }?; - } - - trace!("Finalizing pending functions"); -- instance_group.finalize_pending_resolutions_from_linker(&pending_resolutions, store)?; -+ { instance_group.finalize_pending_resolutions_from_linker(&pending_resolutions, store) }?; - - trace!("Applying externally-requested function table entries"); -- instance_group.apply_requested_symbols_from_linker(store, &linker_state)?; -+ { instance_group.apply_requested_symbols_from_linker(store, &linker_state) }?; - - let linker = Self { - linker_state: self.linker_state.clone(), -@@ -1419,25 +1761,28 @@ impl Linker { - dl_operation_pending: self.dl_operation_pending.clone(), - }; - -+ let mut main_module_instance_handles = WasiModuleInstanceHandles::new( -+ memory.clone(), -+ store, -+ main_instance.clone(), -+ Some(indirect_function_table.clone()), -+ ); -+ main_module_instance_handles.stack_pointer = Some(stack_pointer); - let module_handles = WasiModuleTreeHandles::Dynamic { - linker: linker.clone(), -- main_module_instance_handles: WasiModuleInstanceHandles::new( -- memory.clone(), -- store, -- main_instance.clone(), -- Some(indirect_function_table.clone()), -- ), -+ main_module_instance_handles, - }; - -- func_env -- .initialize_handles_and_layout( -+ { -+ func_env.initialize_handles_and_layout( - store, - main_instance.clone(), - module_handles, - None, - false, - ) -- .map_err(LinkError::MainModuleHandleInitFailed)?; -+ } -+ .map_err(LinkError::MainModuleHandleInitFailed)?; - - trace!("Instance group spawned successfully"); - -@@ -2056,6 +2401,26 @@ impl Linker { - } - } - -+/// Reserves a fresh exec image's initial address window only after proving -+/// that inherited shared mappings can remain attached for the memory's full -+/// lifetime. The caller still owns rollback authority when this runs. -+fn reserve_exec_memory_window( -+ memory: &Memory, -+ store: &mut impl AsStoreMut, -+ initial_memory_end: u64, -+ has_exec_reservations: bool, -+) -> Result<(), LinkError> { -+ if has_exec_reservations && !memory.supports_persistent_shared_fixed_remap(store) { -+ return Err(LinkError::SharedMemoryExecReservation( -+ "inherited shared mappings require a supported nonmoving static linear memory" -+ .to_string(), -+ )); -+ } -+ -+ memory.grow_at_least(store, initial_memory_end)?; -+ Ok(()) -+} -+ - impl LinkerState { - fn allocate_memory( - &mut self, -@@ -2203,7 +2568,7 @@ impl LinkerState { - &self, - group: &InstanceGroupState, - import: &ImportType, -- store: &impl AsStoreRef, -+ store: &mut impl AsStoreMut, - ) -> Result { - let ExternType::Function(import_func_ty) = import.ty() else { - return Err(LinkError::ImportMustBeFunction( -@@ -2212,16 +2577,17 @@ impl LinkerState { - )); - }; - -- let export = group.resolve_exported_symbol(import.name()); -+ let export = group.resolve_exported_symbol(store, import.name()); - - match export { - Some((module_handle, export)) => { -+ let export_ty = export.ty(store); - let Extern::Function(export_func) = export else { - return Err(LinkError::ImportTypeMismatch( - "env".to_string(), - import.name().to_string(), - ExternType::Function(import_func_ty.clone()), -- export.ty(store).clone(), -+ export_ty, - )); - }; - -@@ -2230,7 +2596,7 @@ impl LinkerState { - "env".to_string(), - import.name().to_string(), - ExternType::Function(import_func_ty.clone()), -- export.ty(store).clone(), -+ ExternType::Function(export_func.ty(store)), - )); - } - -@@ -2255,18 +2621,19 @@ impl LinkerState { - &self, - group: &InstanceGroupState, - import: &ImportType, -- store: &impl AsStoreRef, -+ store: &mut impl AsStoreMut, - ) -> Result { - let global_type = get_integer_global_type_from_import(import)?; - -- match group.resolve_exported_symbol(import.name()) { -+ match group.resolve_exported_symbol(store, import.name()) { - Some((module_handle, export)) => { -- let ExternType::Global(global_type) = export.ty(store) else { -+ let export_ty = export.ty(store); -+ let ExternType::Global(global_type) = export_ty else { - return Err(LinkError::ImportTypeMismatch( - "GOT.mem".to_string(), - import.name().to_string(), - ExternType::Global(global_type), -- export.ty(store).clone(), -+ export.ty(store), - )); - }; - -@@ -2275,7 +2642,7 @@ impl LinkerState { - "GOT.mem".to_string(), - import.name().to_string(), - ExternType::Global(global_type), -- export.ty(store).clone(), -+ export.ty(store), - )); - } - -@@ -2292,12 +2659,12 @@ impl LinkerState { - &self, - group: &InstanceGroupState, - import: &ImportType, -- store: &impl AsStoreRef, -+ store: &mut impl AsStoreMut, - ) -> Result { - // Ensure the global is the correct type (i32 or i64) - let _ = get_integer_global_type_from_import(import)?; - -- match group.resolve_exported_symbol(import.name()) { -+ match group.resolve_exported_symbol(store, import.name()) { - Some((module_handle, export)) => { - let ExternType::Function(_) = export.ty(store) else { - return Err(LinkError::ExportMustBeFunction( -@@ -2643,7 +3010,12 @@ impl InstanceGroupState { - &well_known_imports, - )?; - -- let instance = Instance::new(store, &module, &imports)?; -+ let instance = Instance::new_with_export_names( -+ store, -+ &module, -+ &imports, -+ DYNAMIC_INSTANCE_INITIAL_EXPORTS, -+ )?; - - let instance_handles = WasiModuleInstanceHandles::new( - self.memory.clone(), -@@ -2756,7 +3128,12 @@ impl InstanceGroupState { - pending_resolutions, - )?; - -- let instance = Instance::new(store, &dl_module.module, &imports)?; -+ let instance = Instance::new_with_export_names( -+ store, -+ &dl_module.module, -+ &imports, -+ DYNAMIC_INSTANCE_INITIAL_EXPORTS, -+ )?; - - // This is a non-main instance of a side module, so it needs a new TLS area - let tls_base = call_initialization_function::(&instance, store, "__wasix_init_tls")? -@@ -2794,18 +3171,19 @@ impl InstanceGroupState { - trace!("Finalizing pending functions"); - - for pending in &pending_resolutions.functions { -- let func = self -- .instance(pending.resolved_from) -- .exports -- .get_function(&pending.name) -- .unwrap_or_else(|e| { -- panic!( -- "Internal error: failed to resolve exported function {}: {e:?}", -- pending.name -- ) -- }); -+ let func = lookup_exported_function( -+ self.instance(pending.resolved_from), -+ store, -+ &pending.name, -+ ) -+ .unwrap_or_else(|e| { -+ panic!( -+ "Internal error: failed to resolve exported function {}: {e:?}", -+ pending.name -+ ) -+ }); - -- self.place_in_function_table_at(store, func.clone(), pending.function_table_index) -+ self.place_in_function_table_at(store, func, pending.function_table_index) - .map_err(LinkError::TableAllocationError)?; - - trace!(?pending, "Placed pending function in table"); -@@ -2849,11 +3227,11 @@ impl InstanceGroupState { - panic!("Internal error: module {resolved_from:?} not loaded by this group") - }); - -- let func = instance.exports.get_function(name).unwrap_or_else(|e| { -+ let func = lookup_exported_function(instance, store, name).unwrap_or_else(|e| { - panic!("Internal error: failed to resolve exported function {name}: {e:?}") - }); - -- self.place_in_function_table_at(store, func.clone(), function_table_index) -+ self.place_in_function_table_at(store, func, function_table_index) - .map_err(LinkError::TableAllocationError)?; - - Ok(()) -@@ -2936,24 +3314,27 @@ impl InstanceGroupState { - Ok(()) - } - -- fn resolve_exported_symbol(&self, symbol: &str) -> Option<(ModuleHandle, &Extern)> { -- if let Some(export) = self -- .main_instance() -- .and_then(|instance| instance.exports.get_extern(symbol)) -+ fn resolve_exported_symbol( -+ &self, -+ store: &mut impl AsStoreMut, -+ symbol: &str, -+ ) -> Option<(ModuleHandle, Extern)> { -+ if let Some(instance) = self.main_instance() -+ && let Some(export) = instance.lookup_export(store, symbol) - { - trace!(symbol, from = ?MAIN_MODULE_HANDLE, ?export, "Resolved exported symbol"); -- Some((MAIN_MODULE_HANDLE, export)) -- } else { -- for (handle, dl_instance) in &self.side_instances { -- if let Some(export) = dl_instance.instance.exports.get_extern(symbol) { -- trace!(symbol, from = ?handle, ?export, "Resolved exported symbol"); -- return Some((*handle, export)); -- } -- } -+ return Some((MAIN_MODULE_HANDLE, export)); -+ } - -- trace!(symbol, "Failed to resolve exported symbol"); -- None -+ for (handle, dl_instance) in &self.side_instances { -+ if let Some(export) = dl_instance.instance.lookup_export(store, symbol) { -+ trace!(symbol, from = ?handle, ?export, "Resolved exported symbol"); -+ return Some((*handle, export)); -+ } - } -+ -+ trace!(symbol, "Failed to resolve exported symbol"); -+ None - } - - // This function populates the imports object for a single module from the given -@@ -3108,11 +3489,12 @@ impl InstanceGroupState { - - match resolution { - InProgressSymbolResolution::Function(module_handle) => { -- let func = self -- .instance(*module_handle) -- .exports -- .get_function(import.name()) -- .expect("Internal error: bad in-progress symbol resolution"); -+ let func = lookup_exported_function( -+ self.instance(*module_handle), -+ store, -+ import.name(), -+ ) -+ .expect("Internal error: bad in-progress symbol resolution"); - imports.define(import.module(), import.name(), func.clone()); - linker_state.symbol_resolution_records.insert( - SymbolResolutionKey::Needed(key.clone()), -@@ -3207,14 +3589,15 @@ impl InstanceGroupState { - } - - InProgressSymbolResolution::FuncGlobal(module_handle) => { -- let func = self -- .instance(*module_handle) -- .exports -- .get_function(import.name()) -- .expect("Internal error: bad in-progress symbol resolution"); -+ let func = lookup_exported_function( -+ self.instance(*module_handle), -+ store, -+ import.name(), -+ ) -+ .expect("Internal error: bad in-progress symbol resolution"); - - let func_handle = self -- .append_to_function_table(store, func.clone()) -+ .append_to_function_table(store, func) - .map_err(LinkError::TableAllocationError)?; - trace!( - ?module_handle, -@@ -3408,11 +3791,8 @@ impl InstanceGroupState { - ?resolved_from, - "Already have instance to resolve from" - ); -- instance -- .exports -- .get_function(import.name()) -+ lookup_exported_function(instance, store, import.name()) - .expect("Internal error: failed to get exported function") -- .clone() - } - // We may be loading a module tree, and the instance from which - // we're supposed to import the function may not exist yet, so -@@ -3452,15 +3832,14 @@ impl InstanceGroupState { - function_table_index, - } => { - let func = self.try_instance(*resolved_from).map(|instance| { -- instance -- .exports -- .get_function(import.name()) -- .unwrap_or_else(|e| { -+ lookup_exported_function(instance, store, import.name()).unwrap_or_else( -+ |e| { - panic!( - "Internal error: failed to resolve function {}: {e:?}", - import.name() - ) -- }) -+ }, -+ ) - }); - match func { - Some(func) => { -@@ -3470,12 +3849,8 @@ impl InstanceGroupState { - function_table_index, - "Placing function pointer into table" - ); -- self.place_in_function_table_at( -- store, -- func.clone(), -- *function_table_index, -- ) -- .map_err(LinkError::TableAllocationError)?; -+ self.place_in_function_table_at(store, func, *function_table_index) -+ .map_err(LinkError::TableAllocationError)?; - } - None => { - trace!( -@@ -3620,7 +3995,7 @@ impl InstanceGroupState { - allow_hidden: bool, - ) -> Result { - trace!(from = ?module_handle, symbol, "Resolving export from instance"); -- let export = instance.exports.get_extern(symbol).ok_or_else(|| { -+ let export = instance.lookup_export(store, symbol).ok_or_else(|| { - trace!(from = ?module_handle, symbol, "Not found"); - ResolveError::MissingExport - })?; -@@ -3638,12 +4013,15 @@ impl InstanceGroupState { - match export.ty(store) { - ExternType::Function(_) => { - trace!(from = ?module_handle, symbol, "Found function"); -- Ok(PartiallyResolvedExport::Function( -- Function::get_self_from_extern(export).unwrap().clone(), -- )) -+ let Extern::Function(function) = export else { -+ unreachable!("export type and value disagreed") -+ }; -+ Ok(PartiallyResolvedExport::Function(function)) - } - ty @ ExternType::Global(_) => { -- let global = Global::get_self_from_extern(export).unwrap(); -+ let Extern::Global(global) = export else { -+ unreachable!("export type and value disagreed") -+ }; - let value = match global.get(store) { - Value::I32(value) => value as u64, - Value::I64(value) => value as u64, -@@ -3712,7 +4090,7 @@ impl InstanceGroupState { - None => { - trace!(?requesting_module, name, "Resolving stub function"); - -- let (data, store) = env.data_and_store_mut(); -+ let (data, mut store) = env.data_and_store_mut(); - let env_inner = data.inner(); - // Safe to unwrap since we already know we're doing DL - let linker = env_inner.linker().unwrap(); -@@ -3777,12 +4155,12 @@ impl InstanceGroupState { - return Err(mk_error()); - } - -- let func = group_state -- .instance(*resolved_from) -- .exports -- .get_function(&name) -- .unwrap() -- .clone(); -+ let func = lookup_exported_function( -+ group_state.instance(*resolved_from), -+ &mut store, -+ &name, -+ ) -+ .unwrap(); - *resolved_guard = Some(Some(func.clone())); - func - } -@@ -3790,7 +4168,7 @@ impl InstanceGroupState { - trace!(?requesting_module, name, "Resolving function"); - - let Some((resolved_from, export)) = -- group_state.resolve_exported_symbol(name.as_str()) -+ group_state.resolve_exported_symbol(&mut store, name.as_str()) - else { - trace!(?requesting_module, name, "Failed to resolve symbol"); - *resolved_guard = Some(None); -@@ -4257,13 +4635,44 @@ fn set_integer_global( - Ok(()) - } - -+/// Resolves a function from either the small eager set or the VM's complete -+/// export metadata. Returning an owned handle is important: deferred lookup -+/// may populate the instance's identity cache while the caller holds only an -+/// immutable `Instance` reference. -+fn lookup_exported_function( -+ instance: &Instance, -+ store: &mut impl AsStoreMut, -+ name: &str, -+) -> Result { -+ match instance.lookup_export(store, name) { -+ Some(Extern::Function(function)) => Ok(function), -+ Some(_) => Err(ExportError::IncompatibleType), -+ None => Err(ExportError::Missing(name.to_string())), -+ } -+} -+ -+fn lookup_exported_global( -+ instance: &Instance, -+ store: &mut impl AsStoreMut, -+ name: &str, -+) -> Result { -+ match instance.lookup_export(store, name) { -+ Some(Extern::Global(global)) => Ok(global), -+ Some(_) => Err(ExportError::IncompatibleType), -+ None => Err(ExportError::Missing(name.to_string())), -+ } -+} -+ - fn call_initialization_function( - instance: &Instance, - store: &mut impl AsStoreMut, - name: &str, - ) -> Result, LinkError> { -- match instance.exports.get_typed_function::<(), Ret>(store, name) { -- Ok(f) => { -+ match lookup_exported_function(instance, store, name) { -+ Ok(function) => { -+ let f = function -+ .typed::<(), Ret>(store) -+ .map_err(|_| LinkError::InitFuncWithInvalidSignature(name.to_string()))?; - let ret = f - .call(store) - .map_err(|e| LinkError::InitFunctionFailed(name.to_string(), e))?; -@@ -4280,7 +4689,7 @@ fn get_tls_base_export( - instance: &Instance, - store: &mut impl AsStoreMut, - ) -> Result, LinkError> { -- match instance.exports.get_global("__tls_base") { -+ match lookup_exported_global(instance, store, "__tls_base") { - Ok(global) => match global.get(store) { - Value::I32(x) => Ok(Some(x as u64)), - Value::I64(x) => Ok(Some(x as u64)), -@@ -4291,6 +4700,143 @@ fn get_tls_base_export( - } - } - -+#[cfg(test)] -+mod dynamic_instance_export_tests { -+ use std::collections::HashSet; -+ -+ use super::DYNAMIC_INSTANCE_INITIAL_EXPORTS; -+ -+ #[test] -+ fn eager_exports_cover_every_runtime_handle_lookup_without_duplicates() { -+ const HANDLE_EXPORTS: &[&str] = &[ -+ "_start", -+ "_initialize", -+ "wasi_thread_start", -+ "__wasm_signal", -+ "asyncify_start_unwind", -+ "asyncify_stop_unwind", -+ "asyncify_start_rewind", -+ "asyncify_stop_rewind", -+ "asyncify_get_state", -+ ]; -+ -+ let eager: HashSet<_> = DYNAMIC_INSTANCE_INITIAL_EXPORTS.iter().copied().collect(); -+ assert_eq!(eager.len(), DYNAMIC_INSTANCE_INITIAL_EXPORTS.len()); -+ for name in HANDLE_EXPORTS { -+ assert!( -+ eager.contains(name), -+ "runtime handle export {name} must be materialized eagerly" -+ ); -+ } -+ } -+} -+ -+#[cfg(test)] -+mod single_slot_broadcast_tests { -+ use std::{sync::Arc, thread}; -+ -+ use super::{SingleSlotBroadcast, SingleSlotBroadcastInner}; -+ -+ #[test] -+ fn delivers_every_broadcast_to_every_existing_reader() { -+ let mut sender = SingleSlotBroadcast::new(); -+ let mut first = sender.add_rx(); -+ let mut second = sender.add_rx(); -+ -+ assert_eq!(sender.try_broadcast(41), Ok(())); -+ assert_eq!(first.recv(), Ok(41)); -+ assert_eq!(second.recv(), Ok(41)); -+ -+ assert_eq!(sender.try_broadcast(42), Ok(())); -+ assert_eq!(second.recv(), Ok(42)); -+ assert_eq!(first.recv(), Ok(42)); -+ } -+ -+ #[test] -+ fn rejects_a_full_slot_without_partial_delivery() { -+ let mut sender = SingleSlotBroadcast::new(); -+ let mut first = sender.add_rx(); -+ let mut second = sender.add_rx(); -+ -+ assert_eq!(sender.try_broadcast("first"), Ok(())); -+ assert_eq!(first.recv(), Ok("first")); -+ assert_eq!(sender.try_broadcast("rejected"), Err("rejected")); -+ assert_eq!(second.recv(), Ok("first")); -+ -+ assert_eq!(sender.try_broadcast("next"), Ok(())); -+ assert_eq!(first.recv(), Ok("next")); -+ assert_eq!(second.recv(), Ok("next")); -+ } -+ -+ #[test] -+ fn reader_added_to_an_occupied_channel_only_sees_future_values() { -+ let mut sender = SingleSlotBroadcast::new(); -+ let mut existing = sender.add_rx(); -+ assert_eq!(sender.try_broadcast(1), Ok(())); -+ -+ let mut added_later = sender.add_rx(); -+ assert_eq!(existing.recv(), Ok(1)); -+ assert_eq!(sender.try_broadcast(2), Ok(())); -+ assert_eq!(added_later.recv(), Ok(2)); -+ assert_eq!(existing.recv(), Ok(2)); -+ } -+ -+ #[test] -+ fn dropping_readers_updates_count_and_releases_the_slot() { -+ let mut sender = SingleSlotBroadcast::new(); -+ let first = sender.add_rx(); -+ let second = sender.add_rx(); -+ assert_eq!(sender.rx_count(), 2); -+ -+ assert_eq!(sender.try_broadcast(1), Ok(())); -+ drop(first); -+ assert_eq!(sender.rx_count(), 1); -+ assert_eq!(sender.try_broadcast(2), Err(2)); -+ -+ drop(second); -+ assert_eq!(sender.rx_count(), 0); -+ assert_eq!(sender.try_broadcast(3), Ok(())); -+ -+ let mut replacement = sender.add_rx(); -+ assert_eq!(sender.rx_count(), 1); -+ assert_eq!(sender.try_broadcast(4), Ok(())); -+ assert_eq!(replacement.recv(), Ok(4)); -+ } -+ -+ #[test] -+ fn closing_sender_wakes_a_blocked_reader() { -+ let mut sender = SingleSlotBroadcast::::new(); -+ let mut receiver = sender.add_rx(); -+ let inner = Arc::clone(&receiver.inner); -+ -+ let waiter = thread::spawn(move || receiver.recv()); -+ wait_until_reader_blocks(&inner); -+ drop(sender); -+ -+ assert_eq!(waiter.join().unwrap(), Err(std::sync::mpsc::RecvError)); -+ } -+ -+ #[test] -+ fn closing_sender_preserves_an_outstanding_delivery() { -+ let mut sender = SingleSlotBroadcast::new(); -+ let mut receiver = sender.add_rx(); -+ assert_eq!(sender.try_broadcast(42), Ok(())); -+ drop(sender); -+ -+ assert_eq!(receiver.recv(), Ok(42)); -+ assert_eq!(receiver.recv(), Err(std::sync::mpsc::RecvError)); -+ } -+ -+ fn wait_until_reader_blocks(inner: &Arc>) { -+ loop { -+ if inner.lock().waiting_readers != 0 { -+ return; -+ } -+ thread::yield_now(); -+ } -+ } -+} -+ - #[cfg(test)] - mod memory_allocator_tests { - use wasmer::{Engine, Memory, Store}; -@@ -4349,3 +4895,32 @@ mod memory_allocator_tests { - assert_eq!(addr, 2 * WASM_PAGE_SIZE + 512); - } - } -+ -+#[cfg(all(test, feature = "sys"))] -+mod shared_exec_memory_capability_tests { -+ use wasmer::sys::{BaseTunables, NativeEngineExt}; -+ use wasmer::{Engine, Memory, MemoryType, Pages, Store, WASM_PAGE_SIZE}; -+ -+ use super::{LinkError, reserve_exec_memory_window}; -+ -+ #[test] -+ fn unsupported_remap_capability_rejects_before_exec_window_growth() { -+ let mut engine = Engine::default(); -+ engine.set_tunables(BaseTunables { -+ static_memory_bound: Pages(0), -+ static_memory_offset_guard_size: 0, -+ dynamic_memory_offset_guard_size: 0, -+ }); -+ let mut store = Store::new(engine); -+ let memory = -+ Memory::new(&mut store, MemoryType::new(Pages(1), Some(Pages(2)), true)).unwrap(); -+ let size_before = memory.size(&store); -+ -+ let error = -+ reserve_exec_memory_window(&memory, &mut store, (WASM_PAGE_SIZE * 2) as u64, true) -+ .expect_err("dynamic memory must be rejected before exec reservation growth"); -+ -+ assert!(matches!(error, LinkError::SharedMemoryExecReservation(_))); -+ assert_eq!(memory.size(&store), size_before); -+ } -+} -diff --git a/lib/wasix/src/state/mod.rs b/lib/wasix/src/state/mod.rs -index fc3463c..c74fda1 100644 ---- a/lib/wasix/src/state/mod.rs -+++ b/lib/wasix/src/state/mod.rs -@@ -21,12 +21,14 @@ mod env; - mod func_env; - mod handles; - mod linker; -+mod preinitialized_memory_image; - mod types; - - use std::{ - collections::{BTreeMap, HashMap}, -+ ops::Bound::{Excluded, Unbounded}, - path::Path, -- sync::Mutex, -+ sync::{Arc, Mutex, Weak}, - task::Waker, - time::Duration, - }; -@@ -42,6 +44,18 @@ pub use self::{ - builder::*, - env::{WasiEnv, WasiEnvInit, WasiModuleInstanceHandles, WasiModuleTreeHandles}, - func_env::WasiFunctionEnv, -+ preinitialized_memory_image::{ -+ DETERMINISTIC_START_ANALYZER_POLICY, DETERMINISTIC_START_GLOBAL_EFFECTS, -+ DETERMINISTIC_START_MEMORY_EFFECTS, DETERMINISTIC_START_MEMORY_READS, -+ DETERMINISTIC_START_PROOF_SCHEMA, DETERMINISTIC_START_TABLE_EFFECTS, -+ DeterministicStartProof, IntrinsicFileImmutability, PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, -+ PREINITIALIZED_MEMORY_IMAGE_PHASE, PreinitializedMemoryImage, -+ PreinitializedMemoryImageBacking, PreinitializedMemoryImageCapture, -+ PreinitializedMemoryImageHandle, PreinitializedMemoryImageLoadAudit, -+ PreinitializedMemoryImageLoader, PreinitializedMemoryImageMetadata, -+ PreinitializedMemoryImageMode, PreinitializedMemoryImageRuntimeAudit, -+ intrinsic_file_immutability, -+ }, - types::*, - }; - pub use crate::fs::{InodeGuard, InodeWeakGuard}; -@@ -121,6 +135,561 @@ pub(crate) struct WasiFutexState { - pub futexes: HashMap, - } - -+#[derive(Debug, Clone, Copy, Hash, Eq, Ord, PartialEq, PartialOrd)] -+pub(crate) struct WasiSharedFileIdentity { -+ device: u64, -+ file: u64, -+} -+ -+impl WasiSharedFileIdentity { -+ #[cfg(unix)] -+ fn for_file(file: &std::fs::File) -> Result { -+ use std::os::unix::fs::MetadataExt; -+ -+ let metadata = file.metadata().map_err(|_| Errno::Io)?; -+ Ok(Self { -+ device: metadata.dev(), -+ file: metadata.ino(), -+ }) -+ } -+ -+ #[cfg(windows)] -+ fn for_file(file: &std::fs::File) -> Result { -+ use std::os::windows::fs::MetadataExt; -+ -+ let metadata = file.metadata().map_err(|_| Errno::Io)?; -+ Ok(Self { -+ device: metadata.volume_serial_number().ok_or(Errno::Notsup)? as u64, -+ file: metadata.file_index().ok_or(Errno::Notsup)?, -+ }) -+ } -+ -+ #[cfg(not(any(unix, windows)))] -+ fn for_file(_file: &std::fs::File) -> Result { -+ Err(Errno::Notsup) -+ } -+} -+ -+/// Number of weak shared-futex registry slots inspected on each lookup. -+/// -+/// Registries normally remove their own slot when the last live mapping or -+/// waiter drops. This bounded sweep is a deterministic backstop for a lookup -+/// racing that final drop without making mmap latency proportional to the -+/// lifetime number of mapped files. -+const SHARED_FUTEX_REGISTRY_PRUNE_BUDGET: usize = 16; -+ -+#[derive(Debug)] -+pub(crate) struct WasiSharedFutexRegistry { -+ futexs: Mutex, -+ -+ // The mapping's live file description is the generation token for the -+ // device/inode identity above. Keeping the same Arc with the wait registry -+ // prevents the backing inode (or Windows file index) from being recycled -+ // while a mapping or an in-flight futex wait can still reach old wait state. -+ _file_anchor: Arc, -+ identity: WasiSharedFileIdentity, -+ generation: Arc<()>, -+ owner: Weak>, -+} -+ -+impl WasiSharedFutexRegistry { -+ #[cfg(test)] -+ fn detached(file_anchor: Arc) -> Arc { -+ Arc::new(Self { -+ futexs: Default::default(), -+ identity: WasiSharedFileIdentity::for_file(&file_anchor).unwrap(), -+ _file_anchor: file_anchor, -+ generation: Arc::new(()), -+ owner: Weak::new(), -+ }) -+ } -+} -+ -+impl Drop for WasiSharedFutexRegistry { -+ fn drop(&mut self) { -+ let Some(owner) = self.owner.upgrade() else { -+ return; -+ }; -+ -+ // Drop must not panic while unwinding. Recovering a poisoned table is -+ // safe here because removing this exact generation is self-contained. -+ let mut owner = owner.lock().unwrap_or_else(|err| err.into_inner()); -+ owner.remove_if_current(self.identity, &self.generation); -+ } -+} -+ -+#[derive(Debug)] -+struct WasiSharedFutexRegistryEntry { -+ generation: Arc<()>, -+ registry: Weak, -+} -+ -+#[derive(Debug, Default)] -+pub(crate) struct WasiSharedFutexRegistries { -+ entries: BTreeMap, -+ prune_cursor: Option, -+} -+ -+impl WasiSharedFutexRegistries { -+ fn registry_for_file( -+ owner: &Arc>, -+ file_anchor: Arc, -+ ) -> Result, Errno> { -+ let identity = WasiSharedFileIdentity::for_file(&file_anchor)?; -+ -+ let mut registries = owner.lock().unwrap_or_else(|err| err.into_inner()); -+ registries.prune_stale(SHARED_FUTEX_REGISTRY_PRUNE_BUDGET); -+ -+ if let Some(registry) = registries -+ .entries -+ .get(&identity) -+ .and_then(|entry| entry.registry.upgrade()) -+ { -+ return Ok(registry); -+ } -+ registries.entries.remove(&identity); -+ -+ let generation = Arc::new(()); -+ let registry = Arc::new(WasiSharedFutexRegistry { -+ futexs: Default::default(), -+ _file_anchor: file_anchor, -+ identity, -+ generation: generation.clone(), -+ owner: Arc::downgrade(owner), -+ }); -+ registries.entries.insert( -+ identity, -+ WasiSharedFutexRegistryEntry { -+ generation, -+ registry: Arc::downgrade(®istry), -+ }, -+ ); -+ Ok(registry) -+ } -+ -+ fn remove_if_current(&mut self, identity: WasiSharedFileIdentity, generation: &Arc<()>) { -+ let is_current = self -+ .entries -+ .get(&identity) -+ .is_some_and(|entry| Arc::ptr_eq(&entry.generation, generation)); -+ if is_current { -+ self.entries.remove(&identity); -+ if self.entries.is_empty() { -+ self.prune_cursor = None; -+ } -+ } -+ } -+ -+ fn prune_stale(&mut self, budget: usize) -> usize { -+ if budget == 0 || self.entries.is_empty() { -+ return 0; -+ } -+ -+ let mut keys = Vec::with_capacity(budget.min(self.entries.len())); -+ if let Some(cursor) = self.prune_cursor { -+ keys.extend( -+ self.entries -+ .range((Excluded(cursor), Unbounded)) -+ .take(budget) -+ .map(|(identity, _)| *identity), -+ ); -+ let remaining = budget.saturating_sub(keys.len()); -+ keys.extend( -+ self.entries -+ .range(..=cursor) -+ .take(remaining) -+ .map(|(identity, _)| *identity), -+ ); -+ } else { -+ keys.extend(self.entries.keys().take(budget).copied()); -+ } -+ -+ self.prune_cursor = keys.last().copied(); -+ let before = self.entries.len(); -+ for identity in keys { -+ let stale = self -+ .entries -+ .get(&identity) -+ .is_some_and(|entry| entry.registry.strong_count() == 0); -+ if stale { -+ self.entries.remove(&identity); -+ } -+ } -+ if self.entries.is_empty() { -+ self.prune_cursor = None; -+ } -+ before - self.entries.len() -+ } -+ -+ #[cfg(test)] -+ fn counts(&self) -> (usize, usize) { -+ self.entries -+ .values() -+ .fold((0usize, 0usize), |(active, stale), entry| { -+ if entry.registry.strong_count() == 0 { -+ (active, stale + 1) -+ } else { -+ (active + 1, stale) -+ } -+ }) -+ } -+} -+ -+#[derive(Debug, Clone)] -+pub(crate) enum WasiFutexRegistry { -+ Private(Arc), -+ Shared(Arc), -+} -+ -+impl WasiFutexRegistry { -+ pub(crate) fn resolve(state: &Arc, addr: u64) -> Result<(Self, u64), Errno> { -+ if let Some((futexs, key)) = state.shared_futex_registry(addr)? { -+ Ok((Self::Shared(futexs), key)) -+ } else { -+ Ok((Self::Private(state.clone()), addr)) -+ } -+ } -+ -+ pub(crate) fn with(&self, f: impl FnOnce(&mut WasiFutexState) -> R) -> R { -+ match self { -+ Self::Private(state) => { -+ let mut guard = state.futexs.lock().unwrap(); -+ f(&mut guard) -+ } -+ Self::Shared(futexs) => { -+ let mut guard = futexs.futexs.lock().unwrap(); -+ f(&mut guard) -+ } -+ } -+ } -+} -+ -+#[derive(Debug, Clone)] -+pub(crate) struct WasiSharedMemoryMapping { -+ pub start: u64, -+ pub len: u64, -+ pub file: Arc, -+ pub file_offset: u64, -+ pub futexs: Arc, -+} -+ -+impl WasiSharedMemoryMapping { -+ fn end(&self) -> Result { -+ self.start.checked_add(self.len).ok_or(Errno::Overflow) -+ } -+ -+ fn same_exec_reservation(&self, candidate: &Self) -> bool { -+ self.start == candidate.start -+ && self.same_exec_backing(candidate.len, candidate.file_offset, &candidate.futexs) -+ } -+ -+ fn same_exec_backing( -+ &self, -+ len: u64, -+ file_offset: u64, -+ futexs: &Arc, -+ ) -> bool { -+ self.len == len && self.file_offset == file_offset && Arc::ptr_eq(&self.futexs, futexs) -+ } -+} -+ -+#[derive(Debug, Default, Clone)] -+pub(crate) struct WasiSharedMemoryMappings { -+ mappings: Vec, -+} -+ -+impl WasiSharedMemoryMappings { -+ pub fn snapshot(&self) -> Vec { -+ self.mappings.clone() -+ } -+ -+ fn max_end(&self) -> Result, Errno> { -+ self.mappings.iter().try_fold(None, |maximum, mapping| { -+ let end = mapping.end()?; -+ Ok(Some(maximum.map_or(end, |old: u64| old.max(end)))) -+ }) -+ } -+ -+ fn overlaps(&self, start: u64, len: u64) -> Result { -+ if len == 0 { -+ return Err(Errno::Inval); -+ } -+ let end = start.checked_add(len).ok_or(Errno::Overflow)?; -+ for mapping in &self.mappings { -+ let mapping_end = mapping.end()?; -+ if mapping.start < end && start < mapping_end { -+ return Ok(true); -+ } -+ } -+ Ok(false) -+ } -+ -+ fn covers(&self, start: u64, len: u64) -> Result { -+ if len == 0 { -+ return Err(Errno::Inval); -+ } -+ let end = start.checked_add(len).ok_or(Errno::Overflow)?; -+ let mut covered_until = start; -+ for mapping in &self.mappings { -+ let mapping_end = mapping.end()?; -+ if mapping_end <= covered_until { -+ continue; -+ } -+ if mapping.start > covered_until { -+ return Ok(false); -+ } -+ covered_until = mapping_end; -+ if covered_until >= end { -+ return Ok(true); -+ } -+ } -+ Ok(false) -+ } -+ -+ fn exact_exec_reservation_index(&self, candidate: &WasiSharedMemoryMapping) -> Option { -+ self.mappings -+ .iter() -+ .position(|mapping| mapping.same_exec_reservation(candidate)) -+ } -+ -+ fn exec_backing_reservation_index( -+ &self, -+ len: u64, -+ file_offset: u64, -+ futexs: &Arc, -+ ) -> Option { -+ self.mappings -+ .iter() -+ .position(|mapping| mapping.same_exec_backing(len, file_offset, futexs)) -+ } -+ -+ fn replacing(&self, mapping: WasiSharedMemoryMapping) -> Result { -+ if mapping.len == 0 { -+ return Err(Errno::Inval); -+ } -+ -+ let start = mapping.start; -+ let end = mapping.end()?; -+ let mut updated = Vec::with_capacity(self.mappings.len() + 1); -+ -+ for old in &self.mappings { -+ let old_end = old.end()?; -+ if old_end <= start || old.start >= end { -+ updated.push(old.clone()); -+ continue; -+ } -+ -+ if old.start < start { -+ updated.push(WasiSharedMemoryMapping { -+ start: old.start, -+ len: start - old.start, -+ file: old.file.clone(), -+ file_offset: old.file_offset, -+ futexs: old.futexs.clone(), -+ }); -+ } -+ -+ if old_end > end { -+ updated.push(WasiSharedMemoryMapping { -+ start: end, -+ len: old_end - end, -+ file: old.file.clone(), -+ file_offset: old -+ .file_offset -+ .checked_add(end - old.start) -+ .ok_or(Errno::Overflow)?, -+ futexs: old.futexs.clone(), -+ }); -+ } -+ } -+ -+ updated.push(mapping); -+ updated.sort_by_key(|mapping| mapping.start); -+ Ok(Self { mappings: updated }) -+ } -+ -+ pub fn replace(&mut self, mapping: WasiSharedMemoryMapping) -> Result<(), Errno> { -+ *self = self.replacing(mapping)?; -+ Ok(()) -+ } -+ -+ fn removing(&self, start: u64, len: u64) -> Result { -+ if len == 0 { -+ return Err(Errno::Inval); -+ } -+ -+ let end = start.checked_add(len).ok_or(Errno::Overflow)?; -+ let mut updated = Vec::with_capacity(self.mappings.len()); -+ -+ for old in &self.mappings { -+ let old_end = old.end()?; -+ if old_end <= start || old.start >= end { -+ updated.push(old.clone()); -+ continue; -+ } -+ -+ if old.start < start { -+ updated.push(WasiSharedMemoryMapping { -+ start: old.start, -+ len: start - old.start, -+ file: old.file.clone(), -+ file_offset: old.file_offset, -+ futexs: old.futexs.clone(), -+ }); -+ } -+ -+ if old_end > end { -+ updated.push(WasiSharedMemoryMapping { -+ start: end, -+ len: old_end - end, -+ file: old.file.clone(), -+ file_offset: old -+ .file_offset -+ .checked_add(end - old.start) -+ .ok_or(Errno::Overflow)?, -+ futexs: old.futexs.clone(), -+ }); -+ } -+ } -+ -+ updated.sort_by_key(|mapping| mapping.start); -+ Ok(Self { mappings: updated }) -+ } -+ -+ #[cfg(test)] -+ pub fn remove(&mut self, start: u64, len: u64) -> Result<(), Errno> { -+ *self = self.removing(start, len)?; -+ Ok(()) -+ } -+ -+ fn shared_futex_registry(&self, addr: u64) -> Option<(Arc, u64)> { -+ let idx = self -+ .mappings -+ .partition_point(|mapping| mapping.start <= addr); -+ let mapping = self.mappings.get(idx.checked_sub(1)?)?; -+ let end = mapping.end().ok()?; -+ if addr >= end { -+ return None; -+ } -+ -+ let key = mapping.file_offset.checked_add(addr - mapping.start)?; -+ Some((mapping.futexs.clone(), key)) -+ } -+} -+ -+/// Identifies the fixed-address path. Runtime-selected mappings use a separate -+/// API that atomically reserves fresh WebAssembly pages and therefore never -+/// need permission to replace existing private memory. -+#[derive(Debug, Clone, Copy, Eq, PartialEq)] -+pub(crate) enum WasiSharedMemoryMapOrigin { -+ GuestFixed, -+} -+ -+#[derive(Debug, Clone, Copy, Eq, PartialEq)] -+pub(crate) enum WasiSharedMemoryRuntimePlacement { -+ Fresh { minimum_start: u64 }, -+ Inherited { start: u64 }, -+} -+ -+/// Serializes the address-space handoff from an old process image to a fresh -+/// EXEC_BACKEND image. The two independent completion conditions are recorded -+/// explicitly: the linker must install the reserved address window, and the -+/// task manager must commit successor admission. Exact reattachment alone can -+/// therefore never discard rollback authority. -+#[derive(Debug, Clone, Copy, Eq, PartialEq)] -+pub(crate) enum WasiSharedMemoryExecPhase { -+ Normal, -+ Transition { -+ reservations_installed: bool, -+ successor_committed: bool, -+ }, -+} -+ -+impl Default for WasiSharedMemoryExecPhase { -+ fn default() -> Self { -+ Self::Normal -+ } -+} -+ -+/// Rollback guard for the point at which an old process image gives its active -+/// shared mappings to a fresh exec image as address-only reservations. Spawn -+/// failures restore the old process-visible registry exactly; a successfully -+/// accepted successor commits the transition. -+#[derive(Debug)] -+pub(crate) struct WasiSharedMemoryExecTransition { -+ state: Arc, -+ original_mappings: Option>, -+} -+ -+impl WasiSharedMemoryExecTransition { -+ pub(crate) fn commit(mut self) { -+ if self.original_mappings.is_some() { -+ self.state.commit_shared_memory_exec_transition(); -+ } -+ self.original_mappings.take(); -+ } -+} -+ -+impl Drop for WasiSharedMemoryExecTransition { -+ fn drop(&mut self) { -+ let Some(original_mappings) = self.original_mappings.take() else { -+ return; -+ }; -+ -+ // Use the same lock order as every multi-registry mapping operation. -+ let mut phase = self -+ .state -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if !matches!(*phase, WasiSharedMemoryExecPhase::Transition { .. }) { -+ tracing::error!( -+ ?phase, -+ "shared-memory exec rollback refused after phase advanced" -+ ); -+ return; -+ } -+ let mut reservations = self -+ .state -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let mut active = self -+ .state -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ // Linker pre-growth marks reservations installed before synchronous -+ // Instance::new can fail. A start function may also have consumed -+ // exact reservations before trapping. Rollback is still safe when the -+ // reservation+active partition is exactly the original set: all host -+ // remaps occurred in the discarded fresh memory, while the old image's -+ // mapping never moved. -+ let mut current = reservations.snapshot(); -+ current.extend(active.snapshot()); -+ current.sort_by_key(|mapping| (mapping.start, mapping.len, mapping.file_offset)); -+ let mut original = original_mappings.clone(); -+ original.sort_by_key(|mapping| (mapping.start, mapping.len, mapping.file_offset)); -+ let exact_partition = current.len() == original.len() -+ && current -+ .iter() -+ .zip(&original) -+ .all(|(current, original)| original.same_exec_reservation(current)); -+ if !exact_partition { -+ tracing::error!( -+ reservations = reservations.mappings.len(), -+ active = active.mappings.len(), -+ original = original_mappings.len(), -+ "shared-memory exec rollback refused for a non-exact successor partition" -+ ); -+ return; -+ } -+ reservations.mappings.clear(); -+ active.mappings = original_mappings; -+ *phase = WasiSharedMemoryExecPhase::Normal; -+ } -+} -+ - /// Top level data type containing all* the state with which WASI can - /// interact. - /// -@@ -139,7 +708,14 @@ pub(crate) struct WasiState { - pub args: Mutex>, - pub envs: Mutex>>, - pub signals: Mutex>, -- -+ #[cfg_attr(feature = "enable-serde", serde(skip))] -+ pub(crate) shared_futex_registries: Arc>, -+ #[cfg_attr(feature = "enable-serde", serde(skip))] -+ pub(crate) shared_memory_mappings: Mutex, -+ #[cfg_attr(feature = "enable-serde", serde(skip))] -+ pub(crate) shared_memory_exec_phase: Mutex, -+ #[cfg_attr(feature = "enable-serde", serde(skip))] -+ pub(crate) shared_memory_exec_reservations: Mutex, - // TODO: should not be here, since this requires active work to resolve. - // State should only hold active runtime state that can be reproducibly re-created. - pub preopen: Vec, -@@ -210,6 +786,465 @@ impl WasiState { - self.fs.root_fs.new_open_options() - } - -+ #[cfg(test)] -+ pub(crate) fn shared_memory_mappings(&self) -> Vec { -+ self.shared_memory_mappings.lock().unwrap().snapshot() -+ } -+ -+ /// Converts the current image's active shared mappings into reservations -+ /// for a fresh exec image. Reservations deliberately do not participate in -+ /// futex address resolution until the child proves and installs the exact -+ /// backing object again. -+ pub(crate) fn prepare_shared_memory_for_exec( -+ self: &Arc, -+ ) -> Result { -+ // Global order: phase, reservations, active mappings. -+ let mut phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if *phase != WasiSharedMemoryExecPhase::Normal { -+ return Err(Errno::Busy); -+ } -+ let mut reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if !reservations.mappings.is_empty() { -+ return Err(Errno::Busy); -+ } -+ let mut active = self -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let original_mappings = active.snapshot(); -+ if original_mappings.is_empty() { -+ return Ok(WasiSharedMemoryExecTransition { -+ state: self.clone(), -+ original_mappings: None, -+ }); -+ } -+ *phase = WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: false, -+ successor_committed: false, -+ }; -+ reservations.mappings = std::mem::take(&mut active.mappings); -+ drop(active); -+ drop(reservations); -+ drop(phase); -+ -+ Ok(WasiSharedMemoryExecTransition { -+ state: self.clone(), -+ original_mappings: Some(original_mappings), -+ }) -+ } -+ -+ #[cfg(test)] -+ pub(crate) fn shared_memory_exec_reservation_end(&self) -> Result, Errno> { -+ self.shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()) -+ .max_end() -+ } -+ -+ pub(crate) fn shared_memory_exec_reservations(&self) -> Vec { -+ self.shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()) -+ .snapshot() -+ } -+ -+ pub(crate) fn has_shared_memory_exec_reservations(&self) -> bool { -+ !self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()) -+ .mappings -+ .is_empty() -+ } -+ -+ /// Advances a rollback-capable exec handoff only after the linker has -+ /// grown the fresh memory above every reserved range. Once installed, the -+ /// guest may consume reservations only through exact fixed reattachments. -+ pub(crate) fn mark_shared_memory_exec_reservations_installed(&self) -> Result<(), Errno> { -+ let mut phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if reservations.mappings.is_empty() { -+ return Err(Errno::Inval); -+ } -+ match *phase { -+ WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: false, -+ successor_committed, -+ } => { -+ *phase = WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: true, -+ successor_committed, -+ }; -+ } -+ _ => return Err(Errno::Busy), -+ } -+ Ok(()) -+ } -+ -+ fn commit_shared_memory_exec_transition(&self) { -+ let mut phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ match *phase { -+ WasiSharedMemoryExecPhase::Transition { -+ reservations_installed, -+ .. -+ } if reservations_installed && reservations.mappings.is_empty() => { -+ *phase = WasiSharedMemoryExecPhase::Normal; -+ } -+ WasiSharedMemoryExecPhase::Transition { -+ reservations_installed, -+ .. -+ } => { -+ *phase = WasiSharedMemoryExecPhase::Transition { -+ reservations_installed, -+ successor_committed: true, -+ }; -+ } -+ WasiSharedMemoryExecPhase::Normal => { -+ // A transition with inherited mappings cannot reach Normal -+ // before this commit. Keep the task live, but surface an -+ // internal invariant violation loudly in debug telemetry. -+ tracing::error!("shared-memory exec commit observed no active transition"); -+ } -+ } -+ } -+ -+ /// Runs the host remap and publishes its active mapping as one atomic state -+ /// transition. A reattach must match range, file offset, and the stable -+ /// shared-backing registry identity inherited across exec. Other fixed -+ /// remaps require full coverage by an already active mapping. -+ pub(crate) fn install_shared_memory_mapping( -+ &self, -+ mapping: WasiSharedMemoryMapping, -+ origin: WasiSharedMemoryMapOrigin, -+ remap: impl FnOnce() -> Result<(), Errno>, -+ ) -> Result<(), Errno> { -+ let mut phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let mut reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let mut active = self -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ -+ let reservation_index = match (*phase, origin) { -+ ( -+ WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: false, -+ .. -+ }, -+ _, -+ ) => return Err(Errno::Busy), -+ ( -+ WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: true, -+ .. -+ }, -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ ) => Some( -+ reservations -+ .exact_exec_reservation_index(&mapping) -+ .ok_or(Errno::Inval)?, -+ ), -+ (WasiSharedMemoryExecPhase::Normal, WasiSharedMemoryMapOrigin::GuestFixed) => { -+ if reservations.mappings.is_empty() && active.covers(mapping.start, mapping.len)? { -+ // POSIX also permits MAP_FIXED to replace a known mapping -+ // in the same image. Requiring full active coverage keeps -+ // that behavior without allowing a raw fixed request to -+ // overwrite an untracked heap allocation. -+ None -+ } else { -+ return Err(Errno::Inval); -+ } -+ } -+ }; -+ -+ // Compute every fallible metadata transformation before touching the -+ // host mapping. Publication after remap is an infallible assignment. -+ let updated_active = active.replacing(mapping)?; -+ remap()?; -+ -+ if let Some(index) = reservation_index { -+ reservations.mappings.remove(index); -+ } -+ *active = updated_active; -+ if reservations.mappings.is_empty() -+ && matches!( -+ *phase, -+ WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: true, -+ successor_committed: true, -+ } -+ ) -+ { -+ *phase = WasiSharedMemoryExecPhase::Normal; -+ } -+ Ok(()) -+ } -+ -+ /// Atomically installs an original non-fixed `mmap`. During exec, a -+ /// request for an inherited backing consumes its matching reservation and -+ /// reuses that address; otherwise `reserve_and_remap` must use WebAssembly -+ /// memory.grow and return its old end, which is unique even when another -+ /// guest thread grows the heap concurrently. Holding the mapping locks -+ /// across either operation serializes it with fixed remaps and keeps the -+ /// registry publication atomic. -+ pub(crate) fn install_runtime_selected_shared_memory_mapping( -+ &self, -+ len: u64, -+ file: Arc, -+ file_offset: u64, -+ futexs: Arc, -+ reserve_and_remap: impl FnOnce(WasiSharedMemoryRuntimePlacement) -> Result, -+ ) -> Result { -+ if len == 0 { -+ return Err(Errno::Inval); -+ } -+ -+ let mut phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let mut reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let mut active = self -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ -+ if let WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: true, -+ successor_committed, -+ } = *phase -+ { -+ let reservation_index = reservations -+ .exec_backing_reservation_index(len, file_offset, &futexs) -+ .ok_or(Errno::Busy)?; -+ let start = reservations.mappings[reservation_index].start; -+ let mapping = WasiSharedMemoryMapping { -+ start, -+ len, -+ file, -+ file_offset, -+ futexs, -+ }; -+ let updated_active = active.replacing(mapping)?; -+ let selected = -+ reserve_and_remap(WasiSharedMemoryRuntimePlacement::Inherited { start })?; -+ if selected != start { -+ return Err(Errno::Inval); -+ } -+ reservations.mappings.remove(reservation_index); -+ *active = updated_active; -+ if reservations.mappings.is_empty() && successor_committed { -+ *phase = WasiSharedMemoryExecPhase::Normal; -+ } -+ return Ok(start); -+ } -+ if *phase != WasiSharedMemoryExecPhase::Normal { -+ return Err(Errno::Busy); -+ } -+ -+ let minimum_start = reservations -+ .max_end()? -+ .into_iter() -+ .chain(active.max_end()?) -+ .max() -+ .unwrap_or(0); -+ let start = reserve_and_remap(WasiSharedMemoryRuntimePlacement::Fresh { minimum_start })?; -+ let _end = start.checked_add(len).ok_or(Errno::Overflow)?; -+ if start < minimum_start -+ || reservations.overlaps(start, len)? -+ || active.overlaps(start, len)? -+ { -+ return Err(Errno::Inval); -+ } -+ -+ active.replace(WasiSharedMemoryMapping { -+ start, -+ len, -+ file, -+ file_offset, -+ futexs, -+ })?; -+ Ok(start) -+ } -+ -+ pub(crate) fn futex_registry_for_shared_file( -+ &self, -+ file: Arc, -+ ) -> Result, Errno> { -+ WasiSharedFutexRegistries::registry_for_file(&self.shared_futex_registries, file) -+ } -+ -+ /// Serializes a host-file shrink against creation and lifetime of shared mappings for the -+ /// same backing inode. The returned guard must be kept alive until the shrinking operation -+ /// has completed. Growth is harmless and returns no guard. -+ #[cfg(feature = "host-fs")] -+ pub(crate) fn guard_shared_mapping_file_shrink( -+ &self, -+ file: &std::fs::File, -+ new_len: u64, -+ ) -> Result>, Errno> { -+ let current_len = file.metadata().map_err(|_| Errno::Io)?.len(); -+ if new_len >= current_len { -+ return Ok(None); -+ } -+ -+ let identity = WasiSharedFileIdentity::for_file(file)?; -+ let registries = self -+ .shared_futex_registries -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if registries -+ .entries -+ .get(&identity) -+ .and_then(|entry| entry.registry.upgrade()) -+ .is_some() -+ { -+ return Err(Errno::Busy); -+ } -+ Ok(Some(registries)) -+ } -+ -+ fn shared_futex_registry( -+ &self, -+ addr: u64, -+ ) -> Result, u64)>, Errno> { -+ let phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let active = self -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ -+ if let Some(resolved) = active.shared_futex_registry(addr) { -+ return Ok(Some(resolved)); -+ } -+ if matches!(*phase, WasiSharedMemoryExecPhase::Transition { .. }) -+ && reservations.shared_futex_registry(addr).is_some() -+ { -+ return Err(Errno::Busy); -+ } -+ Ok(None) -+ } -+ -+ #[cfg(test)] -+ pub(crate) fn replace_shared_memory_mapping( -+ &self, -+ mapping: WasiSharedMemoryMapping, -+ ) -> Result<(), Errno> { -+ self.shared_memory_mappings.lock().unwrap().replace(mapping) -+ } -+ -+ /// Runs an operation only while the requested range is fully backed by -+ /// active shared mappings. Exec reservations are deliberately excluded: -+ /// they describe addresses that a fresh image must not touch until it has -+ /// reattached the exact backing object. `Noent` has the ABI-significant -+ /// meaning that the raw requested range has zero runtime ownership, so -+ /// wasix-libc may dispatch to its legacy malloc-backed implementation. -+ pub(crate) fn with_active_shared_memory_mapping( -+ &self, -+ start: u64, -+ requested_len: u64, -+ operation_len: u64, -+ operation: impl FnOnce() -> Result<(), Errno>, -+ ) -> Result<(), Errno> { -+ let phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if *phase != WasiSharedMemoryExecPhase::Normal { -+ return Err(Errno::Busy); -+ } -+ let reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let active = self -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if !active.overlaps(start, requested_len)? { -+ return Err(Errno::Noent); -+ } -+ if reservations.overlaps(start, requested_len)? || !active.covers(start, operation_len)? { -+ return Err(Errno::Inval); -+ } -+ operation() -+ } -+ -+ /// Restores private memory and removes an active shared range as one state -+ /// transition. The host operation runs only after ownership validation and -+ /// while both mapping registries are locked, so a rejected or racing -+ /// `munmap` cannot mutate an exec reservation or unrelated guest memory. -+ /// `Noent` is returned only for zero runtime overlap; partial ownership is -+ /// `Inval` and must never fall through to guest legacy metadata. -+ pub(crate) fn remove_shared_memory_mapping( -+ &self, -+ start: u64, -+ requested_len: u64, -+ operation_len: u64, -+ restore_private: impl FnOnce() -> Result<(), Errno>, -+ ) -> Result<(), Errno> { -+ let phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if *phase != WasiSharedMemoryExecPhase::Normal { -+ return Err(Errno::Busy); -+ } -+ let reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let mut active = self -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if !active.overlaps(start, requested_len)? { -+ return Err(Errno::Noent); -+ } -+ if reservations.overlaps(start, requested_len)? || !active.covers(start, operation_len)? { -+ return Err(Errno::Inval); -+ } -+ let updated_active = active.removing(start, operation_len)?; -+ restore_private()?; -+ *active = updated_active; -+ Ok(()) -+ } -+ - /// Turn the WasiState into bytes - #[cfg(feature = "enable-serde")] - pub fn freeze(&self) -> Option> { -@@ -251,9 +1286,44 @@ impl WasiState { - Ok(ret) - } - -- /// Forking the WasiState is used when either fork or vfork is called -- pub fn fork(&self) -> Self { -- WasiState { -+ /// Creates an owned process-local fork and applies caller-specific state -+ /// before the new identity becomes observable. -+ pub(crate) fn fork_with(&self, prepare: impl FnOnce(&mut Self)) -> Result, Errno> { -+ let mut state = self.fork()?; -+ prepare(&mut state); -+ Ok(Arc::new(state)) -+ } -+ -+ /// Raw fork construction is private to the state module. Runtime callers -+ /// must use the owned fork boundary; unit tests in this module may inspect -+ /// the unwrapped value directly. -+ fn fork(&self) -> Result { -+ // Mapping state is one logical object. Capture it under the global -+ // phase -> reservations -> active lock order, and reject fork while an -+ // exec image handoff is in flight. -+ let phase = self -+ .shared_memory_exec_phase -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if *phase != WasiSharedMemoryExecPhase::Normal { -+ return Err(Errno::Busy); -+ } -+ let reservations = self -+ .shared_memory_exec_reservations -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ if !reservations.mappings.is_empty() { -+ return Err(Errno::Inval); -+ } -+ let active = self -+ .shared_memory_mappings -+ .lock() -+ .unwrap_or_else(|err| err.into_inner()); -+ let active = active.clone(); -+ let reservations = reservations.clone(); -+ drop(phase); -+ -+ Ok(WasiState { - fs: self.fs.fork(), - secret: self.secret, - inodes: self.inodes.clone(), -@@ -262,7 +1332,1169 @@ impl WasiState { - args: Mutex::new(self.args.lock().unwrap().clone()), - envs: Mutex::new(self.envs.lock().unwrap().clone()), - signals: Mutex::new(self.signals.lock().unwrap().clone()), -+ shared_futex_registries: self.shared_futex_registries.clone(), -+ shared_memory_mappings: Mutex::new(active), -+ shared_memory_exec_phase: Default::default(), -+ shared_memory_exec_reservations: Mutex::new(reservations), - preopen: self.preopen.clone(), -+ }) -+ } -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use std::{ -+ sync::{Barrier, mpsc}, -+ thread, -+ time::Instant, -+ }; -+ -+ fn temporary_file() -> Arc { -+ Arc::new(tempfile::tempfile().unwrap()) -+ } -+ -+ fn mapping(start: u64, len: u64, file_offset: u64) -> WasiSharedMemoryMapping { -+ let file = temporary_file(); -+ let futexs = WasiSharedFutexRegistry::detached(file.clone()); -+ -+ WasiSharedMemoryMapping { -+ start, -+ len, -+ file, -+ file_offset, -+ futexs, - } - } -+ -+ #[test] -+ #[cfg(feature = "host-fs")] -+ fn live_shared_mapping_registry_blocks_backing_file_shrink() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-file-shrink-test"); -+ let file = temporary_file(); -+ file.set_len(0x20_000).unwrap(); -+ let registry = state.futex_registry_for_shared_file(file.clone()).unwrap(); -+ -+ assert!(matches!( -+ state.guard_shared_mapping_file_shrink(&file, 0x10_000), -+ Err(Errno::Busy) -+ )); -+ -+ drop(registry); -+ let guard = state -+ .guard_shared_mapping_file_shrink(&file, 0x10_000) -+ .unwrap(); -+ assert!(guard.is_some()); -+ file.set_len(0x10_000).unwrap(); -+ drop(guard); -+ assert_eq!(file.metadata().unwrap().len(), 0x10_000); -+ } -+ -+ fn test_state(program: &str) -> Arc { -+ Arc::new( -+ WasiEnv::builder(program) -+ .engine(wasmer::Engine::default()) -+ .build_init() -+ .unwrap() -+ .state, -+ ) -+ } -+ -+ #[test] -+ fn exec_transition_hides_active_mappings_and_rolls_back_exactly() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-exec-rollback-test"); -+ let file = temporary_file(); -+ let futexs = state.futex_registry_for_shared_file(file.clone()).unwrap(); -+ let original = WasiSharedMemoryMapping { -+ start: 0x20_000, -+ len: 0x30_000, -+ file, -+ file_offset: 0x10_000, -+ futexs: futexs.clone(), -+ }; -+ state -+ .replace_shared_memory_mapping(original.clone()) -+ .unwrap(); -+ assert!(matches!( -+ WasiFutexRegistry::resolve(&state, original.start), -+ Ok((WasiFutexRegistry::Shared(_), _)) -+ )); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ assert!(state.shared_memory_mappings().is_empty()); -+ assert_eq!( -+ state.shared_memory_exec_reservation_end().unwrap(), -+ Some(original.start + original.len) -+ ); -+ assert!(matches!( -+ WasiFutexRegistry::resolve(&state, original.start), -+ Err(Errno::Busy) -+ )); -+ -+ drop(transition); -+ let restored = state.shared_memory_mappings(); -+ assert_eq!(restored.len(), 1); -+ assert_eq!(restored[0].start, original.start); -+ assert_eq!(restored[0].len, original.len); -+ assert_eq!(restored[0].file_offset, original.file_offset); -+ assert!(Arc::ptr_eq(&restored[0].futexs, &futexs)); -+ assert_eq!(state.shared_memory_exec_reservation_end().unwrap(), None); -+ } -+ -+ #[test] -+ fn exec_transition_rolls_back_an_exact_partially_installed_successor() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-exec-installed-rollback-test"); -+ let first = mapping(0x20_000, 0x10_000, 0); -+ let second = mapping(0x40_000, 0x20_000, 0x10_000); -+ state.replace_shared_memory_mapping(first.clone()).unwrap(); -+ state.replace_shared_memory_mapping(second.clone()).unwrap(); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ state -+ .install_shared_memory_mapping( -+ first.clone(), -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || Ok(()), -+ ) -+ .unwrap(); -+ assert_eq!(state.shared_memory_mappings().len(), 1); -+ assert_eq!(state.shared_memory_exec_reservations().len(), 1); -+ -+ // Model synchronous Instance::new failure after linker pre-growth and -+ // a start function's first exact reattachment. -+ drop(transition); -+ let restored = state.shared_memory_mappings(); -+ assert_eq!(restored.len(), 2); -+ assert_eq!( -+ (restored[0].start, restored[0].len), -+ (first.start, first.len) -+ ); -+ assert_eq!( -+ (restored[1].start, restored[1].len), -+ (second.start, second.len) -+ ); -+ assert!(state.shared_memory_exec_reservations().is_empty()); -+ assert_eq!( -+ *state.shared_memory_exec_phase.lock().unwrap(), -+ WasiSharedMemoryExecPhase::Normal -+ ); -+ } -+ -+ #[test] -+ fn exec_transition_resolves_only_reattached_futex_mappings() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-exec-partial-futex-test"); -+ let active = mapping(0x20_000, 0x10_000, 0x1000); -+ let reserved = mapping(0x40_000, 0x10_000, 0x2000); -+ state.replace_shared_memory_mapping(active.clone()).unwrap(); -+ state -+ .replace_shared_memory_mapping(reserved.clone()) -+ .unwrap(); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ state -+ .install_shared_memory_mapping( -+ active.clone(), -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || Ok(()), -+ ) -+ .unwrap(); -+ transition.commit(); -+ -+ let (registry, key) = WasiFutexRegistry::resolve(&state, active.start + 4).unwrap(); -+ let WasiFutexRegistry::Shared(registry) = registry else { -+ panic!("reattached mapping resolved as a private futex"); -+ }; -+ assert!(Arc::ptr_eq(®istry, &active.futexs)); -+ assert_eq!(key, active.file_offset + 4); -+ assert!(matches!( -+ WasiFutexRegistry::resolve(&state, reserved.start + 4), -+ Err(Errno::Busy) -+ )); -+ -+ state -+ .install_shared_memory_mapping(reserved, WasiSharedMemoryMapOrigin::GuestFixed, || { -+ Ok(()) -+ }) -+ .unwrap(); -+ assert_eq!( -+ *state.shared_memory_exec_phase.lock().unwrap(), -+ WasiSharedMemoryExecPhase::Normal -+ ); -+ } -+ -+ #[test] -+ fn exact_reattach_does_not_relinquish_rollback_before_admission() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-exec-reattached-rollback-test"); -+ let original = mapping(0x20_000, 0x10_000, 0); -+ state -+ .replace_shared_memory_mapping(original.clone()) -+ .unwrap(); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ state -+ .install_shared_memory_mapping( -+ original.clone(), -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || Ok(()), -+ ) -+ .unwrap(); -+ -+ assert!(state.shared_memory_exec_reservations().is_empty()); -+ assert!(matches!( -+ *state.shared_memory_exec_phase.lock().unwrap(), -+ WasiSharedMemoryExecPhase::Transition { -+ reservations_installed: true, -+ successor_committed: false, -+ } -+ )); -+ assert!(matches!(state.fork(), Err(Errno::Busy))); -+ let new_file = temporary_file(); -+ let new_futexs = state -+ .futex_registry_for_shared_file(new_file.clone()) -+ .unwrap(); -+ assert_eq!( -+ state.install_runtime_selected_shared_memory_mapping( -+ 0x10_000, -+ new_file, -+ 0, -+ new_futexs, -+ |_| Ok(0x40_000), -+ ), -+ Err(Errno::Busy) -+ ); -+ -+ // Model a start function that reattached every range and then trapped -+ // before task admission. The old image is still restored exactly. -+ drop(transition); -+ assert_eq!(state.shared_memory_mappings().len(), 1); -+ assert!(state.shared_memory_exec_reservations().is_empty()); -+ assert_eq!( -+ *state.shared_memory_exec_phase.lock().unwrap(), -+ WasiSharedMemoryExecPhase::Normal -+ ); -+ } -+ -+ #[test] -+ fn admission_and_exact_reattach_jointly_finish_exec_in_either_order() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-exec-two-condition-commit-test"); -+ let original = mapping(0x20_000, 0x10_000, 0); -+ state -+ .replace_shared_memory_mapping(original.clone()) -+ .unwrap(); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ state -+ .install_shared_memory_mapping(original, WasiSharedMemoryMapOrigin::GuestFixed, || { -+ Ok(()) -+ }) -+ .unwrap(); -+ transition.commit(); -+ -+ assert_eq!( -+ *state.shared_memory_exec_phase.lock().unwrap(), -+ WasiSharedMemoryExecPhase::Normal -+ ); -+ assert!(state.fork().is_ok()); -+ } -+ -+ #[test] -+ fn exec_reattach_consumes_only_exact_range_offset_and_backing_identity() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-exec-identity-test"); -+ let file = temporary_file(); -+ file.set_len(0x80_000).unwrap(); -+ let futexs = state.futex_registry_for_shared_file(file.clone()).unwrap(); -+ let original = WasiSharedMemoryMapping { -+ start: 0x40_000, -+ len: 0x20_000, -+ file: file.clone(), -+ file_offset: 0x10_000, -+ futexs: futexs.clone(), -+ }; -+ state -+ .replace_shared_memory_mapping(original.clone()) -+ .unwrap(); -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ -+ let mut remap_called = false; -+ assert_eq!( -+ state.install_shared_memory_mapping( -+ original.clone(), -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || { -+ remap_called = true; -+ Ok(()) -+ }, -+ ), -+ Err(Errno::Busy) -+ ); -+ assert!(!remap_called); -+ transition.commit(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ -+ let wrong_file = temporary_file(); -+ wrong_file.set_len(0x80_000).unwrap(); -+ let wrong_futexs = state -+ .futex_registry_for_shared_file(wrong_file.clone()) -+ .unwrap(); -+ assert_eq!( -+ state.install_shared_memory_mapping( -+ WasiSharedMemoryMapping { -+ file: wrong_file, -+ futexs: wrong_futexs, -+ ..original.clone() -+ }, -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || { -+ remap_called = true; -+ Ok(()) -+ }, -+ ), -+ Err(Errno::Inval) -+ ); -+ assert!(!remap_called); -+ -+ assert_eq!( -+ state.install_shared_memory_mapping( -+ WasiSharedMemoryMapping { -+ file_offset: original.file_offset + 4096, -+ ..original.clone() -+ }, -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || { -+ remap_called = true; -+ Ok(()) -+ }, -+ ), -+ Err(Errno::Inval) -+ ); -+ assert!(!remap_called); -+ -+ let reopened = Arc::new(file.try_clone().unwrap()); -+ let reopened_futexs = state -+ .futex_registry_for_shared_file(reopened.clone()) -+ .unwrap(); -+ state -+ .install_shared_memory_mapping( -+ WasiSharedMemoryMapping { -+ file: reopened, -+ futexs: reopened_futexs, -+ ..original.clone() -+ }, -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || { -+ remap_called = true; -+ Ok(()) -+ }, -+ ) -+ .unwrap(); -+ assert!(remap_called); -+ assert_eq!(state.shared_memory_exec_reservation_end().unwrap(), None); -+ assert_eq!(state.shared_memory_mappings().len(), 1); -+ } -+ -+ #[test] -+ fn runtime_selected_mapping_starts_after_reservations_and_active_ranges() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-runtime-selection-test"); -+ let original = mapping(0x40_000, 0x20_000, 0); -+ state -+ .replace_shared_memory_mapping(original.clone()) -+ .unwrap(); -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ transition.commit(); -+ -+ let rejected = mapping(0, 0x10_000, 0); -+ let mut remap_called = false; -+ assert_eq!( -+ state.install_runtime_selected_shared_memory_mapping( -+ rejected.len, -+ rejected.file, -+ rejected.file_offset, -+ rejected.futexs, -+ |_| { -+ remap_called = true; -+ Ok(0x80_000) -+ }, -+ ), -+ Err(Errno::Busy) -+ ); -+ assert!(!remap_called); -+ -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ let installed_rejected = mapping(0, 0x10_000, 0); -+ assert_eq!( -+ state.install_runtime_selected_shared_memory_mapping( -+ installed_rejected.len, -+ installed_rejected.file, -+ installed_rejected.file_offset, -+ installed_rejected.futexs, -+ |_| { -+ remap_called = true; -+ Ok(0x80_000) -+ }, -+ ), -+ Err(Errno::Busy) -+ ); -+ assert!(!remap_called); -+ -+ state -+ .install_shared_memory_mapping( -+ original.clone(), -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || Ok(()), -+ ) -+ .unwrap(); -+ -+ let bad_selection = mapping(0, 0x10_000, 0); -+ assert_eq!( -+ state.install_runtime_selected_shared_memory_mapping( -+ bad_selection.len, -+ bad_selection.file, -+ bad_selection.file_offset, -+ bad_selection.futexs, -+ |placement| { -+ assert_eq!( -+ placement, -+ WasiSharedMemoryRuntimePlacement::Fresh { -+ minimum_start: original.start + original.len, -+ } -+ ); -+ Err(Errno::Inval) -+ }, -+ ), -+ Err(Errno::Inval) -+ ); -+ -+ let selected_mapping = mapping(0, 0x10_000, 0); -+ let selected = state -+ .install_runtime_selected_shared_memory_mapping( -+ selected_mapping.len, -+ selected_mapping.file, -+ selected_mapping.file_offset, -+ selected_mapping.futexs, -+ |placement| { -+ assert_eq!( -+ placement, -+ WasiSharedMemoryRuntimePlacement::Fresh { -+ minimum_start: original.start + original.len, -+ } -+ ); -+ remap_called = true; -+ Ok(0x80_000) -+ }, -+ ) -+ .unwrap(); -+ assert!(remap_called); -+ assert_eq!(selected, 0x80_000); -+ assert_eq!(state.shared_memory_mappings()[1].start, selected); -+ } -+ -+ #[test] -+ fn runtime_selected_exec_reattach_consumes_matching_inherited_backing() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-runtime-exec-reattach-test"); -+ let main = mapping(0x20_000, 0x10_000, 0); -+ let dsm = mapping(0x40_000, 0x20_000, 0); -+ state.replace_shared_memory_mapping(main.clone()).unwrap(); -+ state.replace_shared_memory_mapping(dsm.clone()).unwrap(); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ state -+ .install_shared_memory_mapping(main, WasiSharedMemoryMapOrigin::GuestFixed, || Ok(())) -+ .unwrap(); -+ transition.commit(); -+ -+ let mut observed_placement = None; -+ let selected = state -+ .install_runtime_selected_shared_memory_mapping( -+ dsm.len, -+ dsm.file.clone(), -+ dsm.file_offset, -+ dsm.futexs.clone(), -+ |placement| { -+ observed_placement = Some(placement); -+ match placement { -+ WasiSharedMemoryRuntimePlacement::Inherited { start } => Ok(start), -+ WasiSharedMemoryRuntimePlacement::Fresh { .. } => Err(Errno::Inval), -+ } -+ }, -+ ) -+ .unwrap(); -+ -+ assert_eq!(selected, dsm.start); -+ assert_eq!( -+ observed_placement, -+ Some(WasiSharedMemoryRuntimePlacement::Inherited { start: dsm.start }) -+ ); -+ assert!(state.shared_memory_exec_reservations().is_empty()); -+ assert_eq!(state.shared_memory_mappings().len(), 2); -+ assert_eq!( -+ *state.shared_memory_exec_phase.lock().unwrap(), -+ WasiSharedMemoryExecPhase::Normal -+ ); -+ } -+ -+ #[test] -+ fn runtime_selected_exec_reattach_rolls_back_before_admission() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-runtime-exec-rollback-test"); -+ let original = mapping(0x40_000, 0x20_000, 0x10_000); -+ state -+ .replace_shared_memory_mapping(original.clone()) -+ .unwrap(); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ state -+ .install_runtime_selected_shared_memory_mapping( -+ original.len, -+ original.file.clone(), -+ original.file_offset, -+ original.futexs.clone(), -+ |placement| match placement { -+ WasiSharedMemoryRuntimePlacement::Inherited { start } => Ok(start), -+ WasiSharedMemoryRuntimePlacement::Fresh { .. } => Err(Errno::Inval), -+ }, -+ ) -+ .unwrap(); -+ -+ drop(transition); -+ let restored = state.shared_memory_mappings(); -+ assert_eq!(restored.len(), 1); -+ assert!(restored[0].same_exec_reservation(&original)); -+ assert!(state.shared_memory_exec_reservations().is_empty()); -+ assert_eq!( -+ *state.shared_memory_exec_phase.lock().unwrap(), -+ WasiSharedMemoryExecPhase::Normal -+ ); -+ } -+ -+ #[test] -+ fn exec_phase_rejects_fork_until_exact_reattach_finishes() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-exec-fork-exclusion-test"); -+ let original = mapping(0x40_000, 0x20_000, 0); -+ state -+ .replace_shared_memory_mapping(original.clone()) -+ .unwrap(); -+ -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ assert!(matches!(state.fork(), Err(Errno::Busy))); -+ transition.commit(); -+ state -+ .mark_shared_memory_exec_reservations_installed() -+ .unwrap(); -+ assert!(matches!(state.fork(), Err(Errno::Busy))); -+ -+ state -+ .install_shared_memory_mapping(original, WasiSharedMemoryMapOrigin::GuestFixed, || { -+ Ok(()) -+ }) -+ .unwrap(); -+ assert!(state.fork().is_ok()); -+ } -+ -+ #[test] -+ fn guest_fixed_mapping_requires_full_known_coverage_without_exec_reservation() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-fixed-coverage-test"); -+ state -+ .replace_shared_memory_mapping(mapping(0x20_000, 0x20_000, 0)) -+ .unwrap(); -+ -+ let mut remap_called = false; -+ state -+ .install_shared_memory_mapping( -+ mapping(0x28_000, 0x8_000, 0), -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || { -+ remap_called = true; -+ Ok(()) -+ }, -+ ) -+ .unwrap(); -+ assert!(remap_called); -+ -+ remap_called = false; -+ assert_eq!( -+ state.install_shared_memory_mapping( -+ mapping(0x38_000, 0x10_000, 0), -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || { -+ remap_called = true; -+ Ok(()) -+ }, -+ ), -+ Err(Errno::Inval) -+ ); -+ assert!(!remap_called); -+ } -+ -+ #[test] -+ fn shared_mapping_operations_validate_ownership_before_host_mutation() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ let state = test_state("shared-memory-operation-ownership-test"); -+ state -+ .replace_shared_memory_mapping(mapping(0x20_000, 0x20_000, 0)) -+ .unwrap(); -+ -+ let mut operation_called = false; -+ assert_eq!( -+ state.remove_shared_memory_mapping(0x50_000, 0x10_000, 0x10_000, || { -+ operation_called = true; -+ Ok(()) -+ }), -+ Err(Errno::Noent) -+ ); -+ assert!(!operation_called); -+ -+ assert_eq!( -+ state.remove_shared_memory_mapping(0x18_000, 0x10_000, 0x10_000, || { -+ operation_called = true; -+ Ok(()) -+ }), -+ Err(Errno::Inval) -+ ); -+ assert!(!operation_called); -+ -+ assert_eq!( -+ state.remove_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || Err(Errno::Io)), -+ Err(Errno::Io) -+ ); -+ assert_eq!(state.shared_memory_mappings().len(), 1); -+ -+ state -+ .with_active_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || { -+ operation_called = true; -+ Ok(()) -+ }) -+ .unwrap(); -+ assert!(operation_called); -+ -+ operation_called = false; -+ let transition = state.prepare_shared_memory_for_exec().unwrap(); -+ assert_eq!( -+ state.with_active_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || { -+ operation_called = true; -+ Ok(()) -+ }), -+ Err(Errno::Busy) -+ ); -+ assert!(!operation_called); -+ drop(transition); -+ -+ state -+ .remove_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || { -+ operation_called = true; -+ Ok(()) -+ }) -+ .unwrap(); -+ assert!(operation_called); -+ let remaining = state.shared_memory_mappings(); -+ assert_eq!(remaining.len(), 2); -+ assert_eq!((remaining[0].start, remaining[0].len), (0x20_000, 0x8_000)); -+ assert_eq!((remaining[1].start, remaining[1].len), (0x30_000, 0x10_000)); -+ } -+ -+ #[test] -+ fn replacing_shared_memory_mapping_splits_overlapping_ranges() { -+ let mut mappings = WasiSharedMemoryMappings::default(); -+ mappings.replace(mapping(0, 16, 128)).unwrap(); -+ mappings.replace(mapping(4, 4, 512)).unwrap(); -+ -+ let snapshot = mappings.snapshot(); -+ let ranges = snapshot -+ .iter() -+ .map(|mapping| (mapping.start, mapping.len, mapping.file_offset)) -+ .collect::>(); -+ -+ assert_eq!(ranges, vec![(0, 4, 128), (4, 4, 512), (8, 8, 136)]); -+ } -+ -+ #[test] -+ fn removing_shared_memory_mapping_splits_overlapping_ranges() { -+ let mut mappings = WasiSharedMemoryMappings::default(); -+ mappings.replace(mapping(0, 16, 128)).unwrap(); -+ mappings.remove(4, 4).unwrap(); -+ -+ let snapshot = mappings.snapshot(); -+ let ranges = snapshot -+ .iter() -+ .map(|mapping| (mapping.start, mapping.len, mapping.file_offset)) -+ .collect::>(); -+ -+ assert_eq!(ranges, vec![(0, 4, 128), (8, 8, 136)]); -+ } -+ -+ #[test] -+ fn shared_memory_mapping_splits_reuse_futex_registry() { -+ let mut mappings = WasiSharedMemoryMappings::default(); -+ let original = mapping(0, 16, 128); -+ let registry = original.futexs.clone(); -+ mappings.replace(original).unwrap(); -+ mappings.replace(mapping(4, 4, 512)).unwrap(); -+ -+ let snapshot = mappings.snapshot(); -+ assert!(Arc::ptr_eq(&snapshot[0].futexs, ®istry)); -+ assert!(!Arc::ptr_eq(&snapshot[1].futexs, ®istry)); -+ assert!(Arc::ptr_eq(&snapshot[2].futexs, ®istry)); -+ -+ let left = mappings.shared_futex_registry(2).unwrap(); -+ let right = mappings.shared_futex_registry(12).unwrap(); -+ assert!(Arc::ptr_eq(&left.0, &right.0)); -+ assert_eq!(left.1, 130); -+ assert_eq!(right.1, 140); -+ } -+ -+ #[test] -+ fn shared_futex_registry_uses_containing_mapping_only() { -+ let mut mappings = WasiSharedMemoryMappings::default(); -+ mappings.replace(mapping(16, 8, 128)).unwrap(); -+ mappings.replace(mapping(32, 8, 256)).unwrap(); -+ -+ assert!(mappings.shared_futex_registry(15).is_none()); -+ assert!(mappings.shared_futex_registry(24).is_none()); -+ assert!(mappings.shared_futex_registry(31).is_none()); -+ assert!(mappings.shared_futex_registry(40).is_none()); -+ -+ assert_eq!(mappings.shared_futex_registry(16).unwrap().1, 128); -+ assert_eq!(mappings.shared_futex_registry(23).unwrap().1, 135); -+ assert_eq!(mappings.shared_futex_registry(32).unwrap().1, 256); -+ assert_eq!(mappings.shared_futex_registry(39).unwrap().1, 263); -+ } -+ -+ #[test] -+ fn shared_file_identity_survives_file_clone() { -+ let mapping = mapping(0, 4096, 0); -+ let cloned = mapping.file.try_clone().unwrap(); -+ -+ assert_eq!( -+ WasiSharedFileIdentity::for_file(&mapping.file).unwrap(), -+ WasiSharedFileIdentity::for_file(&cloned).unwrap() -+ ); -+ } -+ -+ #[test] -+ fn shared_futex_registry_reuses_same_live_file() { -+ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); -+ let mapping = mapping(0, 4096, 0); -+ let cloned = mapping.file.try_clone().unwrap(); -+ -+ let first = WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) -+ .unwrap(); -+ let second = -+ WasiSharedFutexRegistries::registry_for_file(®istries, Arc::new(cloned)).unwrap(); -+ -+ assert!(Arc::ptr_eq(&first, &second)); -+ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); -+ } -+ -+ #[test] -+ fn shared_futex_registry_pins_file_identity_until_last_live_reference() { -+ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); -+ let mapping = mapping(0, 4096, 0); -+ let expected_identity = WasiSharedFileIdentity::for_file(&mapping.file).unwrap(); -+ -+ let registry = -+ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) -+ .unwrap(); -+ let wait_reference = registry.clone(); -+ assert!(Arc::ptr_eq(®istry._file_anchor, &mapping.file)); -+ assert_eq!( -+ WasiSharedFileIdentity::for_file(®istry._file_anchor).unwrap(), -+ expected_identity -+ ); -+ -+ drop(registry); -+ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); -+ assert_eq!( -+ WasiSharedFileIdentity::for_file(&wait_reference._file_anchor).unwrap(), -+ expected_identity -+ ); -+ } -+ -+ #[test] -+ fn shared_futex_registry_replaces_entry_after_last_live_drop() { -+ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); -+ let mapping = mapping(0, 4096, 0); -+ -+ let first = WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) -+ .unwrap(); -+ let first_weak = Arc::downgrade(&first); -+ drop(first); -+ assert_eq!(registries.lock().unwrap().counts(), (0, 0)); -+ -+ let replacement = -+ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) -+ .unwrap(); -+ assert!(!Weak::ptr_eq(&first_weak, &Arc::downgrade(&replacement))); -+ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); -+ } -+ -+ #[test] -+ fn shared_futex_registry_last_drop_racing_lookup_keeps_exact_replacement() { -+ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); -+ let file = temporary_file(); -+ let identity = WasiSharedFileIdentity::for_file(&file).unwrap(); -+ let old = WasiSharedFutexRegistries::registry_for_file(®istries, file.clone()).unwrap(); -+ let old_weak = Arc::downgrade(&old); -+ let old_generation = old.generation.clone(); -+ -+ // Hold the owner table while the final strong reference starts its -+ // destructor. This makes the old Drop and the replacement lookup -+ // contend on the exact production mutex after the Weak can no longer -+ // be upgraded. -+ let owner_guard = registries.lock().unwrap(); -+ let drop_barrier = Arc::new(Barrier::new(2)); -+ let drop_thread = { -+ let drop_barrier = drop_barrier.clone(); -+ thread::spawn(move || { -+ drop_barrier.wait(); -+ drop(old); -+ }) -+ }; -+ drop_barrier.wait(); -+ -+ let deadline = Instant::now() + Duration::from_secs(5); -+ while old_weak.strong_count() != 0 && Instant::now() < deadline { -+ thread::yield_now(); -+ } -+ assert_eq!( -+ old_weak.strong_count(), -+ 0, -+ "final registry Drop did not reach the owner-table boundary" -+ ); -+ -+ let (lookup_started_tx, lookup_started_rx) = mpsc::channel(); -+ let lookup_thread = { -+ let registries = registries.clone(); -+ let file = file.clone(); -+ thread::spawn(move || { -+ lookup_started_tx.send(()).unwrap(); -+ WasiSharedFutexRegistries::registry_for_file(®istries, file).unwrap() -+ }) -+ }; -+ lookup_started_rx -+ .recv_timeout(Duration::from_secs(5)) -+ .unwrap(); -+ -+ // Either waiter may acquire the table first. Both legal orderings must -+ // leave the replacement installed after the old destructor completes. -+ drop(owner_guard); -+ let replacement = lookup_thread.join().unwrap(); -+ drop_thread.join().unwrap(); -+ -+ assert!(old_weak.upgrade().is_none()); -+ assert!(!Weak::ptr_eq(&old_weak, &Arc::downgrade(&replacement))); -+ assert!(!Arc::ptr_eq(&old_generation, &replacement.generation)); -+ { -+ let registries = registries.lock().unwrap(); -+ let current = registries.entries.get(&identity).unwrap(); -+ assert!(Arc::ptr_eq(¤t.generation, &replacement.generation)); -+ assert!(Arc::ptr_eq( -+ ¤t.registry.upgrade().unwrap(), -+ &replacement -+ )); -+ assert_eq!(registries.counts(), (1, 0)); -+ assert_eq!(registries.entries.len(), 1); -+ } -+ -+ drop(replacement); -+ let registries = registries.lock().unwrap(); -+ assert_eq!(registries.counts(), (0, 0)); -+ assert!(registries.entries.is_empty()); -+ } -+ -+ #[test] -+ fn forked_states_share_mapping_registry_and_return_to_zero_plateau() { -+ #[cfg(not(target_arch = "wasm32"))] -+ let runtime = tokio::runtime::Builder::new_current_thread() -+ .enable_all() -+ .build() -+ .unwrap(); -+ #[cfg(not(target_arch = "wasm32"))] -+ let _runtime_guard = runtime.enter(); -+ -+ let parent = WasiEnv::builder("shared-futex-fork-ownership-test") -+ .engine(wasmer::Engine::default()) -+ .build_init() -+ .unwrap() -+ .state; -+ let file = temporary_file(); -+ let registry = parent.futex_registry_for_shared_file(file.clone()).unwrap(); -+ let start = 0x10_000; -+ let file_offset = 0x20_000; -+ parent -+ .replace_shared_memory_mapping(WasiSharedMemoryMapping { -+ start, -+ len: 4096, -+ file, -+ file_offset, -+ futexs: registry.clone(), -+ }) -+ .unwrap(); -+ -+ let first_fork = Arc::new(parent.fork().unwrap()); -+ let second_fork = Arc::new(parent.fork().unwrap()); -+ let owner = parent.shared_futex_registries.clone(); -+ assert!(Arc::ptr_eq(&owner, &first_fork.shared_futex_registries)); -+ assert!(Arc::ptr_eq(&owner, &second_fork.shared_futex_registries)); -+ -+ let address = start + 512; -+ let (first_registry, first_key) = WasiFutexRegistry::resolve(&first_fork, address).unwrap(); -+ let (second_registry, second_key) = -+ WasiFutexRegistry::resolve(&second_fork, address).unwrap(); -+ let WasiFutexRegistry::Shared(first_registry) = first_registry else { -+ panic!("first fork resolved a shared mapping as a private futex") -+ }; -+ let WasiFutexRegistry::Shared(second_registry) = second_registry else { -+ panic!("second fork resolved a shared mapping as a private futex") -+ }; -+ assert!(Arc::ptr_eq(&first_registry, ®istry)); -+ assert!(Arc::ptr_eq(&second_registry, ®istry)); -+ assert_eq!(first_key, file_offset + 512); -+ assert_eq!(second_key, first_key); -+ { -+ let registries = owner.lock().unwrap(); -+ assert_eq!(registries.counts(), (1, 0)); -+ assert_eq!(registries.entries.len(), 1); -+ } -+ -+ drop(first_registry); -+ drop(second_registry); -+ drop(registry); -+ drop(first_fork); -+ drop(second_fork); -+ drop(parent); -+ -+ let registries = owner.lock().unwrap(); -+ assert_eq!(registries.counts(), (0, 0)); -+ assert!(registries.entries.is_empty()); -+ } -+ -+ #[cfg(target_os = "linux")] -+ fn linux_fd_count_for(identity: WasiSharedFileIdentity) -> usize { -+ use std::os::unix::fs::MetadataExt; -+ -+ std::fs::read_dir("/proc/self/fd") -+ .unwrap() -+ .filter_map(Result::ok) -+ .filter_map(|entry| entry.path().metadata().ok()) -+ .filter(|metadata| metadata.dev() == identity.device && metadata.ino() == identity.file) -+ .count() -+ } -+ -+ #[test] -+ fn repeated_shared_futex_registry_churn_returns_to_slot_and_fd_plateau() { -+ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); -+ let file = temporary_file(); -+ -+ #[cfg(target_os = "linux")] -+ let (identity, baseline_fds) = { -+ let identity = WasiSharedFileIdentity::for_file(&file).unwrap(); -+ let baseline_fds = linux_fd_count_for(identity); -+ assert_eq!(baseline_fds, 1); -+ (identity, baseline_fds) -+ }; -+ -+ for _ in 0..512 { -+ let registry = -+ WasiSharedFutexRegistries::registry_for_file(®istries, file.clone()).unwrap(); -+ { -+ let registries = registries.lock().unwrap(); -+ assert_eq!(registries.counts(), (1, 0)); -+ assert_eq!(registries.entries.len(), 1); -+ } -+ drop(registry); -+ { -+ let registries = registries.lock().unwrap(); -+ assert_eq!(registries.counts(), (0, 0)); -+ assert!(registries.entries.is_empty()); -+ } -+ } -+ -+ #[cfg(target_os = "linux")] -+ assert_eq!(linux_fd_count_for(identity), baseline_fds); -+ } -+ -+ #[test] -+ fn shared_futex_registry_old_generation_cannot_remove_replacement() { -+ let identity = WasiSharedFileIdentity { device: 1, file: 2 }; -+ let old_generation = Arc::new(()); -+ let current_generation = Arc::new(()); -+ let mut registries = WasiSharedFutexRegistries::default(); -+ registries.entries.insert( -+ identity, -+ WasiSharedFutexRegistryEntry { -+ generation: current_generation.clone(), -+ registry: Weak::new(), -+ }, -+ ); -+ -+ registries.remove_if_current(identity, &old_generation); -+ assert!(registries.entries.contains_key(&identity)); -+ -+ registries.remove_if_current(identity, ¤t_generation); -+ assert!(!registries.entries.contains_key(&identity)); -+ } -+ -+ #[test] -+ fn shared_futex_registry_prunes_stale_slots_in_bounded_order() { -+ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); -+ let stale_slots = SHARED_FUTEX_REGISTRY_PRUNE_BUDGET * 3; -+ { -+ let mut registries = registries.lock().unwrap(); -+ for file in 0..stale_slots as u64 { -+ registries.entries.insert( -+ WasiSharedFileIdentity { -+ device: u64::MAX, -+ file, -+ }, -+ WasiSharedFutexRegistryEntry { -+ generation: Arc::new(()), -+ registry: Weak::new(), -+ }, -+ ); -+ } -+ } -+ -+ let mapping = mapping(0, 4096, 0); -+ let live = WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) -+ .unwrap(); -+ { -+ let registries = registries.lock().unwrap(); -+ assert_eq!( -+ registries.prune_cursor, -+ Some(WasiSharedFileIdentity { -+ device: u64::MAX, -+ file: SHARED_FUTEX_REGISTRY_PRUNE_BUDGET as u64 - 1, -+ }) -+ ); -+ assert_eq!( -+ registries.counts(), -+ (1, stale_slots - SHARED_FUTEX_REGISTRY_PRUNE_BUDGET) -+ ); -+ } -+ -+ let reused = -+ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) -+ .unwrap(); -+ assert!(Arc::ptr_eq(&live, &reused)); -+ assert_eq!( -+ registries.lock().unwrap().counts(), -+ (1, stale_slots - SHARED_FUTEX_REGISTRY_PRUNE_BUDGET * 2) -+ ); -+ -+ let reused = -+ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) -+ .unwrap(); -+ assert!(Arc::ptr_eq(&live, &reused)); -+ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); -+ } - } -diff --git a/lib/wasix/src/state/preinitialized_memory_image.rs b/lib/wasix/src/state/preinitialized_memory_image.rs -new file mode 100644 -index 0000000..86542d1 ---- /dev/null -+++ b/lib/wasix/src/state/preinitialized_memory_image.rs -@@ -0,0 +1,1828 @@ -+use std::{ -+ collections::BTreeMap, -+ fmt, -+ fs::{File, OpenOptions}, -+ io::{Read, Seek, SeekFrom, Write}, -+ path::PathBuf, -+ sync::{ -+ Arc, OnceLock, -+ atomic::{AtomicBool, AtomicU64, Ordering}, -+ }, -+}; -+ -+use serde::{Deserialize, Serialize}; -+use sha2::{Digest, Sha256}; -+use wasmer::{AsStoreMut, AsStoreRef, Memory, MemoryType, Module}; -+use wasmer_types::ModuleHash; -+ -+use super::linker::{DylinkInfo, LinkError, OrdinaryModuleStartCompleted}; -+use crate::runtime::sealed_loader_audit::{ -+ FileAdviceAudit, FileResidencyAudit, advise_file_away, file_residency, -+}; -+ -+/// Stable mapping granularity for sealed preinitialized memory images. -+/// -+/// One WebAssembly page is aligned on the native page sizes used by Linux and -+/// macOS, and on the allocation granularity required by the planned Windows -+/// placeholder-backed implementation. -+pub const PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT: u64 = 64 * 1024; -+ -+/// The exact lifecycle point represented by a preinitialized memory image. -+pub const PREINITIALIZED_MEMORY_IMAGE_PHASE: &str = "post-module-start-pre-link-relocations-v1"; -+ -+/// Static-proof contract that permits one process-wide differential byte -+/// validation instead of repeating it for every fresh instance. -+pub const DETERMINISTIC_START_PROOF_SCHEMA: &str = -+ "oliphaunt.wasix-postmaster.deterministic-start-proof.v1"; -+pub const DETERMINISTIC_START_ANALYZER_POLICY: &str = -+ "llvm-shared-memory-init-restricted-effects.v1"; -+pub const DETERMINISTIC_START_MEMORY_READS: &str = "fresh-zero-atomic-guard-only"; -+pub const DETERMINISTIC_START_MEMORY_EFFECTS: &str = -+ "passive-data-init-zero-fill-atomic-guard-only"; -+pub const DETERMINISTIC_START_GLOBAL_EFFECTS: &str = "local-numeric-relocations-only"; -+pub const DETERMINISTIC_START_TABLE_EFFECTS: &str = "none"; -+ -+/// Carrier-bound static evidence that ordinary module start is deterministic -+/// for a newly allocated, zeroed memory with the receipt-bound layout. -+/// -+/// This does not authorize skipping WebAssembly start. Every instance still -+/// executes ordinary Wasm instantiation. It only allows later instances to -+/// reuse the first instance's exact post-start image comparison. -+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -+#[serde(rename_all = "kebab-case", deny_unknown_fields)] -+pub struct DeterministicStartProof { -+ pub schema: String, -+ pub analyzer_policy: String, -+ pub module_sha256: String, -+ pub proof_sha256: String, -+ pub start_function_index: u32, -+ pub start_function_export: String, -+ pub transitive_function_indices: Vec, -+ pub imported_function_calls: u32, -+ pub memory_reads: String, -+ pub memory_effects: String, -+ pub global_effects: String, -+ pub table_effects: String, -+ pub requires_fresh_zeroed_memory: bool, -+ pub ordinary_start_execution_per_instance: bool, -+ pub first_instance_full_byte_validation: bool, -+} -+ -+/// Versioned, portable metadata bound into the sealed carrier manifest. -+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -+#[serde(rename_all = "kebab-case", deny_unknown_fields)] -+pub struct PreinitializedMemoryImageMetadata { -+ pub schema: String, -+ pub module_sha256: String, -+ pub runtime_abi_id: String, -+ pub phase: String, -+ pub mapping_alignment: u64, -+ pub mapped_size: u64, -+ pub memory_minimum_pages: u32, -+ pub memory_maximum_pages: Option, -+ pub memory_shared: bool, -+ pub memory_base: u64, -+ pub dylink_memory_size: u32, -+ pub dylink_memory_alignment: u32, -+ pub stack_low: u64, -+ #[serde(default, skip_serializing_if = "Option::is_none")] -+ pub deterministic_start_proof: Option, -+ #[serde(default, skip_serializing_if = "Option::is_none")] -+ pub deterministic_start_proof_output_sha256: Option, -+} -+ -+impl PreinitializedMemoryImageMetadata { -+ pub const SCHEMA: &'static str = "oliphaunt.wasix-postmaster.memory-image.v1"; -+ pub const ATTESTED_SCHEMA: &'static str = "oliphaunt.wasix-postmaster.memory-image.v2"; -+} -+ -+/// An immutable backing file and its carrier-attested initialization metadata. -+#[derive(Debug)] -+pub struct PreinitializedMemoryImage { -+ file: File, -+ module_hash: ModuleHash, -+ metadata: PreinitializedMemoryImageMetadata, -+ backing: PreinitializedMemoryImageBacking, -+ load_audit: PreinitializedMemoryImageLoadAudit, -+ attested_runtime_validation: OnceLock>>, -+ runtime_audit: PreinitializedMemoryImageRuntimeCounters, -+} -+ -+#[derive(Debug, Default)] -+struct PreinitializedMemoryImageRuntimeCounters { -+ ordinary_start_completed_instances: AtomicU64, -+ fresh_zeroed_instances: AtomicU64, -+ nonfresh_instances: AtomicU64, -+ validation_attempts: AtomicU64, -+ full_compare_attempts: AtomicU64, -+ full_compare_successes: AtomicU64, -+ full_compare_failures: AtomicU64, -+ compared_bytes: AtomicU64, -+ reuse_successes: AtomicU64, -+ reuse_failures: AtomicU64, -+ skipped_bytes: AtomicU64, -+ remap_successes: AtomicU64, -+ remap_failures: AtomicU64, -+ counter_overflow: AtomicBool, -+} -+ -+/// Terminal, non-forcing audit snapshot for one attested memory image. -+/// -+/// The product executor reads this only after the complete WASIX process tree -+/// has joined. It is intentionally bounded: no per-instance records, paths, -+/// timestamps, or general tracing state are retained. -+#[derive(Debug, Clone, PartialEq, Eq, Serialize)] -+pub struct PreinitializedMemoryImageRuntimeAudit { -+ pub module_sha256: String, -+ pub memory_image_schema: String, -+ pub proof_sha256: String, -+ pub proof_output_sha256: String, -+ pub mapped_size: u64, -+ pub ordinary_start_completed_instances: u64, -+ pub fresh_zeroed_instances: u64, -+ pub nonfresh_instances: u64, -+ pub validation_attempts: u64, -+ pub full_compare_attempts: u64, -+ pub full_compare_successes: u64, -+ pub full_compare_failures: u64, -+ pub compared_bytes: u64, -+ pub reuse_successes: u64, -+ pub reuse_failures: u64, -+ pub skipped_bytes: u64, -+ pub remap_successes: u64, -+ pub remap_failures: u64, -+ pub counter_overflow: bool, -+} -+ -+impl PreinitializedMemoryImageRuntimeCounters { -+ fn add(&self, counter: &AtomicU64, value: u64) { -+ if value == 0 { -+ return; -+ } -+ if counter -+ .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { -+ current.checked_add(value) -+ }) -+ .is_err() -+ { -+ self.counter_overflow.store(true, Ordering::Release); -+ } -+ } -+ -+ fn record_start_boundary(&self, fresh_zeroed_memory: bool) { -+ self.add(&self.ordinary_start_completed_instances, 1); -+ self.add(&self.validation_attempts, 1); -+ if fresh_zeroed_memory { -+ self.add(&self.fresh_zeroed_instances, 1); -+ } else { -+ self.add(&self.nonfresh_instances, 1); -+ } -+ } -+ -+ fn record_validation(&self, compared: bool, succeeded: bool, mapped_size: u64) { -+ match (compared, succeeded) { -+ (true, true) => { -+ self.add(&self.full_compare_attempts, 1); -+ self.add(&self.full_compare_successes, 1); -+ self.add(&self.compared_bytes, mapped_size); -+ } -+ (true, false) => { -+ self.add(&self.full_compare_attempts, 1); -+ self.add(&self.full_compare_failures, 1); -+ } -+ (false, true) => { -+ self.add(&self.reuse_successes, 1); -+ self.add(&self.skipped_bytes, mapped_size); -+ } -+ (false, false) => self.add(&self.reuse_failures, 1), -+ } -+ } -+ -+ fn record_remap(&self, succeeded: bool) { -+ if succeeded { -+ self.add(&self.remap_successes, 1); -+ } else { -+ self.add(&self.remap_failures, 1); -+ } -+ } -+} -+ -+/// Loader evidence captured while the verified source mapping is still live. -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub struct PreinitializedMemoryImageLoadAudit { -+ pub source_residency_after_hash_inspect: Option, -+ pub mapping_cache_eviction: FileAdviceAudit, -+} -+ -+/// Storage retained behind a preinitialized memory image. -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub enum PreinitializedMemoryImageBacking { -+ /// Runtime-owned, sealed anonymous storage populated from a mutable source. -+ SealedCopy, -+ /// The carrier inode itself, admitted only after kernel-backed immutability. -+ DirectIntrinsic(IntrinsicFileImmutability), -+} -+ -+impl PreinitializedMemoryImageBacking { -+ pub const fn audit_mode(self) -> &'static str { -+ match self { -+ Self::SealedCopy => "streamed-copy-sealed-backing", -+ Self::DirectIntrinsic(IntrinsicFileImmutability::ReadOnlyFilesystem) => { -+ "direct-read-only-filesystem" -+ } -+ Self::DirectIntrinsic(IntrinsicFileImmutability::ImmutableInode) => { -+ "direct-immutable-inode" -+ } -+ } -+ } -+} -+ -+/// Kernel property that makes a live file descriptor safe as immutable mmap -+/// backing. Unix permission bits are intentionally not sufficient. -+#[derive(Debug, Clone, Copy, PartialEq, Eq)] -+pub enum IntrinsicFileImmutability { -+ /// The opened inode resides on SquashFS or EROFS. -+ ReadOnlyFilesystem, -+ /// `FS_IMMUTABLE_FL` is set and this process cannot clear it. -+ ImmutableInode, -+} -+ -+/// Deferred, single-flight source for a preinitialized memory image. -+/// -+/// Sealed carriers use this indirection so an exec-only image does not consume -+/// RSS or copy bytes until the corresponding executable is actually selected. -+/// Implementations must return the same immutable image (or the same failure) -+/// to every caller. -+pub trait PreinitializedMemoryImageLoader: fmt::Debug + Send + Sync + 'static { -+ fn load(&self) -> Result, String>; -+ -+ /// Return terminal audit counters only when this loader has already -+ /// materialized an attested image. This must never force image loading. -+ fn runtime_audit_if_loaded(&self) -> Option { -+ None -+ } -+} -+ -+/// Cloneable handle to a deferred preinitialized memory image. -+#[derive(Clone)] -+pub struct PreinitializedMemoryImageHandle { -+ loader: Arc, -+} -+ -+impl PreinitializedMemoryImageHandle { -+ pub fn new(loader: Arc) -> Self { -+ Self { loader } -+ } -+ -+ pub fn load(&self) -> Result, String> { -+ self.loader.load() -+ } -+ -+ /// Snapshot an already-loaded attested image without activating the -+ /// underlying loader or faulting image bytes. -+ pub fn runtime_audit_if_loaded(&self) -> Option { -+ self.loader.runtime_audit_if_loaded() -+ } -+} -+ -+impl fmt::Debug for PreinitializedMemoryImageHandle { -+ fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { -+ formatter -+ .debug_struct("PreinitializedMemoryImageHandle") -+ .finish_non_exhaustive() -+ } -+} -+ -+impl PreinitializedMemoryImage { -+ /// Validates carrier metadata without reading or materializing image bytes. -+ pub fn validate_metadata( -+ metadata: &PreinitializedMemoryImageMetadata, -+ module_hash: ModuleHash, -+ ) -> Result<(), String> { -+ validate_metadata_shape(metadata, module_hash) -+ } -+ -+ /// Copies an image into runtime-owned immutable storage while verifying its -+ /// carrier-attested digest. The source file is never used as mmap backing, -+ /// so a concurrent carrier-path replacement or later source write cannot -+ /// change the bytes between comparison and remapping. -+ pub fn from_verified_reader( -+ mut source: impl Read, -+ expected_image_sha256: [u8; 32], -+ module_hash: ModuleHash, -+ metadata: PreinitializedMemoryImageMetadata, -+ ) -> Result { -+ validate_metadata_shape(&metadata, module_hash)?; -+ -+ let mut file = new_immutable_snapshot_backing()?; -+ let mut hasher = Sha256::new(); -+ let mut copied = 0_u64; -+ let mut buffer = [0_u8; 128 * 1024]; -+ loop { -+ let count = source -+ .read(&mut buffer) -+ .map_err(|err| format!("read preinitialized memory image: {err}"))?; -+ if count == 0 { -+ break; -+ } -+ copied = copied -+ .checked_add(count as u64) -+ .ok_or_else(|| "preinitialized memory image size overflow".to_string())?; -+ if copied > metadata.mapped_size { -+ return Err(format!( -+ "preinitialized memory image exceeds metadata size: metadata={} copied={copied}", -+ metadata.mapped_size -+ )); -+ } -+ hasher.update(&buffer[..count]); -+ file.write_all(&buffer[..count]) -+ .map_err(|err| format!("copy preinitialized memory image: {err}"))?; -+ } -+ if copied != metadata.mapped_size { -+ return Err(format!( -+ "preinitialized memory image size mismatch: metadata={} copied={copied}", -+ metadata.mapped_size, -+ )); -+ } -+ let actual: [u8; 32] = hasher.finalize().into(); -+ if actual != expected_image_sha256 { -+ return Err(format!( -+ "preinitialized memory image SHA-256 mismatch: expected={} actual={}", -+ hex::encode(expected_image_sha256), -+ hex::encode(actual) -+ )); -+ } -+ // This is a private, live-FD runtime snapshot, not a durable -+ // publication. Completed writes are coherent with the later mapping; -+ // hashing plus sealing supplies integrity. A filesystem sync would add -+ // latency (and possibly flash traffic for an unlinked temp backing) -+ // without strengthening correctness. -+ seal_immutable_snapshot(&file)?; -+ let mapping = shared_buffer::OwnedBuffer::from_file(&file) -+ .map_err(|err| format!("map sealed preinitialized memory image: {err}"))?; -+ let mapped_len = u64::try_from(mapping.len()) -+ .map_err(|_| "sealed memory-image mapping length exceeds u64".to_string())?; -+ if mapped_len != metadata.mapped_size { -+ return Err(format!( -+ "sealed preinitialized memory image mapping size mismatch: metadata={} mapping={mapped_len}", -+ metadata.mapped_size -+ )); -+ } -+ let backing_digest: [u8; 32] = Sha256::digest(mapping.as_slice()).into(); -+ if backing_digest != expected_image_sha256 { -+ return Err(format!( -+ "sealed preinitialized memory image backing SHA-256 mismatch: expected={} actual={}", -+ hex::encode(expected_image_sha256), -+ hex::encode(backing_digest) -+ )); -+ } -+ let mapping_cache_eviction = advise_mapping_away(&mapping); -+ file.seek(SeekFrom::Start(0)) -+ .map_err(|err| format!("rewind preinitialized memory image: {err}"))?; -+ -+ Ok(Self { -+ file, -+ module_hash, -+ metadata, -+ backing: PreinitializedMemoryImageBacking::SealedCopy, -+ load_audit: PreinitializedMemoryImageLoadAudit { -+ source_residency_after_hash_inspect: None, -+ mapping_cache_eviction, -+ }, -+ attested_runtime_validation: OnceLock::new(), -+ runtime_audit: PreinitializedMemoryImageRuntimeCounters::default(), -+ }) -+ } -+ -+ /// Retains an intrinsically immutable carrier inode as the MAP_PRIVATE -+ /// backing after hashing the exact file mapping. This avoids an anonymous -+ /// copy while preserving a fresh private mapping for every instance. -+ #[cfg(target_os = "linux")] -+ pub fn from_verified_immutable_file( -+ file: File, -+ expected_image_sha256: [u8; 32], -+ module_hash: ModuleHash, -+ metadata: PreinitializedMemoryImageMetadata, -+ ) -> Result { -+ (move || { -+ validate_metadata_shape(&metadata, module_hash)?; -+ let immutable = intrinsic_file_immutability(&file)?; -+ let actual_size = file -+ .metadata() -+ .map_err(|err| format!("stat immutable preinitialized memory image: {err}"))? -+ .len(); -+ if actual_size != metadata.mapped_size { -+ return Err(format!( -+ "preinitialized memory image size mismatch: metadata={} actual={actual_size}", -+ metadata.mapped_size -+ )); -+ } -+ -+ let mapping = shared_buffer::OwnedBuffer::from_file(&file) -+ .map_err(|err| format!("map immutable preinitialized memory image: {err}"))?; -+ let mapped_len = u64::try_from(mapping.len()) -+ .map_err(|_| "immutable memory-image mapping length exceeds u64".to_string())?; -+ if mapped_len != metadata.mapped_size { -+ return Err(format!( -+ "immutable preinitialized memory image mapping size mismatch: metadata={} mapping={}", -+ metadata.mapped_size, mapped_len -+ )); -+ } -+ let actual: [u8; 32] = Sha256::digest(mapping.as_slice()).into(); -+ if actual != expected_image_sha256 { -+ return Err(format!( -+ "preinitialized memory image SHA-256 mismatch: expected={} actual={}", -+ hex::encode(expected_image_sha256), -+ hex::encode(actual) -+ )); -+ } -+ let source_residency_after_hash_inspect = file_residency(&file, metadata.mapped_size); -+ let mapping_cache_eviction = advise_mapping_away(&mapping); -+ -+ Ok(Self { -+ file, -+ module_hash, -+ metadata, -+ backing: PreinitializedMemoryImageBacking::DirectIntrinsic(immutable), -+ load_audit: PreinitializedMemoryImageLoadAudit { -+ source_residency_after_hash_inspect: Some(source_residency_after_hash_inspect), -+ mapping_cache_eviction, -+ }, -+ attested_runtime_validation: OnceLock::new(), -+ runtime_audit: PreinitializedMemoryImageRuntimeCounters::default(), -+ }) -+ })() -+ } -+ -+ #[cfg(not(target_os = "linux"))] -+ pub fn from_verified_immutable_file( -+ _file: File, -+ _expected_image_sha256: [u8; 32], -+ _module_hash: ModuleHash, -+ _metadata: PreinitializedMemoryImageMetadata, -+ ) -> Result { -+ Err("direct immutable memory-image backing is only implemented on Linux".to_string()) -+ } -+ -+ pub fn metadata(&self) -> &PreinitializedMemoryImageMetadata { -+ &self.metadata -+ } -+ -+ pub fn backing(&self) -> PreinitializedMemoryImageBacking { -+ self.backing -+ } -+ -+ pub fn load_audit(&self) -> PreinitializedMemoryImageLoadAudit { -+ self.load_audit -+ } -+ -+ /// Return the bounded process-lifetime counters for a v2 attested image. -+ /// Legacy v1 images deliberately have no summary in the attested schema. -+ pub fn runtime_audit(&self) -> Option { -+ let proof = self.metadata.deterministic_start_proof.as_ref()?; -+ let proof_output_sha256 = self -+ .metadata -+ .deterministic_start_proof_output_sha256 -+ .as_ref()?; -+ Some(PreinitializedMemoryImageRuntimeAudit { -+ module_sha256: self.module_hash.to_string().to_ascii_lowercase(), -+ memory_image_schema: self.metadata.schema.clone(), -+ proof_sha256: proof.proof_sha256.clone(), -+ proof_output_sha256: proof_output_sha256.clone(), -+ mapped_size: self.metadata.mapped_size, -+ ordinary_start_completed_instances: self -+ .runtime_audit -+ .ordinary_start_completed_instances -+ .load(Ordering::Acquire), -+ fresh_zeroed_instances: self -+ .runtime_audit -+ .fresh_zeroed_instances -+ .load(Ordering::Acquire), -+ nonfresh_instances: self -+ .runtime_audit -+ .nonfresh_instances -+ .load(Ordering::Acquire), -+ validation_attempts: self -+ .runtime_audit -+ .validation_attempts -+ .load(Ordering::Acquire), -+ full_compare_attempts: self -+ .runtime_audit -+ .full_compare_attempts -+ .load(Ordering::Acquire), -+ full_compare_successes: self -+ .runtime_audit -+ .full_compare_successes -+ .load(Ordering::Acquire), -+ full_compare_failures: self -+ .runtime_audit -+ .full_compare_failures -+ .load(Ordering::Acquire), -+ compared_bytes: self.runtime_audit.compared_bytes.load(Ordering::Acquire), -+ reuse_successes: self.runtime_audit.reuse_successes.load(Ordering::Acquire), -+ reuse_failures: self.runtime_audit.reuse_failures.load(Ordering::Acquire), -+ skipped_bytes: self.runtime_audit.skipped_bytes.load(Ordering::Acquire), -+ remap_successes: self.runtime_audit.remap_successes.load(Ordering::Acquire), -+ remap_failures: self.runtime_audit.remap_failures.load(Ordering::Acquire), -+ counter_overflow: self.runtime_audit.counter_overflow.load(Ordering::Acquire), -+ }) -+ } -+ -+ pub(crate) fn apply( -+ &self, -+ _ordinary_start_completed: OrdinaryModuleStartCompleted, -+ main_module: &Module, -+ memory: &Memory, -+ store: &mut impl AsStoreMut, -+ memory_type: MemoryType, -+ dylink: &DylinkInfo, -+ memory_base: u64, -+ stack_low: u64, -+ fresh_zeroed_memory: bool, -+ ) -> Result<(), LinkError> { -+ // `Linker::new` can call this method only after -+ // `Instance::new_with_export_names` has synchronously completed the -+ // module start function. This is therefore the exact ordinary-start -+ // completion boundary for every image-backed fresh backend. -+ self.runtime_audit -+ .record_start_boundary(fresh_zeroed_memory); -+ validate_runtime_layout( -+ &self.metadata, -+ self.module_hash, -+ main_module, -+ memory_type, -+ dylink, -+ memory_base, -+ stack_low, -+ fresh_zeroed_memory, -+ ) -+ .map_err(LinkError::PreinitializedMemoryImage)?; -+ -+ #[cfg(unix)] -+ { -+ let mapped_size = usize::try_from(self.metadata.mapped_size).map_err(|_| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "preinitialized memory image exceeds host address width: {}", -+ self.metadata.mapped_size -+ )) -+ })?; -+ let runtime_bytes = self.validate_runtime_bytes(fresh_zeroed_memory, || { -+ compare_memory_to_file(memory, store, &self.file, self.metadata.mapped_size) -+ }); -+ self.runtime_audit.record_validation( -+ runtime_bytes.compared_this_instance, -+ runtime_bytes.validation.is_ok(), -+ self.metadata.mapped_size, -+ ); -+ // Comparison necessarily faults the complete image source, but a -+ // backend may touch only a fraction of that prefix. Release those -+ // validation-only cache references before installing the private -+ // file mapping so its steady-state working set is demand-paged. -+ // This deliberately trades at most one later refault per used clean -+ // page for not pinning every compared page. DONTNEED does not -+ // modify the still-live anonymous guest memory, and correctness -+ // remains independent of whether the kernel accepts or acts on it. -+ // An attested cache hit faulted no validation-only pages and must -+ // not evict clean pages that active private mappings may share. -+ let source_cache_eviction = if runtime_bytes.compared_this_instance { -+ release_validation_faulted_source_pages(self.backing, &self.file) -+ } else { -+ FileAdviceAudit::not_applicable() -+ }; -+ let compared_bytes = if runtime_bytes.compared_this_instance { -+ self.metadata.mapped_size -+ } else { -+ 0 -+ }; -+ let skipped_bytes = -+ if runtime_bytes.validation.is_ok() && !runtime_bytes.compared_this_instance { -+ self.metadata.mapped_size -+ } else { -+ 0 -+ }; -+ let validation_state = if self.metadata.deterministic_start_proof.is_none() { -+ "compared-per-instance" -+ } else if runtime_bytes.compared_this_instance { -+ "attested-first-instance-compared" -+ } else if runtime_bytes.validation.is_ok() { -+ "attested-validation-reused" -+ } else { -+ "attested-failure-reused" -+ }; -+ tracing::debug!( -+ target: "wasmer_wasix::preinitialized_memory_image_audit", -+ audit_schema = "oliphaunt.wasix-postmaster.sealed-loader-audit.v2", -+ artifact_kind = "preinitialized-memory-runtime-validation", -+ module_sha256 = %self.module_hash, -+ validation_state, -+ validation_succeeded = runtime_bytes.validation.is_ok(), -+ compared_bytes, -+ skipped_bytes, -+ backing_mode = self.backing.audit_mode(), -+ source_cache_eviction_supported = source_cache_eviction.supported, -+ source_cache_eviction_calls = source_cache_eviction.calls, -+ source_cache_eviction_successes = source_cache_eviction.successes, -+ source_cache_eviction_errno = source_cache_eviction.first_errno, -+ "validated deterministic module-start bytes" -+ ); -+ runtime_bytes -+ .validation -+ .map_err(LinkError::PreinitializedMemoryImage)?; -+ // SAFETY: module start has returned, the linker exclusively owns -+ // the store, and no guest thread can run before linking returns. -+ // `self.file` is runtime-owned and immutable for this image's -+ // lifetime. -+ let remap = -+ unsafe { memory.remap_private_file_fixed(store, 0, mapped_size, &self.file, 0) }; -+ self.runtime_audit.record_remap(remap.is_ok()); -+ remap.map_err(|err| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "private file remapping failed: {err}" -+ )) -+ })?; -+ Ok(()) -+ } -+ -+ #[cfg(not(unix))] -+ { -+ let _ = (memory, store); -+ Err(LinkError::PreinitializedMemoryImage( -+ "private preinitialized memory-image mapping is not implemented on this host" -+ .to_string(), -+ )) -+ } -+ } -+ -+ /// Run ordinary per-instance validation for v1 images. For an analyzer- -+ /// proven v2 image, execute one process-wide differential validation and -+ /// reuse its immutable success or failure. The proof contract requires a -+ /// freshly allocated zeroed memory and ordinary start on every instance. -+ fn validate_runtime_bytes( -+ &self, -+ fresh_zeroed_memory: bool, -+ compare: impl FnOnce() -> Result<(), String>, -+ ) -> RuntimeByteValidation { -+ if self.metadata.deterministic_start_proof.is_none() { -+ return RuntimeByteValidation { -+ validation: compare(), -+ compared_this_instance: true, -+ }; -+ } -+ if !fresh_zeroed_memory { -+ return RuntimeByteValidation { -+ validation: Err( -+ "attested deterministic-start validation requires fresh zeroed memory" -+ .to_string(), -+ ), -+ compared_this_instance: false, -+ }; -+ } -+ -+ let mut compared_this_instance = false; -+ let validation = self.attested_runtime_validation.get_or_init(|| { -+ compared_this_instance = true; -+ compare().map_err(Arc::::from) -+ }); -+ RuntimeByteValidation { -+ validation: validation.clone().map_err(|message| message.to_string()), -+ compared_this_instance, -+ } -+ } -+} -+ -+#[derive(Debug)] -+struct RuntimeByteValidation { -+ validation: Result<(), String>, -+ compared_this_instance: bool, -+} -+ -+/// A builder-only request to capture the exact post-start memory prefix. -+#[derive(Debug, Clone)] -+pub struct PreinitializedMemoryImageCapture { -+ pub image_path: PathBuf, -+ pub receipt_path: PathBuf, -+ pub module_hash: ModuleHash, -+} -+ -+impl PreinitializedMemoryImageCapture { -+ pub(crate) fn capture( -+ &self, -+ main_module: &Module, -+ memory: &Memory, -+ store: &impl AsStoreRef, -+ memory_type: MemoryType, -+ dylink: &DylinkInfo, -+ memory_base: u64, -+ stack_low: u64, -+ ) -> Result<(), LinkError> { -+ let module_hash = main_module.info().hash.ok_or_else(|| { -+ LinkError::PreinitializedMemoryImage( -+ "captured module has no embedded raw module hash".to_string(), -+ ) -+ })?; -+ if module_hash != self.module_hash { -+ return Err(LinkError::PreinitializedMemoryImage(format!( -+ "capture module hash mismatch: request={} module={module_hash}", -+ self.module_hash -+ ))); -+ } -+ -+ let mapped_size = stack_low / PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT -+ * PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT; -+ if mapped_size == 0 { -+ return Err(LinkError::PreinitializedMemoryImage( -+ "capture layout has no full 64-KiB page before the guest stack".to_string(), -+ )); -+ } -+ let view = memory.view(store); -+ if mapped_size > view.data_size() { -+ return Err(LinkError::PreinitializedMemoryImage(format!( -+ "capture range exceeds current memory: range={mapped_size} memory={}", -+ view.data_size() -+ ))); -+ } -+ -+ let mut image = create_new_regular_file(&self.image_path) -+ .map_err(LinkError::PreinitializedMemoryImage)?; -+ // SAFETY: module start has returned and no guest code can execute while -+ // the linker owns the store. The view is dropped before any remapping. -+ let initialized = unsafe { view.data_unchecked() }; -+ let mapped_size_usize = usize::try_from(mapped_size).map_err(|_| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "captured memory image exceeds host address width: {mapped_size}" -+ )) -+ })?; -+ image -+ .write_all(&initialized[..mapped_size_usize]) -+ .map_err(|err| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "write captured memory image {}: {err}", -+ self.image_path.display() -+ )) -+ })?; -+ image.sync_all().map_err(|err| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "sync captured memory image {}: {err}", -+ self.image_path.display() -+ )) -+ })?; -+ drop(image); -+ -+ let metadata = PreinitializedMemoryImageMetadata { -+ schema: PreinitializedMemoryImageMetadata::SCHEMA.to_string(), -+ module_sha256: module_hash.to_string().to_ascii_lowercase(), -+ runtime_abi_id: option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID") -+ .unwrap_or("") -+ .to_string(), -+ phase: PREINITIALIZED_MEMORY_IMAGE_PHASE.to_string(), -+ mapping_alignment: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, -+ mapped_size, -+ memory_minimum_pages: memory_type.minimum.0, -+ memory_maximum_pages: memory_type.maximum.map(|pages| pages.0), -+ memory_shared: memory_type.shared, -+ memory_base, -+ dylink_memory_size: dylink.mem_info.memory_size, -+ dylink_memory_alignment: dylink.mem_info.memory_alignment, -+ stack_low, -+ deterministic_start_proof: None, -+ deterministic_start_proof_output_sha256: None, -+ }; -+ validate_metadata_shape(&metadata, module_hash) -+ .map_err(LinkError::PreinitializedMemoryImage)?; -+ -+ let mut receipt = create_new_regular_file(&self.receipt_path) -+ .map_err(LinkError::PreinitializedMemoryImage)?; -+ serde_json::to_writer_pretty(&mut receipt, &metadata).map_err(|err| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "write memory image receipt {}: {err}", -+ self.receipt_path.display() -+ )) -+ })?; -+ receipt.write_all(b"\n").map_err(|err| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "finish memory image receipt {}: {err}", -+ self.receipt_path.display() -+ )) -+ })?; -+ receipt.sync_all().map_err(|err| { -+ LinkError::PreinitializedMemoryImage(format!( -+ "sync memory image receipt {}: {err}", -+ self.receipt_path.display() -+ )) -+ })?; -+ Ok(()) -+ } -+} -+ -+/// Operation performed at the post-module-start linker boundary. -+#[derive(Debug, Clone)] -+pub enum PreinitializedMemoryImageMode { -+ Apply(Arc), -+ Capture(PreinitializedMemoryImageCapture), -+} -+ -+fn validate_metadata_shape( -+ metadata: &PreinitializedMemoryImageMetadata, -+ module_hash: ModuleHash, -+) -> Result<(), String> { -+ if metadata.schema != PreinitializedMemoryImageMetadata::SCHEMA -+ && metadata.schema != PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA -+ { -+ return Err(format!( -+ "preinitialized memory image schema mismatch: {}", -+ metadata.schema -+ )); -+ } -+ if metadata.phase != PREINITIALIZED_MEMORY_IMAGE_PHASE { -+ return Err(format!( -+ "preinitialized memory image phase mismatch: {}", -+ metadata.phase -+ )); -+ } -+ if metadata.module_sha256 != module_hash.to_string().to_ascii_lowercase() { -+ return Err(format!( -+ "preinitialized memory image module hash mismatch: metadata={} module={module_hash}", -+ metadata.module_sha256 -+ )); -+ } -+ if metadata.runtime_abi_id != option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID").unwrap_or("") { -+ return Err(format!( -+ "preinitialized memory image runtime ABI mismatch: metadata={} runtime={}", -+ metadata.runtime_abi_id, -+ option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID").unwrap_or("") -+ )); -+ } -+ if metadata.mapping_alignment != PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT -+ || metadata.mapped_size == 0 -+ || !metadata -+ .mapped_size -+ .is_multiple_of(PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT) -+ { -+ return Err("preinitialized memory image has an invalid 64-KiB mapping range".to_string()); -+ } -+ usize::try_from(metadata.mapped_size).map_err(|_| { -+ format!( -+ "preinitialized memory image exceeds host address width: {}", -+ metadata.mapped_size -+ ) -+ })?; -+ match ( -+ metadata.schema.as_str(), -+ metadata.deterministic_start_proof.as_ref(), -+ metadata.deterministic_start_proof_output_sha256.as_deref(), -+ ) { -+ (PreinitializedMemoryImageMetadata::SCHEMA, None, None) => {} -+ (PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA, Some(proof), Some(output_sha256)) => { -+ validate_deterministic_start_proof(proof, output_sha256, module_hash)?; -+ } -+ (PreinitializedMemoryImageMetadata::SCHEMA, _, _) => { -+ return Err( -+ "v1 preinitialized memory images cannot contain start-proof evidence".to_string(), -+ ); -+ } -+ (PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA, _, _) => { -+ return Err( -+ "v2 preinitialized memory images require a start proof and canonical output digest" -+ .to_string(), -+ ); -+ } -+ _ => unreachable!("memory image schema was checked above"), -+ } -+ Ok(()) -+} -+ -+fn validate_deterministic_start_proof( -+ proof: &DeterministicStartProof, -+ output_sha256: &str, -+ module_hash: ModuleHash, -+) -> Result<(), String> { -+ if proof.schema != DETERMINISTIC_START_PROOF_SCHEMA -+ || proof.analyzer_policy != DETERMINISTIC_START_ANALYZER_POLICY -+ || proof.module_sha256 != module_hash.to_string().to_ascii_lowercase() -+ || proof.imported_function_calls != 0 -+ || proof.memory_reads != DETERMINISTIC_START_MEMORY_READS -+ || proof.memory_effects != DETERMINISTIC_START_MEMORY_EFFECTS -+ || proof.global_effects != DETERMINISTIC_START_GLOBAL_EFFECTS -+ || proof.table_effects != DETERMINISTIC_START_TABLE_EFFECTS -+ || !proof.requires_fresh_zeroed_memory -+ || !proof.ordinary_start_execution_per_instance -+ || !proof.first_instance_full_byte_validation -+ { -+ return Err( -+ "preinitialized memory image has an invalid deterministic-start proof policy" -+ .to_string(), -+ ); -+ } -+ if proof.start_function_export.is_empty() -+ || proof.transitive_function_indices.is_empty() -+ || !proof -+ .transitive_function_indices -+ .contains(&proof.start_function_index) -+ || proof -+ .transitive_function_indices -+ .windows(2) -+ .any(|pair| pair[0] >= pair[1]) -+ { -+ return Err("deterministic-start proof has an invalid function closure".to_string()); -+ } -+ if proof.proof_sha256.len() != 64 -+ || !proof -+ .proof_sha256 -+ .bytes() -+ .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) -+ { -+ return Err("deterministic-start proof digest is not canonical SHA-256".to_string()); -+ } -+ if output_sha256.len() != 64 -+ || !output_sha256 -+ .bytes() -+ .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) -+ || output_sha256 != canonical_deterministic_start_proof_sha256(proof)? -+ { -+ return Err( -+ "deterministic-start proof canonical output digest does not match the proof" -+ .to_string(), -+ ); -+ } -+ Ok(()) -+} -+ -+fn canonical_deterministic_start_proof_sha256( -+ proof: &DeterministicStartProof, -+) -> Result { -+ let value = serde_json::to_value(proof) -+ .map_err(|err| format!("serialize deterministic-start proof: {err}"))?; -+ let object = value -+ .as_object() -+ .ok_or_else(|| "deterministic-start proof did not serialize as an object".to_string())?; -+ let canonical = object -+ .iter() -+ .map(|(key, value)| (key.clone(), value.clone())) -+ .collect::>(); -+ let bytes = serde_json::to_vec(&canonical) -+ .map_err(|err| format!("canonicalize deterministic-start proof: {err}"))?; -+ Ok(hex::encode(Sha256::digest(bytes))) -+} -+ -+#[allow(clippy::too_many_arguments)] -+fn validate_runtime_layout( -+ metadata: &PreinitializedMemoryImageMetadata, -+ module_hash: ModuleHash, -+ main_module: &Module, -+ memory_type: MemoryType, -+ dylink: &DylinkInfo, -+ memory_base: u64, -+ stack_low: u64, -+ fresh_zeroed_memory: bool, -+) -> Result<(), String> { -+ validate_metadata_shape(metadata, module_hash)?; -+ if main_module.info().hash != Some(module_hash) { -+ return Err("preinitialized memory image is paired with a different module".to_string()); -+ } -+ let expected_mapped_size = -+ stack_low / PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT * PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT; -+ if metadata.mapped_size != expected_mapped_size -+ || metadata.memory_minimum_pages != memory_type.minimum.0 -+ || metadata.memory_maximum_pages != memory_type.maximum.map(|pages| pages.0) -+ || metadata.memory_shared != memory_type.shared -+ || metadata.memory_base != memory_base -+ || metadata.dylink_memory_size != dylink.mem_info.memory_size -+ || metadata.dylink_memory_alignment != dylink.mem_info.memory_alignment -+ || metadata.stack_low != stack_low -+ { -+ return Err( -+ "preinitialized memory image layout does not match the linked module".to_string(), -+ ); -+ } -+ if let Some(proof) = &metadata.deterministic_start_proof { -+ if !fresh_zeroed_memory { -+ return Err( -+ "attested deterministic-start image requires fresh zeroed memory".to_string(), -+ ); -+ } -+ let expected_start = wasmer_types::FunctionIndex::from_u32(proof.start_function_index); -+ if main_module.info().start_function != Some(expected_start) -+ || main_module.info().exports.get(&proof.start_function_export) -+ != Some(&wasmer_types::ExportIndex::Function(expected_start)) -+ { -+ return Err( -+ "deterministic-start proof does not match the module start function".to_string(), -+ ); -+ } -+ } -+ Ok(()) -+} -+ -+#[cfg(unix)] -+fn compare_memory_to_file( -+ memory: &Memory, -+ store: &impl AsStoreRef, -+ file: &File, -+ len: u64, -+) -> Result<(), String> { -+ use std::os::unix::fs::FileExt; -+ -+ let view = memory.view(store); -+ if len > view.data_size() { -+ return Err(format!( -+ "preinitialized memory image exceeds current memory: image={len} memory={}", -+ view.data_size() -+ )); -+ } -+ // SAFETY: module start has returned and the linker exclusively owns the -+ // store. No guest code runs until this function returns. -+ let len = usize::try_from(len) -+ .map_err(|_| format!("preinitialized memory image exceeds host address width: {len}"))?; -+ let current = unsafe { &view.data_unchecked()[..len] }; -+ let mut buffer = vec![0_u8; PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT as usize]; -+ let mut offset = 0_usize; -+ while offset < current.len() { -+ let count = buffer.len().min(current.len() - offset); -+ let mut read = 0_usize; -+ while read < count { -+ let got = file -+ .read_at(&mut buffer[read..count], (offset + read) as u64) -+ .map_err(|err| format!("read preinitialized memory image: {err}"))?; -+ if got == 0 { -+ return Err("preinitialized memory image ended during comparison".to_string()); -+ } -+ read += got; -+ } -+ if current[offset..offset + count] != buffer[..count] { -+ let mismatch = current[offset..offset + count] -+ .iter() -+ .zip(&buffer[..count]) -+ .position(|(actual, expected)| actual != expected) -+ .unwrap_or(0); -+ return Err(format!( -+ "preinitialized memory image differs from module-start state at byte {}", -+ offset + mismatch -+ )); -+ } -+ offset += count; -+ } -+ Ok(()) -+} -+ -+fn create_new_regular_file(path: &PathBuf) -> Result { -+ let mut options = OpenOptions::new(); -+ options.create_new(true).read(true).write(true); -+ #[cfg(unix)] -+ { -+ use std::os::unix::fs::OpenOptionsExt; -+ options.mode(0o600).custom_flags(libc::O_CLOEXEC); -+ } -+ options -+ .open(path) -+ .map_err(|err| format!("create {}: {err}", path.display())) -+} -+ -+#[cfg(target_os = "linux")] -+fn new_immutable_snapshot_backing() -> Result { -+ use std::{ffi::CString, os::fd::FromRawFd}; -+ -+ let name = CString::new("oliphaunt-wasix-preinitialized-memory") -+ .map_err(|err| format!("construct memory-image backing name: {err}"))?; -+ // SAFETY: the name is NUL terminated and the flags are valid for -+ // memfd_create. A successful descriptor is transferred into File below. -+ let descriptor = unsafe { -+ libc::syscall( -+ libc::SYS_memfd_create, -+ name.as_ptr(), -+ libc::MFD_CLOEXEC | libc::MFD_ALLOW_SEALING, -+ ) -+ }; -+ if descriptor < 0 { -+ return Err(format!( -+ "create sealed memory-image backing: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ // SAFETY: memfd_create returned a new owned descriptor on success. -+ Ok(unsafe { File::from_raw_fd(descriptor as i32) }) -+} -+ -+#[cfg(not(target_os = "linux"))] -+fn new_immutable_snapshot_backing() -> Result { -+ tempfile::tempfile().map_err(|err| format!("create unlinked memory-image backing: {err}")) -+} -+ -+#[cfg(target_os = "linux")] -+fn seal_immutable_snapshot(file: &File) -> Result<(), String> { -+ use std::os::fd::AsRawFd; -+ -+ let seals = libc::F_SEAL_WRITE | libc::F_SEAL_GROW | libc::F_SEAL_SHRINK | libc::F_SEAL_SEAL; -+ // SAFETY: fcntl is called with F_ADD_SEALS on a live memfd descriptor. -+ let result = unsafe { libc::fcntl(file.as_raw_fd(), libc::F_ADD_SEALS, seals) }; -+ if result != 0 { -+ return Err(format!( -+ "seal memory-image backing: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ Ok(()) -+} -+ -+#[cfg(not(target_os = "linux"))] -+fn seal_immutable_snapshot(_file: &File) -> Result<(), String> { -+ // The backing is an unlinked file and its only writable handle is retained -+ // privately by this type. Mapping support still fails closed on non-Unix. -+ Ok(()) -+} -+ -+/// Proves that the opened Linux inode cannot be modified through ordinary -+/// runtime privileges. The decision is tied to the live descriptor, not its -+/// path or advisory Unix mode bits. -+#[cfg(target_os = "linux")] -+pub fn intrinsic_file_immutability(file: &File) -> Result { -+ use std::os::fd::AsRawFd; -+ -+ if !file -+ .metadata() -+ .map_err(|err| format!("stat immutable-file candidate: {err}"))? -+ .is_file() -+ { -+ return Err("immutable-file candidate is not a regular file".to_string()); -+ } -+ -+ let mut stat = std::mem::MaybeUninit::::uninit(); -+ // SAFETY: `stat` points to writable storage and the descriptor remains live. -+ let result = unsafe { libc::fstatfs(file.as_raw_fd(), stat.as_mut_ptr()) }; -+ if result != 0 { -+ return Err(format!( -+ "identify immutable-file filesystem: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ // SAFETY: successful fstatfs initialized the value. -+ let filesystem_type = unsafe { stat.assume_init() }.f_type as u64; -+ if let Some(read_only) = classify_intrinsic_immutability(filesystem_type, None, true) { -+ return Ok(read_only); -+ } -+ -+ let mut inode_flags: libc::c_long = 0; -+ // SAFETY: FS_IOC_GETFLAGS only reads flags from this live descriptor into -+ // the suitably sized output word. -+ let flags_result = -+ unsafe { libc::ioctl(file.as_raw_fd(), libc::FS_IOC_GETFLAGS, &mut inode_flags) }; -+ let flags_error = (flags_result != 0).then(std::io::Error::last_os_error); -+ let inode_flags = (flags_result == 0).then_some(inode_flags as u32); -+ let can_clear_immutable = process_has_permitted_linux_immutable()?; -+ -+ classify_intrinsic_immutability(filesystem_type, inode_flags, can_clear_immutable) -+ .ok_or_else(|| { -+ if flags_result == 0 { -+ "file is neither on SquashFS/EROFS nor protected by an immutable inode flag the runtime cannot clear".to_string() -+ } else { -+ format!( -+ "file is not on SquashFS/EROFS and inode flags are unavailable: {}", -+ flags_error.expect("failed ioctl must retain its error") -+ ) -+ } -+ }) -+} -+ -+#[cfg(not(target_os = "linux"))] -+pub fn intrinsic_file_immutability(_file: &File) -> Result { -+ Err("intrinsically immutable file detection is only implemented on Linux".to_string()) -+} -+ -+#[cfg(target_os = "linux")] -+fn classify_intrinsic_immutability( -+ filesystem_type: u64, -+ inode_flags: Option, -+ can_clear_immutable: bool, -+) -> Option { -+ const SQUASHFS_MAGIC: u64 = 0x7371_7368; -+ const EROFS_SUPER_MAGIC: u64 = 0xe0f5_e1e2; -+ const FS_IMMUTABLE_FL: u32 = 0x0000_0010; -+ -+ if matches!(filesystem_type, SQUASHFS_MAGIC | EROFS_SUPER_MAGIC) { -+ Some(IntrinsicFileImmutability::ReadOnlyFilesystem) -+ } else if inode_flags.is_some_and(|flags| flags & FS_IMMUTABLE_FL != 0) && !can_clear_immutable -+ { -+ Some(IntrinsicFileImmutability::ImmutableInode) -+ } else { -+ None -+ } -+} -+ -+#[cfg(target_os = "linux")] -+fn process_has_permitted_linux_immutable() -> Result { -+ #[repr(C)] -+ struct CapabilityHeader { -+ version: u32, -+ pid: i32, -+ } -+ #[repr(C)] -+ #[derive(Clone, Copy)] -+ struct CapabilityData { -+ effective: u32, -+ permitted: u32, -+ inheritable: u32, -+ } -+ -+ const LINUX_CAPABILITY_VERSION_3: u32 = 0x2008_0522; -+ const CAP_LINUX_IMMUTABLE: u32 = 9; -+ let mut header = CapabilityHeader { -+ version: LINUX_CAPABILITY_VERSION_3, -+ pid: 0, -+ }; -+ let mut data = [CapabilityData { -+ effective: 0, -+ permitted: 0, -+ inheritable: 0, -+ }; 2]; -+ // SAFETY: capget writes exactly two v3 data words for the calling thread. -+ let result = unsafe { -+ libc::syscall( -+ libc::SYS_capget, -+ &mut header as *mut CapabilityHeader, -+ data.as_mut_ptr(), -+ ) -+ }; -+ if result != 0 { -+ return Err(format!( -+ "read process capabilities for immutable-file admission: {}", -+ std::io::Error::last_os_error() -+ )); -+ } -+ let word = (CAP_LINUX_IMMUTABLE / 32) as usize; -+ let bit = 1_u32 << (CAP_LINUX_IMMUTABLE % 32); -+ Ok(data[word].permitted & bit != 0) -+} -+ -+#[cfg(unix)] -+fn advise_mapping_away(mapping: &shared_buffer::OwnedBuffer) -> FileAdviceAudit { -+ if mapping.is_empty() { -+ FileAdviceAudit::not_applicable() -+ } else { -+ // SAFETY: the read-only mapping remains live for this call and the hint -+ // does not affect file contents. -+ let result = unsafe { -+ libc::madvise( -+ mapping.as_ptr().cast_mut().cast(), -+ mapping.len(), -+ libc::MADV_DONTNEED, -+ ) -+ }; -+ let errno = (result != 0).then(|| { -+ std::io::Error::last_os_error() -+ .raw_os_error() -+ .unwrap_or(libc::EIO) -+ }); -+ FileAdviceAudit { -+ supported: true, -+ calls: 1, -+ successes: u64::from(result == 0), -+ first_errno: errno, -+ } -+ } -+} -+ -+fn release_validation_faulted_source_pages( -+ backing: PreinitializedMemoryImageBacking, -+ file: &File, -+) -> FileAdviceAudit { -+ if matches!( -+ backing, -+ PreinitializedMemoryImageBacking::DirectIntrinsic(_) -+ ) { -+ advise_file_away(file) -+ } else { -+ FileAdviceAudit::not_applicable() -+ } -+} -+ -+#[cfg(not(unix))] -+fn advise_mapping_away(_mapping: &shared_buffer::OwnedBuffer) -> FileAdviceAudit { -+ FileAdviceAudit::unsupported() -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use std::io::Cursor; -+ -+ fn metadata(module_hash: ModuleHash) -> PreinitializedMemoryImageMetadata { -+ PreinitializedMemoryImageMetadata { -+ schema: PreinitializedMemoryImageMetadata::SCHEMA.to_string(), -+ module_sha256: module_hash.to_string().to_ascii_lowercase(), -+ runtime_abi_id: option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID") -+ .unwrap_or("") -+ .to_string(), -+ phase: PREINITIALIZED_MEMORY_IMAGE_PHASE.to_string(), -+ mapping_alignment: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, -+ mapped_size: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, -+ memory_minimum_pages: 1, -+ memory_maximum_pages: None, -+ memory_shared: true, -+ memory_base: 4096, -+ dylink_memory_size: 1, -+ dylink_memory_alignment: 12, -+ stack_low: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, -+ deterministic_start_proof: None, -+ deterministic_start_proof_output_sha256: None, -+ } -+ } -+ -+ fn deterministic_start_proof(module_hash: ModuleHash) -> DeterministicStartProof { -+ DeterministicStartProof { -+ schema: DETERMINISTIC_START_PROOF_SCHEMA.to_string(), -+ analyzer_policy: DETERMINISTIC_START_ANALYZER_POLICY.to_string(), -+ module_sha256: module_hash.to_string().to_ascii_lowercase(), -+ proof_sha256: "ab".repeat(32), -+ start_function_index: 7, -+ start_function_export: "__wasm_init_memory".to_string(), -+ transitive_function_indices: vec![7, 8], -+ imported_function_calls: 0, -+ memory_reads: DETERMINISTIC_START_MEMORY_READS.to_string(), -+ memory_effects: DETERMINISTIC_START_MEMORY_EFFECTS.to_string(), -+ global_effects: DETERMINISTIC_START_GLOBAL_EFFECTS.to_string(), -+ table_effects: DETERMINISTIC_START_TABLE_EFFECTS.to_string(), -+ requires_fresh_zeroed_memory: true, -+ ordinary_start_execution_per_instance: true, -+ first_instance_full_byte_validation: true, -+ } -+ } -+ -+ #[test] -+ fn metadata_rejects_mapping_ranges_that_are_not_wasm_page_aligned() { -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let mut metadata = metadata(module_hash); -+ metadata.mapped_size -= 1; -+ assert!(validate_metadata_shape(&metadata, module_hash).is_err()); -+ } -+ -+ #[test] -+ fn verified_reader_rejects_short_long_and_corrupt_images() { -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let metadata = metadata(module_hash); -+ let bytes = vec![0x5a; metadata.mapped_size as usize]; -+ let digest: [u8; 32] = Sha256::digest(&bytes).into(); -+ -+ assert!( -+ PreinitializedMemoryImage::from_verified_reader( -+ Cursor::new(&bytes[..bytes.len() - 1]), -+ digest, -+ module_hash, -+ metadata.clone(), -+ ) -+ .is_err() -+ ); -+ -+ let mut long = bytes.clone(); -+ long.push(0); -+ assert!( -+ PreinitializedMemoryImage::from_verified_reader( -+ Cursor::new(long), -+ digest, -+ module_hash, -+ metadata.clone(), -+ ) -+ .is_err() -+ ); -+ -+ let mut corrupt = bytes.clone(); -+ corrupt[1] ^= 1; -+ assert!( -+ PreinitializedMemoryImage::from_verified_reader( -+ Cursor::new(corrupt), -+ digest, -+ module_hash, -+ metadata, -+ ) -+ .is_err() -+ ); -+ } -+ -+ fn verified_image() -> Arc { -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let metadata = metadata(module_hash); -+ let bytes = vec![0x5a; metadata.mapped_size as usize]; -+ let digest: [u8; 32] = Sha256::digest(&bytes).into(); -+ Arc::new( -+ PreinitializedMemoryImage::from_verified_reader( -+ Cursor::new(bytes), -+ digest, -+ module_hash, -+ metadata, -+ ) -+ .unwrap(), -+ ) -+ } -+ -+ fn verified_attested_image() -> Arc { -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let mut metadata = metadata(module_hash); -+ metadata.schema = PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA.to_string(); -+ let proof = deterministic_start_proof(module_hash); -+ metadata.deterministic_start_proof_output_sha256 = -+ Some(canonical_deterministic_start_proof_sha256(&proof).unwrap()); -+ metadata.deterministic_start_proof = Some(proof); -+ let bytes = vec![0x5a; metadata.mapped_size as usize]; -+ let digest: [u8; 32] = Sha256::digest(&bytes).into(); -+ Arc::new( -+ PreinitializedMemoryImage::from_verified_reader( -+ Cursor::new(bytes), -+ digest, -+ module_hash, -+ metadata, -+ ) -+ .unwrap(), -+ ) -+ } -+ -+ #[test] -+ fn attested_runtime_audit_has_exact_terminal_conservation() { -+ let image = verified_attested_image(); -+ let mapped_size = image.metadata.mapped_size; -+ -+ image.runtime_audit.record_start_boundary(true); -+ image -+ .runtime_audit -+ .record_validation(true, true, mapped_size); -+ image.runtime_audit.record_remap(true); -+ for _ in 0..3 { -+ image.runtime_audit.record_start_boundary(true); -+ image -+ .runtime_audit -+ .record_validation(false, true, mapped_size); -+ image.runtime_audit.record_remap(true); -+ } -+ -+ let audit = image.runtime_audit().unwrap(); -+ assert_eq!(audit.ordinary_start_completed_instances, 4); -+ assert_eq!(audit.fresh_zeroed_instances, 4); -+ assert_eq!(audit.nonfresh_instances, 0); -+ assert_eq!(audit.validation_attempts, 4); -+ assert_eq!(audit.full_compare_attempts, 1); -+ assert_eq!(audit.full_compare_successes, 1); -+ assert_eq!(audit.full_compare_failures, 0); -+ assert_eq!(audit.compared_bytes, mapped_size); -+ assert_eq!(audit.reuse_successes, 3); -+ assert_eq!(audit.reuse_failures, 0); -+ assert_eq!(audit.skipped_bytes, 3 * mapped_size); -+ assert_eq!(audit.remap_successes, 4); -+ assert_eq!(audit.remap_failures, 0); -+ assert!(!audit.counter_overflow); -+ } -+ -+ #[test] -+ fn attested_runtime_audit_reports_counter_overflow_without_wrapping() { -+ let image = verified_attested_image(); -+ image -+ .runtime_audit -+ .skipped_bytes -+ .store(u64::MAX, Ordering::Release); -+ image -+ .runtime_audit -+ .record_validation(false, true, image.metadata.mapped_size); -+ -+ let audit = image.runtime_audit().unwrap(); -+ assert_eq!(audit.skipped_bytes, u64::MAX); -+ assert_eq!(audit.reuse_successes, 1); -+ assert!(audit.counter_overflow); -+ } -+ -+ #[test] -+ fn metadata_requires_proof_and_attested_schema_as_a_pair() { -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let mut metadata = metadata(module_hash); -+ let proof = deterministic_start_proof(module_hash); -+ metadata.deterministic_start_proof_output_sha256 = -+ Some(canonical_deterministic_start_proof_sha256(&proof).unwrap()); -+ metadata.deterministic_start_proof = Some(proof); -+ assert!(validate_metadata_shape(&metadata, module_hash).is_err()); -+ -+ metadata.schema = PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA.to_string(); -+ metadata.deterministic_start_proof = None; -+ assert!(validate_metadata_shape(&metadata, module_hash).is_err()); -+ -+ metadata.deterministic_start_proof = Some(deterministic_start_proof(module_hash)); -+ validate_metadata_shape(&metadata, module_hash).unwrap(); -+ } -+ -+ #[test] -+ fn metadata_rejects_valid_proof_for_another_module_and_output_digest_drift() { -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let other_hash = ModuleHash::from_bytes([0x43; 32]); -+ let proof = deterministic_start_proof(other_hash); -+ let mut metadata = metadata(module_hash); -+ metadata.schema = PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA.to_string(); -+ metadata.deterministic_start_proof_output_sha256 = -+ Some(canonical_deterministic_start_proof_sha256(&proof).unwrap()); -+ metadata.deterministic_start_proof = Some(proof); -+ let error = validate_metadata_shape(&metadata, module_hash).unwrap_err(); -+ assert!(error.contains("proof policy"), "unexpected error: {error}"); -+ -+ let proof = deterministic_start_proof(module_hash); -+ metadata.deterministic_start_proof_output_sha256 = Some("cd".repeat(32)); -+ metadata.deterministic_start_proof = Some(proof); -+ let error = validate_metadata_shape(&metadata, module_hash).unwrap_err(); -+ assert!( -+ error.contains("canonical output digest"), -+ "unexpected error: {error}" -+ ); -+ } -+ -+ #[test] -+ fn attested_runtime_validation_is_single_flight() { -+ use std::sync::{ -+ Barrier, -+ atomic::{AtomicUsize, Ordering}, -+ }; -+ -+ let image = verified_attested_image(); -+ let comparisons = Arc::new(AtomicUsize::new(0)); -+ let barrier = Arc::new(Barrier::new(8)); -+ let threads = (0..8) -+ .map(|_| { -+ let image = image.clone(); -+ let comparisons = comparisons.clone(); -+ let barrier = barrier.clone(); -+ std::thread::spawn(move || { -+ barrier.wait(); -+ image.validate_runtime_bytes(true, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Ok(()) -+ }) -+ }) -+ }) -+ .collect::>(); -+ let results = threads -+ .into_iter() -+ .map(|thread| thread.join().unwrap()) -+ .collect::>(); -+ -+ assert!(results.iter().all(|result| result.validation.is_ok())); -+ assert_eq!(comparisons.load(Ordering::Relaxed), 1); -+ assert_eq!( -+ results -+ .iter() -+ .filter(|result| result.compared_this_instance) -+ .count(), -+ 1 -+ ); -+ } -+ -+ #[test] -+ fn attested_runtime_validation_caches_failure_and_rejects_reused_memory() { -+ use std::sync::atomic::{AtomicUsize, Ordering}; -+ -+ let image = verified_attested_image(); -+ let comparisons = AtomicUsize::new(0); -+ let reused = image.validate_runtime_bytes(false, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Ok(()) -+ }); -+ assert!( -+ reused -+ .validation -+ .unwrap_err() -+ .contains("fresh zeroed memory") -+ ); -+ assert!(!reused.compared_this_instance); -+ assert_eq!(comparisons.load(Ordering::Relaxed), 0); -+ -+ let first = image.validate_runtime_bytes(true, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Err("captured image mismatch".to_string()) -+ }); -+ assert_eq!(first.validation.unwrap_err(), "captured image mismatch"); -+ assert!(first.compared_this_instance); -+ -+ let cached = image.validate_runtime_bytes(true, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Ok(()) -+ }); -+ assert_eq!(cached.validation.unwrap_err(), "captured image mismatch"); -+ assert!(!cached.compared_this_instance); -+ assert_eq!(comparisons.load(Ordering::Relaxed), 1); -+ } -+ -+ #[test] -+ fn runtime_byte_validation_runs_for_every_fresh_instance() { -+ use std::sync::{ -+ Barrier, -+ atomic::{AtomicUsize, Ordering}, -+ }; -+ -+ let image = verified_image(); -+ let comparisons = Arc::new(AtomicUsize::new(0)); -+ let barrier = Arc::new(Barrier::new(8)); -+ let threads = (0..8) -+ .map(|_| { -+ let image = image.clone(); -+ let comparisons = comparisons.clone(); -+ let barrier = barrier.clone(); -+ std::thread::spawn(move || { -+ barrier.wait(); -+ image -+ .validate_runtime_bytes(true, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Ok(()) -+ }) -+ .validation -+ }) -+ }) -+ .collect::>(); -+ let results = threads -+ .into_iter() -+ .map(|thread| thread.join().unwrap()) -+ .collect::>(); -+ -+ assert!(results.iter().all(Result::is_ok)); -+ assert_eq!(comparisons.load(Ordering::Relaxed), 8); -+ -+ image -+ .validate_runtime_bytes(true, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Ok(()) -+ }) -+ .validation -+ .unwrap(); -+ assert_eq!(comparisons.load(Ordering::Relaxed), 9); -+ } -+ -+ #[test] -+ fn runtime_byte_validation_does_not_cache_a_prior_mismatch() { -+ use std::sync::{ -+ Barrier, -+ atomic::{AtomicUsize, Ordering}, -+ }; -+ -+ let image = verified_image(); -+ let comparisons = Arc::new(AtomicUsize::new(0)); -+ let barrier = Arc::new(Barrier::new(8)); -+ let threads = (0..8) -+ .map(|_| { -+ let image = image.clone(); -+ let comparisons = comparisons.clone(); -+ let barrier = barrier.clone(); -+ std::thread::spawn(move || { -+ barrier.wait(); -+ image -+ .validate_runtime_bytes(true, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Err("deterministic module-start mismatch".to_string()) -+ }) -+ .validation -+ }) -+ }) -+ .collect::>(); -+ let results = threads -+ .into_iter() -+ .map(|thread| thread.join().unwrap()) -+ .collect::>(); -+ -+ assert!(results.iter().all(|result| { -+ result.as_ref().unwrap_err() == "deterministic module-start mismatch" -+ })); -+ assert_eq!(comparisons.load(Ordering::Relaxed), 8); -+ -+ image -+ .validate_runtime_bytes(true, || { -+ comparisons.fetch_add(1, Ordering::Relaxed); -+ Ok(()) -+ }) -+ .validation -+ .unwrap(); -+ assert_eq!(comparisons.load(Ordering::Relaxed), 9); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[test] -+ fn validation_cache_release_is_direct_only_and_preserves_file_bytes() { -+ use std::os::unix::fs::FileExt; -+ -+ let mut source = tempfile::tempfile().unwrap(); -+ let bytes = vec![0x5a; PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT as usize]; -+ source.write_all(&bytes).unwrap(); -+ let direct = release_validation_faulted_source_pages( -+ PreinitializedMemoryImageBacking::DirectIntrinsic( -+ IntrinsicFileImmutability::ReadOnlyFilesystem, -+ ), -+ &source, -+ ); -+ assert!(direct.supported); -+ assert_eq!(direct.calls, 1); -+ assert_eq!(direct.successes, 1); -+ assert_eq!(direct.first_errno, None); -+ -+ let copied = release_validation_faulted_source_pages( -+ PreinitializedMemoryImageBacking::SealedCopy, -+ &source, -+ ); -+ assert_eq!(copied, FileAdviceAudit::not_applicable()); -+ -+ let mut retained = vec![0_u8; bytes.len()]; -+ source.read_at(&mut retained, 0).unwrap(); -+ assert_eq!(retained, bytes); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[test] -+ fn intrinsic_immutability_classifier_is_fail_closed() { -+ const EXT4_SUPER_MAGIC: u64 = 0xef53; -+ const SQUASHFS_MAGIC: u64 = 0x7371_7368; -+ const EROFS_SUPER_MAGIC: u64 = 0xe0f5_e1e2; -+ const FS_IMMUTABLE_FL: u32 = 0x10; -+ -+ assert_eq!( -+ classify_intrinsic_immutability(SQUASHFS_MAGIC, None, true), -+ Some(IntrinsicFileImmutability::ReadOnlyFilesystem) -+ ); -+ assert_eq!( -+ classify_intrinsic_immutability(EROFS_SUPER_MAGIC, None, true), -+ Some(IntrinsicFileImmutability::ReadOnlyFilesystem) -+ ); -+ assert_eq!( -+ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, Some(FS_IMMUTABLE_FL), false), -+ Some(IntrinsicFileImmutability::ImmutableInode) -+ ); -+ assert_eq!( -+ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, Some(FS_IMMUTABLE_FL), true), -+ None -+ ); -+ assert_eq!( -+ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, Some(0), false), -+ None -+ ); -+ assert_eq!( -+ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, None, false), -+ None -+ ); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[test] -+ fn direct_image_constructor_rejects_a_mutable_inode() { -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let metadata = metadata(module_hash); -+ let bytes = vec![0x5a; metadata.mapped_size as usize]; -+ let digest: [u8; 32] = Sha256::digest(&bytes).into(); -+ let mut source = tempfile::tempfile().unwrap(); -+ source.write_all(&bytes).unwrap(); -+ -+ let error = PreinitializedMemoryImage::from_verified_immutable_file( -+ source, -+ digest, -+ module_hash, -+ metadata, -+ ) -+ .unwrap_err(); -+ assert!(error.contains("neither on SquashFS/EROFS nor protected")); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[test] -+ fn sealed_copy_is_isolated_from_later_source_mutation() { -+ use std::os::unix::fs::FileExt; -+ -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let metadata = metadata(module_hash); -+ let bytes = vec![0x5a; metadata.mapped_size as usize]; -+ let digest: [u8; 32] = Sha256::digest(&bytes).into(); -+ let mut source = tempfile::tempfile().unwrap(); -+ source.write_all(&bytes).unwrap(); -+ source.seek(SeekFrom::Start(0)).unwrap(); -+ let image = -+ PreinitializedMemoryImage::from_verified_reader(&source, digest, module_hash, metadata) -+ .unwrap(); -+ -+ source.write_at(&[0x33], 0).unwrap(); -+ let mut retained = [0_u8; 1]; -+ image.file.read_at(&mut retained, 0).unwrap(); -+ assert_eq!(retained, [0x5a]); -+ assert_eq!( -+ image.backing(), -+ PreinitializedMemoryImageBacking::SealedCopy -+ ); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[test] -+ fn verified_reader_owns_a_sealed_backing() { -+ use std::os::{fd::AsRawFd, unix::fs::FileExt}; -+ -+ let module_hash = ModuleHash::from_bytes([0x42; 32]); -+ let metadata = metadata(module_hash); -+ let bytes = vec![0x5a; metadata.mapped_size as usize]; -+ let digest: [u8; 32] = Sha256::digest(&bytes).into(); -+ let image = PreinitializedMemoryImage::from_verified_reader( -+ Cursor::new(bytes), -+ digest, -+ module_hash, -+ metadata, -+ ) -+ .unwrap(); -+ -+ // SAFETY: F_GET_SEALS does not mutate the live memfd descriptor. -+ let seals = unsafe { libc::fcntl(image.file.as_raw_fd(), libc::F_GET_SEALS) }; -+ let expected = -+ libc::F_SEAL_WRITE | libc::F_SEAL_GROW | libc::F_SEAL_SHRINK | libc::F_SEAL_SEAL; -+ assert_eq!(seals & expected, expected); -+ assert!(image.file.write_at(&[0], 0).is_err()); -+ } -+} -diff --git a/lib/wasix/src/syscalls/mod.rs b/lib/wasix/src/syscalls/mod.rs -index a3d58df..a9e55fe 100644 ---- a/lib/wasix/src/syscalls/mod.rs -+++ b/lib/wasix/src/syscalls/mod.rs -@@ -136,8 +136,8 @@ pub(crate) use crate::{ - }, - runtime::SpawnType, - state::{ -- self, InodeGuard, InodeWeakGuard, PollEvent, PollEventBuilder, WasiFutex, WasiState, -- iterate_poll_events, -+ self, InodeGuard, InodeWeakGuard, PollEvent, PollEventBuilder, WasiFutex, -+ WasiFutexRegistry, WasiSharedMemoryMapping, WasiState, iterate_poll_events, - }, - utils::{self, map_io_err}, - }; -@@ -341,10 +341,23 @@ where - return Poll::Ready(Ok(res)); - } - if let Some(signals) = self.ctx.data().thread.pop_signals_or_subscribe(cx.waker()) { -- if let Err(err) = WasiEnv::process_signals_internal(self.ctx, signals) { -+ let processed_guest_signal = -+ match WasiEnv::process_signals_internal(self.ctx, signals) { -+ Ok(processed) => processed, -+ Err(err) => { -+ return Poll::Ready(Err(err)); -+ } -+ }; -+ if let Err(err) = WasiEnv::do_pending_link_operations(self.ctx, false) { - return Poll::Ready(Err(err)); - } -- return Poll::Ready(Ok(Err(Errno::Intr))); -+ if let Poll::Ready(res) = Pin::new(&mut self.pinned_work).poll(cx) { -+ return Poll::Ready(Ok(res)); -+ } -+ if processed_guest_signal { -+ return Poll::Ready(Ok(Err(Errno::Intr))); -+ } -+ self.ctx.data().thread.signals_subscribe(cx.waker()); - } - Poll::Pending - } -@@ -461,12 +474,12 @@ pub(crate) fn maybe_backoff( - // Determine if we need to do a backoff, if so lets do one - if let Some(backoff) = env.process.acquire_cpu_backoff_token(env.tasks()) { - tracing::trace!("exponential CPU backoff {:?}", backoff.backoff_time()); -- if let AsyncifyAction::Finish(mut ctx, _) = -- __asyncify_with_deep_sleep::(ctx, backoff)? -- { -- Ok(Ok(ctx)) -- } else { -- Ok(Err(Errno::Success)) -+ match __asyncify(&mut ctx, None, async move { -+ backoff.await; -+ Ok(()) -+ })? { -+ Ok(()) => Ok(Ok(ctx)), -+ Err(err) => Ok(Err(err)), - } - } else { - Ok(Ok(ctx)) -@@ -612,8 +625,8 @@ where - let mut work = { - let inode = fd_entry.inode.clone(); - let tasks = env.tasks().clone(); -- let mut guard = inode.write(); -- match guard.deref_mut() { -+ let guard = inode.read(); -+ match guard.deref() { - Kind::Socket { socket } => { - // Clone the socket and release the lock - let socket = socket.clone(); -@@ -654,8 +667,8 @@ where - } - - let inode = fd_entry.inode.clone(); -- let mut guard = inode.write(); -- match guard.deref_mut() { -+ let guard = inode.read(); -+ match guard.deref() { - Kind::Socket { socket } => { - // Clone the socket and release the lock - let socket = socket.clone(); -@@ -672,43 +685,49 @@ where - } - } - --/// Performs an immutable operation on the socket while running in an asynchronous runtime --/// This has built in signal support --pub(crate) fn __sock_actor( -- ctx: &mut FunctionEnvMut<'_, WasiEnv>, -+/// Performs a synchronous operation on the socket without entering asyncify. -+pub(crate) fn __sock_actor_env( -+ env: &WasiEnv, - sock: WasiFd, - rights: Rights, - actor: F, - ) -> Result - where -- T: 'static, - F: FnOnce(crate::net::socket::InodeSocket, Fd) -> Result, - { -- let env = ctx.data(); -- let tasks = env.tasks().clone(); -- - let fd_entry = env.state.fs.get_fd(sock)?; - if !rights.is_empty() && !fd_entry.inner.rights.contains(rights) { - return Err(Errno::Access); - } - - let inode = fd_entry.inode.clone(); -- -- let tasks = env.tasks().clone(); -- let mut guard = inode.write(); -- match guard.deref_mut() { -+ let guard = inode.read(); -+ match guard.deref() { - Kind::Socket { socket } => { -- // Clone the socket and release the lock - let socket = socket.clone(); - drop(guard); -- -- // Start the work using the socket - actor(socket, fd_entry) - } - _ => Err(Errno::Notsock), - } - } - -+/// Performs an immutable operation on the socket while running in an asynchronous runtime -+/// This has built in signal support -+pub(crate) fn __sock_actor( -+ ctx: &mut FunctionEnvMut<'_, WasiEnv>, -+ sock: WasiFd, -+ rights: Rights, -+ actor: F, -+) -> Result -+where -+ T: 'static, -+ F: FnOnce(crate::net::socket::InodeSocket, Fd) -> Result, -+{ -+ let env = ctx.data(); -+ __sock_actor_env(env, sock, rights, actor) -+} -+ - /// Performs mutable work on a socket under an asynchronous runtime with - /// built in signal processing - pub(crate) fn __sock_actor_mut( -@@ -730,8 +749,8 @@ where - } - - let inode = fd_entry.inode.clone(); -- let mut guard = inode.write(); -- match guard.deref_mut() { -+ let guard = inode.read(); -+ match guard.deref() { - Kind::Socket { socket } => { - // Clone the socket and release the lock - let socket = socket.clone(); -@@ -1017,7 +1036,7 @@ pub(crate) fn deep_sleep( - trigger: Pin>, - ) -> Result<(), WasiError> { - // Grab all the globals and serialize them -- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) -+ let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut())? - .serialize() - .unwrap(); - let store_data = Bytes::from(store_data); -@@ -1033,9 +1052,12 @@ pub(crate) fn deep_sleep( - // If journal'ing is enabled then we dump the stack into the journal - if ctx.data().enable_journal { - // Grab all the globals and serialize them -- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) -- .serialize() -- .unwrap(); -+ let snapshot = -+ match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { -+ Ok(snapshot) => snapshot, -+ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), -+ }; -+ let store_data = snapshot.serialize().unwrap(); - let store_data = Bytes::from(store_data); - - tracing::trace!( -@@ -1158,8 +1180,9 @@ where - let asyncify_data = wasi_try_ok!(unwind_pointer.try_into().map_err(|_| Errno::Overflow)); - if let Some(asyncify_start_unwind) = env - .inner() -- .static_module_instance_handles() -- .and_then(|handles| handles.asyncify_start_unwind.clone()) -+ .main_module_instance_handles() -+ .asyncify_start_unwind -+ .clone() - { - asyncify_start_unwind.call(&mut ctx, asyncify_data); - } else { -@@ -1222,10 +1245,8 @@ where - .map_err(|err| format!("failed to read stack: {err}"))?; - - // Notify asyncify that we are no longer unwinding -- if let Some(asyncify_stop_unwind) = env -- .inner() -- .static_module_instance_handles() -- .and_then(|i| i.asyncify_stop_unwind.clone()) -+ if let Some(asyncify_stop_unwind) = -+ env.inner().main_module_instance_handles().asyncify_stop_unwind.clone() - { - asyncify_stop_unwind.call(&mut ctx); - } else { -@@ -1283,11 +1304,14 @@ pub fn rewind_ext( - let store_snapshot = match StoreSnapshot::deserialize(&store_data[..]) { - Ok(a) => a, - Err(err) => { -- warn!("snapshot restore failed - the store snapshot could not be deserialized"); -+ warn!("snapshot restore failed - the store snapshot could not be deserialized: {err}"); - return Errno::Unknown; - } - }; -- crate::utils::store::restore_store_snapshot(ctx, &store_snapshot); -+ if let Err(err) = crate::utils::store::restore_store_snapshot(ctx, &store_snapshot) { -+ warn!("snapshot restore failed - destination store is incompatible: {err}"); -+ return Errno::Unknown; -+ } - let env = ctx.data(); - let memory = match env.try_memory_view(&ctx) { - Some(v) => v, -@@ -1340,8 +1364,9 @@ pub fn rewind_ext( - let asyncify_data = wasi_try!(rewind_pointer.try_into().map_err(|_| Errno::Overflow)); - if let Some(asyncify_start_rewind) = env - .inner() -- .static_module_instance_handles() -- .and_then(|a| a.asyncify_start_rewind.clone()) -+ .main_module_instance_handles() -+ .asyncify_start_rewind -+ .clone() - { - asyncify_start_rewind.call(ctx, asyncify_data); - } else { -@@ -1443,8 +1468,9 @@ where - let env = ctx.data(); - if let Some(asyncify_stop_rewind) = env - .inner() -- .static_module_instance_handles() -- .and_then(|handles| handles.asyncify_stop_rewind.clone()) -+ .main_module_instance_handles() -+ .asyncify_stop_rewind -+ .clone() - { - asyncify_stop_rewind.call(ctx); - } else { -diff --git a/lib/wasix/src/syscalls/wasi/fd_advise.rs b/lib/wasix/src/syscalls/wasi/fd_advise.rs -index 288a84a..0e5299c 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_advise.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_advise.rs -@@ -1,5 +1,6 @@ - use super::*; - use crate::syscalls::*; -+use virtual_fs::FileAdvice; - - /// ### `fd_advise()` - /// Advise the system about how a file will be used -@@ -22,7 +23,9 @@ pub fn fd_advise( - ) -> Result { - WasiEnv::do_pending_operations(&mut ctx)?; - -- wasi_try_ok!(fd_advise_internal(&mut ctx, fd, offset, len, advice)); -+ if let Err(error) = fd_advise_internal(&mut ctx, fd, offset, len, advice) { -+ return Ok(error); -+ } - let env = ctx.data(); - - #[cfg(feature = "journal")] -@@ -43,19 +46,206 @@ pub(crate) fn fd_advise_internal( - len: Filesize, - advice: Advice, - ) -> Result<(), Errno> { -- // Instead of unconditionally returning OK. This barebones implementation -- // only performs basic fd and rights checks. -- - let env = ctx.data(); - let (_, mut state) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; - let fd_entry = state.fs.get_fd(fd)?; - let inode = fd_entry.inode; - -- if !fd_entry.inner.rights.contains(Rights::FD_ADVISE) { -- return Err(Errno::Access); -+ require_fd_advise_right(fd_entry.inner.rights)?; -+ -+ delegate_fd_advice(offset, len, advice, |offset, len, advice| { -+ let guard = inode.read(); -+ advise_inode_kind(guard.deref(), offset, len, advice) -+ }) -+} -+ -+fn require_fd_advise_right(rights: Rights) -> Result<(), Errno> { -+ if rights.contains(Rights::FD_ADVISE) { -+ Ok(()) -+ } else { -+ Err(Errno::Notcapable) - } -+} - -- let _end = offset.checked_add(len).ok_or(Errno::Inval)?; -+fn advise_inode_kind( -+ kind: &Kind, -+ offset: Filesize, -+ len: Filesize, -+ advice: FileAdvice, -+) -> Result<(), Errno> { -+ match kind { -+ Kind::File { -+ handle: Some(handle), -+ .. -+ } => handle -+ .read() -+ .map_err(|_| Errno::Fault)? -+ .advise(offset, len, advice) -+ .map_err(fd_advise_io_error), -+ Kind::File { handle: None, .. } => Err(Errno::Badf), -+ Kind::Buffer { .. } => Err(Errno::Notsup), -+ Kind::PipeRx { .. } | Kind::PipeTx { .. } | Kind::DuplexPipe { .. } => Err(Errno::Spipe), -+ // Preview1 fd_advise classifies directories as invalid descriptors -+ // even though Linux posix_fadvise happens to accept directory fds. -+ Kind::Dir { .. } | Kind::Root { .. } => Err(Errno::Badf), -+ Kind::Socket { .. } -+ | Kind::Symlink { .. } -+ | Kind::EventNotifications { .. } -+ | Kind::Epoll { .. } => Err(Errno::Badf), -+ } -+} - -- Ok(()) -+fn delegate_fd_advice( -+ offset: Filesize, -+ len: Filesize, -+ advice: Advice, -+ delegate: impl FnOnce(Filesize, Filesize, FileAdvice) -> Result<(), Errno>, -+) -> Result<(), Errno> { -+ offset.checked_add(len).ok_or(Errno::Inval)?; -+ let advice = match advice { -+ Advice::Normal => FileAdvice::Normal, -+ Advice::Sequential => FileAdvice::Sequential, -+ Advice::Random => FileAdvice::Random, -+ Advice::Willneed => FileAdvice::WillNeed, -+ Advice::Dontneed => FileAdvice::DontNeed, -+ Advice::Noreuse => FileAdvice::NoReuse, -+ Advice::Unknown => return Err(Errno::Inval), -+ _ => return Err(Errno::Inval), -+ }; -+ delegate(offset, len, advice) -+} -+ -+fn fd_advise_io_error(error: io::Error) -> Errno { -+ #[cfg(target_os = "linux")] -+ if let Some(error) = error.raw_os_error() { -+ return match error { -+ libc::EACCES => Errno::Access, -+ libc::EBADF => Errno::Badf, -+ libc::EINVAL => Errno::Inval, -+ libc::EIO => Errno::Io, -+ libc::ENOSYS => Errno::Nosys, -+ libc::EOPNOTSUPP => Errno::Notsup, -+ libc::EOVERFLOW => Errno::Overflow, -+ libc::EPERM => Errno::Perm, -+ libc::ESPIPE => Errno::Spipe, -+ _ => return map_io_err(io::Error::from_raw_os_error(error)), -+ }; -+ } -+ -+ match error.kind() { -+ io::ErrorKind::InvalidInput => Errno::Inval, -+ io::ErrorKind::Unsupported => Errno::Notsup, -+ _ => map_io_err(error), -+ } -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ -+ #[test] -+ fn fd_advise_maps_every_wasi_advice_exactly() { -+ let cases = [ -+ (Advice::Normal, FileAdvice::Normal), -+ (Advice::Sequential, FileAdvice::Sequential), -+ (Advice::Random, FileAdvice::Random), -+ (Advice::Willneed, FileAdvice::WillNeed), -+ (Advice::Dontneed, FileAdvice::DontNeed), -+ (Advice::Noreuse, FileAdvice::NoReuse), -+ ]; -+ -+ for (wasi_advice, expected) in cases { -+ let mut observed = None; -+ delegate_fd_advice(17, 23, wasi_advice, |offset, len, advice| { -+ observed = Some((offset, len, advice)); -+ Ok(()) -+ }) -+ .unwrap(); -+ assert_eq!(observed, Some((17, 23, expected))); -+ } -+ } -+ -+ #[test] -+ fn fd_advise_delegates_willneed_and_dontneed() { -+ let mut observed = Vec::new(); -+ for advice in [Advice::Willneed, Advice::Dontneed] { -+ delegate_fd_advice(4096, 8192, advice, |offset, len, advice| { -+ observed.push((offset, len, advice)); -+ Ok(()) -+ }) -+ .unwrap(); -+ } -+ -+ assert_eq!( -+ observed, -+ vec![ -+ (4096, 8192, FileAdvice::WillNeed), -+ (4096, 8192, FileAdvice::DontNeed), -+ ] -+ ); -+ } -+ -+ #[test] -+ fn fd_advise_rejects_invalid_ranges_and_unknown_advice_without_delegating() { -+ let mut calls = 0; -+ let overflow = delegate_fd_advice(u64::MAX, 1, Advice::Willneed, |_, _, _| { -+ calls += 1; -+ Ok(()) -+ }); -+ assert_eq!(overflow, Err(Errno::Inval)); -+ -+ let unknown = delegate_fd_advice(0, 0, Advice::Unknown, |_, _, _| { -+ calls += 1; -+ Ok(()) -+ }); -+ assert_eq!(unknown, Err(Errno::Inval)); -+ assert_eq!(calls, 0); -+ } -+ -+ #[test] -+ fn fd_advise_maps_unsupported_backend_to_notsup() { -+ let result = delegate_fd_advice(0, 4096, Advice::Dontneed, |_, _, _| { -+ Err(fd_advise_io_error(io::ErrorKind::Unsupported.into())) -+ }); -+ -+ assert_eq!(result, Err(Errno::Notsup)); -+ } -+ -+ #[test] -+ fn fd_advise_missing_right_is_notcapable() { -+ assert_eq!( -+ require_fd_advise_right(Rights::empty()), -+ Err(Errno::Notcapable) -+ ); -+ assert_eq!(require_fd_advise_right(Rights::FD_ADVISE), Ok(())); -+ } -+ -+ #[test] -+ fn fd_advise_directory_is_badf() { -+ let directory = Kind::Root { -+ entries: Default::default(), -+ }; -+ assert_eq!( -+ advise_inode_kind(&directory, 0, 4096, FileAdvice::WillNeed), -+ Err(Errno::Badf) -+ ); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[test] -+ fn fd_advise_preserves_posix_fadvise_errnos() { -+ let cases = [ -+ (libc::EBADF, Errno::Badf), -+ (libc::EINVAL, Errno::Inval), -+ (libc::EOVERFLOW, Errno::Overflow), -+ (libc::ESPIPE, Errno::Spipe), -+ ]; -+ -+ for (host_errno, expected) in cases { -+ assert_eq!( -+ fd_advise_io_error(io::Error::from_raw_os_error(host_errno)), -+ expected -+ ); -+ } -+ } - } -diff --git a/lib/wasix/src/syscalls/wasi/fd_datasync.rs b/lib/wasix/src/syscalls/wasi/fd_datasync.rs -index f799d90..6187bf6 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_datasync.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_datasync.rs -@@ -16,9 +16,11 @@ pub fn fd_datasync(mut ctx: FunctionEnvMut<'_, WasiEnv>, fd: WasiFd) -> Result { - if let Some(handle) = handle { - let mut handle = handle.write().unwrap(); -+ #[cfg(feature = "host-fs")] -+ let _shared_mapping_guard = { -+ let host_file = handle -+ .upcast_any_ref() -+ .downcast_ref::() -+ .map(|file| file.try_clone_std_file()) -+ .transpose() -+ .map_err(crate::utils::map_io_err)?; -+ host_file -+ .as_ref() -+ .map(|file| state.guard_shared_mapping_file_shrink(file, st_size)) -+ .transpose()? -+ .flatten() -+ }; - handle.set_len(st_size).map_err(fs_error_into_wasi_err)?; - } else { - return Err(Errno::Badf); -diff --git a/lib/wasix/src/syscalls/wasi/fd_read.rs b/lib/wasix/src/syscalls/wasi/fd_read.rs -index f5de216..400441b 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_read.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_read.rs -@@ -1,6 +1,6 @@ - use std::{collections::VecDeque, task::Waker}; - --use virtual_fs::{AsyncReadExt, DeviceFile, ReadBuf}; -+use virtual_fs::{AsyncReadExt, DeviceFile, ReadBuf, VirtualFile}; - - use super::*; - use crate::{ -@@ -131,6 +131,88 @@ pub(crate) fn fd_read_internal_handler( - Ok(ret) - } - -+fn read_file_iovs_at( -+ memory: &wasmer::MemoryView<'_>, -+ iovs: WasmPtr<__wasi_iovec_t, M>, -+ iovs_len: M::Offset, -+ offset: usize, -+ mut read_at: impl FnMut(&mut [u8], u64) -> std::io::Result, -+) -> Result { -+ let mut total_read = 0usize; -+ let mut read_offset = offset as u64; -+ let iovs_arr = iovs.slice(memory, iovs_len).map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ for iovs in iovs_arr.iter() { -+ let mut buf = WasmPtr::::new(iovs.buf) -+ .slice(memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ if buf.as_ref().is_empty() { -+ continue; -+ } -+ let local_read = match read_at(buf.as_mut(), read_offset) { -+ Ok(s) => s, -+ Err(_) if total_read > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ total_read += local_read; -+ read_offset += local_read as u64; -+ if local_read != buf.len() { -+ break; -+ } -+ } -+ -+ Ok(total_read) -+} -+ -+fn try_read_file_positioned_shared_blocking( -+ handle: &(dyn VirtualFile + Send + Sync), -+ memory: &wasmer::MemoryView<'_>, -+ iovs: WasmPtr<__wasi_iovec_t, M>, -+ iovs_len: M::Offset, -+ offset: usize, -+) -> Result, Errno> { -+ if !handle.has_blocking_read_at_shared() { -+ return Ok(None); -+ } -+ -+ read_file_iovs_at(memory, iovs, iovs_len, offset, |buf, read_offset| { -+ handle.read_at_blocking_shared(buf, read_offset) -+ }) -+ .map(Some) -+} -+ -+fn try_read_file_blocking( -+ handle: &mut (dyn VirtualFile + Send + Sync), -+ memory: &wasmer::MemoryView<'_>, -+ iovs: WasmPtr<__wasi_iovec_t, M>, -+ iovs_len: M::Offset, -+ offset: usize, -+ positioned_read: bool, -+) -> Result, Errno> { -+ if positioned_read { -+ if !handle.has_blocking_read_at() { -+ return Ok(None); -+ } -+ return read_file_iovs_at(memory, iovs, iovs_len, offset, |buf, read_offset| { -+ handle.read_at_blocking(buf, read_offset) -+ }) -+ .map(Some); -+ } -+ -+ if !handle.has_blocking_seek() || !handle.has_blocking_read() { -+ return Ok(None); -+ } -+ handle -+ .seek_blocking(std::io::SeekFrom::Start(offset as u64)) -+ .map_err(map_io_err)?; -+ read_file_iovs_at(memory, iovs, iovs_len, offset, |buf, _| { -+ handle.read_blocking(buf) -+ }) -+ .map(Some) -+} -+ - #[allow(clippy::await_holding_lock)] - pub(crate) fn fd_read_internal( - ctx: &mut FunctionEnvMut<'_, WasiEnv>, -@@ -157,8 +239,8 @@ pub(crate) fn fd_read_internal( - let fd_flags = fd_entry.inner.flags; - - let (bytes_read, can_update_cursor) = { -- let mut guard = inode.write(); -- match guard.deref_mut() { -+ let guard = inode.read(); -+ match guard.deref() { - Kind::File { handle, .. } => { - let Some(handle) = handle else { - tracing::warn!("fd_read: file handle is None"); -@@ -168,66 +250,126 @@ pub(crate) fn fd_read_internal( - - drop(guard); - -- let res = __asyncify_light( -- env, -- if fd_flags.contains(Fdflags::NONBLOCK) { -- Some(Duration::ZERO) -+ let positioned_read = !is_stdio; -+ let blocking_read = if !is_stdio { -+ if positioned_read { -+ let handle_guard = match handle.read() { -+ Ok(a) => a, -+ Err(_) => return Ok(Err(Errno::Fault)), -+ }; -+ match try_read_file_positioned_shared_blocking::( -+ handle_guard.as_ref(), -+ &memory, -+ iovs, -+ iovs_len, -+ offset, -+ ) { -+ Ok(Some(read)) => Some(read), -+ Ok(None) => { -+ drop(handle_guard); -+ let mut handle = match handle.write() { -+ Ok(a) => a, -+ Err(_) => return Ok(Err(Errno::Fault)), -+ }; -+ match try_read_file_blocking::( -+ handle.as_mut(), -+ &memory, -+ iovs, -+ iovs_len, -+ offset, -+ true, -+ ) { -+ Ok(read) => read, -+ Err(err) => return Ok(Err(err)), -+ } -+ } -+ Err(err) => return Ok(Err(err)), -+ } - } else { -- None -- }, -- async move { - let mut handle = match handle.write() { - Ok(a) => a, -- Err(_) => return Err(Errno::Fault), -+ Err(_) => return Ok(Err(Errno::Fault)), - }; -- if !is_stdio { -- handle -- .seek(std::io::SeekFrom::Start(offset as u64)) -- .await -- .map_err(map_io_err)?; -+ match try_read_file_blocking::( -+ handle.as_mut(), -+ &memory, -+ iovs, -+ iovs_len, -+ offset, -+ false, -+ ) { -+ Ok(read) => read, -+ Err(err) => return Ok(Err(err)), - } -+ } -+ } else { -+ None -+ }; - -- let mut total_read = 0usize; -+ let read = if let Some(read) = blocking_read { -+ read -+ } else { -+ let res = __asyncify_light( -+ env, -+ if fd_flags.contains(Fdflags::NONBLOCK) { -+ Some(Duration::ZERO) -+ } else { -+ None -+ }, -+ async move { -+ let mut handle = match handle.write() { -+ Ok(a) => a, -+ Err(_) => return Err(Errno::Fault), -+ }; -+ if !is_stdio { -+ handle -+ .seek(std::io::SeekFrom::Start(offset as u64)) -+ .await -+ .map_err(map_io_err)?; -+ } - -- let iovs_arr = -- iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; -- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -- for iovs in iovs_arr.iter() { -- let mut buf = WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi)? -- .access() -- .map_err(mem_error_to_wasi)?; -- let r = handle.read(buf.as_mut()).await.map_err(|err| { -- let err = From::::from(err); -- match err { -- Errno::Again => { -- if is_stdio { -- Errno::Badf -- } else { -- Errno::Again -+ let mut total_read = 0usize; -+ -+ let iovs_arr = -+ iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ for iovs in iovs_arr.iter() { -+ let mut buf = WasmPtr::::new(iovs.buf) -+ .slice(&memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ let r = handle.read(buf.as_mut()).await.map_err(|err| { -+ let err = From::::from(err); -+ match err { -+ Errno::Again => { -+ if is_stdio { -+ Errno::Badf -+ } else { -+ Errno::Again -+ } - } -+ a => a, - } -- a => a, -+ }); -+ let local_read = match r { -+ Ok(s) => s, -+ Err(_) if total_read > 0 => break, -+ Err(err) => return Err(err), -+ }; -+ total_read += local_read; -+ if local_read != buf.len() { -+ break; - } -- }); -- let local_read = match r { -- Ok(s) => s, -- Err(_) if total_read > 0 => break, -- Err(err) => return Err(err), -- }; -- total_read += local_read; -- if local_read != buf.len() { -- break; - } -- } -- Ok(total_read) -- }, -- ); -- let read = wasi_try_ok_ok!(res?.map_err(|err| match err { -- Errno::Timedout => Errno::Again, -- a => a, -- })); -+ Ok(total_read) -+ }, -+ ); -+ wasi_try_ok_ok!(res?.map_err(|err| match err { -+ Errno::Timedout => Errno::Again, -+ a => a, -+ })) -+ }; - (read, true) - } - Kind::Socket { socket } => { -diff --git a/lib/wasix/src/syscalls/wasi/fd_readdir.rs b/lib/wasix/src/syscalls/wasi/fd_readdir.rs -index 5672ab5..fba763a 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_readdir.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_readdir.rs -@@ -37,83 +37,105 @@ pub fn fd_readdir( - let working_dir = wasi_try_ok!(state.fs.get_fd(fd)); - let mut buf_idx = 0usize; - -- let entries: Vec<(String, Filetype, u64)> = { -- let guard = working_dir.inode.read(); -- match guard.deref() { -- Kind::Dir { path, entries, .. } => { -- trace!("reading dir {:?}", path); -- // TODO: refactor this code -- // we need to support multiple calls, -- // simple and obviously correct implementation for now: -- // maintain consistent order via lexacographic sorting -- let fs_info = wasi_try_ok!( -- wasi_try_ok!(state.fs_read_dir(path)) -- .collect::, _>>() -- .map_err(fs_error_into_wasi_err) -- ); -- let mut entry_vec = wasi_try_ok!( -- fs_info -- .into_iter() -- .map(|entry| { -- let filename = entry.file_name().to_string_lossy().to_string(); -- trace!("getting file: {:?}", filename); -- let filetype = virtual_file_type_to_wasi_file_type( -- entry.file_type().map_err(fs_error_into_wasi_err)?, -- ); -- Ok(( -- filename, filetype, 0, // TODO: inode -- )) -- }) -- .collect::, _>>() -- ); -- entry_vec.extend(entries.iter().filter(|(_, inode)| inode.is_preopened).map( -- |(name, inode)| { -- let stat = inode.stat.read().unwrap(); -- ( -- inode.name.read().unwrap().to_string(), -- stat.st_filetype, -- stat.st_ino, -- ) -- }, -- )); -- // adding . and .. special folders -- // TODO: inode -- entry_vec.push((".".to_string(), Filetype::Directory, 0)); -- entry_vec.push(("..".to_string(), Filetype::Directory, 0)); -- entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); -- entry_vec -- } -- Kind::Root { entries } => { -- trace!("reading root"); -- let sorted_entries = { -- let mut entry_vec: Vec<(String, InodeGuard)> = entries -- .iter() -- .map(|(a, b)| (a.clone(), b.clone())) -- .collect(); -- entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); -- entry_vec -- }; -- sorted_entries -- .into_iter() -- .map(|(name, inode)| { -- let stat = inode.stat.read().unwrap(); -- ( -- format!("/{}", inode.name.read().unwrap().as_ref()), -- stat.st_filetype, -- stat.st_ino, -- ) -- }) -- .collect() -- } -- Kind::File { .. } -- | Kind::Symlink { .. } -- | Kind::Buffer { .. } -- | Kind::Socket { .. } -- | Kind::PipeRx { .. } -- | Kind::PipeTx { .. } -- | Kind::DuplexPipe { .. } -- | Kind::EventNotifications { .. } -- | Kind::Epoll { .. } => return Ok(Errno::Notdir), -+ let entries = { -+ let cached = if cookie == 0 { -+ None -+ } else { -+ working_dir -+ .inner -+ .readdir_cache -+ .read() -+ .unwrap() -+ .as_ref() -+ .cloned() -+ }; -+ if let Some(entries) = cached { -+ entries -+ } else { -+ let entries = { -+ let guard = working_dir.inode.read(); -+ match guard.deref() { -+ Kind::Dir { path, entries, .. } => { -+ trace!("reading dir {:?}", path); -+ // TODO: refactor this code -+ // we need to support multiple calls, -+ // simple and obviously correct implementation for now: -+ // maintain consistent order via lexacographic sorting -+ let fs_info = wasi_try_ok!( -+ wasi_try_ok!(state.fs_read_dir(path)) -+ .collect::, _>>() -+ .map_err(fs_error_into_wasi_err) -+ ); -+ let mut entry_vec = wasi_try_ok!( -+ fs_info -+ .into_iter() -+ .map(|entry| { -+ let filename = entry.file_name().to_string_lossy().to_string(); -+ trace!("getting file: {:?}", filename); -+ let filetype = virtual_file_type_to_wasi_file_type( -+ entry.file_type().map_err(fs_error_into_wasi_err)?, -+ ); -+ Ok(( -+ filename, filetype, 0, // TODO: inode -+ )) -+ }) -+ .collect::, _>>() -+ ); -+ entry_vec.extend( -+ entries.iter().filter(|(_, inode)| inode.is_preopened).map( -+ |(name, inode)| { -+ let stat = inode.stat.read().unwrap(); -+ ( -+ inode.name.read().unwrap().to_string(), -+ stat.st_filetype, -+ stat.st_ino, -+ ) -+ }, -+ ), -+ ); -+ // adding . and .. special folders -+ // TODO: inode -+ entry_vec.push((".".to_string(), Filetype::Directory, 0)); -+ entry_vec.push(("..".to_string(), Filetype::Directory, 0)); -+ entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); -+ entry_vec -+ } -+ Kind::Root { entries } => { -+ trace!("reading root"); -+ let sorted_entries = { -+ let mut entry_vec: Vec<(String, InodeGuard)> = entries -+ .iter() -+ .map(|(a, b)| (a.clone(), b.clone())) -+ .collect(); -+ entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); -+ entry_vec -+ }; -+ sorted_entries -+ .into_iter() -+ .map(|(name, inode)| { -+ let stat = inode.stat.read().unwrap(); -+ ( -+ format!("/{}", inode.name.read().unwrap().as_ref()), -+ stat.st_filetype, -+ stat.st_ino, -+ ) -+ }) -+ .collect() -+ } -+ Kind::File { .. } -+ | Kind::Symlink { .. } -+ | Kind::Buffer { .. } -+ | Kind::Socket { .. } -+ | Kind::PipeRx { .. } -+ | Kind::PipeTx { .. } -+ | Kind::DuplexPipe { .. } -+ | Kind::EventNotifications { .. } -+ | Kind::Epoll { .. } => return Ok(Errno::Notdir), -+ } -+ }; -+ let entries = std::sync::Arc::new(entries); -+ *working_dir.inner.readdir_cache.write().unwrap() = Some(entries.clone()); -+ entries - } - }; - -diff --git a/lib/wasix/src/syscalls/wasi/fd_renumber.rs b/lib/wasix/src/syscalls/wasi/fd_renumber.rs -index 4dc700f..c7253f5 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_renumber.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_renumber.rs -@@ -54,9 +54,6 @@ pub(crate) fn fd_renumber_internal( - from: WasiFd, - to: WasiFd, - ) -> Result { -- if from == to { -- return Ok(Errno::Success); -- } - let env = ctx.data(); - let (_, mut state) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; - -@@ -68,7 +65,11 @@ pub(crate) fn fd_renumber_internal( - let mut fd_map = state.fs.fd_map.write().unwrap(); - - // Validate the source first. If `from` is invalid we must not mutate `to`. -- let fd_entry = wasi_try_ok!(fd_map.get(from).ok_or(Errno::Badf)); -+ wasi_try_ok!(fd_map.get(from).ok_or(Errno::Badf)); -+ -+ if from == to { -+ return Ok(Errno::Success); -+ } - - // Never allow renumbering over preopens. - if let Some(target_fd) = fd_map.get(to) -@@ -79,28 +80,11 @@ pub(crate) fn fd_renumber_internal( - return Ok(Errno::Notsup); - } - -- let new_fd_entry = Fd { -- inner: FdInner { -- offset: fd_entry.inner.offset.clone(), -- rights: fd_entry.inner.rights_inheriting, -- fd_flags: { -- let mut f = fd_entry.inner.fd_flags; -- f.set(Fdflagsext::CLOEXEC, false); -- f -- }, -- ..fd_entry.inner -- }, -- inode: fd_entry.inode.clone(), -- ..*fd_entry -- }; -- -- // Remove the target FD under the same lock (replaces the separate -- // close_fd call which would acquire its own lock). -- old_fd = fd_map.remove(to); -- -- if !fd_map.insert(true, to, new_fd_entry) { -- panic!("Internal error: expected FD {to} to be free after closing in fd_renumber"); -- } -+ // This is a move, not dup-to: source ownership remains accounted for -+ // and the source slot ceases to exist in the same critical section. -+ old_fd = fd_map -+ .renumber_deferred(from, to) -+ .expect("source was validated while holding the fd-map write lock"); - } - // Flush and drop the old FD outside the lock. The flush is best-effort: - // failures are intentionally ignored so fd_renumber result depends only on -@@ -114,6 +98,9 @@ pub(crate) fn fd_renumber_internal( - _ => None, - } - }); -+ if let Some(old_fd) = old_fd.as_ref() { -+ old_fd.release_descriptor(); -+ } - drop(old_fd); - - if let Some(file) = flush_target { -diff --git a/lib/wasix/src/syscalls/wasi/fd_seek.rs b/lib/wasix/src/syscalls/wasi/fd_seek.rs -index 35fdf7a..38b3cd5 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_seek.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_seek.rs -@@ -1,4 +1,5 @@ - use super::*; -+use crate::WasiFs; - use crate::syscalls::*; - - /// ### `fd_seek()` -@@ -55,7 +56,6 @@ pub(crate) fn fd_seek_internal( - ) -> Result, WasiError> { - let env = ctx.data(); - let state = env.state.clone(); -- let (memory, _) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; - let fd_entry = wasi_try_ok_ok!(state.fs.get_fd(fd)); - - if !fd_entry.inner.rights.contains(Rights::FD_SEEK) { -@@ -85,55 +85,15 @@ pub(crate) fn fd_seek_internal( - } - } - Whence::End => { -- use std::io::SeekFrom; -- let mut guard = fd_entry.inode.write(); -- let deref_mut = guard.deref_mut(); -- match deref_mut { -- Kind::File { handle, .. } => { -- // TODO: remove allow once inodes are refactored (see comments on [`WasiState`]) -- #[allow(clippy::await_holding_lock)] -- if let Some(handle) = handle { -- let handle = handle.clone(); -- let fd_offset = fd_entry.inner.offset.clone(); -- drop(guard); -- -- wasi_try_ok_ok!(__asyncify(ctx, None, async move { -- let mut handle = handle.write().unwrap(); -- let end = handle -- .seek(SeekFrom::End(offset)) -- .await -- .map_err(map_io_err)?; -- -- // Keep updating the original file description offset even if this -- // numeric FD was concurrently closed or reused. -- fd_offset.store(end, Ordering::Release); -- Ok(()) -- })?); -- } else { -- return Ok(Err(Errno::Inval)); -- } -- } -- Kind::Symlink { .. } => { -- unimplemented!("wasi::fd_seek not implemented for symlinks") -- } -- Kind::Dir { .. } -- | Kind::Root { .. } -- | Kind::Socket { .. } -- | Kind::PipeRx { .. } -- | Kind::PipeTx { .. } -- | Kind::DuplexPipe { .. } -- | Kind::EventNotifications { .. } -- | Kind::Epoll { .. } => { -- // TODO: check this -- return Ok(Err(Errno::Inval)); -- } -- Kind::Buffer { .. } => { -- // seeking buffers probably makes sense -- // FIXME: implement this -- return Ok(Err(Errno::Inval)); -- } -- } -- fd_entry.inner.offset.load(Ordering::Acquire) -+ let end = WasiFs::authoritative_fd_size(&fd_entry); -+ let new_offset = if offset >= 0 { -+ end.checked_add(offset as u64).ok_or(Errno::Overflow) -+ } else { -+ end.checked_sub(offset.unsigned_abs()).ok_or(Errno::Inval) -+ }; -+ let new_offset = wasi_try_ok_ok!(new_offset); -+ fd_entry.inner.offset.store(new_offset, Ordering::Release); -+ new_offset - } - Whence::Set => { - let offset: u64 = wasi_try_ok_ok!(u64::try_from(offset).map_err(|_| Errno::Inval)); -diff --git a/lib/wasix/src/syscalls/wasi/fd_sync.rs b/lib/wasix/src/syscalls/wasi/fd_sync.rs -index d16db5a..3e460d2 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_sync.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_sync.rs -@@ -15,55 +15,15 @@ pub fn fd_sync(mut ctx: FunctionEnvMut<'_, WasiEnv>, fd: WasiFd) -> Result { -- if let Some(handle) = handle { -- let handle = handle.clone(); -- drop(guard); -- -- // TODO: remove allow once inodes are refactored (see comments on [`WasiState`]) -- #[allow(clippy::await_holding_lock)] -- let size = { -- wasi_try_ok!(__asyncify(&mut ctx, None, async move { -- // TODO: remove allow once inodes are refactored (see comments on [`WasiState`]) -- #[allow(clippy::await_holding_lock)] -- let mut handle = handle.write().unwrap(); -- handle.flush().await.map_err(map_io_err)?; -- Ok(handle.size()) -- })?) -- }; -- -- // Update FileStat to reflect the correct current size. -- // TODO: don't lock twice - currently needed to not keep a lock on all inodes -- { -- let mut guard = inode.stat.write().unwrap(); -- guard.st_size = size; -- } -- } else { -- return Ok(Errno::Inval); -- } -- } -- Kind::Root { .. } | Kind::Dir { .. } => return Ok(Errno::Isdir), -- Kind::Buffer { .. } -- | Kind::Symlink { .. } -- | Kind::Socket { .. } -- | Kind::PipeTx { .. } -- | Kind::PipeRx { .. } -- | Kind::DuplexPipe { .. } -- | Kind::EventNotifications { .. } -- | Kind::Epoll { .. } => return Ok(Errno::Inval), -- } -+ if let Some(()) = wasi_try_ok!(state.fs.sync_blocking(fd, true)) { -+ return Ok(Errno::Success); - } -- -- Ok(Errno::Success) -+ Ok(wasi_try_ok!(__asyncify(&mut ctx, None, async move { -+ state.fs.sync(fd, true).await.map(|_| Errno::Success) -+ })?)) - } -diff --git a/lib/wasix/src/syscalls/wasi/fd_write.rs b/lib/wasix/src/syscalls/wasi/fd_write.rs -index 68e50b5..362f917 100644 ---- a/lib/wasix/src/syscalls/wasi/fd_write.rs -+++ b/lib/wasix/src/syscalls/wasi/fd_write.rs -@@ -1,4 +1,8 @@ --use std::task::Waker; -+use std::{ -+ sync::{Arc, RwLock}, -+ task::Waker, -+ time::Duration, -+}; - - use super::*; - #[cfg(feature = "journal")] -@@ -7,6 +11,7 @@ use crate::{ - utils::map_snapshot_err, - }; - use crate::{net::socket::TimeType, syscalls::*}; -+use virtual_fs::VirtualFile; - - /// ### `fd_write()` - /// Write data to the file descriptor -@@ -105,7 +110,6 @@ pub fn fd_pwrite( - )?); - - Span::current().record("nwritten", bytes_written); -- - let mut env = ctx.data(); - let memory = unsafe { env.memory_view(&ctx) }; - let nwritten_ref = nwritten.deref(&memory); -@@ -124,6 +128,605 @@ pub(crate) enum FdWriteSource<'a, M: MemorySize> { - Buffer(Cow<'a, [u8]>), - } - -+fn blocking_sync_available(handle: &(dyn VirtualFile + Send + Sync), fd_flags: Fdflags) -> bool { -+ if fd_flags.contains(Fdflags::SYNC) { -+ handle.is_sync_on_write(true) || handle.has_blocking_sync_all_to_disk() -+ } else if fd_flags.contains(Fdflags::DSYNC) { -+ handle.is_sync_on_write(false) || handle.has_blocking_sync_data_to_disk() -+ } else { -+ true -+ } -+} -+ -+fn sync_after_blocking_write( -+ handle: &mut (dyn VirtualFile + Send + Sync), -+ fd_flags: Fdflags, -+) -> Result<(), Errno> { -+ if fd_flags.contains(Fdflags::SYNC) { -+ if !handle.is_sync_on_write(true) { -+ handle.sync_all_to_disk_blocking().map_err(map_io_err)?; -+ } -+ } else if fd_flags.contains(Fdflags::DSYNC) && !handle.is_sync_on_write(false) { -+ handle.sync_data_to_disk_blocking().map_err(map_io_err)?; -+ } -+ Ok(()) -+} -+ -+fn positioned_write_needs_explicit_sync(fd_flags: Fdflags) -> bool { -+ fd_flags.intersects(Fdflags::SYNC | Fdflags::DSYNC) -+} -+ -+fn is_all_zero(buf: &[u8]) -> bool { -+ buf.iter().all(|byte| *byte == 0) -+} -+ -+fn try_write_file_positioned_shared_blocking( -+ handle: &(dyn VirtualFile + Send + Sync), -+ memory: &wasmer::MemoryView<'_>, -+ data: &FdWriteSource<'_, M>, -+ offset: u64, -+ fd_flags: Fdflags, -+) -> Result, Errno> { -+ if positioned_write_needs_explicit_sync(fd_flags) || !handle.has_blocking_write_at_shared() { -+ return Ok(None); -+ } -+ -+ let mut written = 0usize; -+ let mut write_offset = offset; -+ -+ match data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ if *iovs_len == M::ONE { -+ let iov = iovs.read(memory).map_err(mem_error_to_wasi)?; -+ let buf = WasmPtr::::new(iov.buf) -+ .slice(memory, iov.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ if buf.as_ref().is_empty() { -+ return Ok(Some(0)); -+ } -+ -+ if handle.has_blocking_write_zeroes_at_shared() && is_all_zero(buf.as_ref()) { -+ let zeroes_written = match handle -+ .write_zeroes_at_blocking_shared(buf.as_ref().len() as u64, offset) -+ { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ return Ok(Some(zeroes_written)); -+ } -+ -+ let local_written = match handle.write_at_blocking_shared(buf.as_ref(), offset) { -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ return Ok(Some(local_written)); -+ } -+ -+ if handle.has_blocking_write_vectored_at_shared() { -+ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ let mut non_empty_count = 0usize; -+ let mut repeated_iov = None; -+ let mut all_non_empty_iovs_match = true; -+ for iov in iovs_arr.iter() { -+ if iov.buf_len == M::ZERO { -+ continue; -+ } -+ non_empty_count += 1; -+ match repeated_iov { -+ Some((buf, buf_len)) if buf == iov.buf && buf_len == iov.buf_len => {} -+ Some(_) => all_non_empty_iovs_match = false, -+ None => repeated_iov = Some((iov.buf, iov.buf_len)), -+ } -+ } -+ -+ if non_empty_count == 0 { -+ return Ok(Some(0)); -+ } -+ -+ if non_empty_count > 1 { -+ if all_non_empty_iovs_match { -+ let (buf, buf_len) = repeated_iov.expect("non-empty iovec missing"); -+ let access_guard = WasmPtr::::new(buf) -+ .slice(memory, buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ if handle.has_blocking_write_zeroes_at_shared() -+ && is_all_zero(access_guard.as_ref()) -+ { -+ let buf_len = from_offset::(buf_len)?; -+ let total_len = buf_len -+ .checked_mul(non_empty_count) -+ .ok_or(Errno::Overflow)?; -+ let zeroes_written = match handle -+ .write_zeroes_at_blocking_shared(total_len as u64, offset) -+ { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ return Ok(Some(zeroes_written)); -+ } -+ -+ let repeated = std::io::IoSlice::new(access_guard.as_ref()); -+ let bufs = std::iter::repeat(repeated) -+ .take(non_empty_count) -+ .collect::>(); -+ let vectored_written = -+ match handle.write_vectored_at_blocking_shared(&bufs, offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ return Ok(Some(vectored_written)); -+ } -+ -+ let mut access_guards = Vec::with_capacity(non_empty_count); -+ for iov in iovs_arr.iter() { -+ if iov.buf_len == M::ZERO { -+ continue; -+ } -+ let buf = WasmPtr::::new(iov.buf) -+ .slice(memory, iov.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ access_guards.push(buf); -+ } -+ -+ let bufs = access_guards -+ .iter() -+ .map(|buf| std::io::IoSlice::new(buf.as_ref())) -+ .collect::>(); -+ let vectored_written = -+ match handle.write_vectored_at_blocking_shared(&bufs, offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ return Ok(Some(vectored_written)); -+ } -+ } -+ -+ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ for iovs in iovs_arr.iter() { -+ let buf = WasmPtr::::new(iovs.buf) -+ .slice(memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ let local_written = -+ match handle.write_at_blocking_shared(buf.as_ref(), write_offset) { -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ written += local_written; -+ write_offset += local_written as u64; -+ if local_written != buf.len() { -+ break; -+ } -+ } -+ } -+ FdWriteSource::Buffer(data) => { -+ while written < data.len() { -+ let local_written = -+ match handle.write_at_blocking_shared(&data[written..], write_offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ written += local_written; -+ write_offset += local_written as u64; -+ } -+ } -+ } -+ -+ Ok(Some(written)) -+} -+ -+fn try_write_file_blocking( -+ handle: &mut (dyn VirtualFile + Send + Sync), -+ memory: &wasmer::MemoryView<'_>, -+ data: &FdWriteSource<'_, M>, -+ offset: u64, -+ positioned_write: bool, -+ fd_flags: Fdflags, -+) -> Result, Errno> { -+ let has_blocking_write = if positioned_write { -+ handle.has_blocking_write_at() -+ } else { -+ handle.has_blocking_seek() && handle.has_blocking_write() -+ }; -+ -+ if !has_blocking_write || !blocking_sync_available(handle, fd_flags) { -+ return Ok(None); -+ } -+ -+ if !positioned_write { -+ handle -+ .seek_blocking(std::io::SeekFrom::Start(offset)) -+ .map_err(map_io_err)?; -+ } -+ -+ let mut written = 0usize; -+ let mut write_offset = offset; -+ -+ match data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ if *iovs_len == M::ONE { -+ let iov = iovs.read(memory).map_err(mem_error_to_wasi)?; -+ let buf = WasmPtr::::new(iov.buf) -+ .slice(memory, iov.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ if buf.as_ref().is_empty() { -+ return Ok(Some(0)); -+ } -+ -+ if positioned_write -+ && handle.has_blocking_write_zeroes_at() -+ && is_all_zero(buf.as_ref()) -+ { -+ let zeroes_written = -+ match handle.write_zeroes_at_blocking(buf.as_ref().len() as u64, offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ sync_after_blocking_write(handle, fd_flags)?; -+ return Ok(Some(zeroes_written)); -+ } -+ -+ let local_written = if positioned_write { -+ match handle.write_at_blocking(buf.as_ref(), offset) { -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ } else { -+ match handle.write_blocking(buf.as_ref()) { -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ }; -+ if local_written > 0 { -+ sync_after_blocking_write(handle, fd_flags)?; -+ } -+ return Ok(Some(local_written)); -+ } -+ -+ if positioned_write && handle.has_blocking_write_vectored_at() { -+ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ let mut non_empty_count = 0usize; -+ let mut repeated_iov = None; -+ let mut all_non_empty_iovs_match = true; -+ for iov in iovs_arr.iter() { -+ if iov.buf_len == M::ZERO { -+ continue; -+ } -+ non_empty_count += 1; -+ match repeated_iov { -+ Some((buf, buf_len)) if buf == iov.buf && buf_len == iov.buf_len => {} -+ Some(_) => all_non_empty_iovs_match = false, -+ None => repeated_iov = Some((iov.buf, iov.buf_len)), -+ } -+ } -+ -+ if non_empty_count == 0 { -+ return Ok(Some(0)); -+ } -+ -+ if non_empty_count > 1 { -+ if all_non_empty_iovs_match { -+ let (buf, buf_len) = repeated_iov.expect("non-empty iovec missing"); -+ let access_guard = WasmPtr::::new(buf) -+ .slice(memory, buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ if handle.has_blocking_write_zeroes_at() -+ && is_all_zero(access_guard.as_ref()) -+ { -+ let buf_len = from_offset::(buf_len)?; -+ let total_len = buf_len -+ .checked_mul(non_empty_count) -+ .ok_or(Errno::Overflow)?; -+ let zeroes_written = -+ match handle.write_zeroes_at_blocking(total_len as u64, offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ sync_after_blocking_write(handle, fd_flags)?; -+ return Ok(Some(zeroes_written)); -+ } -+ -+ let repeated = std::io::IoSlice::new(access_guard.as_ref()); -+ let bufs = std::iter::repeat(repeated) -+ .take(non_empty_count) -+ .collect::>(); -+ let vectored_written = -+ match handle.write_vectored_at_blocking(&bufs, offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ sync_after_blocking_write(handle, fd_flags)?; -+ return Ok(Some(vectored_written)); -+ } -+ -+ let mut access_guards = Vec::with_capacity(non_empty_count); -+ for iov in iovs_arr.iter() { -+ if iov.buf_len == M::ZERO { -+ continue; -+ } -+ let buf = WasmPtr::::new(iov.buf) -+ .slice(memory, iov.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ access_guards.push(buf); -+ } -+ -+ let bufs = access_guards -+ .iter() -+ .map(|buf| std::io::IoSlice::new(buf.as_ref())) -+ .collect::>(); -+ let vectored_written = match handle.write_vectored_at_blocking(&bufs, offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ sync_after_blocking_write(handle, fd_flags)?; -+ return Ok(Some(vectored_written)); -+ } -+ } -+ -+ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ for iovs in iovs_arr.iter() { -+ let buf = WasmPtr::::new(iovs.buf) -+ .slice(memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ let local_written = if positioned_write { -+ match handle.write_at_blocking(buf.as_ref(), write_offset) { -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ } else { -+ match handle.write_blocking(buf.as_ref()) { -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ }; -+ written += local_written; -+ write_offset += local_written as u64; -+ if local_written != buf.len() { -+ break; -+ } -+ } -+ } -+ FdWriteSource::Buffer(data) => { -+ while written < data.len() { -+ let local_written = if positioned_write { -+ match handle.write_at_blocking(&data[written..], write_offset) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ } else { -+ match handle.write_blocking(&data[written..]) { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ }; -+ written += local_written; -+ write_offset += local_written as u64; -+ } -+ } -+ } -+ -+ if written > 0 { -+ sync_after_blocking_write(handle, fd_flags)?; -+ } -+ -+ Ok(Some(written)) -+} -+ -+fn try_write_file_positioned_blocking( -+ handle: &mut (dyn VirtualFile + Send + Sync), -+ memory: &wasmer::MemoryView<'_>, -+ data: &FdWriteSource<'_, M>, -+ offset: u64, -+ fd_flags: Fdflags, -+) -> Result, Errno> { -+ try_write_file_blocking(handle, memory, data, offset, true, fd_flags) -+} -+ -+fn try_write_file_cursor_blocking( -+ handle: &mut (dyn VirtualFile + Send + Sync), -+ memory: &wasmer::MemoryView<'_>, -+ data: &FdWriteSource<'_, M>, -+ offset: u64, -+ fd_flags: Fdflags, -+) -> Result, Errno> { -+ try_write_file_blocking(handle, memory, data, offset, false, fd_flags) -+} -+ -+#[allow(clippy::too_many_arguments)] -+fn write_regular_file( -+ env: &WasiEnv, -+ fd_entry: &Fd, -+ is_stdio: bool, -+ fd_flags: Fdflags, -+ memory: &wasmer::MemoryView<'_>, -+ data: &FdWriteSource<'_, M>, -+ offset: &mut u64, -+ should_update_cursor: bool, -+ handle: Arc>>, -+) -> Result, WasiError> { -+ let positioned_write = !is_stdio && !should_update_cursor; -+ if !is_stdio && fd_flags.contains(Fdflags::APPEND) { -+ // `fdflags::append` means we need to seek to the end before writing. -+ *offset = fd_entry.inode.stat.read().unwrap().st_size; -+ fd_entry.inner.offset.store(*offset, Ordering::Release); -+ } -+ -+ let blocking_written = if !is_stdio { -+ if positioned_write { -+ let handle_guard = handle.read().unwrap(); -+ let shared_write = try_write_file_positioned_shared_blocking::( -+ handle_guard.as_ref(), -+ memory, -+ data, -+ *offset, -+ fd_flags, -+ ); -+ match shared_write { -+ Ok(Some(written)) => Some(written), -+ Ok(None) => { -+ drop(handle_guard); -+ let mut handle = handle.write().unwrap(); -+ match try_write_file_positioned_blocking::( -+ handle.as_mut(), -+ memory, -+ data, -+ *offset, -+ fd_flags, -+ ) { -+ Ok(written) => written, -+ Err(err) => return Ok(Err(err)), -+ } -+ } -+ Err(err) => return Ok(Err(err)), -+ } -+ } else { -+ let mut handle = handle.write().unwrap(); -+ match try_write_file_cursor_blocking::( -+ handle.as_mut(), -+ memory, -+ data, -+ *offset, -+ fd_flags, -+ ) { -+ Ok(written) => written, -+ Err(err) => return Ok(Err(err)), -+ } -+ } -+ } else { -+ None -+ }; -+ -+ if let Some(written) = blocking_written { -+ return Ok(Ok(written)); -+ } -+ -+ let res = __asyncify_light( -+ env, -+ if fd_entry.inner.flags.contains(Fdflags::NONBLOCK) { -+ Some(Duration::ZERO) -+ } else { -+ None -+ }, -+ async { -+ let mut handle = handle.write().unwrap(); -+ let positioned_write = !is_stdio && !should_update_cursor; -+ if !is_stdio && !positioned_write { -+ handle -+ .seek(std::io::SeekFrom::Start(*offset)) -+ .await -+ .map_err(map_io_err)?; -+ } -+ -+ let mut written = 0usize; -+ let mut write_offset = *offset; -+ -+ match data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ for iovs in iovs_arr.iter() { -+ let buf = WasmPtr::::new(iovs.buf) -+ .slice(memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi)?; -+ let local_written = if positioned_write { -+ match handle.write_at(buf.as_ref(), write_offset).await { -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ } else { -+ match handle.write(buf.as_ref()).await { -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ } -+ }; -+ written += local_written; -+ write_offset += local_written as u64; -+ if local_written != buf.len() { -+ break; -+ } -+ } -+ } -+ FdWriteSource::Buffer(data) => { -+ if positioned_write { -+ while written < data.len() { -+ let local_written = -+ match handle.write_at(&data[written..], write_offset).await { -+ Ok(0) => return Err(Errno::Io), -+ Ok(s) => s, -+ Err(_) if written > 0 => break, -+ Err(err) => return Err(map_io_err(err)), -+ }; -+ written += local_written; -+ write_offset += local_written as u64; -+ } -+ } else { -+ handle.write_all(data).await?; -+ written += data.len(); -+ } -+ } -+ } -+ -+ if is_stdio { -+ handle.flush().await.map_err(map_io_err)?; -+ } else if written > 0 { -+ if fd_flags.contains(Fdflags::SYNC) { -+ if !handle.is_sync_on_write(true) { -+ handle.sync_all_to_disk().await.map_err(map_io_err)?; -+ } -+ } else if fd_flags.contains(Fdflags::DSYNC) && !handle.is_sync_on_write(false) { -+ handle.sync_data_to_disk().await.map_err(map_io_err)?; -+ } -+ } -+ Ok(written) -+ }, -+ ); -+ let written = res?.map_err(|err| match err { -+ Errno::Timedout => Errno::Again, -+ a => a, -+ }); -+ Ok(written) -+} -+ - #[allow(clippy::await_holding_lock)] - pub(crate) fn fd_write_internal( - mut ctx: &mut FunctionEnvMut<'_, WasiEnv>, -@@ -149,352 +752,332 @@ pub(crate) fn fd_write_internal( - - let (bytes_written, is_file, can_snapshot) = { - let (mut memory, _) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; -- let mut guard = fd_entry.inode.write(); -- match guard.deref_mut() { -- Kind::File { handle, .. } => { -- if let Some(handle) = handle { -- let handle = handle.clone(); -+ if let Some(handle) = { -+ let guard = fd_entry.inode.read(); -+ match guard.deref() { -+ Kind::File { -+ handle: Some(handle), -+ .. -+ } => Some(handle.clone()), -+ Kind::File { handle: None, .. } => return Ok(Err(Errno::Inval)), -+ _ => None, -+ } -+ } { -+ let written = wasi_try_ok_ok!(write_regular_file::( -+ env, -+ &fd_entry, -+ is_stdio, -+ fd_flags, -+ &memory, -+ &data, -+ &mut offset, -+ should_update_cursor, -+ handle, -+ )?); -+ (written, true, true) -+ } else { -+ let mut guard = fd_entry.inode.write(); -+ match guard.deref_mut() { -+ Kind::File { handle, .. } => { -+ if let Some(handle) = handle { -+ let handle = handle.clone(); -+ drop(guard); -+ let written = wasi_try_ok_ok!(write_regular_file::( -+ env, -+ &fd_entry, -+ is_stdio, -+ fd_flags, -+ &memory, -+ &data, -+ &mut offset, -+ should_update_cursor, -+ handle, -+ )?); -+ (written, true, true) -+ } else { -+ return Ok(Err(Errno::Inval)); -+ } -+ } -+ Kind::Socket { socket } => { -+ let socket = socket.clone(); - drop(guard); - -- let res = __asyncify_light( -- env, -- if fd_entry.inner.flags.contains(Fdflags::NONBLOCK) { -- Some(Duration::ZERO) -- } else { -- None -- }, -- async { -- let mut handle = handle.write().unwrap(); -- if !is_stdio { -- if fd_entry.inner.flags.contains(Fdflags::APPEND) { -- // `fdflags::append` means we need to seek to the end before writing. -- offset = fd_entry.inode.stat.read().unwrap().st_size; -- fd_entry.inner.offset.store(offset, Ordering::Release); -- } -+ let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); -+ let timeout = socket -+ .opt_time(TimeType::WriteTimeout) -+ .ok() -+ .flatten() -+ .unwrap_or(Duration::from_secs(30)); - -- handle -- .seek(std::io::SeekFrom::Start(offset)) -- .await -- .map_err(map_io_err)?; -- } -+ let tasks = env.tasks().clone(); - -- let mut written = 0usize; -+ let res = __asyncify_light(env, None, async { -+ let mut sent = 0usize; - -- match &data { -- FdWriteSource::Iovs { iovs, iovs_len } => { -- let iovs_arr = iovs -- .slice(&memory, *iovs_len) -+ match &data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ let iovs_arr = iovs -+ .slice(&memory, *iovs_len) -+ .map_err(mem_error_to_wasi)?; -+ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ for iovs in iovs_arr.iter() { -+ let buf = WasmPtr::::new(iovs.buf) -+ .slice(&memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() - .map_err(mem_error_to_wasi)?; -- let iovs_arr = -- iovs_arr.access().map_err(mem_error_to_wasi)?; -- for iovs in iovs_arr.iter() { -- let buf = WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi)? -- .access() -- .map_err(mem_error_to_wasi)?; -- let local_written = -- match handle.write(buf.as_ref()).await { -- Ok(s) => s, -- Err(_) if written > 0 => break, -- Err(err) => return Err(map_io_err(err)), -- }; -- written += local_written; -- if local_written != buf.len() { -- break; -- } -+ let local_sent = socket -+ .send( -+ tasks.deref(), -+ buf.as_ref(), -+ Some(timeout), -+ nonblocking, -+ ) -+ .await?; -+ sent += local_sent; -+ if local_sent != buf.len() { -+ break; - } - } -- FdWriteSource::Buffer(data) => { -- handle.write_all(data).await?; -- written += data.len(); -- } -- } -- -- if is_stdio { -- handle.flush().await.map_err(map_io_err)?; - } -- Ok(written) -- }, -- ); -- let written = wasi_try_ok_ok!(res?.map_err(|err| match err { -- Errno::Timedout => Errno::Again, -- a => a, -- })); -- -- (written, true, true) -- } else { -- return Ok(Err(Errno::Inval)); -- } -- } -- Kind::Socket { socket } => { -- let socket = socket.clone(); -- drop(guard); -- -- let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); -- let timeout = socket -- .opt_time(TimeType::WriteTimeout) -- .ok() -- .flatten() -- .unwrap_or(Duration::from_secs(30)); -- -- let tasks = env.tasks().clone(); -- -- let res = __asyncify_light(env, None, async { -- let mut sent = 0usize; -- -- match &data { -- FdWriteSource::Iovs { iovs, iovs_len } => { -- let iovs_arr = -- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi)?; -- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -- for iovs in iovs_arr.iter() { -- let buf = WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi)? -- .access() -- .map_err(mem_error_to_wasi)?; -- let local_sent = socket -+ FdWriteSource::Buffer(data) => { -+ sent += socket - .send( - tasks.deref(), -- buf.as_ref(), -+ data.as_ref(), - Some(timeout), - nonblocking, - ) - .await?; -- sent += local_sent; -- if local_sent != buf.len() { -- break; -- } - } - } -- FdWriteSource::Buffer(data) => { -- sent += socket -- .send(tasks.deref(), data.as_ref(), Some(timeout), nonblocking) -- .await?; -- } -- } -- Ok(sent) -- }); -- let written = wasi_try_ok_ok!(res?); -- (written, false, false) -- } -- Kind::PipeRx { .. } => { -- return Ok(Err(Errno::Badf)); -- } -- Kind::PipeTx { tx } => { -- let mut written = 0usize; -- -- match &data { -- FdWriteSource::Iovs { iovs, iovs_len } => { -- let mut raise_sigpipe = false; -- let iovs_arr = wasi_try_ok_ok!( -- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) -- ); -- let iovs_arr = -- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -- for iovs in iovs_arr.iter() { -- let buf = wasi_try_ok_ok!( -- WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi) -+ Ok(sent) -+ }); -+ let written = wasi_try_ok_ok!(res?); -+ (written, false, false) -+ } -+ Kind::PipeRx { .. } => { -+ return Ok(Err(Errno::Badf)); -+ } -+ Kind::PipeTx { tx } => { -+ let mut written = 0usize; -+ -+ match &data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ let mut raise_sigpipe = false; -+ let iovs_arr = wasi_try_ok_ok!( -+ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) - ); -- let buf = wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); -- let write_result = std::io::Write::write(tx, buf.as_ref()); -- let local_written = match write_result { -- Ok(w) => w, -- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -- // Need to do this to avoid double borrow on ctx with iovs_arr -- raise_sigpipe = true; -+ let iovs_arr = -+ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -+ for iovs in iovs_arr.iter() { -+ let buf = wasi_try_ok_ok!( -+ WasmPtr::::new(iovs.buf) -+ .slice(&memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi) -+ ); -+ let buf = -+ wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); -+ let write_result = std::io::Write::write(tx, buf.as_ref()); -+ let local_written = match write_result { -+ Ok(w) => w, -+ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -+ // Need to do this to avoid double borrow on ctx with iovs_arr -+ raise_sigpipe = true; -+ break; -+ } -+ Err(e) => return Ok(Err(map_io_err(e))), -+ }; -+ -+ written += local_written; -+ if local_written != buf.len() { - break; - } -- Err(e) => return Ok(Err(map_io_err(e))), -- }; -- -- written += local_written; -- if local_written != buf.len() { -- break; - } -- } - -- drop(iovs_arr); -+ drop(iovs_arr); - -- if raise_sigpipe { -- env.process.signal_process(Signal::Sigpipe); -- wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); -- return Ok(Err(Errno::Pipe)); -- } -- } -- FdWriteSource::Buffer(data) => { -- match std::io::Write::write_all(tx, data) { -- Ok(()) => (), -- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -+ if raise_sigpipe { - env.process.signal_process(Signal::Sigpipe); - wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); - return Ok(Err(Errno::Pipe)); - } -- Err(e) => return Ok(Err(map_io_err(e))), -- }; -- written += data.len(); -+ } -+ FdWriteSource::Buffer(data) => { -+ match std::io::Write::write_all(tx, data) { -+ Ok(()) => (), -+ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -+ env.process.signal_process(Signal::Sigpipe); -+ wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); -+ return Ok(Err(Errno::Pipe)); -+ } -+ Err(e) => return Ok(Err(map_io_err(e))), -+ }; -+ written += data.len(); -+ } - } -+ -+ (written, false, true) - } -+ Kind::DuplexPipe { pipe } => { -+ let mut written = 0usize; - -- (written, false, true) -- } -- Kind::DuplexPipe { pipe } => { -- let mut written = 0usize; -- -- match &data { -- FdWriteSource::Iovs { iovs, iovs_len } => { -- let mut raise_sigpipe = false; -- let iovs_arr = wasi_try_ok_ok!( -- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) -- ); -- let iovs_arr = -- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -- for iovs in iovs_arr.iter() { -- let buf = wasi_try_ok_ok!( -- WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi) -+ match &data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ let mut raise_sigpipe = false; -+ let iovs_arr = wasi_try_ok_ok!( -+ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) - ); -- let buf = wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); -- let write_result = std::io::Write::write(pipe, buf.as_ref()); -- let local_written = match write_result { -- Ok(w) => w, -- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -- // Need to do this to avoid double borrow on ctx with iovs_arr -- raise_sigpipe = true; -+ let iovs_arr = -+ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -+ for iovs in iovs_arr.iter() { -+ let buf = wasi_try_ok_ok!( -+ WasmPtr::::new(iovs.buf) -+ .slice(&memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi) -+ ); -+ let buf = -+ wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); -+ let write_result = std::io::Write::write(pipe, buf.as_ref()); -+ let local_written = match write_result { -+ Ok(w) => w, -+ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -+ // Need to do this to avoid double borrow on ctx with iovs_arr -+ raise_sigpipe = true; -+ break; -+ } -+ Err(e) => return Ok(Err(map_io_err(e))), -+ }; -+ -+ written += local_written; -+ if local_written != buf.len() { - break; - } -- Err(e) => return Ok(Err(map_io_err(e))), -- }; -- -- written += local_written; -- if local_written != buf.len() { -- break; - } -- } - -- drop(iovs_arr); -+ drop(iovs_arr); - -- if raise_sigpipe { -- env.process.signal_process(Signal::Sigpipe); -- wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); -- return Ok(Err(Errno::Pipe)); -- } -- } -- FdWriteSource::Buffer(data) => { -- match std::io::Write::write_all(pipe, data) { -- Ok(()) => (), -- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -+ if raise_sigpipe { - env.process.signal_process(Signal::Sigpipe); - wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); - return Ok(Err(Errno::Pipe)); - } -- Err(e) => return Ok(Err(map_io_err(e))), -- }; -- written += data.len(); -+ } -+ FdWriteSource::Buffer(data) => { -+ match std::io::Write::write_all(pipe, data) { -+ Ok(()) => (), -+ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { -+ env.process.signal_process(Signal::Sigpipe); -+ wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); -+ return Ok(Err(Errno::Pipe)); -+ } -+ Err(e) => return Ok(Err(map_io_err(e))), -+ }; -+ written += data.len(); -+ } - } -+ -+ (written, false, true) -+ } -+ Kind::Dir { .. } | Kind::Root { .. } => { -+ // TODO: verify -+ return Ok(Err(Errno::Isdir)); - } -+ Kind::EventNotifications { inner } => { -+ let mut written = 0usize; - -- (written, false, true) -- } -- Kind::Dir { .. } | Kind::Root { .. } => { -- // TODO: verify -- return Ok(Err(Errno::Isdir)); -- } -- Kind::EventNotifications { inner } => { -- let mut written = 0usize; -- -- match &data { -- FdWriteSource::Iovs { iovs, iovs_len } => { -- let iovs_arr = wasi_try_ok_ok!( -- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) -- ); -- let iovs_arr = -- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -- for iovs in iovs_arr.iter() { -- let buf_len: usize = wasi_try_ok_ok!( -- iovs.buf_len.try_into().map_err(|_| Errno::Inval) -+ match &data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ let iovs_arr = wasi_try_ok_ok!( -+ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) - ); -- let will_be_written = buf_len; -+ let iovs_arr = -+ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -+ for iovs in iovs_arr.iter() { -+ let buf_len: usize = wasi_try_ok_ok!( -+ iovs.buf_len.try_into().map_err(|_| Errno::Inval) -+ ); -+ let will_be_written = buf_len; - -- let val_cnt = buf_len / std::mem::size_of::(); -- let val_cnt: M::Offset = -- wasi_try_ok_ok!(val_cnt.try_into().map_err(|_| Errno::Inval)); -+ let val_cnt = buf_len / std::mem::size_of::(); -+ let val_cnt: M::Offset = wasi_try_ok_ok!( -+ val_cnt.try_into().map_err(|_| Errno::Inval) -+ ); - -- let vals = wasi_try_ok_ok!( -- WasmPtr::::new(iovs.buf) -- .slice(&memory, val_cnt as M::Offset) -- .map_err(mem_error_to_wasi) -- ); -- let vals = -- wasi_try_ok_ok!(vals.access().map_err(mem_error_to_wasi)); -- for val in vals.iter() { -- inner.write(*val); -- } -+ let vals = wasi_try_ok_ok!( -+ WasmPtr::::new(iovs.buf) -+ .slice(&memory, val_cnt as M::Offset) -+ .map_err(mem_error_to_wasi) -+ ); -+ let vals = -+ wasi_try_ok_ok!(vals.access().map_err(mem_error_to_wasi)); -+ for val in vals.iter() { -+ inner.write(*val); -+ } - -- written += will_be_written; -+ written += will_be_written; -+ } - } -- } -- FdWriteSource::Buffer(data) => { -- let cnt = data.len() / std::mem::size_of::(); -- for n in 0..cnt { -- let start = n * std::mem::size_of::(); -- let data = [ -- data[start], -- data[start + 1], -- data[start + 2], -- data[start + 3], -- data[start + 4], -- data[start + 5], -- data[start + 6], -- data[start + 7], -- ]; -- inner.write(u64::from_ne_bytes(data)); -+ FdWriteSource::Buffer(data) => { -+ let cnt = data.len() / std::mem::size_of::(); -+ for n in 0..cnt { -+ let start = n * std::mem::size_of::(); -+ let data = [ -+ data[start], -+ data[start + 1], -+ data[start + 2], -+ data[start + 3], -+ data[start + 4], -+ data[start + 5], -+ data[start + 6], -+ data[start + 7], -+ ]; -+ inner.write(u64::from_ne_bytes(data)); -+ } - } - } -+ -+ (written, false, true) - } -+ Kind::Symlink { .. } | Kind::Epoll { .. } => return Ok(Err(Errno::Inval)), -+ Kind::Buffer { buffer } => { -+ let mut written = 0usize; - -- (written, false, true) -- } -- Kind::Symlink { .. } | Kind::Epoll { .. } => return Ok(Err(Errno::Inval)), -- Kind::Buffer { buffer } => { -- let mut written = 0usize; -- -- match &data { -- FdWriteSource::Iovs { iovs, iovs_len } => { -- let iovs_arr = wasi_try_ok_ok!( -- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) -- ); -- let iovs_arr = -- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -- for iovs in iovs_arr.iter() { -- let buf = wasi_try_ok_ok!( -- WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi) -- ); -- let buf = wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); -- let local_written = wasi_try_ok_ok!( -- std::io::Write::write(buffer, buf.as_ref()).map_err(map_io_err) -+ match &data { -+ FdWriteSource::Iovs { iovs, iovs_len } => { -+ let iovs_arr = wasi_try_ok_ok!( -+ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) - ); -- written += local_written; -- if local_written != buf.len() { -- break; -+ let iovs_arr = -+ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); -+ for iovs in iovs_arr.iter() { -+ let buf = wasi_try_ok_ok!( -+ WasmPtr::::new(iovs.buf) -+ .slice(&memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi) -+ ); -+ let buf = -+ wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); -+ let local_written = wasi_try_ok_ok!( -+ std::io::Write::write(buffer, buf.as_ref()) -+ .map_err(map_io_err) -+ ); -+ written += local_written; -+ if local_written != buf.len() { -+ break; -+ } - } - } -+ FdWriteSource::Buffer(data) => { -+ wasi_try_ok_ok!( -+ std::io::Write::write_all(buffer, data).map_err(map_io_err) -+ ); -+ written += data.len(); -+ } - } -- FdWriteSource::Buffer(data) => { -- wasi_try_ok_ok!( -- std::io::Write::write_all(buffer, data).map_err(map_io_err) -- ); -- written += data.len(); -- } -- } - -- (written, false, true) -+ (written, false, true) -+ } - } - } - }; -@@ -512,9 +1095,6 @@ pub(crate) fn fd_write_internal( - })?; - } - -- env = ctx.data(); -- memory = unsafe { env.memory_view(&ctx) }; -- - // reborrow and update the size - if !is_stdio { - let curr_offset = if is_file && should_update_cursor { -@@ -529,10 +1109,6 @@ pub(crate) fn fd_write_internal( - fd_entry.inner.offset.load(Ordering::Acquire) - }; - -- // we set the size but we don't return any errors if it fails as -- // pipes and sockets will not do anything with this -- let (mut memory, _, inodes) = -- unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; - if is_file { - let mut stat = fd_entry.inode.stat.write().unwrap(); - if should_update_cursor { -diff --git a/lib/wasix/src/syscalls/wasi/poll_oneoff.rs b/lib/wasix/src/syscalls/wasi/poll_oneoff.rs -index 5cd9781..1cc026c 100644 ---- a/lib/wasix/src/syscalls/wasi/poll_oneoff.rs -+++ b/lib/wasix/src/syscalls/wasi/poll_oneoff.rs -@@ -474,14 +474,9 @@ where - return Ok(Errno::Success); - } - -- // We use asyncify with a deep sleep to wait on new IO events -- let res = __asyncify_with_deep_sleep::, Errno>, _>( -- ctx, -- Box::pin(trigger), -- )?; -- if let AsyncifyAction::Finish(mut ctx, events) = res { -- let events = events.map(|events| events.into_iter().map(EventResult::into_event).collect()); -- process_events(&ctx, events); -- } -+ // Wait on new IO events while still processing signals. -+ let events = __asyncify(&mut ctx, None, Box::pin(trigger))?; -+ let events = events.map(|events| events.into_iter().map(EventResult::into_event).collect()); -+ process_events(&ctx, events); - Ok(Errno::Success) - } -diff --git a/lib/wasix/src/syscalls/wasix/epoll_ctl.rs b/lib/wasix/src/syscalls/wasix/epoll_ctl.rs -index 14cf332..2302ca1 100644 ---- a/lib/wasix/src/syscalls/wasix/epoll_ctl.rs -+++ b/lib/wasix/src/syscalls/wasix/epoll_ctl.rs -@@ -3,12 +3,84 @@ use wasmer_wasix_types::wasi::{EpollCtl, EpollEvent, EpollEventCtl, Subscription - use super::*; - use crate::{ - WasiInodes, -- fs::{InodeValFilePollGuard, InodeValFilePollGuardJoin}, -- os::epoll::register_epoll_handler, -+ fs::{Fd, InodeValFilePollGuard, InodeValFilePollGuardJoin}, -+ os::epoll::{EpollFd, EpollState, EpollSubState, EpollSubscriptionKey, register_epoll_handler}, - state::PollEventSet, - syscalls::*, - }; - -+fn install_subscription( -+ target: &Fd, -+ key: EpollSubscriptionKey, -+ epoll_fd: &EpollFd, -+ state: &Arc, -+ subscription: &Arc, -+) -> Result<(), Errno> { -+ let close_registration = target -+ .inner -+ .ofd -+ .register_epoll(state, subscription, key) -+ .ok_or(Errno::Badf)?; -+ subscription -+ .attach_close_registration(close_registration) -+ .map_err(|_| Errno::Badf)?; -+ -+ if let Some(join) = -+ register_epoll_handler(target, key, epoll_fd, state.clone(), subscription.clone())? -+ { -+ subscription.add_join(join).map_err(|_| Errno::Badf)?; -+ } -+ Ok(()) -+} -+ -+fn add_subscription( -+ target: &Fd, -+ key: EpollSubscriptionKey, -+ state: &Arc, -+ event: &EpollEventCtl, -+) -> Result<(), Errno> { -+ let (epoll_fd, subscription) = state.prepare_add(key, event)?; -+ match install_subscription(target, key, &epoll_fd, state, &subscription) { -+ Ok(()) => Ok(()), -+ Err(err) => { -+ state.rollback_registration(key, &subscription); -+ Err(err) -+ } -+ } -+} -+ -+fn modify_subscription( -+ target: &Fd, -+ key: EpollSubscriptionKey, -+ state: &Arc, -+ event: &EpollEventCtl, -+) -> Result<(), Errno> { -+ let (epoll_fd, subscription, old_subscription) = state.prepare_mod(key, event)?; -+ let old_epoll_fd = old_subscription.fd_meta(); -+ old_subscription.deactivate_and_detach(); -+ -+ match install_subscription(target, key, &epoll_fd, state, &subscription) { -+ Ok(()) => Ok(()), -+ Err(err) => { -+ state.rollback_registration(key, &subscription); -+ let restored = state.prepare_restore(key, old_epoll_fd.clone())?; -+ if let Err(reinstall_err) = -+ install_subscription(target, key, &old_epoll_fd, state, &restored) -+ { -+ state.rollback_registration(key, &restored); -+ tracing::warn!( -+ fd = key.fd(), -+ ?err, -+ ?reinstall_err, -+ "failed to reinstall previous epoll handler after MOD failure" -+ ); -+ return Err(reinstall_err); -+ } -+ Err(err) -+ } -+ } -+} -+ - /// ### `epoll_ctl()` - /// Modifies an epoll interest list - /// Output: -@@ -69,99 +141,176 @@ pub(crate) fn epoll_ctl_internal( - event_ctl: Option<&EpollEventCtl>, - ) -> Result, WasiError> { - let env = ctx.data(); -- let fd_entry = wasi_try_ok_ok!(env.state.fs.get_fd(epfd)); -- -- let mut inode_guard = fd_entry.inode.read(); -- match inode_guard.deref() { -- Kind::Epoll { state } => { -- let res = match op { -- EpollCtl::Add => { -- let Some(event) = event_ctl else { -- return Ok(Err(Errno::Inval)); -- }; -- let (epoll_fd, sub_state) = match state.prepare_add(fd, event) { -- Ok(v) => v, -- Err(err) => return Ok(Err(err)), -- }; -- -- match register_epoll_handler( -- &env.state, -- &epoll_fd, -- state.clone(), -- sub_state.clone(), -- ) { -- Ok(fd_guard) => { -- if let Some(fd_guard) = fd_guard { -- sub_state.add_join(fd_guard); -- } -- Ok(()) -- } -- Err(err) => { -- state.rollback_registration(fd, None); -- Err(err) -- } -- } -- } -- EpollCtl::Mod => { -- let Some(event) = event_ctl else { -- return Ok(Err(Errno::Inval)); -- }; -- let (epoll_fd, sub_state, old_subscription) = match state.prepare_mod(fd, event) -- { -- Ok(v) => v, -- Err(err) => return Ok(Err(err)), -- }; -- // Detach the previous generation before installing the new -- // handler so dropping old guards cannot remove the new one. -- old_subscription.detach_joins(); -- -- match register_epoll_handler( -- &env.state, -- &epoll_fd, -- state.clone(), -- sub_state.clone(), -- ) { -- Ok(fd_guard) => { -- if let Some(fd_guard) = fd_guard { -- sub_state.add_join(fd_guard); -- } -- Ok(()) -- } -- Err(err) => { -- state.rollback_registration(fd, Some(old_subscription.clone())); -- let old_epoll_fd = old_subscription.fd_meta(); -- match register_epoll_handler( -- &env.state, -- &old_epoll_fd, -- state.clone(), -- old_subscription.clone(), -- ) { -- Ok(fd_guard) => { -- if let Some(fd_guard) = fd_guard { -- old_subscription.add_join(fd_guard); -- } -- } -- Err(reinstall_err) => { -- // Do not leave a restored subscription without handlers. -- state.rollback_registration(fd, None); -- tracing::warn!( -- fd, -- ?err, -- ?reinstall_err, -- "failed to reinstall previous epoll handler after MOD failure" -- ); -- return Ok(Err(reinstall_err)); -- } -- } -- Err(err) -- } -- } -- } -- EpollCtl::Del => state.apply_del(fd), -- EpollCtl::Unknown => Err(Errno::Inval), -- }; -- Ok(res) -+ let epoll_entry = wasi_try_ok_ok!(env.state.fs.get_fd(epfd)); -+ let state = { -+ let inode_guard = epoll_entry.inode.read(); -+ match inode_guard.deref() { -+ Kind::Epoll { state } => state.clone(), -+ _ => return Ok(Err(Errno::Inval)), -+ } -+ }; -+ if epfd == fd { -+ return Ok(Err(Errno::Inval)); -+ } -+ -+ // Capture the target once. Numeric fd reuse after this point must not make -+ // registration attach to a different open file description. -+ let target = wasi_try_ok_ok!(env.state.fs.get_fd(fd)); -+ let key = EpollSubscriptionKey::new(fd, target.inner.ofd.id()); -+ -+ let result = state.with_ctl_transaction(|| match op { -+ EpollCtl::Add => event_ctl -+ .ok_or(Errno::Inval) -+ .and_then(|event| add_subscription(&target, key, &state, event)), -+ EpollCtl::Mod => event_ctl -+ .ok_or(Errno::Inval) -+ .and_then(|event| modify_subscription(&target, key, &state, event)), -+ EpollCtl::Del => state.apply_del(key), -+ EpollCtl::Unknown => Err(Errno::Inval), -+ }); -+ Ok(result) -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use crate::{ -+ fs::{FdInner, InodeVal, OpenFileDescription}, -+ net::socket::{InodeSocket, InodeSocketKind}, -+ os::epoll::drain_ready_events, -+ }; -+ use std::{ -+ borrow::Cow, -+ net::{Ipv4Addr, SocketAddr}, -+ sync::{Arc, Mutex, RwLock, atomic::AtomicU64}, -+ task::{Context, Poll}, -+ }; -+ use virtual_net::{ -+ InterestHandler, NetworkError, Result as NetResult, VirtualIoSource, VirtualTcpListener, -+ VirtualTcpSocket, -+ }; -+ use wasmer_wasix_types::wasi::EpollType; -+ -+ #[derive(Debug)] -+ struct FailSecondHandlerInstall { -+ calls: Arc, -+ handler: Arc>>>, -+ } -+ -+ impl VirtualIoSource for FailSecondHandlerInstall { -+ fn remove_handler(&mut self) { -+ self.handler.lock().unwrap().take(); -+ } -+ -+ fn poll_read_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { -+ Poll::Pending -+ } -+ -+ fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { -+ Poll::Pending -+ } -+ } -+ -+ impl VirtualTcpListener for FailSecondHandlerInstall { -+ fn try_accept(&mut self) -> NetResult<(Box, SocketAddr)> { -+ Err(NetworkError::WouldBlock) -+ } -+ -+ fn set_handler( -+ &mut self, -+ handler: Box, -+ ) -> NetResult<()> { -+ let call = self.calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst) + 1; -+ if call == 2 { -+ return Err(NetworkError::IOError); -+ } -+ *self.handler.lock().unwrap() = Some(handler); -+ Ok(()) - } -- _ => Ok(Err(Errno::Inval)), -+ -+ fn addr_local(&self) -> NetResult { -+ Ok(SocketAddr::from((Ipv4Addr::LOCALHOST, 0))) -+ } -+ -+ fn set_ttl(&mut self, _ttl: u8) -> NetResult<()> { -+ Ok(()) -+ } -+ -+ fn ttl(&self) -> NetResult { -+ Ok(64) -+ } -+ } -+ -+ #[test] -+ fn failed_mod_rebuilds_active_old_subscription_with_fresh_identity() { -+ let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); -+ let installed_handler = Arc::new(Mutex::new(None)); -+ let socket = InodeSocket::new(InodeSocketKind::TcpListener { -+ socket: Box::new(FailSecondHandlerInstall { -+ calls: calls.clone(), -+ handler: installed_handler.clone(), -+ }), -+ accept_timeout: None, -+ }); -+ let inodes = WasiInodes::new(); -+ let inode = inodes.add_inode_val(InodeVal { -+ stat: RwLock::new(Default::default()), -+ is_preopened: false, -+ name: RwLock::new(Cow::Borrowed("mod-rollback-listener")), -+ kind: RwLock::new(Kind::Socket { socket }), -+ }); -+ let ofd = OpenFileDescription::new(); -+ let rights = Rights::POLL_FD_READWRITE | Rights::FD_READ; -+ let target = Fd { -+ inner: FdInner { -+ rights, -+ rights_inheriting: rights, -+ flags: Fdflags::empty(), -+ offset: Arc::new(AtomicU64::new(0)), -+ ofd: ofd.clone(), -+ readdir_cache: Default::default(), -+ fd_flags: Fdflagsext::empty(), -+ }, -+ open_flags: 0, -+ inode, -+ is_stdio: false, -+ }; -+ target.acquire_descriptor(); -+ -+ let state = Arc::new(EpollState::new()); -+ let key = EpollSubscriptionKey::new(42, ofd.id()); -+ let old_event = EpollEventCtl { -+ events: EpollType::EPOLLIN, -+ ptr: 0, -+ fd: 42, -+ data1: 11, -+ data2: 0, -+ }; -+ let new_event = EpollEventCtl { -+ data1: 22, -+ ..old_event -+ }; -+ add_subscription(&target, key, &state, &old_event).unwrap(); -+ -+ assert_eq!( -+ modify_subscription(&target, key, &state, &new_event), -+ Err(Errno::Pipe) -+ ); -+ assert_eq!(calls.load(std::sync::atomic::Ordering::SeqCst), 3); -+ assert_eq!(state.subscription_count(), 1); -+ assert_eq!(ofd.epoll_registration_count(), 1); -+ -+ installed_handler -+ .lock() -+ .unwrap() -+ .as_mut() -+ .unwrap() -+ .push_interest(virtual_mio::InterestType::Error); -+ let events = drain_ready_events(&state, 8); -+ assert_eq!(events.len(), 1); -+ assert_eq!(events[0].0.data1(), 11); -+ -+ target.release_descriptor(); -+ assert_eq!(state.subscription_count(), 0); - } - } -diff --git a/lib/wasix/src/syscalls/wasix/epoll_wait.rs b/lib/wasix/src/syscalls/wasix/epoll_wait.rs -index d1b56a3..d99411f 100644 ---- a/lib/wasix/src/syscalls/wasix/epoll_wait.rs -+++ b/lib/wasix/src/syscalls/wasix/epoll_wait.rs -@@ -4,7 +4,7 @@ use super::*; - use crate::{ - WasiInodes, - fs::{InodeValFilePollGuard, InodeValFilePollGuardJoin}, -- os::epoll::{EpollFd, drain_ready_events}, -+ os::epoll::{EpollFd, wait_for_ready_events}, - state::PollEventSet, - syscalls::*, - }; -@@ -49,19 +49,7 @@ pub fn epoll_wait( - // We enter a controlled loop that will continuously poll and react to - // epoll events until something of interest needs to be returned to the - // caller or a timeout happens -- let work = { -- async move { -- // Loop until some events of interest are returned -- loop { -- let ret = drain_ready_events(&epoll_state, maxevents); -- if !ret.is_empty() { -- return Ok(ret); -- } -- -- epoll_state.wait().await; -- } -- } -- }; -+ let work = async move { Ok(wait_for_ready_events(&epoll_state, maxevents).await) }; - - // Build the trigger using the timeout - let trigger = { -@@ -128,6 +116,10 @@ pub fn epoll_wait( - wasi_try_mem!(ret_nevents.write(&memory, M::ZERO)); - Errno::Success - } -+ Err(Errno::Intr) => { -+ tracing::trace!("epoll interrupted by signal"); -+ Errno::Intr -+ } - Err(err) => { - tracing::warn!("failed to epoll during deep sleep - {}", err); - err -@@ -143,14 +135,7 @@ pub fn epoll_wait( - return Ok(process_events(&ctx, events)); - } - -- // We use asyncify with a deep sleep to wait on new IO events -- let res = __asyncify_with_deep_sleep::, Errno>, _>( -- ctx, -- Box::pin(trigger), -- )?; -- if let AsyncifyAction::Finish(mut ctx, events) = res { -- Ok(process_events(&ctx, events)) -- } else { -- Ok(Errno::Success) -- } -+ // Wait on new IO events while still processing signals. -+ let events = __asyncify(&mut ctx, None, Box::pin(trigger))?; -+ Ok(process_events(&ctx, events)) - } -diff --git a/lib/wasix/src/syscalls/wasix/fd_sync_range.rs b/lib/wasix/src/syscalls/wasix/fd_sync_range.rs -new file mode 100644 -index 0000000..eb479ab ---- /dev/null -+++ b/lib/wasix/src/syscalls/wasix/fd_sync_range.rs -@@ -0,0 +1,252 @@ -+use super::*; -+use crate::syscalls::*; -+use virtual_fs::FileWritebackFlags; -+ -+const SYNC_FILE_RANGE_WAIT_BEFORE: u32 = 1; -+const SYNC_FILE_RANGE_WRITE: u32 = 2; -+const SYNC_FILE_RANGE_WAIT_AFTER: u32 = 4; -+const SYNC_FILE_RANGE_WAIT_BEFORE_WRITE: u32 = SYNC_FILE_RANGE_WAIT_BEFORE | SYNC_FILE_RANGE_WRITE; -+const SYNC_FILE_RANGE_WAIT_BEFORE_WAIT_AFTER: u32 = -+ SYNC_FILE_RANGE_WAIT_BEFORE | SYNC_FILE_RANGE_WAIT_AFTER; -+const SYNC_FILE_RANGE_WRITE_WAIT_AFTER: u32 = SYNC_FILE_RANGE_WRITE | SYNC_FILE_RANGE_WAIT_AFTER; -+const SYNC_FILE_RANGE_VALID_FLAGS: u32 = -+ SYNC_FILE_RANGE_WAIT_BEFORE | SYNC_FILE_RANGE_WRITE | SYNC_FILE_RANGE_WAIT_AFTER; -+ -+/// ### `fd_sync_range()` -+/// -+/// Starts and/or waits for writeback of a byte range without implying crash -+/// durability. This is the WASIX counterpart of Linux `sync_file_range(2)`; -+/// callers must still use `fd_datasync` or `fd_sync` at durability boundaries. -+#[instrument( -+ level = "trace", -+ skip_all, -+ fields(%fd, %offset, %len, flags = format_args!("{flags:#x}")), -+ ret -+)] -+pub fn fd_sync_range( -+ mut ctx: FunctionEnvMut<'_, WasiEnv>, -+ fd: WasiFd, -+ offset: i64, -+ len: i64, -+ flags: u32, -+) -> Result { -+ WasiEnv::do_pending_operations(&mut ctx)?; -+ -+ match fd_sync_range_internal(&mut ctx, fd, offset, len, flags) { -+ Ok(()) => Ok(Errno::Success), -+ Err(error) => Ok(error), -+ } -+} -+ -+pub(crate) fn fd_sync_range_internal( -+ ctx: &mut FunctionEnvMut<'_, WasiEnv>, -+ fd: WasiFd, -+ offset: i64, -+ len: i64, -+ flags: u32, -+) -> Result<(), Errno> { -+ let env = ctx.data(); -+ let (_, mut state) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; -+ let fd_entry = state.fs.get_fd(fd)?; -+ let inode = fd_entry.inode; -+ -+ require_fd_sync_range_right(fd_entry.inner.rights)?; -+ -+ delegate_fd_sync_range(offset, len, flags, |offset, len, flags| { -+ let guard = inode.read(); -+ sync_range_inode_kind(guard.deref(), offset, len, flags) -+ }) -+} -+ -+fn sync_range_inode_kind( -+ kind: &Kind, -+ offset: u64, -+ len: u64, -+ flags: FileWritebackFlags, -+) -> Result<(), Errno> { -+ match kind { -+ Kind::File { -+ handle: Some(handle), -+ .. -+ } => handle -+ .read() -+ .map_err(|_| Errno::Fault)? -+ .writeback_range(offset, len, flags) -+ .map_err(fd_sync_range_io_error), -+ Kind::File { handle: None, .. } => Err(Errno::Badf), -+ Kind::Buffer { .. } => Err(Errno::Nosys), -+ Kind::PipeRx { .. } | Kind::PipeTx { .. } | Kind::DuplexPipe { .. } => Err(Errno::Spipe), -+ // The product ABI exposes range writeback only for regular -+ // VirtualFile handles. Linux may accept native directory fds, but that -+ // is outside PostgreSQL's use and is not emulated here. -+ Kind::Dir { .. } | Kind::Root { .. } => Err(Errno::Badf), -+ Kind::Socket { .. } => Err(Errno::Spipe), -+ Kind::Symlink { .. } | Kind::EventNotifications { .. } | Kind::Epoll { .. } => { -+ Err(Errno::Badf) -+ } -+ } -+} -+ -+fn require_fd_sync_range_right(rights: Rights) -> Result<(), Errno> { -+ // sync_file_range schedules or waits for writeback; it is not a -+ // durability boundary. PostgreSQL intentionally invokes it on O_RDONLY -+ // descriptors, so require the range-advice capability rather than a write -+ // or data-sync capability. -+ if rights.contains(Rights::FD_ADVISE) { -+ Ok(()) -+ } else { -+ Err(Errno::Notcapable) -+ } -+} -+ -+fn delegate_fd_sync_range( -+ offset: i64, -+ len: i64, -+ flags: u32, -+ delegate: impl FnOnce(u64, u64, FileWritebackFlags) -> Result<(), Errno>, -+) -> Result<(), Errno> { -+ if offset < 0 || len < 0 || flags & !SYNC_FILE_RANGE_VALID_FLAGS != 0 { -+ return Err(Errno::Inval); -+ } -+ -+ // Linux requires a representable signed exclusive end. len=0 is the -+ // through-EOF sentinel and therefore has no finite end to check. -+ if len > 0 { -+ offset.checked_add(len).ok_or(Errno::Inval)?; -+ } -+ -+ let flags = FileWritebackFlags::from_bits(flags).ok_or(Errno::Inval)?; -+ delegate(offset as u64, len as u64, flags) -+} -+ -+fn fd_sync_range_io_error(error: io::Error) -> Errno { -+ #[cfg(target_os = "linux")] -+ if let Some(error) = error.raw_os_error() { -+ return match error { -+ libc::EACCES => Errno::Access, -+ libc::EBADF => Errno::Badf, -+ libc::EINVAL => Errno::Inval, -+ libc::EIO => Errno::Io, -+ libc::ENOMEM => Errno::Nomem, -+ libc::ENOSPC => Errno::Nospc, -+ libc::ENOSYS => Errno::Nosys, -+ libc::EOPNOTSUPP => Errno::Notsup, -+ libc::EOVERFLOW => Errno::Overflow, -+ libc::EPERM => Errno::Perm, -+ libc::ESPIPE => Errno::Spipe, -+ _ => return map_io_err(io::Error::from_raw_os_error(error)), -+ }; -+ } -+ -+ match error.kind() { -+ io::ErrorKind::InvalidInput => Errno::Inval, -+ io::ErrorKind::Unsupported => Errno::Nosys, -+ _ => map_io_err(error), -+ } -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ -+ #[test] -+ fn fd_sync_range_maps_all_exact_flag_combinations() { -+ for raw_flags in 0..=SYNC_FILE_RANGE_VALID_FLAGS { -+ let mut observed = None; -+ delegate_fd_sync_range(17, 23, raw_flags, |offset, len, flags| { -+ observed = Some((offset, len, flags.bits())); -+ Ok(()) -+ }) -+ .unwrap(); -+ -+ assert_eq!(observed, Some((17, 23, raw_flags))); -+ } -+ } -+ -+ #[test] -+ fn fd_sync_range_rejects_negative_overflowing_and_unknown_ranges() { -+ let cases = [ -+ (-1, 0, 0), -+ (0, -1, 0), -+ (i64::MAX, 1, 0), -+ (i64::MAX - 1, 2, 0), -+ (0, 0, SYNC_FILE_RANGE_VALID_FLAGS + 1), -+ ]; -+ -+ for (offset, len, flags) in cases { -+ let mut calls = 0; -+ let result = delegate_fd_sync_range(offset, len, flags, |_, _, _| { -+ calls += 1; -+ Ok(()) -+ }); -+ assert_eq!(result, Err(Errno::Inval)); -+ assert_eq!(calls, 0); -+ } -+ } -+ -+ #[test] -+ fn fd_sync_range_preserves_zero_length_and_maximum_finite_range() { -+ let mut observed = Vec::new(); -+ for (offset, len) in [(9, 0), (i64::MAX - 1, 1)] { -+ delegate_fd_sync_range(offset, len, 0, |offset, len, flags| { -+ observed.push((offset, len, flags.bits())); -+ Ok(()) -+ }) -+ .unwrap(); -+ } -+ -+ assert_eq!(observed, vec![(9, 0, 0), (i64::MAX as u64 - 1, 1, 0)]); -+ } -+ -+ #[test] -+ fn fd_sync_range_maps_unsupported_backends_to_nosys() { -+ let result = delegate_fd_sync_range(0, 4096, SYNC_FILE_RANGE_WRITE, |_, _, _| { -+ Err(fd_sync_range_io_error(io::ErrorKind::Unsupported.into())) -+ }); -+ -+ assert_eq!(result, Err(Errno::Nosys)); -+ } -+ -+ #[test] -+ fn fd_sync_range_read_only_advice_right_is_accepted() { -+ assert_eq!( -+ require_fd_sync_range_right(Rights::FD_READ), -+ Err(Errno::Notcapable) -+ ); -+ assert_eq!( -+ require_fd_sync_range_right(Rights::FD_READ | Rights::FD_ADVISE), -+ Ok(()) -+ ); -+ } -+ -+ #[test] -+ fn fd_sync_range_directory_is_badf() { -+ let directory = Kind::Root { -+ entries: Default::default(), -+ }; -+ assert_eq!( -+ sync_range_inode_kind(&directory, 0, 0, FileWritebackFlags::empty()), -+ Err(Errno::Badf) -+ ); -+ } -+ -+ #[cfg(target_os = "linux")] -+ #[test] -+ fn fd_sync_range_preserves_linux_writeback_errnos() { -+ let cases = [ -+ (libc::EBADF, Errno::Badf), -+ (libc::EINVAL, Errno::Inval), -+ (libc::EIO, Errno::Io), -+ (libc::ENOMEM, Errno::Nomem), -+ (libc::ENOSPC, Errno::Nospc), -+ (libc::ESPIPE, Errno::Spipe), -+ ]; -+ -+ for (host_errno, expected) in cases { -+ assert_eq!( -+ fd_sync_range_io_error(io::Error::from_raw_os_error(host_errno)), -+ expected -+ ); -+ } -+ } -+} -diff --git a/lib/wasix/src/syscalls/wasix/futex_wait.rs b/lib/wasix/src/syscalls/wasix/futex_wait.rs -index c9e78a9..7154d3d 100644 ---- a/lib/wasix/src/syscalls/wasix/futex_wait.rs -+++ b/lib/wasix/src/syscalls/wasix/futex_wait.rs -@@ -1,37 +1,36 @@ --use std::task::Waker; -+use std::{task::Waker, time::Instant}; - - use super::*; - use crate::syscalls::*; - - /// Poller returns true if its triggered and false if it times out - struct FutexPoller { -- state: Arc, -+ registry: WasiFutexRegistry, - poller_idx: u64, - futex_idx: u64, -- expected: u32, - timeout: Option + Send + Sync + 'static>>>, - } - impl Future for FutexPoller { - type Output = bool; - fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll { -- let mut guard = self.state.futexs.lock().unwrap(); -- -- // If the futex itself is no longer registered then it was likely -- // woken by a wake call -- let futex = match guard.futexes.get_mut(&self.futex_idx) { -- Some(f) => f, -- None => return Poll::Ready(true), -- }; -- let waker = match futex.wakers.get_mut(&self.poller_idx) { -- Some(w) => w, -- None => return Poll::Ready(true), -- }; -- -- // Register the waker -- waker.replace(cx.waker().clone()); -- -- // Check for timeout -- drop(guard); -+ let poll_state = self.registry.with(|guard| { -+ // If the futex itself is no longer registered then it was likely -+ // woken by a wake call. -+ let Some(futex) = guard.futexes.get_mut(&self.futex_idx) else { -+ return 0; -+ }; -+ let Some(waker) = futex.wakers.get_mut(&self.poller_idx) else { -+ return 1; -+ }; -+ -+ // Register the waker. -+ waker.replace(cx.waker().clone()); -+ 2 -+ }); -+ if poll_state != 2 { -+ return Poll::Ready(true); -+ } -+ - if let Some(timeout) = self.timeout.as_mut() { - let timeout = timeout.as_mut(); - if timeout.poll(cx).is_ready() { -@@ -46,17 +45,22 @@ impl Future for FutexPoller { - } - impl Drop for FutexPoller { - fn drop(&mut self) { -- let mut guard = self.state.futexs.lock().unwrap(); -- -- let mut should_remove = false; -- if let Some(futex) = guard.futexes.get_mut(&self.futex_idx) { -- if let Some(Some(waker)) = futex.wakers.remove(&self.poller_idx) { -- waker.wake(); -+ let waker_to_wake = self.registry.with(|guard| { -+ let mut should_remove = false; -+ let mut waker_to_wake = None; -+ if let Some(futex) = guard.futexes.get_mut(&self.futex_idx) { -+ if let Some(Some(waker)) = futex.wakers.remove(&self.poller_idx) { -+ waker_to_wake = Some(waker); -+ } -+ should_remove = futex.wakers.is_empty(); - } -- should_remove = futex.wakers.is_empty(); -- } -- if should_remove { -- guard.futexes.remove(&self.futex_idx); -+ if should_remove { -+ guard.futexes.remove(&self.futex_idx); -+ } -+ waker_to_wake -+ }); -+ if let Some(waker) = waker_to_wake { -+ waker.wake(); - } - } - } -@@ -115,8 +119,8 @@ pub(super) fn futex_wait_internal( - }; - Span::current().record("timeout", format!("{timeout:?}")); - -- let state = env.state.clone(); -- let futex_idx: u64 = futex_ptr.offset().into(); -+ let futex_addr: u64 = futex_ptr.offset().into(); -+ let (registry, futex_idx) = wasi_try_ok!(WasiFutexRegistry::resolve(&env.state, futex_addr)); - Span::current().record("futex_idx", futex_idx); - - // We generate a new poller which also registers in the -@@ -125,24 +129,23 @@ pub(super) fn futex_wait_internal( - // removed whenever the wake call is invoked (which could - // be before the poller is polled). - let poller = { -- let mut guard = env.state.futexs.lock().unwrap(); -- guard.poller_seed += 1; -- let poller_idx = guard.poller_seed; -- -- // Create the timeout if one exists - let timeout = timeout.map(|timeout| env.tasks().sleep_now(timeout)); -+ let poller_idx = registry.with(|guard| { -+ guard.poller_seed += 1; -+ let poller_idx = guard.poller_seed; - -- // We insert the futex before we check the condition variable to avoid -- // certain race conditions -- let futex = guard.futexes.entry(futex_idx).or_default(); -- futex.wakers.insert(poller_idx, Default::default()); -+ // We insert the futex before we check the condition variable to avoid -+ // certain race conditions. -+ let futex = guard.futexes.entry(futex_idx).or_default(); -+ futex.wakers.insert(poller_idx, Default::default()); -+ poller_idx -+ }); - - Span::current().record("poller_idx", poller_idx); - FutexPoller { -- state: env.state.clone(), -+ registry, - poller_idx, - futex_idx, -- expected, - timeout, - } - }; -@@ -160,17 +163,16 @@ pub(super) fn futex_wait_internal( - // then the value is not set) - the poller will set it to true - wasi_try_mem_ok!(ret_woken.write(&memory, Bool::False)); - -- // We use asyncify on the poller and potentially go into deep sleep -+ // Wait on the futex while still processing signals. - tracing::trace!("wait on {futex_idx}"); -- let res = __asyncify_with_deep_sleep::(ctx, Box::pin(poller))?; -- if let AsyncifyAction::Finish(ctx, res) = res { -- let mut env = ctx.data(); -- let memory = unsafe { env.memory_view(&ctx) }; -- if res { -- wasi_try_mem_ok!(ret_woken.write(&memory, Bool::True)); -- } else { -- wasi_try_mem_ok!(ret_woken.write(&memory, Bool::False)); -- } -+ let res = __asyncify(&mut ctx, None, async move { Ok(poller.await) })?; -+ let res = wasi_try_ok!(res); -+ let env = ctx.data(); -+ let memory = unsafe { env.memory_view(&ctx) }; -+ if res { -+ wasi_try_mem_ok!(ret_woken.write(&memory, Bool::True)); -+ } else { -+ wasi_try_mem_ok!(ret_woken.write(&memory, Bool::False)); - } - Ok(Errno::Success) - } -diff --git a/lib/wasix/src/syscalls/wasix/futex_wake.rs b/lib/wasix/src/syscalls/wasix/futex_wake.rs -index ae2020d..b63f1ba 100644 ---- a/lib/wasix/src/syscalls/wasix/futex_wake.rs -+++ b/lib/wasix/src/syscalls/wasix/futex_wake.rs -@@ -1,6 +1,19 @@ - use super::*; - use crate::syscalls::*; - -+fn remove_waiter_to_wake(futex: &mut WasiFutex) -> Option<(u64, Option)> { -+ let registered = futex -+ .wakers -+ .iter() -+ .find_map(|(id, waker)| waker.as_ref().map(|_| *id)); -+ if let Some(id) = registered { -+ return futex.wakers.remove(&id).map(|waker| (id, waker)); -+ } -+ -+ let first = futex.wakers.keys().copied().next()?; -+ futex.wakers.remove(&first).map(|waker| (first, waker)) -+} -+ - /// Wake up one thread that's blocked on futex_wait on this futex. - /// Returns true if this actually woke up such a thread, - /// or false if no thread was waiting on this futex. -@@ -18,31 +31,34 @@ pub fn futex_wake( - - let env = ctx.data(); - let memory = unsafe { env.memory_view(&ctx) }; -- let state = env.state.deref(); -- -- let pointer: u64 = futex_ptr.offset().into(); -- Span::current().record("futex_idx", pointer); -- -- let mut woken = false; -- let woken = { -- let mut guard = state.futexs.lock().unwrap(); -- if let Some(futex) = guard.futexes.get_mut(&pointer) { -- let first = futex.wakers.keys().copied().next(); -- if let Some(id) = first -- && let Some(Some(w)) = futex.wakers.remove(&id) -- { -- w.wake(); -+ -+ let futex_addr: u64 = futex_ptr.offset().into(); -+ let (registry, futex_idx) = wasi_try_ok!(WasiFutexRegistry::resolve(&env.state, futex_addr)); -+ Span::current().record("futex_idx", futex_idx); -+ let (woken, waker_to_wake) = registry.with(|guard| { -+ if let Some(futex) = guard.futexes.get_mut(&futex_idx) { -+ let mut waker_to_wake = None; -+ if let Some((_poller_idx, waker)) = remove_waiter_to_wake(futex) { -+ match waker { -+ Some(w) => { -+ waker_to_wake = Some(w); -+ } -+ None => {} -+ } - } - if futex.wakers.is_empty() { -- guard.futexes.remove(&pointer); -+ guard.futexes.remove(&futex_idx); - } -- tracing::trace!("wake(hit) on {pointer}"); -- true -+ tracing::trace!("wake(hit) on {futex_idx}"); -+ (true, waker_to_wake) - } else { -- tracing::trace!("wake(miss) on {pointer}"); -- true -+ tracing::trace!("wake(miss) on {futex_idx}"); -+ (false, None) - } -- }; -+ }); -+ if let Some(waker) = waker_to_wake { -+ waker.wake(); -+ } - Span::current().record("woken", woken); - - let woken = match woken { -@@ -53,3 +69,32 @@ pub fn futex_wake( - - Ok(Errno::Success) - } -+ -+#[cfg(test)] -+mod tests { -+ use futures::task::noop_waker; -+ -+ use super::*; -+ -+ #[test] -+ fn futex_wake_prefers_registered_waker() { -+ let mut futex = WasiFutex::default(); -+ futex.wakers.insert(1, None); -+ futex.wakers.insert(2, Some(noop_waker())); -+ -+ assert!(remove_waiter_to_wake(&mut futex).unwrap().1.is_some()); -+ assert!(futex.wakers.contains_key(&1)); -+ assert!(!futex.wakers.contains_key(&2)); -+ } -+ -+ #[test] -+ fn futex_wake_consumes_unregistered_waiter_when_no_waker_exists() { -+ let mut futex = WasiFutex::default(); -+ futex.wakers.insert(1, None); -+ futex.wakers.insert(2, None); -+ -+ assert!(remove_waiter_to_wake(&mut futex).unwrap().1.is_none()); -+ assert!(!futex.wakers.contains_key(&1)); -+ assert!(futex.wakers.contains_key(&2)); -+ } -+} -diff --git a/lib/wasix/src/syscalls/wasix/futex_wake_all.rs b/lib/wasix/src/syscalls/wasix/futex_wake_all.rs -index 8a74714..88621cc 100644 ---- a/lib/wasix/src/syscalls/wasix/futex_wake_all.rs -+++ b/lib/wasix/src/syscalls/wasix/futex_wake_all.rs -@@ -16,28 +16,29 @@ pub fn futex_wake_all( - - let env = ctx.data(); - let memory = unsafe { env.memory_view(&ctx) }; -- let state = env.state.deref(); - -- let pointer: u64 = futex_ptr.offset().into(); -- //Span::current().record("futex_idx", pointer); -- -- let mut woken = false; -- let woken = { -- let mut guard = state.futexs.lock().unwrap(); -- if let Some(futex) = guard.futexes.remove(&pointer) { -- for waker in futex.wakers { -- if let Some(waker) = waker.1 { -- waker.wake(); -+ let futex_addr: u64 = futex_ptr.offset().into(); -+ let (registry, futex_idx) = wasi_try_ok!(WasiFutexRegistry::resolve(&env.state, futex_addr)); -+ Span::current().record("futex_idx", futex_idx); -+ let (woken, wakers_to_wake) = registry.with(|guard| { -+ if let Some(futex) = guard.futexes.remove(&futex_idx) { -+ let mut wakers_to_wake = Vec::new(); -+ for (poller_idx, waker) in futex.wakers { -+ if let Some(waker) = waker { -+ wakers_to_wake.push(waker); - } - } -- tracing::trace!("wake_all (hit) on {pointer}"); -- true -+ tracing::trace!("wake_all (hit) on {futex_idx}"); -+ (true, wakers_to_wake) - } else { -- tracing::trace!("wake_all (miss) on {pointer}"); -- true -+ tracing::trace!("wake_all (miss) on {futex_idx}"); -+ (false, Vec::new()) - } -- }; -- //Span::current().record("woken", woken); -+ }); -+ for waker in wakers_to_wake { -+ waker.wake(); -+ } -+ Span::current().record("woken", woken); - - let woken = match woken { - false => Bool::False, -diff --git a/lib/wasix/src/syscalls/wasix/mem_mmap.rs b/lib/wasix/src/syscalls/wasix/mem_mmap.rs -new file mode 100644 -index 0000000..7a197de ---- /dev/null -+++ b/lib/wasix/src/syscalls/wasix/mem_mmap.rs -@@ -0,0 +1,453 @@ -+use super::*; -+use crate::state::{WasiSharedMemoryMapOrigin, WasiSharedMemoryRuntimePlacement}; -+use crate::syscalls::*; -+ -+const PROT_READ: u32 = 0x01; -+const PROT_WRITE: u32 = 0x02; -+const PROT_ALLOWED: u32 = PROT_READ | PROT_WRITE; -+ -+const MAP_SHARED: u32 = 0x01; -+const MAP_PRIVATE: u32 = 0x02; -+const MAP_TYPE: u32 = 0x0f; -+const MAP_FIXED: u32 = 0x10; -+const MAP_ANON: u32 = 0x20; -+const MAP_ALLOWED: u32 = MAP_TYPE | MAP_FIXED; -+const WASM_PAGE_SIZE: usize = 65_536; -+ -+const MS_ASYNC: u32 = 0x01; -+const MS_INVALIDATE: u32 = 0x02; -+const MS_SYNC: u32 = 0x04; -+ -+/// ### `mem_mmap()` -+/// -+/// Remaps a guest memory range to a host file-backed shared mapping. -+/// -+/// File-backed `MAP_SHARED` supports both runtime-selected and fixed mappings. -+/// A non-fixed request atomically claims new WebAssembly pages using the old -+/// size returned by memory.grow, so it cannot race a concurrent sbrk. A fixed -+/// request may replace only a fully tracked active mapping or an exact inherited -+/// exec reservation whose range, offset, and backing identity all match. -+#[instrument( -+ level = "trace", -+ skip_all, -+ fields(%addr, %len, %prot, %flags, %fd, %offset, ret_addr = field::Empty), -+ ret -+)] -+pub fn mem_mmap( -+ mut ctx: FunctionEnvMut<'_, WasiEnv>, -+ addr: M::Offset, -+ len: M::Offset, -+ prot: u32, -+ flags: u32, -+ fd: WasiFd, -+ offset: Filesize, -+ ret_addr: WasmPtr, -+) -> Result { -+ WasiEnv::do_pending_operations(&mut ctx)?; -+ -+ let requested_addr = wasi_try_ok!(from_offset::(addr)); -+ let len = wasi_try_ok!(from_offset::(len)); -+ let file_offset = wasi_try_ok!(offset.try_into().map_err(|_| Errno::Overflow)); -+ let page_size = host_page_size(); -+ let mapped_len = wasi_try_ok!(round_up_to_page(len, page_size)); -+ -+ // The sys backend currently installs a read/write host map. Enforce that -+ // exact contract at the ABI boundary too: callers may import mem_mmap -+ // directly and bypass wasix-libc's validation. -+ if len == 0 || prot != PROT_ALLOWED { -+ return Ok(Errno::Inval); -+ } -+ if file_offset & (page_size - 1) != 0 { -+ return Ok(Errno::Inval); -+ } -+ -+ if (flags & !MAP_ALLOWED) != 0 { -+ return Ok(Errno::Notsup); -+ } -+ -+ if (flags & MAP_TYPE) != MAP_SHARED || (flags & MAP_PRIVATE) != 0 { -+ return Ok(Errno::Notsup); -+ } -+ -+ if (flags & MAP_ANON) != 0 { -+ return Ok(Errno::Notsup); -+ } -+ let fixed = (flags & MAP_FIXED) != 0; -+ if fixed && requested_addr & (page_size - 1) != 0 { -+ return Ok(Errno::Inval); -+ } -+ -+ // Validate the result slot before memory.grow or a host remap. Linear -+ // memory never shrinks, so a pointer readable now remains writable at the -+ // final publication point; a bad pointer cannot strand a live mapping. -+ { -+ let memory_view = unsafe { ctx.data().memory_view(&ctx) }; -+ wasi_try_mem_ok!(ret_addr.read(&memory_view)); -+ } -+ -+ let memory = { -+ let env = ctx.data(); -+ env.inner().main_module_instance_handles().memory_clone() -+ }; -+ // Reject before fd/registry mutation and, critically, before the -+ // runtime-selected path uses memory.grow as its reservation primitive. -+ if !memory.supports_persistent_shared_fixed_remap(&ctx.as_store_ref()) { -+ return Ok(Errno::Notsup); -+ } -+ -+ let file = Arc::new(wasi_try_ok!(mappable_file(&ctx, fd, prot))); -+ let futexs = wasi_try_ok!( -+ ctx.data() -+ .state() -+ .futex_registry_for_shared_file(file.clone()) -+ ); -+ -+ let state = ctx.data().state.clone(); -+ let mapped_addr = if fixed { -+ let mapped_end = wasi_try_ok!( -+ requested_addr -+ .checked_add(mapped_len) -+ .ok_or(Errno::Overflow) -+ ); -+ let mapping = WasiSharedMemoryMapping { -+ start: requested_addr as u64, -+ len: mapped_len as u64, -+ file: file.clone(), -+ file_offset: offset, -+ futexs, -+ }; -+ wasi_try_ok!(state.install_shared_memory_mapping( -+ mapping, -+ WasiSharedMemoryMapOrigin::GuestFixed, -+ || { -+ // Validation happens before this closure. Rejected fixed -+ // requests therefore cannot grow or otherwise mutate memory. -+ memory -+ .grow_at_least(&mut ctx.as_store_mut(), mapped_end as u64) -+ .map_err(|err| { -+ tracing::warn!( -+ error = &err as &dyn std::error::Error, -+ mapped_end, -+ "failed to grow guest memory for fixed shared mapping" -+ ); -+ Errno::Nomem -+ })?; -+ // SAFETY: the process execution lease admits only this continuation for the -+ // image, every temporary MemoryView was dropped before entering the mapping -+ // registry lock, and the registry retains the backing file. WASIX truncation -+ // paths reject shrinking an inode while its shared registry is live; the sealed -+ // runtime additionally owns its HostFS mount for the mapping lifetime. -+ unsafe { -+ memory.remap_shared_file_fixed( -+ &mut ctx.as_store_mut(), -+ requested_addr, -+ mapped_len, -+ &file, -+ file_offset, -+ ) -+ } -+ .map_err(|err| { -+ tracing::warn!( -+ error = &err as &dyn std::error::Error, -+ "failed to remap guest memory to shared file backing" -+ ); -+ Errno::Inval -+ }) -+ } -+ )); -+ requested_addr -+ } else { -+ let delta_pages = wasi_try_ok!( -+ mapped_len -+ .checked_add(WASM_PAGE_SIZE - 1) -+ .ok_or(Errno::Overflow) -+ .and_then(|bytes| { -+ u32::try_from(bytes / WASM_PAGE_SIZE).map_err(|_| Errno::Overflow) -+ }) -+ ); -+ let selected = wasi_try_ok!(state.install_runtime_selected_shared_memory_mapping( -+ mapped_len as u64, -+ file.clone(), -+ offset, -+ futexs, -+ |placement| { -+ let (selected, selected_end) = match placement { -+ WasiSharedMemoryRuntimePlacement::Fresh { minimum_start } => { -+ // memory.grow is the reservation primitive: its returned old -+ // page count is unique with respect to concurrent guest growth. -+ let old_pages = -+ memory -+ .grow(&mut ctx.as_store_mut(), delta_pages) -+ .map_err(|err| { -+ tracing::warn!( -+ error = &err as &dyn std::error::Error, -+ delta_pages, -+ "failed to reserve guest pages for shared mapping" -+ ); -+ Errno::Nomem -+ })?; -+ let selected = u64::from(old_pages.0) -+ .checked_mul(WASM_PAGE_SIZE as u64) -+ .ok_or(Errno::Overflow)?; -+ let selected_end = selected -+ .checked_add(mapped_len as u64) -+ .ok_or(Errno::Overflow)?; -+ if selected < minimum_start { -+ tracing::error!( -+ selected, -+ selected_end, -+ minimum_start, -+ "WebAssembly memory end is below tracked shared mappings" -+ ); -+ return Err(Errno::Inval); -+ } -+ (selected, selected_end) -+ } -+ WasiSharedMemoryRuntimePlacement::Inherited { start } => { -+ let selected_end = start -+ .checked_add(mapped_len as u64) -+ .ok_or(Errno::Overflow)?; -+ memory -+ .grow_at_least(&mut ctx.as_store_mut(), selected_end) -+ .map_err(|err| { -+ tracing::warn!( -+ error = &err as &dyn std::error::Error, -+ selected_end, -+ "failed to cover inherited shared mapping reservation" -+ ); -+ Errno::Nomem -+ })?; -+ (start, selected_end) -+ } -+ }; -+ let selected_usize: usize = selected.try_into().map_err(|_| Errno::Overflow)?; -+ // SAFETY: memory.grow reserved this unique range for the currently leased -+ // continuation, no MemoryView is live, and the mapping registry retains the file -+ // and prevents guest-visible truncation while the mapping exists. -+ unsafe { -+ memory.remap_shared_file_fixed( -+ &mut ctx.as_store_mut(), -+ selected_usize, -+ mapped_len, -+ &file, -+ file_offset, -+ ) -+ } -+ .map_err(|err| { -+ tracing::warn!( -+ error = &err as &dyn std::error::Error, -+ selected, -+ selected_end, -+ "failed to install runtime-selected shared mapping" -+ ); -+ Errno::Inval -+ })?; -+ Ok(selected) -+ } -+ )); -+ wasi_try_ok!(selected.try_into().map_err(|_| Errno::Overflow)) -+ }; -+ -+ let memory_view = unsafe { ctx.data().memory_view(&ctx) }; -+ let addr_ret = wasi_try_ok!(to_offset::(mapped_addr)); -+ wasi_try_mem_ok!(ret_addr.write(&memory_view, addr_ret)); -+ Span::current().record("ret_addr", mapped_addr); -+ -+ Ok(Errno::Success) -+} -+ -+/// ### `mem_munmap()` -+/// -+/// Replaces a guest shared file mapping with private zero-filled memory and -+/// removes it from the fork replay registry. -+#[instrument( -+ level = "trace", -+ skip_all, -+ fields(%addr, %len), -+ ret -+)] -+pub fn mem_munmap( -+ mut ctx: FunctionEnvMut<'_, WasiEnv>, -+ addr: M::Offset, -+ len: M::Offset, -+) -> Result { -+ WasiEnv::do_pending_operations(&mut ctx)?; -+ -+ let addr = wasi_try_ok!(from_offset::(addr)); -+ let len = wasi_try_ok!(from_offset::(len)); -+ let page_size = host_page_size(); -+ let mapped_len = wasi_try_ok!(round_up_to_page(len, page_size)); -+ -+ if len == 0 { -+ return Ok(Errno::Inval); -+ } -+ -+ let memory = { -+ let env = ctx.data(); -+ env.inner().main_module_instance_handles().memory_clone() -+ }; -+ -+ let state = ctx.data().state.clone(); -+ wasi_try_ok!(state.remove_shared_memory_mapping( -+ addr as u64, -+ len as u64, -+ mapped_len as u64, -+ || { -+ // SAFETY: the process execution lease excludes another continuation, no MemoryView -+ // is live, and the mapping registry lock validates and owns the complete range until -+ // the fixed replacement succeeds. -+ unsafe { memory.remap_private_fixed(&mut ctx.as_store_mut(), addr, mapped_len) } -+ .map_err(|err| { -+ tracing::warn!( -+ error = &err as &dyn std::error::Error, -+ "failed to restore private guest memory over shared mapping" -+ ); -+ Errno::Inval -+ }) -+ } -+ )); -+ -+ Ok(Errno::Success) -+} -+ -+/// ### `mem_msync()` -+/// -+/// Synchronizes a shared guest mapping with its host file backing. -+#[instrument( -+ level = "trace", -+ skip_all, -+ fields(%addr, %len, %flags), -+ ret -+)] -+pub fn mem_msync( -+ mut ctx: FunctionEnvMut<'_, WasiEnv>, -+ addr: M::Offset, -+ len: M::Offset, -+ flags: u32, -+) -> Result { -+ WasiEnv::do_pending_operations(&mut ctx)?; -+ -+ if (flags & !(MS_ASYNC | MS_INVALIDATE | MS_SYNC)) != 0 -+ || (flags & MS_ASYNC) != 0 && (flags & MS_SYNC) != 0 -+ { -+ return Ok(Errno::Inval); -+ } -+ -+ let addr = wasi_try_ok!(from_offset::(addr)); -+ let len = wasi_try_ok!(from_offset::(len)); -+ let page_size = host_page_size(); -+ let mapped_len = wasi_try_ok!(round_up_to_page(len, page_size)); -+ -+ if len == 0 { -+ return Ok(Errno::Success); -+ } -+ -+ let host_flags = host_msync_flags(flags); -+ let memory = { -+ let env = ctx.data(); -+ env.inner().main_module_instance_handles().memory_clone() -+ }; -+ let state = ctx.data().state.clone(); -+ wasi_try_ok!(state.with_active_shared_memory_mapping( -+ addr as u64, -+ len as u64, -+ mapped_len as u64, -+ || { -+ memory -+ .msync(&ctx.as_store_ref(), addr, mapped_len, host_flags) -+ .map_err(|err| { -+ tracing::warn!( -+ error = &err as &dyn std::error::Error, -+ "failed to synchronize shared guest memory mapping" -+ ); -+ Errno::Inval -+ }) -+ }, -+ )); -+ -+ Ok(Errno::Success) -+} -+ -+fn round_up_to_page(len: usize, page_size: usize) -> Result { -+ debug_assert!(page_size.is_power_of_two()); -+ len.checked_add(page_size - 1) -+ .map(|value| value & !(page_size - 1)) -+ .ok_or(Errno::Overflow) -+} -+ -+fn host_page_size() -> usize { -+ #[cfg(not(target_os = "windows"))] -+ { -+ let page_size = unsafe { libc::sysconf(libc::_SC_PAGESIZE) }; -+ if page_size > 0 { -+ return page_size as usize; -+ } -+ } -+ -+ 65536 -+} -+ -+fn host_msync_flags(flags: u32) -> i32 { -+ #[cfg(not(target_os = "windows"))] -+ { -+ let mut host_flags = 0; -+ if (flags & MS_ASYNC) != 0 { -+ host_flags |= libc::MS_ASYNC; -+ } -+ if (flags & MS_INVALIDATE) != 0 { -+ host_flags |= libc::MS_INVALIDATE; -+ } -+ if (flags & MS_SYNC) != 0 { -+ host_flags |= libc::MS_SYNC; -+ } -+ host_flags -+ } -+ -+ #[cfg(target_os = "windows")] -+ { -+ let _ = flags; -+ 0 -+ } -+} -+ -+#[cfg(feature = "host-fs")] -+fn mappable_file( -+ ctx: &FunctionEnvMut<'_, WasiEnv>, -+ fd: WasiFd, -+ prot: u32, -+) -> Result { -+ let fd_entry = ctx.data().state().fs.get_fd(fd)?; -+ -+ if !fd_entry.inner.rights.contains(Rights::FD_READ) { -+ return Err(Errno::Access); -+ } -+ if (prot & PROT_WRITE) != 0 && !fd_entry.inner.rights.contains(Rights::FD_WRITE) { -+ return Err(Errno::Access); -+ } -+ -+ let guard = fd_entry.inode.read(); -+ let Kind::File { -+ handle: Some(handle), -+ .. -+ } = guard.deref() -+ else { -+ return Err(Errno::Badf); -+ }; -+ -+ let handle = handle.read().map_err(|_| Errno::Fault)?; -+ let host_file = handle -+ .upcast_any_ref() -+ .downcast_ref::() -+ .ok_or(Errno::Notsup)?; -+ -+ host_file.try_clone_std_file().map_err(map_io_err) -+} -+ -+#[cfg(not(feature = "host-fs"))] -+fn mappable_file( -+ _ctx: &FunctionEnvMut<'_, WasiEnv>, -+ _fd: WasiFd, -+ _prot: u32, -+) -> Result { -+ Err(Errno::Notsup) -+} -diff --git a/lib/wasix/src/syscalls/wasix/mod.rs b/lib/wasix/src/syscalls/wasix/mod.rs -index bc2bf06..19bc10f 100644 ---- a/lib/wasix/src/syscalls/wasix/mod.rs -+++ b/lib/wasix/src/syscalls/wasix/mod.rs -@@ -17,10 +17,12 @@ mod fd_dup2; - mod fd_fdflags_get; - mod fd_fdflags_set; - mod fd_pipe; -+mod fd_sync_range; - mod futex_wait; - mod futex_wake; - mod futex_wake_all; - mod getcwd; -+mod mem_mmap; - mod path_open2; - mod port_addr_add; - mod port_addr_clear; -@@ -44,6 +46,7 @@ mod proc_fork_env; - mod proc_id; - mod proc_join; - mod proc_parent; -+mod proc_rlimit_get; - mod proc_signal; - mod proc_signals_get; - mod proc_signals_sizes_get; -@@ -109,10 +112,12 @@ pub use fd_dup2::*; - pub use fd_fdflags_get::*; - pub use fd_fdflags_set::*; - pub use fd_pipe::*; -+pub use fd_sync_range::*; - pub use futex_wait::*; - pub use futex_wake::*; - pub use futex_wake_all::*; - pub use getcwd::*; -+pub use mem_mmap::*; - pub use path_open2::*; - pub use port_addr_add::*; - pub use port_addr_clear::*; -@@ -136,6 +141,7 @@ pub use proc_fork_env::*; - pub use proc_id::*; - pub use proc_join::*; - pub use proc_parent::*; -+pub use proc_rlimit_get::*; - pub use proc_signal::*; - pub use proc_signals_get::*; - pub use proc_signals_sizes_get::*; -diff --git a/lib/wasix/src/syscalls/wasix/path_open2.rs b/lib/wasix/src/syscalls/wasix/path_open2.rs -index 83db5f2..eaa083d 100644 ---- a/lib/wasix/src/syscalls/wasix/path_open2.rs -+++ b/lib/wasix/src/syscalls/wasix/path_open2.rs -@@ -46,8 +46,7 @@ pub fn path_open2( - Span::current().record("follow_symlinks", true); - } - let env = ctx.data(); -- let (memory, mut state, mut inodes) = -- unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; -+ let memory = unsafe { env.memory_view(&ctx) }; - /* TODO: find actual upper bound on name size (also this is a path, not a name :think-fish:) */ - let path_len64: u64 = path_len.into(); - if path_len64 > 1024u64 * 1024u64 { -@@ -102,8 +101,7 @@ pub fn path_open2( - } - - let env = ctx.data(); -- let (memory, mut state, mut inodes) = -- unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; -+ let memory = unsafe { env.memory_view(&ctx) }; - - Span::current().record("ret_fd", out_fd); - -@@ -149,19 +147,20 @@ pub(crate) fn path_open_internal( - } - - let mut open_flags = 0; -- // TODO: traverse rights of dirs properly -- // COMMENTED OUT: WASI isn't giving appropriate rights here when opening -- // TODO: look into this; file a bug report if this is a bug -- // -- // Maximum rights: should be the working dir rights -- // Minimum rights: whatever rights are provided -- let adjusted_rights = /*fs_rights_base &*/ working_dir_rights_inheriting; -+ let sync_on_write = fs_flags.contains(Fdflags::SYNC); -+ let data_sync_on_write = fs_flags.contains(Fdflags::DSYNC); -+ // The directory's inheriting rights are an upper bound, not an access -+ // request. Promoting a read-only open to FD_WRITE makes immutable host -+ // mounts unusable and grants the resulting descriptor rights the guest -+ // never requested. -+ let adjusted_rights = constrain_requested_rights(fs_rights_base, working_dir_rights_inheriting); -+ let adjusted_inheriting = -+ constrain_requested_rights(fs_rights_inheriting, working_dir_rights_inheriting); - let mut open_options = state.fs_new_open_options(); -+ let write_permission = adjusted_rights.contains(Rights::FD_WRITE); - - let target_rights = match maybe_inode { - Ok(_) => { -- let write_permission = adjusted_rights.contains(Rights::FD_WRITE); -- - // append, truncate, and create all require the permission to write - let (append_permission, truncate_permission, create_permission) = if write_permission { - ( -@@ -174,21 +173,25 @@ pub(crate) fn path_open_internal( - }; - - virtual_fs::OpenOptionsConfig { -- read: fs_rights_base.contains(Rights::FD_READ), -+ read: adjusted_rights.contains(Rights::FD_READ), - write: write_permission, - create_new: create_permission && o_flags.contains(Oflags::EXCL), - create: create_permission, - append: append_permission, - truncate: truncate_permission, -+ sync: sync_on_write, -+ data_sync: data_sync_on_write, - } - } - Err(_) => virtual_fs::OpenOptionsConfig { - append: fs_flags.contains(Fdflags::APPEND), -- write: fs_rights_base.contains(Rights::FD_WRITE), -- read: fs_rights_base.contains(Rights::FD_READ), -+ write: write_permission, -+ read: adjusted_rights.contains(Rights::FD_READ), - create_new: o_flags.contains(Oflags::CREATE) && o_flags.contains(Oflags::EXCL), - create: o_flags.contains(Oflags::CREATE), - truncate: o_flags.contains(Oflags::TRUNC), -+ sync: sync_on_write, -+ data_sync: data_sync_on_write, - }, - }; - -@@ -201,133 +204,249 @@ pub(crate) fn path_open_internal( - create: true, - append: true, - truncate: true, -+ sync: true, -+ data_sync: true, - }; - - let minimum_rights = target_rights.minimum_rights(&parent_rights); - - open_options.options(minimum_rights.clone()); - -+ let handle_satisfies_open = -+ |handle: &(dyn virtual_fs::VirtualFile + Send + Sync), -+ requested_config: &virtual_fs::OpenOptionsConfig| { -+ let read_ok = !requested_config.read || handle.open_read().unwrap_or(true); -+ let write_requested = requested_config.write || requested_config.append; -+ let write_ok = !write_requested || handle.open_write().unwrap_or(false); -+ read_ok && write_ok -+ }; -+ - let orig_path = path; -+ let existing_file_requested_config = virtual_fs::OpenOptionsConfig { -+ append: false, -+ ..minimum_rights.clone() -+ }; -+ -+ let record_file_open_flags = |open_flags: &mut u16| { -+ if minimum_rights.read { -+ *open_flags |= Fd::READ; -+ } -+ if minimum_rights.write { -+ *open_flags |= Fd::WRITE; -+ } -+ if minimum_rights.create { -+ *open_flags |= Fd::CREATE; -+ } -+ if minimum_rights.truncate { -+ *open_flags |= Fd::TRUNCATE; -+ } -+ }; - -+ let mut handle_reservation = None; - let inode = if let Ok(inode) = maybe_inode { - // Happy path, we found the file we're trying to open - let processing_inode = inode.clone(); -- let mut guard = processing_inode.write(); -- -- let deref_mut = guard.deref_mut(); - - if o_flags.contains(Oflags::EXCL) && o_flags.contains(Oflags::CREATE) { - return Ok(Err(Errno::Exist)); - } - -- match deref_mut { -- Kind::File { -- handle, path, fd, .. -- } => { -- if let Some(special_fd) = fd { -- // short circuit if we're dealing with a special file -- assert!(handle.is_some()); -- return Ok(Ok(*special_fd)); -- } -- if o_flags.contains(Oflags::DIRECTORY) || orig_path.ends_with('/') { -- return Ok(Err(Errno::Notdir)); -- } -- -- let open_options = open_options -- .write(minimum_rights.write) -- .create(minimum_rights.create) -- .append(false) -- .truncate(minimum_rights.truncate); -- -- if minimum_rights.read { -- open_flags |= Fd::READ; -- } -- if minimum_rights.write { -- open_flags |= Fd::WRITE; -- } -- if minimum_rights.create { -- open_flags |= Fd::CREATE; -- } -- if minimum_rights.truncate { -- open_flags |= Fd::TRUNCATE; -- } -- // Keep a stable shared handle per inode whenever possible, but reopen it -- // when this open requires stronger rights than the existing handle may have. -- let requires_stronger_handle = -- minimum_rights.write || minimum_rights.truncate || minimum_rights.create; -- if handle.is_none() { -- *handle = Some(Arc::new(std::sync::RwLock::new(wasi_try_ok_ok!( -- open_options.open(&path).map_err(fs_error_into_wasi_err) -- )))); -- } else if requires_stronger_handle { -- let mut file = handle.as_ref().unwrap().write().unwrap(); -- *file = -- wasi_try_ok_ok!(open_options.open(&path).map_err(fs_error_into_wasi_err)); -- } -+ // If the cached regular-file handle already satisfies this open, -+ // creating a new WASIX fd is the only side effect. Keep that path on a -+ // shared inode lock; create/truncate/reopen/symlink handling still -+ // falls through to the write-locked path below. -+ let cached_inode = { -+ let guard = processing_inode.read(); -+ match guard.deref() { -+ Kind::File { -+ handle: Some(handle), -+ fd, -+ .. -+ } if !minimum_rights.truncate => { -+ if let Some(special_fd) = fd { -+ return Ok(Ok(*special_fd)); -+ } -+ if o_flags.contains(Oflags::DIRECTORY) || orig_path.ends_with('/') { -+ return Ok(Err(Errno::Notdir)); -+ } - -- if let Some(handle) = handle { - let handle = handle.read().unwrap(); - if let Some(fd) = handle.get_special_fd() { -- // We clone the file descriptor so that when its closed -- // nothing bad happens - let dup_fd = wasi_try_ok_ok!(state.fs.clone_fd(fd)); - trace!( - %dup_fd - ); -- -- // some special files will return a constant FD rather than -- // actually open the file (/dev/stdin, /dev/stdout, /dev/stderr) - return Ok(Ok(dup_fd)); - } -+ -+ handle_satisfies_open(handle.as_ref(), &existing_file_requested_config).then( -+ || { -+ record_file_open_flags(&mut open_flags); -+ handle_reservation = Some(processing_inode.reserve_handle()); -+ processing_inode.clone() -+ }, -+ ) - } -+ _ => None, - } -- Kind::Buffer { .. } => unimplemented!("wasi::path_open for Buffer type files"), -- Kind::Root { .. } => { -- if !o_flags.contains(Oflags::DIRECTORY) { -- return Ok(Err(Errno::Isdir)); -+ }; -+ if let Some(inode) = cached_inode { -+ inode -+ } else { -+ let mut guard = processing_inode.write(); -+ -+ let deref_mut = guard.deref_mut(); -+ -+ match deref_mut { -+ Kind::File { -+ handle, path, fd, .. -+ } => { -+ if let Some(special_fd) = fd { -+ // short circuit if we're dealing with a special file -+ assert!(handle.is_some()); -+ return Ok(Ok(*special_fd)); -+ } -+ if o_flags.contains(Oflags::DIRECTORY) || orig_path.ends_with('/') { -+ return Ok(Err(Errno::Notdir)); -+ } -+ -+ let requested_config = open_options -+ .write(minimum_rights.write) -+ .create(minimum_rights.create) -+ .append(false) -+ .truncate(minimum_rights.truncate) -+ .get_config(); -+ -+ record_file_open_flags(&mut open_flags); -+ // Keep a stable shared handle per inode whenever possible, but reopen when -+ // the existing handle cannot satisfy the requested access. Truncate opens the -+ // exact target without O_TRUNC, locks that opened inode's shared-mapping -+ // registry, and only then shrinks it before publication. -+ let should_reopen_handle = if handle.is_none() || minimum_rights.truncate { -+ true -+ } else { -+ let file = handle.as_ref().unwrap().read().unwrap(); -+ !handle_satisfies_open(file.as_ref(), &existing_file_requested_config) -+ }; -+ if should_reopen_handle { -+ let mut requested_open_options = state.fs_new_open_options(); -+ let mut shared_open_options = state.fs_new_open_options(); -+ let manual_truncate = requested_config.truncate; -+ let open_config = virtual_fs::OpenOptionsConfig { -+ truncate: false, -+ ..requested_config.clone() -+ }; -+ let shared_config = virtual_fs::OpenOptionsConfig { -+ read: true, -+ write: true, -+ ..open_config.clone() -+ }; -+ let new_handle = shared_open_options -+ .options(shared_config) -+ .open(path.as_path()) -+ .or_else(|_| { -+ requested_open_options -+ .options(open_config) -+ .open(path.as_path()) -+ }) -+ .map_err(fs_error_into_wasi_err); -+ let mut new_handle = wasi_try_ok_ok!(new_handle); -+ if manual_truncate { -+ #[cfg(feature = "host-fs")] -+ let truncation_file = new_handle -+ .upcast_any_ref() -+ .downcast_ref::() -+ .map(|file| file.try_clone_std_file()) -+ .transpose() -+ .map_err(crate::utils::map_io_err); -+ #[cfg(feature = "host-fs")] -+ let truncation_file = wasi_try_ok_ok!(truncation_file); -+ #[cfg(feature = "host-fs")] -+ let _shared_mapping_guard = truncation_file -+ .as_ref() -+ .map(|file| state.guard_shared_mapping_file_shrink(file, 0)) -+ .transpose(); -+ #[cfg(feature = "host-fs")] -+ let _shared_mapping_guard = -+ wasi_try_ok_ok!(_shared_mapping_guard).flatten(); -+ -+ wasi_try_ok_ok!(new_handle.set_len(0).map_err(fs_error_into_wasi_err)); -+ } -+ match handle { -+ Some(handle) => { -+ *handle.write().unwrap() = new_handle; -+ } -+ None => { -+ *handle = Some(Arc::new(std::sync::RwLock::new(new_handle))); -+ } -+ } -+ } -+ -+ if let Some(handle) = handle { -+ let handle = handle.read().unwrap(); -+ if let Some(fd) = handle.get_special_fd() { -+ // We clone the file descriptor so that when its closed -+ // nothing bad happens -+ let dup_fd = wasi_try_ok_ok!(state.fs.clone_fd(fd)); -+ trace!( -+ %dup_fd -+ ); -+ -+ // some special files will return a constant FD rather than -+ // actually open the file (/dev/stdin, /dev/stdout, /dev/stderr) -+ return Ok(Ok(dup_fd)); -+ } -+ } -+ handle_reservation = Some(processing_inode.reserve_handle()); - } -- } -- Kind::Dir { .. } => { -- if fs_rights_base.contains(Rights::FD_WRITE) { -- return Ok(Err(Errno::Isdir)); -+ Kind::Buffer { .. } => unimplemented!("wasi::path_open for Buffer type files"), -+ Kind::Root { .. } => { -+ if !o_flags.contains(Oflags::DIRECTORY) { -+ return Ok(Err(Errno::Isdir)); -+ } -+ } -+ Kind::Dir { .. } => { -+ if fs_rights_base.contains(Rights::FD_WRITE) { -+ return Ok(Err(Errno::Isdir)); -+ } -+ } -+ Kind::Socket { .. } -+ | Kind::PipeTx { .. } -+ | Kind::PipeRx { .. } -+ | Kind::DuplexPipe { .. } -+ | Kind::EventNotifications { .. } -+ | Kind::Epoll { .. } => {} -+ Kind::Symlink { -+ base_po_dir, -+ path_to_symlink, -+ relative_path, -+ } => { -+ // Resolve the symlink via the existing path traversal logic and restart -+ // path_open with lookup-follow semantics for this resolved path. -+ let (resolved_base_fd, resolved_path) = if relative_path.is_absolute() { -+ (VIRTUAL_ROOT_FD, relative_path.clone()) -+ } else { -+ let mut resolved_path = path_to_symlink.clone(); -+ resolved_path.pop(); -+ resolved_path.push(relative_path); -+ (*base_po_dir, resolved_path) -+ }; -+ return path_open_internal( -+ env, -+ resolved_base_fd, -+ __WASI_LOOKUP_SYMLINK_FOLLOW, -+ &resolved_path.to_string_lossy(), -+ o_flags, -+ fs_rights_base, -+ fs_rights_inheriting, -+ fs_flags, -+ fd_flags, -+ with_fd, -+ ); - } - } -- Kind::Socket { .. } -- | Kind::PipeTx { .. } -- | Kind::PipeRx { .. } -- | Kind::DuplexPipe { .. } -- | Kind::EventNotifications { .. } -- | Kind::Epoll { .. } => {} -- Kind::Symlink { -- base_po_dir, -- path_to_symlink, -- relative_path, -- } => { -- // Resolve the symlink via the existing path traversal logic and restart -- // path_open with lookup-follow semantics for this resolved path. -- let (resolved_base_fd, resolved_path) = if relative_path.is_absolute() { -- (VIRTUAL_ROOT_FD, relative_path.clone()) -- } else { -- let mut resolved_path = path_to_symlink.clone(); -- resolved_path.pop(); -- resolved_path.push(relative_path); -- (*base_po_dir, resolved_path) -- }; -- return path_open_internal( -- env, -- resolved_base_fd, -- __WASI_LOOKUP_SYMLINK_FOLLOW, -- &resolved_path.to_string_lossy(), -- o_flags, -- fs_rights_base, -- fs_rights_inheriting, -- fs_flags, -- fd_flags, -- with_fd, -- ); -- } -+ inode - } -- inode - } else { - // less-happy path, we have to try to create the file - if o_flags.contains(Oflags::CREATE) { -@@ -417,6 +536,8 @@ pub(crate) fn path_open_internal( - ) - }; - -+ handle_reservation = Some(new_inode.reserve_handle()); -+ - { - let mut guard = parent_inode.write(); - if let Kind::Dir { entries, .. } = guard.deref_mut() { -@@ -432,12 +553,12 @@ pub(crate) fn path_open_internal( - - // TODO: check and reduce these - // TODO: ensure a mutable fd to root can never be opened -- let out_fd = wasi_try_ok_ok!(if let Some(fd) = with_fd { -+ let out_fd = if let Some(fd) = with_fd { - state - .fs - .with_fd( - adjusted_rights, -- fs_rights_inheriting, -+ adjusted_inheriting, - fs_flags, - fd_flags, - open_flags, -@@ -448,13 +569,53 @@ pub(crate) fn path_open_internal( - } else { - state.fs.create_fd( - adjusted_rights, -- fs_rights_inheriting, -+ adjusted_inheriting, - fs_flags, - fd_flags, - open_flags, - inode, - ) -- }); -+ }; -+ drop(handle_reservation); -+ let out_fd = wasi_try_ok_ok!(out_fd); - - Ok(Ok(out_fd)) - } -+ -+fn constrain_requested_rights(requested: Rights, parent_inheriting: Rights) -> Rights { -+ requested & parent_inheriting -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ -+ #[test] -+ fn read_only_open_never_inherits_parent_write_access() { -+ let parent = Rights::FD_READ | Rights::FD_WRITE | Rights::FD_SEEK; -+ let adjusted = constrain_requested_rights(Rights::FD_READ | Rights::FD_SEEK, parent); -+ -+ assert!(adjusted.contains(Rights::FD_READ)); -+ assert!(adjusted.contains(Rights::FD_SEEK)); -+ assert!(!adjusted.contains(Rights::FD_WRITE)); -+ } -+ -+ #[test] -+ fn parent_capabilities_remain_an_upper_bound() { -+ let requested = Rights::FD_READ | Rights::FD_WRITE; -+ let adjusted = constrain_requested_rights(requested, Rights::FD_READ); -+ -+ assert_eq!(adjusted, Rights::FD_READ); -+ } -+ -+ #[test] -+ fn descriptor_inheriting_rights_are_capped_independently() { -+ let parent = Rights::FD_READ | Rights::FD_SEEK; -+ let adjusted = constrain_requested_rights( -+ Rights::FD_READ | Rights::FD_WRITE | Rights::FD_SEEK, -+ parent, -+ ); -+ -+ assert_eq!(adjusted, parent); -+ } -+} -diff --git a/lib/wasix/src/syscalls/wasix/proc_exec3.rs b/lib/wasix/src/syscalls/wasix/proc_exec3.rs -index ad76a85..e01ab2e 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_exec3.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_exec3.rs -@@ -152,18 +152,17 @@ pub fn proc_exec3( - // Record the stack offsets before we give up ownership of the wasi_env - let stack_lower = wasi_env.layout.stack_lower; - let stack_upper = wasi_env.layout.stack_upper; -+ let child_tasks = wasi_env.tasks().clone(); - - // Spawn a new process with this current execution environment - let mut err_exit_code: ExitCode = Errno::Success.into(); - - let spawn_result = { - let bin_factory = Box::new(ctx.data().bin_factory.clone()); -- let tasks = wasi_env.tasks().clone(); -- - let mut config = Some(wasi_env); - - match bin_factory.try_built_in(name.clone(), Some(&ctx), &mut config) { -- Ok(a) => Ok(()), -+ Ok(handle) => Ok(handle), - Err(err) => { - if !err.is_not_found() { - error!("builtin failed - {}", err); -@@ -175,9 +174,9 @@ pub fn proc_exec3( - __asyncify_light(ctx.data(), None, async { - let ret = bin_factory.spawn(name_inner, env).await; - match ret { -- Ok(ret) => { -+ Ok(handle) => { - trace!(%child_pid, "spawned sub-process"); -- Ok(()) -+ Ok(handle) - } - Err(err) => { - err_exit_code = conv_spawn_err_to_exit_code(&err); -@@ -203,10 +202,30 @@ pub fn proc_exec3( - ctx.data_mut().vfork = Some(vfork); - return Ok(e); - } -- Ok(()) => { -+ Ok(task_handle) => { -+ // The accepted exec successor now owns the child's command -+ // lifetime. Retain the in-place vfork lease until its returned -+ // task handle is terminal; monitor-admission failure forces and -+ // waits for terminal state before releasing ownership. -+ if let Err(err) = vfork -+ .child_execution -+ .clone() -+ .monitor(task_handle, &child_tasks) -+ { -+ error!(%child_pid, "failed to monitor vfork exec successor: {err}"); -+ } -+ - // We spawned a new process - put the parent env back - ctx.data_mut().swap_inner(&mut vfork.env); - std::mem::swap(ctx.data_mut(), &mut vfork.env); -+ // The ordinary vfork/exec path returns inside the original -+ // parent TaskWasm, so its accepted lease supersedes the -+ // supplemental switch guard. If a deep-sleep continuation -+ // changed that ownership, retain the one guard that remains -+ // authoritative; a later cycle will hand off its duplicate -+ // instead of growing this vector per backend. -+ ctx.data_mut() -+ .restore_parent_execution_guard(vfork.parent_execution); - - let Some(asyncify_info) = vfork.asyncify else { - // vfork without asyncify only forks the WasiEnv, which we have restored -@@ -248,7 +267,7 @@ pub fn proc_exec3( - // on the new module - else { - // Prepare the environment -- let mut wasi_env = ctx.data().clone(); -+ let mut wasi_env = ctx.data_mut().take_for_same_thread_continuation(); - _prepare_wasi(&mut wasi_env, Some(args), envs, None); - - // Get a reference to the runtime -@@ -276,29 +295,13 @@ pub fn proc_exec3( - - match process { - Ok(mut process) => { -- // If we support deep sleeping then we switch to deep sleep mode -- let env = ctx.data(); -- -- let thread = env.thread.clone(); -- -- // The poller will wait for the process to actually finish -- let res = __asyncify_with_deep_sleep::(ctx, async move { -- process -- .wait_finished() -- .await -- .unwrap_or_else(|_| Errno::Child.into()) -- .to_native() -- })?; -- match res { -- AsyncifyAction::Finish(mut ctx, result) => { -- // When we arrive here the process should already be terminated -- let exit_code = ExitCode::from_native(result); -- ctx.data().process.terminate(exit_code); -- WasiEnv::process_signals_and_exit(&mut ctx)?; -- Err(WasiError::Exit(Errno::Unknown.into())) -- } -- AsyncifyAction::Unwind => Ok(Errno::Success), -- } -+ let result = block_on(process.wait_finished()) -+ .unwrap_or_else(|_| Errno::Child.into()) -+ .to_native(); -+ let exit_code = ExitCode::from_native(result); -+ ctx.data().process.terminate(exit_code); -+ WasiEnv::process_signals_and_exit(&mut ctx)?; -+ Err(WasiError::Exit(Errno::Unknown.into())) - } - Err(err) => { - warn!( -diff --git a/lib/wasix/src/syscalls/wasix/proc_exit2.rs b/lib/wasix/src/syscalls/wasix/proc_exit2.rs -index 5ded2f0..d1c3369 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_exit2.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_exit2.rs -@@ -40,9 +40,14 @@ pub fn proc_exit2( - ctx.data_mut().swap_inner(parent_env.as_mut()); - let mut child_env = std::mem::replace(ctx.data_mut(), *parent_env); - -- // Terminate the child process -+ // Parent execution ownership follows the restored environment. The normal -+ // return path is still owned by the original parent TaskWasm; retain the -+ // supplemental switch guard only when a deep-sleep continuation made it -+ // the last authoritative parent owner. -+ ctx.data_mut() -+ .restore_parent_execution_guard(vfork.parent_execution); - child_env.owned_handles.push(vfork.handle); -- child_env.process.terminate(code); -+ vfork.child_execution.finish(Ok(code)); - - let Some(asyncify_info) = vfork.asyncify else { - // vfork without asyncify only forks the WasiEnv, which we have restored -diff --git a/lib/wasix/src/syscalls/wasix/proc_fork.rs b/lib/wasix/src/syscalls/wasix/proc_fork.rs -index 0b282be..a1f7de4 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_fork.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_fork.rs -@@ -1,13 +1,7 @@ - use super::*; --use crate::{ -- WasiThreadHandle, WasiVForkAsyncify, capture_store_snapshot, -- os::task::OwnedTaskStatus, -- runtime::task_manager::{TaskWasm, TaskWasmRunProperties}, -- state::context_switching::ContextSwitchingEnvironment, -- syscalls::*, --}; -+use crate::{WasiVForkAsyncify, syscalls::*}; - use serde::{Deserialize, Serialize}; --use wasmer::Memory; -+use wasmer::AsStoreMut; - - #[derive(Serialize, Deserialize)] - pub(crate) struct ForkResult { -@@ -22,16 +16,11 @@ pub(crate) struct ForkResult { - #[instrument(level = "trace", skip_all, fields(pid = ctx.data().process.pid().raw()), ret)] - pub fn proc_fork( - mut ctx: FunctionEnvMut<'_, WasiEnv>, -- mut copy_memory: Bool, -+ copy_memory: Bool, - pid_ptr: WasmPtr, - ) -> Result { - WasiEnv::do_pending_operations(&mut ctx)?; - -- wasi_try_ok!(ctx.data().ensure_static_module().map_err(|_| { -- warn!("process forking not supported for dynamically linked modules"); -- Errno::Notsup -- })); -- - if let Some(context_switching_environment) = ctx.data().context_switching_environment.as_ref() - && context_switching_environment.active_context_id() - != context_switching_environment.main_context_id() -@@ -56,6 +45,11 @@ pub fn proc_fork( - wasi_try_mem_ok!(pid_ptr.write(&memory, result.pid)); - return Ok(result.ret); - } -+ -+ if copy_memory == Bool::True { -+ warn!("copied-memory fork is unsupported; use vfork followed by exec"); -+ return Ok(Errno::Notsup); -+ } - trace!(%copy_memory, "capturing"); - - if let Some(vfork) = ctx.data().vfork.as_ref() { -@@ -63,11 +57,27 @@ pub fn proc_fork( - return Ok(Errno::Notsup); - } - -- // Fork the environment which will copy all the open file handlers -- // and associate a new context but otherwise shares things like the -- // file system interface. The handle to the forked process is stored -- // in the parent process context -- let (mut child_env, mut child_handle) = match ctx.data().fork() { -+ let supports_asyncify = ctx -+ .data() -+ .inner() -+ .main_module_instance_handles() -+ .supports_asyncify_stack_rewind(); -+ if !supports_asyncify { -+ warn!("process forking requires complete Asyncify stack-rewind exports"); -+ return Ok(Errno::Notsup); -+ } -+ trace!("using Asyncify vfork continuation backend"); -+ -+ let env = ctx.data(); -+ let memory = unsafe { env.memory_view(&ctx) }; -+ -+ // Seed the child return slot before stack capture. The parent continuation -+ // overwrites it with the child pid when it rewinds. -+ wasi_try_mem_ok!(pid_ptr.write(&memory, 0)); -+ -+ // Fork the environment for a vfork continuation. The child shares the -+ // current memory until proc_exec installs a fresh process image. -+ let (mut child_env, child_handle, mut child_registration) = match ctx.data().fork_guarded() { - Ok(p) => p, - Err(err) => { - debug!("could not fork process: {err}"); -@@ -75,279 +85,92 @@ pub fn proc_fork( - return Ok(Errno::Perm); - } - }; -- let child_pid = child_env.process.pid(); -- let child_finished = child_env.process.finished.clone(); -- -- // We write a zero to the PID before we capture the stack -- // so that this is what will be returned to the child -- { -- let mut inner = ctx.data().process.lock(); -- inner.children.push(child_env.process.clone()); -- } -- let env = ctx.data(); -- let memory = unsafe { env.memory_view(&ctx) }; -- -- // Setup some properties in the child environment -- wasi_try_mem_ok!(pid_ptr.write(&memory, 0)); -- let pid = child_env.pid(); -- let tid = child_env.tid(); -- -- // Pass some offsets to the unwind function -- let pid_offset = pid_ptr.offset(); -- -- // If we are not copying the memory then we act like a `vfork` -- // instead which will pretend to be the new process for a period -- // of time until `proc_exec` is called at which point the fork -- // actually occurs -- if copy_memory == Bool::False { -- // Perform the unwind action -- return unwind::(ctx, move |mut ctx, mut memory_stack, rewind_stack| { -- // Grab all the globals and serialize them -- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) -- .serialize() -- .unwrap(); -- let store_data = Bytes::from(store_data); -- -- // We first fork the environment and replace the current environment -- // so that the process can continue to prepare for the real fork as -- // if it had actually forked -- child_env.swap_inner(ctx.data_mut()); -- std::mem::swap(ctx.data_mut(), &mut child_env); -- let previous_vfork = ctx.data_mut().vfork.replace(WasiVFork { -- asyncify: Some(WasiVForkAsyncify { -- rewind_stack: rewind_stack.clone(), -- store_data: store_data.clone(), -- is_64bit: M::is_64bit(), -- }), -- env: Box::new(child_env), -- handle: child_handle, -- }); -- assert!(previous_vfork.is_none()); // Already checked above -- -- // Carry on as if the fork had taken place (which basically means -- // it prevents to be the new process with the old one suspended) -- // Rewind the stack and carry on -- match rewind::( -- ctx, -- Some(memory_stack.freeze()), -- rewind_stack.freeze(), -- store_data, -- ForkResult { -- pid: 0, -- ret: Errno::Success, -- }, -- ) { -- Errno::Success => OnCalledAction::InvokeAgain, -- err => { -- warn!("failed - could not rewind the stack - errno={}", err); -- OnCalledAction::Trap(Box::new(WasiError::Exit(err.into()))) -- } -- } -- }); -- } -- -- // Create the thread that will back this forked process -- let state = env.state.clone(); -- let bin_factory = env.bin_factory.clone(); -- -- // Perform the unwind action -- let snapshot = capture_store_snapshot(&mut ctx.as_store_mut()); -- unwind::(ctx, move |mut ctx, mut memory_stack, rewind_stack| { -- let tasks = ctx.data().tasks().clone(); -- let span = debug_span!( -- "unwind", -- memory_stack_len = memory_stack.len(), -- rewind_stack_len = rewind_stack.len() -- ); -- let _span_guard = span.enter(); -- let memory_stack = memory_stack.freeze(); -- let rewind_stack = rewind_stack.freeze(); -- -+ unwind::(ctx, move |mut ctx, memory_stack, rewind_stack| { - // Grab all the globals and serialize them -- let store_data = snapshot.serialize().unwrap(); -+ let snapshot = match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { -+ Ok(snapshot) => snapshot, -+ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), -+ }; -+ let store_data = { snapshot.serialize().unwrap() }; - let store_data = Bytes::from(store_data); -- -- // Now we use the environment and memory references -- let runtime = child_env.runtime.clone(); -- let tasks = child_env.tasks().clone(); -- let child_memory_stack = memory_stack.clone(); -- let child_rewind_stack = rewind_stack.clone(); -- -- let env_inner = ctx.data().inner(); -- let instance_handles = env_inner.static_module_instance_handles().unwrap(); -- let module = instance_handles.module_clone(); -- let memory = instance_handles.memory_clone(); -- let spawn_type = SpawnType::CopyMemory(memory, ctx.as_store_ref()); -- -- // Spawn a new process with this current execution environment -- let signaler = Box::new(child_env.process.clone()); -- { -- let runtime = runtime.clone(); -- let tasks = tasks.clone(); -- let tasks_outer = tasks.clone(); -- let store_data = store_data.clone(); -- -- let run = move |mut props: TaskWasmRunProperties| { -- let ctx = props.ctx; -- let mut store = props.store; -- -- // Rewind the stack and carry on -- { -- trace!("rewinding child"); -- let mut ctx = ctx.env.clone().into_mut(&mut store); -- let (data, mut store) = ctx.data_and_store_mut(); -- match rewind::( -- ctx, -- Some(child_memory_stack), -- child_rewind_stack, -- store_data.clone(), -- ForkResult { -- pid: 0, -- ret: Errno::Success, -- }, -- ) { -- Errno::Success => OnCalledAction::InvokeAgain, -- err => { -- warn!( -- "wasm rewind failed - could not rewind the stack - errno={}", -- err -- ); -- return; -- } -- }; -- } -- -- // Invoke the start function -- run::(ctx, store, child_handle, None); -- }; -- -- tasks_outer -- .task_wasm( -- TaskWasm::new(Box::new(run), child_env, module, false, false) -- .with_globals(snapshot) -- .with_memory(spawn_type), -- ) -- .map_err(|err| { -- warn!( -- "failed to fork as the process could not be spawned - {}", -- err -- ); -- err -- }) -- .ok(); -+ let parent_tasks = ctx.data().tasks().clone(); -+ -+ // Publish and lease both sides before the in-place process switch. -+ // The current TaskWasm continues to own its original parent lease; -+ // these guards make both identities explicit across deep sleep and -+ // the later parent-resume transition. -+ if let Err(err) = child_registration.commit_child() { -+ warn!("failed to publish forked process: {err}"); -+ return OnCalledAction::Trap(Box::new(WasiError::Exit(Errno::Perm.into()))); -+ } -+ let child_execution = match child_env.process.acquire_execution_guard() { -+ Ok(execution) => execution, -+ Err(err) => { -+ warn!("failed to lease forked child execution: {err}"); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ return OnCalledAction::Trap(Box::new(WasiError::Exit(Errno::Perm.into()))); -+ } -+ }; -+ let parent_execution = match ctx.data().acquire_parent_execution_guard() { -+ Ok(execution) => execution, -+ Err(err) => { -+ warn!("failed to retain vfork parent execution: {err}"); -+ child_execution.finish(Ok(Errno::Canceled.into())); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ return OnCalledAction::Trap(Box::new(WasiError::Exit(Errno::Perm.into()))); -+ } - }; - -+ // Replace the current environment only after both process leases -+ // coexist. The parent environment remains local until rewind has -+ // succeeded, making failure rollback non-blocking and exact. -+ child_env.swap_inner(ctx.data_mut()); -+ std::mem::swap(ctx.data_mut(), &mut child_env); -+ -+ // Carry on as if the fork had taken place (which basically means -+ // it prevents to be the new process with the old one suspended) - // Rewind the stack and carry on - match rewind::( -- ctx, -- Some(memory_stack), -- rewind_stack, -- store_data, -+ ctx.as_mut(), -+ Some(memory_stack.freeze()), -+ rewind_stack.clone().freeze(), -+ store_data.clone(), - ForkResult { -- pid: child_pid.raw() as Pid, -+ pid: 0, - ret: Errno::Success, - }, - ) { -- Errno::Success => OnCalledAction::InvokeAgain, -+ Errno::Success => { -+ parent_execution.arm_fail_closed(); -+ let previous_vfork = ctx.data_mut().vfork.replace(WasiVFork { -+ asyncify: Some(WasiVForkAsyncify { -+ rewind_stack: rewind_stack.clone(), -+ store_data: store_data.clone(), -+ is_64bit: M::is_64bit(), -+ }), -+ env: Box::new(child_env), -+ handle: child_handle, -+ parent_execution, -+ child_execution, -+ }); -+ assert!(previous_vfork.is_none()); // Already checked above -+ child_registration.complete_child_launch(&parent_tasks); -+ OnCalledAction::InvokeAgain -+ } - err => { - warn!("failed - could not rewind the stack - errno={}", err); -+ // Restore the parent before releasing either side of the -+ // failed switch. The original parent TaskWasm still owns -+ // execution, so its supplemental guard is a safe handoff. -+ ctx.data_mut().swap_inner(&mut child_env); -+ let mut failed_child = std::mem::replace(ctx.data_mut(), child_env); -+ failed_child.owned_handles.push(child_handle); -+ child_execution.finish(Ok(err.into())); -+ ctx.data_mut() -+ .restore_parent_execution_guard(parent_execution); -+ child_registration.rollback_child(err.into()); - OnCalledAction::Trap(Box::new(WasiError::Exit(err.into()))) - } - } - }) - } -- --fn run( -- ctx: WasiFunctionEnv, -- mut store: Store, -- child_handle: WasiThreadHandle, -- rewind_state: Option<(RewindState, RewindResultType)>, --) -> ExitCode { -- let env = ctx.data(&store); -- let tasks = env.tasks().clone(); -- let pid = env.pid(); -- let tid = env.tid(); -- -- // If we need to rewind then do so -- if let Some((rewind_state, rewind_result)) = rewind_state { -- let mut ctx = ctx.env.clone().into_mut(&mut store); -- let res = rewind_ext::( -- &mut ctx, -- Some(rewind_state.memory_stack), -- rewind_state.rewind_stack, -- rewind_state.store_data, -- rewind_result, -- ); -- if res != Errno::Success { -- return res.into(); -- } -- } -- -- let mut ret: ExitCode = Errno::Success.into(); -- let (mut store, err) = if ctx.data(&store).thread.is_main() { -- trace!(%pid, %tid, "re-invoking main"); -- let start = ctx -- .data(&store) -- .inner() -- .static_module_instance_handles() -- .unwrap() -- .start -- .clone() -- .unwrap(); -- ContextSwitchingEnvironment::run_main_context(&ctx, store, start.into(), vec![]) -- } else { -- trace!(%pid, %tid, "re-invoking thread_spawn"); -- let start = ctx -- .data(&store) -- .inner() -- .static_module_instance_handles() -- .unwrap() -- .thread_spawn -- .clone() -- .unwrap(); -- let params = vec![0i32.into(), 0i32.into()]; -- ContextSwitchingEnvironment::run_main_context(&ctx, store, start.into(), params) -- }; -- if let Err(err) = err { -- match err.downcast::() { -- Ok(WasiError::Exit(exit_code)) => { -- ret = exit_code; -- } -- Ok(WasiError::DeepSleep(deep)) => { -- trace!(%pid, %tid, "entered a deep sleep"); -- -- // Create the respawn function -- let respawn = { -- let tasks = tasks.clone(); -- let rewind_state = deep.rewind; -- move |ctx, store, rewind_result| { -- run::( -- ctx, -- store, -- child_handle, -- Some(( -- rewind_state, -- RewindResultType::RewindWithResult(rewind_result), -- )), -- ); -- } -- }; -- -- /// Spawns the WASM process after a trigger -- unsafe { -- tasks.resume_wasm_after_poller(Box::new(respawn), ctx, store, deep.trigger) -- }; -- return Errno::Success.into(); -- } -- _ => {} -- } -- } -- trace!(%pid, %tid, "child exited (code = {})", ret); -- -- // Clean up the environment and return the result -- ctx.on_exit((&mut store), Some(ret)); -- -- // We drop the handle at the last moment which will close the thread -- drop(child_handle); -- ret --} -diff --git a/lib/wasix/src/syscalls/wasix/proc_fork_env.rs b/lib/wasix/src/syscalls/wasix/proc_fork_env.rs -index 0c948ee..62ad17d 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_fork_env.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_fork_env.rs -@@ -43,7 +43,7 @@ pub fn proc_fork_env( - // and associate a new context but otherwise shares things like the - // file system interface. The handle to the forked process is stored - // in the parent process context -- let (mut child_env, child_handle) = match env.fork() { -+ let (mut child_env, child_handle, mut child_registration) = match env.fork_guarded() { - Ok(p) => p, - Err(err) => { - tracing::error!("Could not fork process: {err}"); -@@ -55,25 +55,54 @@ pub fn proc_fork_env( - // Write the child's PID to the provided pointer - let memory = unsafe { env.memory_view(&ctx) }; - wasi_try_mem_ok!(child_pid_ptr.write(&memory, child_env.pid().raw())); -+ drop(memory); - -+ // Add the child to the parent's list of children and notify the parent -+ // when the child exits. -+ let commit_result = { child_registration.commit_child() }; -+ if let Err(err) = commit_result { -+ tracing::error!("Could not publish forked process: {err}"); -+ // The ABI promises that failure does not leave a usable child PID. -+ let memory = unsafe { ctx.data().memory_view(&ctx) }; -+ wasi_try_mem_ok!(child_pid_ptr.write(&memory, 0)); -+ return Ok(Errno::Perm); -+ } -+ let child_execution = match child_env.process.acquire_execution_guard() { -+ Ok(execution) => execution, -+ Err(err) => { -+ tracing::error!("Could not lease forked child execution: {err}"); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ let memory = unsafe { ctx.data().memory_view(&ctx) }; -+ wasi_try_mem_ok!(child_pid_ptr.write(&memory, 0)); -+ return Ok(Errno::Perm); -+ } -+ }; -+ let parent_execution = match ctx.data().acquire_parent_execution_guard() { -+ Ok(execution) => execution, -+ Err(err) => { -+ tracing::error!("Could not retain vfork parent execution: {err}"); -+ child_execution.finish(Ok(Errno::Canceled.into())); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ let memory = unsafe { ctx.data().memory_view(&ctx) }; -+ wasi_try_mem_ok!(child_pid_ptr.write(&memory, 0)); -+ return Ok(Errno::Perm); -+ } -+ }; - let parent_env = ctx.data_mut(); -- -- // Add the child to the parent's list of children -- parent_env -- .process -- .lock() -- .children -- .push(child_env.process.clone()); - // Swap the current environment with the child environment - child_env.swap_inner(parent_env); - std::mem::swap(parent_env, &mut child_env); - -+ parent_execution.arm_fail_closed(); - let previous_vfork = parent_env.vfork.replace(WasiVFork { - asyncify: None, - env: Box::new(child_env), - handle: child_handle, -+ parent_execution, -+ child_execution, - }); - assert!(previous_vfork.is_none()); // Already checked at the start of the function -+ child_registration.complete_child_launch(ctx.data().tasks()); - - Ok(Errno::Success) - } -diff --git a/lib/wasix/src/syscalls/wasix/proc_join.rs b/lib/wasix/src/syscalls/wasix/proc_join.rs -index 87c12ea..688b574 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_join.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_join.rs -@@ -7,13 +7,48 @@ use wasmer_wasix_types::wasi::{JoinFlags, JoinStatus, JoinStatusType, JoinStatus - use super::*; - use crate::{WasiProcess, syscalls::*}; - --#[derive(Serialize, Deserialize)] -+#[derive(Debug, PartialEq, Eq, Serialize, Deserialize)] - enum JoinStatusResult { - Nothing, - ExitNormal(WasiProcessId, ExitCode), - Err(Errno), - } - -+async fn wait_for_any_child(mut process: WasiProcess) -> JoinStatusResult { -+ match process.join_any_child().await { -+ Ok(Some((pid, exit_code))) => { -+ tracing::trace!(%pid, %exit_code, "triggered child join"); -+ trace!(ret_id = pid.raw(), exit_code = exit_code.raw()); -+ JoinStatusResult::ExitNormal(pid, exit_code) -+ } -+ Ok(None) => { -+ tracing::trace!("triggered child join (no child)"); -+ JoinStatusResult::Err(Errno::Child) -+ } -+ Err(err) => { -+ tracing::trace!(%err, "error triggered on child join"); -+ JoinStatusResult::Err(err) -+ } -+ } -+} -+ -+async fn wait_for_explicit_process( -+ parent: Option, -+ process: WasiProcess, -+ pid: WasiProcessId, -+) -> JoinStatusResult { -+ let Some(status) = process.claim_join().await else { -+ return JoinStatusResult::Nothing; -+ }; -+ if let Some(parent) = parent { -+ parent.remove_child_if_same(&process); -+ } -+ process.reap(); -+ let exit_code = status.unwrap_or_else(|_| Errno::Child.into()); -+ tracing::trace!(%exit_code, "triggered child join"); -+ JoinStatusResult::ExitNormal(pid, exit_code) -+} -+ - /// ### `proc_join()` - /// Joins the child process, blocking this one until the other finishes - /// -@@ -118,27 +153,31 @@ pub(super) fn proc_join_internal( - None => { - let mut process = ctx.data_mut().process.clone(); - -- // We wait for any process to exit (if it takes too long -- // then we go into a deep sleep) -- let res = __asyncify_with_deep_sleep::(ctx, async move { -- let child_exit = process.join_any_child().await; -- match child_exit { -+ if flags.contains(JoinFlags::NON_BLOCKING) { -+ let result = match process.try_join_any_child() { - Ok(Some((pid, exit_code))) => { - tracing::trace!(%pid, %exit_code, "triggered child join"); - trace!(ret_id = pid.raw(), exit_code = exit_code.raw()); - JoinStatusResult::ExitNormal(pid, exit_code) - } - Ok(None) => { -- tracing::trace!("triggered child join (no child)"); -- JoinStatusResult::Err(Errno::Child) -+ tracing::trace!("nonblocking child join found no exited child"); -+ JoinStatusResult::Nothing - } - Err(err) => { - tracing::trace!(%err, "error triggered on child join"); - JoinStatusResult::Err(err) - } -- } -- })?; -- return match res { -+ }; -+ return ret_result(ctx, result); -+ } -+ -+ // A postmaster wait must release the instance and linear memory -+ // after the deep-sleep threshold rather than pinning both for the -+ // lifetime of a child. The future performs claim/removal/reap -+ // before its serializable result is captured for rewind. -+ let action = __asyncify_with_deep_sleep::(ctx, wait_for_any_child(process))?; -+ return match action { - AsyncifyAction::Finish(ctx, result) => ret_result(ctx, result), - AsyncifyAction::Unwind => Ok(Errno::Success), - }; -@@ -151,16 +190,16 @@ pub(super) fn proc_join_internal( - - // Waiting for a process that is an explicit child will join it - // meaning it will no longer be a sub-process of the main process -- let mut process = { -- let mut inner = ctx.data().process.lock(); -+ let (mut process, is_child_process) = { -+ let inner = ctx.data().process.lock(); - let process = inner - .children - .iter() - .filter(|c| c.pid == pid) - .map(Clone::clone) - .next(); -- inner.children.retain(|c| c.pid != pid); -- process -+ let is_child_process = process.is_some(); -+ (process, is_child_process) - }; - - // Otherwise it could be the case that we are waiting for a process -@@ -180,21 +219,24 @@ pub(super) fn proc_join_internal( - )); - - if flags.contains(JoinFlags::NON_BLOCKING) { -- if let Some(status) = process.try_join() { -+ if let Some(status) = process.try_claim_join() { -+ if is_child_process { -+ ctx.data().process.remove_child_if_same(&process); -+ } - let exit_code = status.unwrap_or_else(|_| Errno::Child.into()); -+ process.reap(); - ret_result(ctx, JoinStatusResult::ExitNormal(pid, exit_code)) - } else { - ret_result(ctx, JoinStatusResult::Nothing) - } - } else { - // Wait for the process to finish -- let process2 = process.clone(); -- let res = __asyncify_with_deep_sleep::(ctx, async move { -- let exit_code = process.join().await.unwrap_or_else(|_| Errno::Child.into()); -- tracing::trace!(%exit_code, "triggered child join"); -- JoinStatusResult::ExitNormal(pid, exit_code) -- })?; -- match res { -+ let parent = is_child_process.then(|| ctx.data().process.clone()); -+ let action = __asyncify_with_deep_sleep::( -+ ctx, -+ wait_for_explicit_process(parent, process, pid), -+ )?; -+ match action { - AsyncifyAction::Finish(ctx, result) => ret_result(ctx, result), - AsyncifyAction::Unwind => Ok(Errno::Success), - } -@@ -204,3 +246,112 @@ pub(super) fn proc_join_internal( - ret_result(ctx, JoinStatusResult::Nothing) - } - } -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use crate::{WasiControlPlane, os::task::thread::WasiMemoryLayout}; -+ use wasmer_types::ModuleHash; -+ -+ #[test] -+ fn explicit_blocking_wait_claims_and_reaps_before_serialization() { -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, _child_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let child_pid = child.pid(); -+ parent.lock().children.push(child.clone()); -+ child.terminate(Errno::Success.into()); -+ -+ let result = virtual_mio::block_on(wait_for_explicit_process( -+ Some(parent.clone()), -+ child.clone(), -+ child_pid, -+ )); -+ -+ assert_eq!( -+ result, -+ JoinStatusResult::ExitNormal(child_pid, Errno::Success.into()) -+ ); -+ assert!(parent.lock().children.is_empty()); -+ assert!(plane.get_process(child_pid).is_none()); -+ -+ let encoded = bincode::serde::encode_to_vec(&result, bincode::config::legacy()).unwrap(); -+ let (decoded, _): (JoinStatusResult, usize) = -+ bincode::serde::decode_from_slice(&encoded, bincode::config::legacy()).unwrap(); -+ assert_eq!(decoded, result); -+ -+ assert_eq!( -+ virtual_mio::block_on(wait_for_explicit_process(Some(parent), child, child_pid,)), -+ JoinStatusResult::Nothing -+ ); -+ } -+ -+ #[test] -+ fn any_child_blocking_wait_returns_one_serializable_claim() { -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, _child_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let child_pid = child.pid(); -+ parent.lock().children.push(child.clone()); -+ child.terminate(Errno::Success.into()); -+ -+ let result = virtual_mio::block_on(wait_for_any_child(parent.clone())); -+ assert_eq!( -+ result, -+ JoinStatusResult::ExitNormal(child_pid, Errno::Success.into()) -+ ); -+ assert!(parent.lock().children.is_empty()); -+ assert!(plane.get_process(child_pid).is_none()); -+ } -+ -+ #[test] -+ fn blocking_join_payload_survives_beyond_the_deep_sleep_threshold() { -+ use std::{ -+ thread, -+ time::{Duration, Instant}, -+ }; -+ -+ let plane = WasiControlPlane::default(); -+ let (parent, _parent_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let (child, _child_main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ let child_pid = child.pid(); -+ parent.lock().children.push(child.clone()); -+ -+ let terminating_child = child.clone(); -+ let terminator = thread::spawn(move || { -+ thread::sleep(Duration::from_millis(75)); -+ terminating_child.terminate(Errno::Success.into()); -+ }); -+ let started = Instant::now(); -+ let result = virtual_mio::block_on(wait_for_any_child(parent.clone())); -+ let elapsed = started.elapsed(); -+ terminator.join().unwrap(); -+ -+ assert!( -+ elapsed >= Duration::from_millis(50), -+ "join completed before the deep-sleep threshold: {elapsed:?}" -+ ); -+ assert_eq!( -+ result, -+ JoinStatusResult::ExitNormal(child_pid, Errno::Success.into()) -+ ); -+ let encoded = bincode::serde::encode_to_vec(&result, bincode::config::legacy()).unwrap(); -+ let (decoded, _): (JoinStatusResult, usize) = -+ bincode::serde::decode_from_slice(&encoded, bincode::config::legacy()).unwrap(); -+ assert_eq!(decoded, result); -+ assert!(parent.lock().children.is_empty()); -+ assert!(plane.get_process(child_pid).is_none()); -+ } -+} -diff --git a/lib/wasix/src/syscalls/wasix/proc_rlimit_get.rs b/lib/wasix/src/syscalls/wasix/proc_rlimit_get.rs -new file mode 100644 -index 0000000..ecb4021 ---- /dev/null -+++ b/lib/wasix/src/syscalls/wasix/proc_rlimit_get.rs -@@ -0,0 +1,38 @@ -+use super::*; -+use crate::syscalls::*; -+ -+const RLIMIT_STACK: u32 = 3; -+const RLIMIT_CORE: u32 = 4; -+const RLIMIT_NOFILE: u32 = 7; -+const RLIM_INFINITY: u64 = u64::MAX; -+ -+/// ### `proc_rlimit_get()` -+/// Returns the current and maximum resource limit for the current process. -+#[instrument(level = "trace", skip_all, fields(%resource, cur = field::Empty, max = field::Empty), ret)] -+pub fn proc_rlimit_get( -+ ctx: FunctionEnvMut<'_, WasiEnv>, -+ resource: u32, -+ ret_cur: WasmPtr, -+ ret_max: WasmPtr, -+) -> Errno { -+ let limit = match resource { -+ RLIMIT_STACK => ctx -+ .data() -+ .runtime -+ .resource_limits() -+ .stack -+ .unwrap_or(RLIM_INFINITY), -+ RLIMIT_CORE => 0, -+ RLIMIT_NOFILE => RLIM_INFINITY, -+ _ => return Errno::Inval, -+ }; -+ -+ Span::current().record("cur", limit); -+ Span::current().record("max", limit); -+ -+ let env = ctx.data(); -+ let memory = unsafe { env.memory_view(&ctx) }; -+ wasi_try_mem!(ret_cur.write(&memory, limit)); -+ wasi_try_mem!(ret_max.write(&memory, limit)); -+ Errno::Success -+} -diff --git a/lib/wasix/src/syscalls/wasix/proc_signal.rs b/lib/wasix/src/syscalls/wasix/proc_signal.rs -index 9c80640..acaa5c2 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_signal.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_signal.rs -@@ -1,4 +1,5 @@ - use super::*; -+use crate::WasiControlPlane; - use crate::syscalls::*; - - /// ### `proc_signal()` -@@ -14,15 +15,69 @@ pub fn proc_signal( - pid: Pid, - sig: Signal, - ) -> Result { -- let process = { -- let pid: WasiProcessId = pid.into(); -- ctx.data().control_plane.get_process(pid) -+ let result = signal_registered_process(&ctx.data().control_plane, pid, sig); -+ -+ WasiEnv::do_pending_operations(&mut ctx)?; -+ -+ Ok(result) -+} -+ -+fn signal_registered_process(control_plane: &WasiControlPlane, pid: Pid, sig: Signal) -> Errno { -+ let pid: WasiProcessId = pid.into(); -+ let Some(process) = control_plane.get_process(pid) else { -+ return Errno::Srch; - }; -- if let Some(process) = process { -+ -+ // POSIX signal zero is a liveness/permission probe. WASIX has no process -+ // permission model here, so registry presence is the complete answer and -+ // no signal may enter the guest's pending queue. -+ if sig != Signal::Signone { - process.signal_process(sig); - } -+ Errno::Success -+} - -- WasiEnv::do_pending_operations(&mut ctx)?; -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use crate::os::task::thread::WasiMemoryLayout; -+ use wasmer_types::ModuleHash; - -- Ok(Errno::Success) -+ #[test] -+ fn absent_pid_returns_srch_for_liveness_probe() { -+ let plane = WasiControlPlane::default(); -+ -+ assert_eq!( -+ signal_registered_process(&plane, 42, Signal::Signone), -+ Errno::Srch -+ ); -+ } -+ -+ #[test] -+ fn signal_zero_observes_existing_pid_without_delivery() { -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ -+ assert_eq!( -+ signal_registered_process(&plane, process.pid().raw(), Signal::Signone), -+ Errno::Success -+ ); -+ assert!(main.pop_signals().is_empty()); -+ } -+ -+ #[test] -+ fn real_signal_delivery_is_unchanged() { -+ let plane = WasiControlPlane::default(); -+ let (process, main) = plane -+ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) -+ .unwrap(); -+ -+ assert_eq!( -+ signal_registered_process(&plane, process.pid().raw(), Signal::Sigusr1), -+ Errno::Success -+ ); -+ assert_eq!(main.pop_signals(), vec![Signal::Sigusr1]); -+ } - } -diff --git a/lib/wasix/src/syscalls/wasix/proc_spawn.rs b/lib/wasix/src/syscalls/wasix/proc_spawn.rs -index 8847178..39348b1 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_spawn.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_spawn.rs -@@ -106,22 +106,25 @@ pub fn proc_spawn_internal( - let env = ctx.data(); - - // Fork the current environment and set the new arguments -- let (mut child_env, handle) = match ctx.data().fork() { -+ let (mut child_env, child_handle, mut child_registration) = match ctx.data().fork_guarded() { - Ok(x) => x, - Err(err) => { - // TODO: evaluate the appropriate error code, document it in the spec. - return Ok(Err(Errno::Access)); - } - }; -- let child_process = child_env.process.clone(); - if let Some(args) = args { -- let mut child_state = env.state.fork(); -- child_state.args = std::sync::Mutex::new(args); -- child_env.state = Arc::new(child_state); -+ child_env.state = match env.state.fork_with(move |state| { -+ state.args = std::sync::Mutex::new(args); -+ }) { -+ Ok(state) => state, -+ Err(errno) => { -+ warn!(%errno, "could not fork spawned-process state"); -+ return Ok(Err(errno)); -+ } -+ }; - } - -- // Take ownership of this child -- ctx.data_mut().owned_handles.push(handle); - let env = ctx.data(); - - // Preopen -@@ -230,9 +233,26 @@ pub fn proc_spawn_internal( - // Create the new process - let bin_factory = Box::new(ctx.data().bin_factory.clone()); - let child_pid = child_env.pid(); -+ let child_process = child_env.process.clone(); - - let mut builder = Some(child_env); - -+ // Publish the embryonic child before entering arbitrary built-in code or -+ // Wasm instantiation (which may execute a module start function). -+ let tasks = ctx.data().tasks().clone(); -+ if let Err(err) = child_registration.commit_child() { -+ error!(child_pid = %child_pid, "failed to publish child process: {err}"); -+ return Ok(Err(Errno::Perm)); -+ } -+ let launch_execution = match child_process.acquire_execution_guard() { -+ Ok(execution) => execution, -+ Err(err) => { -+ error!(child_pid = %child_pid, "failed to lease child launch: {err}"); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ return Ok(Err(Errno::Perm)); -+ } -+ }; -+ - // First we try the built in commands - let mut process = match bin_factory.try_built_in(name.clone(), Some(&ctx), &mut builder) { - Ok(a) => a, -@@ -241,23 +261,29 @@ pub fn proc_spawn_internal( - error!("builtin failed - {}", err); - } - // Now we actually spawn the process -- let child_work = bin_factory.spawn(name, builder.take().unwrap()); -+ let env = builder.take().unwrap(); - -- match __asyncify(&mut ctx, None, async move { Ok(child_work.await) })? -- .map_err(|err| Errno::Unknown) -- { -- Ok(Ok(a)) => a, -- Ok(Err(err)) => return Ok(Err(conv_spawn_err_to_errno(&err))), -- Err(err) => return Ok(Err(err)), -+ match block_on(bin_factory.spawn(name, env)) { -+ Ok(a) => a, -+ Err(err) => { -+ let errno = conv_spawn_err_to_errno(&err); -+ launch_execution.finish(Ok(errno.into())); -+ child_registration.rollback_child(errno.into()); -+ return Ok(Err(errno)); -+ } - } - } - }; - -- // Add the process to the environment state -- { -- let mut inner = ctx.data().process.lock(); -- inner.children.push(child_process); -+ if let Err(err) = launch_execution.monitor(process.clone(), &tasks) { -+ error!(child_pid = %child_pid, "failed to monitor child launch: {err}"); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ return Ok(Err(Errno::Canceled)); - } -+ child_registration.complete_child_launch(&tasks); -+ // `child_env` owns its main handle for the full task lifetime. The -+ // syscall's construction handle must not leak into the parent epoch. -+ drop(child_handle); - let env = ctx.data(); - let memory = unsafe { env.memory_view(&ctx) }; - -diff --git a/lib/wasix/src/syscalls/wasix/proc_spawn2.rs b/lib/wasix/src/syscalls/wasix/proc_spawn2.rs -index e447e64..6a813a5 100644 ---- a/lib/wasix/src/syscalls/wasix/proc_spawn2.rs -+++ b/lib/wasix/src/syscalls/wasix/proc_spawn2.rs -@@ -123,7 +123,7 @@ pub fn proc_spawn2( - // and associate a new context but otherwise shares things like the - // file system interface. The handle to the forked process is stored - // in the parent process context -- let (mut child_env, mut child_handle) = match ctx.data().fork() { -+ let (mut child_env, child_handle, mut child_registration) = match ctx.data().fork_guarded() { - Ok(p) => p, - Err(err) => { - debug!("could not fork process: {err}"); -@@ -132,14 +132,10 @@ pub fn proc_spawn2( - } - }; - -- { -- let mut inner = ctx.data().process.lock(); -- inner.children.push(child_env.process.clone()); -- } -- - // Setup some properties in the child environment - let pid = child_env.pid(); - let tid = child_env.tid(); -+ let child_process = child_env.process.clone(); - wasi_try_mem_ok!(ret.write(&memory, pid.raw())); - Span::current() - .record("pid", pid.raw()) -@@ -156,6 +152,25 @@ pub fn proc_spawn2( - - let mut builder = Some(child_env); - -+ // Instance creation can run guest start/initializer code. Make the exact -+ // child identity visible in its parent and registry before resolving any -+ // launch implementation, including arbitrary built-ins. -+ let tasks = ctx.data().tasks().clone(); -+ if let Err(err) = child_registration.commit_child() { -+ error!(child_pid = %pid, "failed to publish child process: {err}"); -+ let _ = ret.write(&memory, 0); -+ return Ok(Errno::Perm); -+ } -+ let launch_execution = match child_process.acquire_execution_guard() { -+ Ok(execution) => execution, -+ Err(err) => { -+ error!(child_pid = %pid, "failed to lease child launch: {err}"); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ let _ = ret.write(&memory, 0); -+ return Ok(Errno::Perm); -+ } -+ }; -+ - let process = match bin_factory.try_built_in(name.clone(), Some(&ctx), &mut builder) { - Ok(a) => Ok(a), - Err(err) => { -@@ -171,8 +186,17 @@ pub fn proc_spawn2( - }; - - match process { -- Ok(_) => { -- ctx.data_mut().owned_handles.push(child_handle); -+ Ok(handle) => { -+ if let Err(err) = launch_execution.monitor(handle, &tasks) { -+ error!(child_pid = %pid, "failed to monitor child launch: {err}"); -+ child_registration.rollback_child(Errno::Canceled.into()); -+ let _ = ret.write(&memory, 0); -+ return Ok(Errno::Canceled); -+ } -+ child_registration.complete_child_launch(&tasks); -+ // The launched child environment owns its main handle. Keeping a -+ // clone in the parent makes RSS and task slots grow per spawn. -+ drop(child_handle); - trace!(child_pid = %pid, "spawned sub-process"); - Ok(Errno::Success) - } -@@ -181,6 +205,9 @@ pub fn proc_spawn2( - - debug!(child_pid = %pid, "process failed with (err={})", err_exit_code); - -+ launch_execution.finish(Ok(err_exit_code)); -+ child_registration.rollback_child(err_exit_code); -+ let _ = ret.write(&memory, 0); - Ok(Errno::Noexec) - } - } -@@ -225,6 +252,8 @@ fn apply_fd_op( - // TODO: verify this is correct - inner: FdInner { - offset: fd_entry.inner.offset.clone(), -+ ofd: fd_entry.inner.ofd.clone(), -+ readdir_cache: fd_entry.inner.readdir_cache.clone(), - rights: fd_entry.inner.rights_inheriting, - fd_flags: { - let mut f = fd_entry.inner.fd_flags; -diff --git a/lib/wasix/src/syscalls/wasix/sock_accept.rs b/lib/wasix/src/syscalls/wasix/sock_accept.rs -index b851666..76aedb0 100644 ---- a/lib/wasix/src/syscalls/wasix/sock_accept.rs -+++ b/lib/wasix/src/syscalls/wasix/sock_accept.rs -@@ -26,19 +26,18 @@ pub fn sock_accept( - - ctx = wasi_try_ok!(maybe_snapshot::(ctx)?); - -- let env = ctx.data(); -- let (memory, state, _) = unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; -- - let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); - - let (fd, _, _) = wasi_try_ok!(sock_accept_internal( -- env, -+ &mut ctx, - sock, - fd_flags, - nonblocking, - None - )?); - -+ let env = ctx.data(); -+ let memory = unsafe { env.memory_view(&ctx) }; - wasi_try_mem_ok!(ro_fd.write(&memory, fd)); - - Ok(Errno::Success) -@@ -67,13 +66,10 @@ pub fn sock_accept_v2( - ) -> Result { - WasiEnv::do_pending_operations(&mut ctx)?; - -- let env = ctx.data(); -- let (memory, state, _) = unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; -- - let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); - - let (fd, local_addr, peer_addr) = wasi_try_ok!(sock_accept_internal( -- env, -+ &mut ctx, - sock, - fd_flags, - nonblocking, -@@ -98,7 +94,7 @@ pub fn sock_accept_v2( - } - - let env = ctx.data(); -- let (memory, state, _) = unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; -+ let memory = unsafe { env.memory_view(&ctx) }; - wasi_try_mem_ok!(ro_fd.write(&memory, fd)); - wasi_try_ok!(crate::net::write_ip_port( - &memory, -@@ -111,37 +107,46 @@ pub fn sock_accept_v2( - } - - pub(crate) fn sock_accept_internal( -- env: &WasiEnv, -+ ctx: &mut FunctionEnvMut<'_, WasiEnv>, - sock: WasiFd, - mut fd_flags: Fdflags, - mut nonblocking: bool, - with_fd: Option, - ) -> Result, WasiError> { -- let state = env.state(); -- let inodes = &state.inodes; -- -- let tasks = env.tasks().clone(); -- let (child, local_addr, peer_addr, fd_flags) = wasi_try_ok_ok!(__sock_asyncify( -- env, -- sock, -- Rights::SOCK_ACCEPT, -- move |socket, fd| async move { -- if fd.inner.flags.contains(Fdflags::NONBLOCK) { -- fd_flags.set(Fdflags::NONBLOCK, true); -- nonblocking = true; -+ let tasks = ctx.data().tasks().clone(); -+ let accept = { -+ let fd_entry = wasi_try_ok_ok!(ctx.data().state.fs.get_fd(sock)); -+ if !fd_entry.inner.rights.contains(Rights::SOCK_ACCEPT) { -+ return Ok(Err(Errno::Access)); -+ } -+ -+ let inode = fd_entry.inode.clone(); -+ let mut guard = inode.write(); -+ match guard.deref_mut() { -+ Kind::Socket { socket } => { -+ let socket = socket.clone(); -+ drop(guard); -+ -+ async move { -+ if fd_entry.inner.flags.contains(Fdflags::NONBLOCK) { -+ fd_flags.set(Fdflags::NONBLOCK, true); -+ nonblocking = true; -+ } -+ let timeout = socket.opt_time(TimeType::AcceptTimeout).ok().flatten(); -+ let local_addr = socket.addr_local()?; -+ socket -+ .accept(tasks.deref(), nonblocking, timeout) -+ .await -+ .map(|a| (a.0, local_addr, a.1, fd_flags)) -+ } - } -- let timeout = socket -- .opt_time(TimeType::AcceptTimeout) -- .ok() -- .flatten() -- .unwrap_or(Duration::from_secs(30)); -- let local_addr = socket.addr_local()?; -- socket -- .accept(tasks.deref(), nonblocking, Some(timeout)) -- .await -- .map(|a| (a.0, local_addr, a.1, fd_flags)) -- }, -- )); -+ _ => return Ok(Err(Errno::Notsock)), -+ } -+ }; -+ let (child, local_addr, peer_addr, fd_flags) = wasi_try_ok_ok!(__asyncify(ctx, None, accept)?); -+ -+ let state = ctx.data().state(); -+ let inodes = &state.inodes; - - let kind = Kind::Socket { - socket: InodeSocket::new(InodeSocketKind::TcpStream { -@@ -159,11 +164,6 @@ pub(crate) fn sock_accept_internal( - new_flags.set(Fdflags::NONBLOCK, true); - } - -- let mut new_flags = Fdflags::empty(); -- if fd_flags.contains(Fdflags::NONBLOCK) { -- new_flags.set(Fdflags::NONBLOCK, true); -- } -- - let rights = Rights::all_socket(); - let fd = wasi_try_ok_ok!(if let Some(fd) = with_fd { - state -diff --git a/lib/wasix/src/syscalls/wasix/sock_recv.rs b/lib/wasix/src/syscalls/wasix/sock_recv.rs -index f507a5b..cf5b823 100644 ---- a/lib/wasix/src/syscalls/wasix/sock_recv.rs -+++ b/lib/wasix/src/syscalls/wasix/sock_recv.rs -@@ -29,11 +29,12 @@ pub fn sock_recv( - WasiEnv::do_pending_operations(&mut ctx)?; - - let env = ctx.data(); -- let fd_entry = wasi_try_ok!(env.state.fs.get_fd(sock)); -- let guard = fd_entry.inode.read(); -- // Some guests route socket-like wakeups through pipe-backed fds. -- let use_read = matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeRx { .. }); -- drop(guard); -+ let fd_entry = wasi_try_ok!({ env.state.fs.get_fd(sock) }); -+ let use_read = { -+ let guard = fd_entry.inode.read(); -+ // Some guests route socket-like wakeups through pipe-backed fds. -+ matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeRx { .. }) -+ }; - if use_read { - fd_read(ctx, sock, ri_data, ri_data_len, ro_data_len) - } else { -@@ -116,23 +117,28 @@ pub(super) fn sock_recv_internal( - - let peek = (ri_flags & __WASI_SOCK_RECV_INPUT_PEEK) != 0; - let nonblocking_flag = (ri_flags & __WASI_SOCK_RECV_INPUT_DONT_WAIT) != 0; -+ - let data = wasi_try_ok_ok!(__sock_asyncify( - env, - sock, - Rights::SOCK_RECV, - |socket, fd| async move { -- let iovs_arr = ri_data -- .slice(&memory, ri_data_len) -- .map_err(mem_error_to_wasi)?; -- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ let iovs_arr = { -+ let iovs_arr = ri_data -+ .slice(&memory, ri_data_len) -+ .map_err(mem_error_to_wasi)?; -+ iovs_arr.access().map_err(mem_error_to_wasi) -+ }?; - - let mut total_read = 0; - for iovs in iovs_arr.iter() { -- let mut buf = WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi)? -- .access() -- .map_err(mem_error_to_wasi)?; -+ let mut buf = { -+ WasmPtr::::new(iovs.buf) -+ .slice(&memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi) -+ }?; - - let nonblocking = nonblocking_flag || fd.inner.flags.contains(Fdflags::NONBLOCK); - let timeout = socket -diff --git a/lib/wasix/src/syscalls/wasix/sock_send.rs b/lib/wasix/src/syscalls/wasix/sock_send.rs -index bfc196b..939e8a4 100644 ---- a/lib/wasix/src/syscalls/wasix/sock_send.rs -+++ b/lib/wasix/src/syscalls/wasix/sock_send.rs -@@ -28,12 +28,13 @@ pub fn sock_send( - WasiEnv::do_pending_operations(&mut ctx)?; - - let env = ctx.data(); -- let fd_entry = wasi_try_ok!(env.state.fs.get_fd(fd)); -+ let fd_entry = wasi_try_ok!({ env.state.fs.get_fd(fd) }); - let enable_journal = env.enable_journal; -- let guard = fd_entry.inode.read(); -- // Some guests route socket-like wakeups through pipe-backed fds. -- let use_write = matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeTx { .. }); -- drop(guard); -+ let use_write = { -+ let guard = fd_entry.inode.read(); -+ // Some guests route socket-like wakeups through pipe-backed fds. -+ matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeTx { .. }) -+ }; - - let bytes_written = if use_write { - let offset = { fd_entry.inner.offset.load(Ordering::Acquire) as usize }; -@@ -108,16 +109,20 @@ pub(crate) fn sock_send_internal( - - match si_data { - FdWriteSource::Iovs { iovs, iovs_len } => { -- let iovs_arr = iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; -- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; -+ let iovs_arr = { -+ let iovs_arr = iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; -+ iovs_arr.access().map_err(mem_error_to_wasi) -+ }?; - - let mut sent = 0usize; - for iovs in iovs_arr.iter() { -- let buf = WasmPtr::::new(iovs.buf) -- .slice(&memory, iovs.buf_len) -- .map_err(mem_error_to_wasi)? -- .access() -- .map_err(mem_error_to_wasi)?; -+ let buf = { -+ WasmPtr::::new(iovs.buf) -+ .slice(&memory, iovs.buf_len) -+ .map_err(mem_error_to_wasi)? -+ .access() -+ .map_err(mem_error_to_wasi) -+ }?; - let local_sent = match socket - .send( - env.tasks().deref(), -diff --git a/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs b/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs -index f7ddbeb..bf020e3 100644 ---- a/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs -+++ b/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs -@@ -48,6 +48,13 @@ pub(crate) fn sock_set_opt_flag_internal( - Ok(o) => o, - Err(_) => return Ok(Err(Errno::Inval)), - }; -+ match (&option, flag) { -+ (crate::net::socket::WasiSocketOption::NoDelay, true) => {} -+ (crate::net::socket::WasiSocketOption::NoDelay, false) => {} -+ (crate::net::socket::WasiSocketOption::KeepAlive, true) => {} -+ (crate::net::socket::WasiSocketOption::KeepAlive, false) => {} -+ _ => {} -+ } - wasi_try_ok_ok!(__sock_actor_mut( - ctx, - sock, -diff --git a/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs b/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs -index db97b42..1e9af3e 100644 ---- a/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs -+++ b/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs -@@ -45,9 +45,11 @@ pub fn stack_checkpoint( - // Perform the unwind action - unwind::(ctx, move |mut ctx, mut memory_stack, rewind_stack| { - // Grab all the globals and serialize them -- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) -- .serialize() -- .unwrap(); -+ let snapshot = match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { -+ Ok(snapshot) => snapshot, -+ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), -+ }; -+ let store_data = snapshot.serialize().unwrap(); - let env = ctx.data(); - let store_data = Bytes::from(store_data); - -diff --git a/lib/wasix/src/syscalls/wasix/thread_join.rs b/lib/wasix/src/syscalls/wasix/thread_join.rs -index 49c3fb9..ae300f4 100644 ---- a/lib/wasix/src/syscalls/wasix/thread_join.rs -+++ b/lib/wasix/src/syscalls/wasix/thread_join.rs -@@ -34,14 +34,16 @@ pub(super) fn thread_join_internal( - let tid: WasiThreadId = join_tid.into(); - let other_thread = env.process.get_thread(&tid); - if let Some(other_thread) = other_thread { -- let res = __asyncify_with_deep_sleep::(ctx, async move { -+ let res = __asyncify(&mut ctx, None, async move { - other_thread - .join() - .await - .map_err(|err| err.as_exit_code().unwrap_or(ExitCode::from(Errno::Unknown))) - .unwrap_or_else(|a| a) -- .raw() -+ .raw(); -+ Ok(()) - })?; -+ wasi_try_ok!(res); - Ok(Errno::Success) - } else { - Ok(Errno::Success) -diff --git a/lib/wasix/src/syscalls/wasix/thread_sleep.rs b/lib/wasix/src/syscalls/wasix/thread_sleep.rs -index 175f079..ec0614f 100644 ---- a/lib/wasix/src/syscalls/wasix/thread_sleep.rs -+++ b/lib/wasix/src/syscalls/wasix/thread_sleep.rs -@@ -40,9 +40,12 @@ pub(crate) fn thread_sleep_internal( - if duration > 0 { - let duration = Duration::from_nanos(duration); - let tasks = env.tasks().clone(); -- let res = __asyncify_with_deep_sleep::(ctx, async move { -+ if let Err(err) = __asyncify(&mut ctx, None, async move { - tasks.sleep_now(duration).await; -- })?; -+ Ok(()) -+ })? { -+ return Ok(err); -+ } - } - Ok(Errno::Success) - } -diff --git a/lib/wasix/src/syscalls/wasix/thread_spawn.rs b/lib/wasix/src/syscalls/wasix/thread_spawn.rs -index 1e0b4bd..d681ecb 100644 ---- a/lib/wasix/src/syscalls/wasix/thread_spawn.rs -+++ b/lib/wasix/src/syscalls/wasix/thread_spawn.rs -@@ -8,7 +8,7 @@ use crate::{ - os::task::thread::WasiMemoryLayout, - runtime::{ - TaintReason, -- task_manager::{TaskWasm, TaskWasmRunProperties}, -+ task_manager::{TaskWasm, TaskWasmAcceptedExecutionGuard, TaskWasmRunProperties}, - }, - state::context_switching::ContextSwitchingEnvironment, - syscalls::*, -@@ -60,6 +60,10 @@ pub fn thread_spawn_internal_from_wasi( - ) -> Result { - // Now we use the environment and memory references - let env = ctx.data(); -+ if env.vfork.is_some() { -+ tracing::warn!("thread_spawn is undefined in a vfork child"); -+ return Err(Errno::Notsup); -+ } - let memory = unsafe { env.memory_view(&ctx) }; - let runtime = env.runtime.clone(); - let tasks = env.tasks().clone(); -@@ -133,10 +137,7 @@ pub fn thread_spawn_internal_using_layout( - let linker = env_inner.linker().cloned(); - - // We capture some local variables -- let state = env.state.clone(); -- let mut thread_env = env.clone(); -- thread_env.thread = thread_handle.as_thread(); -- thread_env.layout = layout; -+ let mut thread_env = env.clone_for_thread_spawn(thread_handle.as_thread(), layout); - - // TODO: Currently asynchronous threading does not work with multi - // threading in JS but it does work for the main thread. This will -@@ -151,9 +152,18 @@ pub fn thread_spawn_internal_using_layout( - // calls into the process - let mut execute_module = { - let thread_handle = thread_handle; -- move |ctx: WasiFunctionEnv, mut store: Store| { -+ move |ctx: WasiFunctionEnv, -+ mut store: Store, -+ execution_guard: Option| { - // Call the thread -- call_module::(ctx, store, start_ptr_offset, thread_handle, rewind_state) -+ call_module::( -+ ctx, -+ store, -+ start_ptr_offset, -+ thread_handle, -+ rewind_state, -+ execution_guard, -+ ) - } - }; - -@@ -172,10 +182,10 @@ pub fn thread_spawn_internal_using_layout( - // Now spawn a thread - trace!("threading: spawning background thread"); - let run = move |props: TaskWasmRunProperties| { -- execute_module(props.ctx, props.store); -+ execute_module(props.ctx, props.store, props.execution_guard); - }; - -- let mut task_wasm = TaskWasm::new(Box::new(run), thread_env, thread_module, false, false) -+ let task_wasm = TaskWasm::new(Box::new(run), thread_env, thread_module, false, false) - .with_memory(spawn_type); - - tasks.task_wasm(task_wasm).map_err(Into::::into)?; -@@ -270,6 +280,10 @@ fn handle_thread_result( - .on_taint(TaintReason::DlSymbolResolutionFailed(symbol.clone())); - Ok(Some(ExitCode::from(129))) - } -+ Ok(WasiError::StoreSnapshot(err)) => { -+ eprintln!("Thread {tid} of process {pid} failed to snapshot store state: {err}"); -+ Ok(Some(ExitCode::from(129))) -+ } - Err(err) => { - eprintln!("Thread {tid} of process {pid} failed with runtime error: {err}"); - env.data(&store) -@@ -287,6 +301,7 @@ fn call_module( - start_ptr_offset: M::Offset, - thread_handle: Arc, - rewind_state: Option<(RewindState, RewindResultType)>, -+ execution_guard: Option, - ) { - let env = ctx.data(&store); - let tasks = env.tasks().clone(); -@@ -302,6 +317,9 @@ fn call_module( - rewind_result, - ); - if res != Errno::Success { -+ let exit_code = res.into(); -+ ctx.data().blocking_on_exit(Some(exit_code)); -+ thread_handle.set_status_finished(Ok(exit_code)); - return; - } - } -@@ -311,11 +329,12 @@ fn call_module( - - // If it went to deep sleep then we need to handle that - if let Err(deep) = ret { -+ let terminal_thread = thread_handle.as_thread(); - // Create the callback that will be invoked when the thread respawns after a deep sleep - let rewind = deep.rewind; - let respawn = { - let tasks = tasks.clone(); -- move |ctx, store, trigger_res| { -+ move |ctx, store, trigger_res, execution_guard| { - // Call the thread - call_module::( - ctx, -@@ -323,14 +342,29 @@ fn call_module( - start_ptr_offset, - thread_handle, - Some((rewind, RewindResultType::RewindWithResult(trigger_res))), -+ execution_guard, - ); - } - }; - -- /// Spawns the WASM process after a trigger -- unsafe { -- tasks.resume_wasm_after_poller(Box::new(respawn), ctx, store, deep.trigger) -- }; -+ // Spawns the WASM process after a trigger. Successor rejection must -+ // terminalize this exact thread before the current callback releases -+ // its execution lease. -+ if let Err(err) = unsafe { -+ tasks.resume_wasm_after_poller( -+ Box::new(respawn), -+ ctx, -+ store, -+ deep.trigger, -+ execution_guard, -+ ) -+ } { -+ tracing::warn!("failed to resume thread after deep sleep: {err}"); -+ terminal_thread.set_status_finished(Err(crate::RuntimeError::new(format!( -+ "failed to resume thread after deep sleep: {err}" -+ )) -+ .into())); -+ } - return; - }; - -diff --git a/lib/wasix/src/utils/store.rs b/lib/wasix/src/utils/store.rs -index d956dba..a4b8db9 100644 ---- a/lib/wasix/src/utils/store.rs -+++ b/lib/wasix/src/utils/store.rs -@@ -1,32 +1,531 @@ - use bincode::config; -+use sha2::{Digest, Sha256}; -+use wasmer_types::Mutability; - --/// A snapshot that captures the runtime state of an instance. --#[derive(Default, serde::Serialize, serde::Deserialize, Clone, Debug)] -+const STORE_SNAPSHOT_MAGIC: &[u8; 8] = b"WXSSTORE"; -+const STORE_SNAPSHOT_VERSION_SPARSE: u8 = 2; -+const STORE_SHAPE_DOMAIN_V2: &[u8] = b"wasix-store-snapshot-shape-v2\0"; -+ -+/// Values of all store globals in the original, unversioned snapshot format. -+/// -+/// Keep this single-field shape byte-compatible with the historical `StoreSnapshot` serde -+/// representation. Snapshots can be journaled, so old dense bytes must remain readable. -+#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] -+struct DenseStoreSnapshotV1 { -+ globals: Vec, -+} -+ -+/// A stable description of the complete store-global layout. -+#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -+struct StoreShapeV2 { -+ global_count: u64, -+ mutable_global_count: u64, -+ descriptor_sha256: [u8; 32], -+} -+ -+/// Version-two wire representation. Mutable values are ordered by their rank among all mutable -+/// globals, while `shape` binds that rank to exact store indexes and WebAssembly value types. -+#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] -+struct SparseStoreSnapshotV2 { -+ shape: StoreShapeV2, -+ mutable_globals: Vec, -+} -+ -+/// A snapshot that captures the runtime state of every store-owned mutable global. -+/// -+/// The sparse representation includes globals created for imports and globals appended by -+/// side-module instantiation because it walks the store, rather than one instance's local-global -+/// list. Dense V1 is retained only to decode already-persisted snapshots; new captures fail closed -+/// on backends that do not expose exact raw WebAssembly values. -+#[derive(Clone, Debug)] -+enum StoreSnapshotRepr { -+ DenseV1(DenseStoreSnapshotV1), -+ SparseV2(SparseStoreSnapshotV2), -+} -+ -+#[derive(Clone, Debug)] - pub struct StoreSnapshot { -- /// Values of all globals, indexed by the same index used in Webassembly. -- pub globals: Vec, -+ repr: StoreSnapshotRepr, -+} -+ -+impl Default for StoreSnapshot { -+ fn default() -> Self { -+ Self { -+ repr: StoreSnapshotRepr::DenseV1(DenseStoreSnapshotV1 { -+ globals: Vec::new(), -+ }), -+ } -+ } - } - - impl StoreSnapshot { - pub fn serialize(&self) -> Result, bincode::error::EncodeError> { -- bincode::serde::encode_to_vec(self, config::legacy()) -+ match &self.repr { -+ StoreSnapshotRepr::DenseV1(snapshot) => { -+ bincode::serde::encode_to_vec(snapshot, config::legacy()) -+ } -+ StoreSnapshotRepr::SparseV2(snapshot) => { -+ let payload = bincode::serde::encode_to_vec(snapshot, config::legacy())?; -+ let mut encoded = -+ Vec::with_capacity(STORE_SNAPSHOT_MAGIC.len() + 1 + payload.len()); -+ encoded.extend_from_slice(STORE_SNAPSHOT_MAGIC); -+ encoded.push(STORE_SNAPSHOT_VERSION_SPARSE); -+ encoded.extend_from_slice(&payload); -+ Ok(encoded) -+ } -+ } - } - - pub fn deserialize(data: &[u8]) -> Result { -- bincode::serde::decode_from_slice(data, config::legacy()).map(|(ret, _)| ret) -+ if data.starts_with(STORE_SNAPSHOT_MAGIC) { -+ let Some((&version, payload)) = data[STORE_SNAPSHOT_MAGIC.len()..].split_first() else { -+ return Err(bincode::error::DecodeError::UnexpectedEnd { additional: 1 }); -+ }; -+ if version != STORE_SNAPSHOT_VERSION_SPARSE { -+ return Err(bincode::error::DecodeError::Other( -+ "unsupported WASIX store snapshot version", -+ )); -+ } -+ return bincode::serde::decode_from_slice(payload, config::legacy()).map( -+ |(snapshot, _)| Self { -+ repr: StoreSnapshotRepr::SparseV2(snapshot), -+ }, -+ ); -+ } -+ -+ // No magic means the historical single-Vec serde structure. This fallback is deliberately -+ // permanent because stack snapshots may outlive the runtime that wrote them. -+ bincode::serde::decode_from_slice(data, config::legacy()).map(|(snapshot, _)| Self { -+ repr: StoreSnapshotRepr::DenseV1(snapshot), -+ }) -+ } -+} -+ -+#[derive(Clone, Debug, thiserror::Error, PartialEq, Eq)] -+pub enum StoreSnapshotCaptureError { -+ #[error("mutable reference global at store index {index} cannot be snapshotted as raw bits")] -+ MutableReferenceGlobal { index: usize }, -+ #[error("the active backend cannot capture an exact store snapshot")] -+ UnsupportedBackend, -+} -+ -+#[derive(Debug, thiserror::Error, PartialEq, Eq)] -+pub enum StoreSnapshotRestoreError { -+ #[error("dense store snapshot has {snapshot} globals, but destination has {store}")] -+ DenseShapeMismatch { snapshot: usize, store: usize }, -+ #[error("sparse store snapshot mutable-value count does not match its shape")] -+ SparseValueCountMismatch, -+ #[error("sparse store snapshot shape does not match the destination store")] -+ SparseShapeMismatch, -+ #[error("the active backend cannot restore an exact sparse store snapshot")] -+ UnsupportedBackend, -+ #[error("reference global at store index {index} cannot be restored from raw bits")] -+ ReferenceGlobal { index: usize }, -+} -+ -+#[cfg(any(feature = "sys", feature = "sys-minimal"))] -+fn sparse_sys_store_shape( -+ store: &wasmer::sys::store::StoreObjects, -+) -> Result { -+ let mut descriptor = Sha256::new(); -+ descriptor.update(STORE_SHAPE_DOMAIN_V2); -+ // Buffer compact (type, mutability) descriptors so a large module does not make one digest -+ // call per global. Sequence position is the store index, so explicit index bytes are redundant. -+ let mut descriptor_buffer = [0_u8; 1024]; -+ let mut descriptor_buffer_len = 0_usize; -+ let mut global_count = 0_u64; -+ let mut mutable_global_count = 0_u64; -+ -+ for (index, global) in store.iter_globals().enumerate() { -+ let global_type = *global.ty(); -+ if global_type.mutability == Mutability::Var && global_type.ty.is_ref() { -+ return Err(StoreSnapshotCaptureError::MutableReferenceGlobal { index }); -+ } -+ let index = u64::try_from(index).expect("store global index does not fit u64"); -+ debug_assert_eq!(index, global_count); -+ descriptor_buffer[descriptor_buffer_len] = global_type.ty as u8; -+ descriptor_buffer[descriptor_buffer_len + 1] = global_type.mutability as u8; -+ descriptor_buffer_len += 2; -+ if descriptor_buffer_len == descriptor_buffer.len() { -+ descriptor.update(descriptor_buffer); -+ descriptor_buffer_len = 0; -+ } -+ global_count += 1; -+ if global_type.mutability == Mutability::Var { -+ mutable_global_count += 1; -+ } -+ } -+ -+ descriptor.update(&descriptor_buffer[..descriptor_buffer_len]); -+ descriptor.update(global_count.to_le_bytes()); -+ descriptor.update(mutable_global_count.to_le_bytes()); -+ Ok(StoreShapeV2 { -+ global_count, -+ mutable_global_count, -+ descriptor_sha256: descriptor.finalize().into(), -+ }) -+} -+ -+fn sparse_store_shape( -+ objects: &wasmer::StoreObjects, -+) -> Result { -+ match objects { -+ #[cfg(any(feature = "sys", feature = "sys-minimal"))] -+ wasmer::StoreObjects::Sys(store) => sparse_sys_store_shape(store), -+ #[allow(unreachable_patterns)] -+ _ => Err(StoreSnapshotCaptureError::UnsupportedBackend), -+ } -+} -+ -+#[cfg(any(feature = "sys", feature = "sys-minimal"))] -+fn capture_raw_mutable_globals( -+ store: &wasmer::sys::store::StoreObjects, -+ capacity: usize, -+) -> Vec { -+ let mut values = Vec::with_capacity(capacity); -+ for global in store -+ .iter_globals() -+ .filter(|global| global.ty().mutability == Mutability::Var) -+ { -+ debug_assert!(!global.ty().ty.is_ref()); -+ // The complete store shape was validated before this second pass, so every mutable -+ // global read here has a numeric/vector representation valid for raw bit capture. -+ values.push(unsafe { global.vmglobal().as_ref().val.u128 }); -+ } -+ values -+} -+ -+pub fn capture_store_snapshot( -+ store: &mut impl wasmer::AsStoreMut, -+) -> Result { -+ let objects = store.objects_mut(); -+ let shape = sparse_store_shape(objects)?; -+ -+ let mutable_capacity = usize::try_from(shape.mutable_global_count) -+ .expect("mutable store-global count does not fit usize"); -+ let mutable_globals = match objects { -+ #[cfg(any(feature = "sys", feature = "sys-minimal"))] -+ wasmer::StoreObjects::Sys(store) => capture_raw_mutable_globals(store, mutable_capacity), -+ #[allow(unreachable_patterns)] -+ _ => return Err(StoreSnapshotCaptureError::UnsupportedBackend), -+ }; -+ debug_assert_eq!(mutable_globals.len(), mutable_capacity); -+ -+ Ok(StoreSnapshot { -+ repr: StoreSnapshotRepr::SparseV2(SparseStoreSnapshotV2 { -+ shape, -+ mutable_globals, -+ }), -+ }) -+} -+ -+#[cfg(any(feature = "sys", feature = "sys-minimal"))] -+fn validate_dense_destination( -+ store: &wasmer::sys::store::StoreObjects, -+ snapshot_len: usize, -+) -> Result<(), StoreSnapshotRestoreError> { -+ let store_len = store.iter_globals().len(); -+ if snapshot_len != store_len { -+ return Err(StoreSnapshotRestoreError::DenseShapeMismatch { -+ snapshot: snapshot_len, -+ store: store_len, -+ }); -+ } -+ if let Some((index, _)) = store -+ .iter_globals() -+ .enumerate() -+ .find(|(_, global)| global.ty().ty.is_ref()) -+ { -+ return Err(StoreSnapshotRestoreError::ReferenceGlobal { index }); -+ } -+ Ok(()) -+} -+ -+#[cfg(any(feature = "sys", feature = "sys-minimal"))] -+unsafe fn restore_dense_numeric_globals(store: &wasmer::sys::store::StoreObjects, values: &[u128]) { -+ for (global, value) in store.iter_globals().zip(values.iter().copied()) { -+ // SAFETY: the caller validated the complete destination shape, including the absence of -+ // reference globals, before performing any write. -+ unsafe { global.vmglobal().as_mut().val.u128 = value }; - } - } - --pub fn capture_store_snapshot(store: &mut impl wasmer::AsStoreMut) -> StoreSnapshot { -- let objs = store.objects_mut(); -- let globals = objs.as_u128_globals(); -- StoreSnapshot { globals } -+#[cfg(any(feature = "sys", feature = "sys-minimal"))] -+unsafe fn restore_sparse_numeric_globals( -+ store: &wasmer::sys::store::StoreObjects, -+ values: &[u128], -+) { -+ for (global, value) in store -+ .iter_globals() -+ .filter(|global| global.ty().mutability == Mutability::Var) -+ .zip(values.iter().copied()) -+ { -+ // SAFETY: the caller validated the exact descriptor hash, mutable count, and absence of -+ // mutable reference globals before performing any write. -+ unsafe { global.vmglobal().as_mut().val.u128 = value }; -+ } -+} -+ -+pub fn restore_store_snapshot( -+ store: &mut impl wasmer::AsStoreMut, -+ snapshot: &StoreSnapshot, -+) -> Result<(), StoreSnapshotRestoreError> { -+ let objects = store.objects_mut(); -+ match &snapshot.repr { -+ StoreSnapshotRepr::DenseV1(snapshot) => { -+ match objects { -+ #[cfg(any(feature = "sys", feature = "sys-minimal"))] -+ wasmer::StoreObjects::Sys(store) => { -+ validate_dense_destination(store, snapshot.globals.len())?; -+ // SAFETY: validation above rejects all reference-bearing destinations and -+ // verifies the value count before the first raw write. -+ unsafe { restore_dense_numeric_globals(store, &snapshot.globals) }; -+ } -+ #[allow(unreachable_patterns)] -+ _ => return Err(StoreSnapshotRestoreError::UnsupportedBackend), -+ } -+ Ok(()) -+ } -+ StoreSnapshotRepr::SparseV2(snapshot) => { -+ if u64::try_from(snapshot.mutable_globals.len()).ok() -+ != Some(snapshot.shape.mutable_global_count) -+ { -+ return Err(StoreSnapshotRestoreError::SparseValueCountMismatch); -+ } -+ let actual_shape = sparse_store_shape(objects).map_err(|err| match err { -+ StoreSnapshotCaptureError::MutableReferenceGlobal { index } => { -+ StoreSnapshotRestoreError::ReferenceGlobal { index } -+ } -+ StoreSnapshotCaptureError::UnsupportedBackend => { -+ StoreSnapshotRestoreError::UnsupportedBackend -+ } -+ })?; -+ if snapshot.shape != actual_shape { -+ return Err(StoreSnapshotRestoreError::SparseShapeMismatch); -+ } -+ match objects { -+ #[cfg(any(feature = "sys", feature = "sys-minimal"))] -+ wasmer::StoreObjects::Sys(store) => { -+ // SAFETY: value count and exact destination shape were validated before the -+ // first write, including rejection of every mutable reference global. -+ unsafe { restore_sparse_numeric_globals(store, &snapshot.mutable_globals) }; -+ } -+ #[allow(unreachable_patterns)] -+ _ => return Err(StoreSnapshotRestoreError::UnsupportedBackend), -+ } -+ Ok(()) -+ } -+ } - } - --pub fn restore_store_snapshot(store: &mut impl wasmer::AsStoreMut, snapshot: &StoreSnapshot) { -- let objs = store.objects_mut(); -+#[cfg(all(test, feature = "sys"))] -+mod tests { -+ use super::*; -+ use wasmer::{Global, Instance, Module, Store, Value, imports}; -+ -+ struct GlobalFixture { -+ store: Store, -+ imported: Global, -+ main: Global, -+ main_constant: Global, -+ side: Global, -+ } -+ -+ impl GlobalFixture { -+ fn new(imported: i32, main: i64, side: f64, extra_side_global: bool) -> Self { -+ let mut store = Store::default(); -+ let imported_global = Global::new_mut(&mut store, Value::I32(imported)); -+ let main_module = Module::new( -+ &store, -+ r#" -+ (module -+ (import "env" "imported" (global $imported (mut i32))) -+ (global (export "main") (mut i64) (i64.const 0)) -+ (global (export "main_constant") i32 (i32.const 77))) -+ "#, -+ ) -+ .unwrap(); -+ let main_instance = Instance::new( -+ &mut store, -+ &main_module, -+ &imports! { -+ "env" => { "imported" => imported_global.clone() } -+ }, -+ ) -+ .unwrap(); -+ let main_global = main_instance.exports.get_global("main").unwrap().clone(); -+ let main_constant = main_instance -+ .exports -+ .get_global("main_constant") -+ .unwrap() -+ .clone(); -+ main_global.set(&mut store, Value::I64(main)).unwrap(); -+ -+ let side_wat = if extra_side_global { -+ r#" -+ (module -+ (global (export "side") (mut f64) (f64.const 0)) -+ (global (mut i32) (i32.const 1))) -+ "# -+ } else { -+ r#" -+ (module -+ (global (export "side") (mut f64) (f64.const 0))) -+ "# -+ }; -+ let side_module = Module::new(&store, side_wat).unwrap(); -+ let side_instance = Instance::new(&mut store, &side_module, &imports! {}).unwrap(); -+ let side_global = side_instance.exports.get_global("side").unwrap().clone(); -+ side_global.set(&mut store, Value::F64(side)).unwrap(); -+ -+ Self { -+ store, -+ imported: imported_global, -+ main: main_global, -+ main_constant, -+ side: side_global, -+ } -+ } -+ -+ fn values(&mut self) -> (Value, Value, Value, Value) { -+ ( -+ self.imported.get(&mut self.store), -+ self.main.get(&mut self.store), -+ self.main_constant.get(&mut self.store), -+ self.side.get(&mut self.store), -+ ) -+ } -+ } -+ -+ #[test] -+ fn sparse_snapshot_restores_imported_main_and_side_module_globals_to_fresh_store() { -+ let mut source = GlobalFixture::new(11, 22, 33.5, false); -+ let snapshot = capture_store_snapshot(&mut source.store).unwrap(); -+ let StoreSnapshotRepr::SparseV2(sparse) = &snapshot.repr else { -+ panic!("sys snapshots must use the sparse V2 representation"); -+ }; -+ assert_eq!(sparse.shape.mutable_global_count, 3); -+ assert_eq!(sparse.mutable_globals.len(), 3); -+ assert_eq!( -+ sparse.mutable_globals.capacity() * std::mem::size_of::(), -+ 48 -+ ); -+ -+ let encoded = snapshot.serialize().unwrap(); -+ assert!(encoded.starts_with(STORE_SNAPSHOT_MAGIC)); -+ assert_eq!( -+ encoded[STORE_SNAPSHOT_MAGIC.len()], -+ STORE_SNAPSHOT_VERSION_SPARSE -+ ); -+ let decoded = StoreSnapshot::deserialize(&encoded).unwrap(); -+ -+ // A distinct Store and fresh instances model EXEC_BACKEND creation. The store-global -+ // order is the same, but every allocation and initial mutable value is different. -+ let mut destination = GlobalFixture::new(101, 202, 303.5, false); -+ restore_store_snapshot(&mut destination.store, &decoded).unwrap(); -+ assert_eq!( -+ destination.values(), -+ ( -+ Value::I32(11), -+ Value::I64(22), -+ Value::I32(77), -+ Value::F64(33.5) -+ ) -+ ); -+ } -+ -+ #[test] -+ fn sparse_snapshot_rejects_shape_mismatch_before_changing_any_global() { -+ let mut source = GlobalFixture::new(11, 22, 33.5, false); -+ let snapshot = capture_store_snapshot(&mut source.store).unwrap(); -+ let mut destination = GlobalFixture::new(101, 202, 303.5, true); -+ let before = destination.values(); -+ -+ assert_eq!( -+ restore_store_snapshot(&mut destination.store, &snapshot), -+ Err(StoreSnapshotRestoreError::SparseShapeMismatch) -+ ); -+ assert_eq!(destination.values(), before); -+ } -+ -+ #[test] -+ fn const_heavy_store_allocation_scales_with_mutable_globals_only() { -+ let mut store = Store::default(); -+ for value in 0..10_907_i32 { -+ let _ = Global::new(&mut store, Value::I32(value)); -+ } -+ for value in 0..4_i32 { -+ let _ = Global::new_mut(&mut store, Value::I32(value)); -+ } -+ -+ let snapshot = capture_store_snapshot(&mut store).unwrap(); -+ let StoreSnapshotRepr::SparseV2(sparse) = &snapshot.repr else { -+ panic!("sys snapshots must use the sparse V2 representation"); -+ }; -+ assert_eq!(sparse.shape.global_count, 10_911); -+ assert_eq!(sparse.shape.mutable_global_count, 4); -+ assert_eq!(sparse.mutable_globals.len(), 4); -+ assert_eq!( -+ sparse.mutable_globals.capacity() * std::mem::size_of::(), -+ 64 -+ ); -+ assert_eq!(snapshot.serialize().unwrap().len(), 129); -+ assert_eq!(10_911 * std::mem::size_of::(), 174_576); -+ } -+ -+ #[test] -+ fn persisted_unversioned_dense_snapshot_still_decodes_and_restores() { -+ let legacy = DenseStoreSnapshotV1 { -+ globals: vec![11_u128, 22_u128, 77_u128, 33.5_f64.to_bits().into()], -+ }; -+ let persisted_bytes = bincode::serde::encode_to_vec(&legacy, config::legacy()).unwrap(); -+ assert!(!persisted_bytes.starts_with(STORE_SNAPSHOT_MAGIC)); -+ -+ let decoded = StoreSnapshot::deserialize(&persisted_bytes).unwrap(); -+ assert!(matches!(decoded.repr, StoreSnapshotRepr::DenseV1(_))); -+ // A decode/encode cycle does not silently migrate or invalidate journal bytes. -+ assert_eq!(decoded.serialize().unwrap(), persisted_bytes); -+ -+ let mut destination = GlobalFixture::new(101, 202, 303.5, false); -+ restore_store_snapshot(&mut destination.store, &decoded).unwrap(); -+ assert_eq!( -+ destination.values(), -+ ( -+ Value::I32(11), -+ Value::I64(22), -+ Value::I32(77), -+ Value::F64(33.5) -+ ) -+ ); -+ } -+ -+ #[test] -+ fn capture_rejects_mutable_reference_global_before_raw_read() { -+ let mut store = Store::default(); -+ let _numeric = Global::new_mut(&mut store, Value::I32(7)); -+ let _reference = Global::new_mut(&mut store, Value::ExternRef(None)); -+ -+ assert!(matches!( -+ capture_store_snapshot(&mut store), -+ Err(StoreSnapshotCaptureError::MutableReferenceGlobal { index: 1 }) -+ )); -+ } -+ -+ #[test] -+ fn dense_restore_rejects_reference_shape_before_any_write() { -+ let mut store = Store::default(); -+ let numeric = Global::new_mut(&mut store, Value::I32(7)); -+ let _reference = Global::new_mut(&mut store, Value::ExternRef(None)); -+ let snapshot = StoreSnapshot { -+ repr: StoreSnapshotRepr::DenseV1(DenseStoreSnapshotV1 { -+ globals: vec![99, 0], -+ }), -+ }; - -- for (index, value) in snapshot.globals.iter().enumerate() { -- objs.set_global_unchecked(index, *value); -+ assert_eq!( -+ restore_store_snapshot(&mut store, &snapshot), -+ Err(StoreSnapshotRestoreError::ReferenceGlobal { index: 1 }) -+ ); -+ assert_eq!(numeric.get(&mut store), Value::I32(7)); - } - } -diff --git a/tests/compilers/issues.rs b/tests/compilers/issues.rs -index 271bd12..6379ce0 100644 ---- a/tests/compilers/issues.rs -+++ b/tests/compilers/issues.rs -@@ -636,6 +636,83 @@ fn compiler_debug_dir_test(mut config: crate::Config) { - assert!(Module::new(&store, wat).is_ok()); - } - -+#[cfg(feature = "llvm")] -+#[test] -+fn llvm_rotates_and_atomic_fence_emit_expected_ir() { -+ use std::path::Path; -+ use tempfile::TempDir; -+ use wasmer_compiler::{CompilerConfig, EngineBuilder}; -+ use wasmer_compiler_llvm::LLVMCallbacks; -+ -+ fn collect_postopt_ir(dir: &Path, out: &mut String) { -+ for entry in std::fs::read_dir(dir).expect("debug directory must be readable") { -+ let entry = entry.expect("debug directory entry must be readable"); -+ let path = entry.path(); -+ if path.is_dir() { -+ collect_postopt_ir(&path, out); -+ } else if path -+ .file_name() -+ .and_then(|name| name.to_str()) -+ .is_some_and(|name| name.ends_with(".postopt.ll")) -+ { -+ out.push_str( -+ &std::fs::read_to_string(&path) -+ .unwrap_or_else(|err| panic!("cannot read {}: {err}", path.display())), -+ ); -+ } -+ } -+ } -+ -+ let mut compiler_config = wasmer_compiler_llvm::LLVM::new(); -+ let temp = TempDir::new().expect("temp folder creation failed"); -+ compiler_config.callbacks(Some(LLVMCallbacks::new(temp.path().to_path_buf()).unwrap())); -+ let store = Store::new(EngineBuilder::new(compiler_config)); -+ -+ let wat = r#" -+ (module -+ (memory 1 1 shared) -+ (func $fence (export "fence") -+ atomic.fence) -+ (func $rotl32 (export "rotl32") (param i32 i32) (result i32) -+ local.get 0 -+ local.get 1 -+ i32.rotl) -+ (func $rotr32 (export "rotr32") (param i32 i32) (result i32) -+ local.get 0 -+ local.get 1 -+ i32.rotr) -+ (func $rotl64 (export "rotl64") (param i64 i64) (result i64) -+ local.get 0 -+ local.get 1 -+ i64.rotl) -+ (func $rotr64 (export "rotr64") (param i64 i64) (result i64) -+ local.get 0 -+ local.get 1 -+ i64.rotr)) -+ "#; -+ -+ Module::new(&store, wat).expect("rotate module must compile"); -+ -+ let mut ir = String::new(); -+ collect_postopt_ir(temp.path(), &mut ir); -+ -+ assert!(ir.contains("call i32 @llvm.fshl.i32")); -+ assert!(ir.contains("call i32 @llvm.fshr.i32")); -+ assert!(ir.contains("call i64 @llvm.fshl.i64")); -+ assert!(ir.contains("call i64 @llvm.fshr.i64")); -+ assert!(ir.contains("fence seq_cst")); -+ -+ let volatile_id = Box::new(wasmer_compiler_llvm::LLVM::new()) -+ .compiler() -+ .deterministic_id(); -+ let mut nonvolatile_config = wasmer_compiler_llvm::LLVM::new(); -+ nonvolatile_config.non_volatile_memops(true); -+ let nonvolatile_id = Box::new(nonvolatile_config).compiler().deterministic_id(); -+ assert_ne!(volatile_id, nonvolatile_id); -+ assert!(volatile_id.contains("-nv0-")); -+ assert!(nonvolatile_id.contains("-nv1-")); -+} -+ - #[compiler_test(issues)] - fn issue_5795_memory_reset_size(mut config: crate::Config) { - let wasm_bytes = wat2wasm( diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0002-virtual-fs-file-description-and-writeback.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0002-virtual-fs-file-description-and-writeback.patch new file mode 100644 index 000000000..ff33fb05d --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0002-virtual-fs-file-description-and-writeback.patch @@ -0,0 +1,2089 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 2/9] virtual-fs: represent file descriptions and writeback + +Preserve open-file-description state and host filesystem operations needed by +PostgreSQL shared mappings, positional I/O, explicit durability and descriptor +finalization. Wrapper files must delegate the same semantics as the underlying +file. This group supplies the filesystem API consumed by the WASIX FD patch; +generic corrections are candidates for independent upstream review. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/lib/virtual-fs/src/arc_fs.rs b/lib/virtual-fs/src/arc_fs.rs +index 507bbc1..d384d2a 100644 +--- a/lib/virtual-fs/src/arc_fs.rs ++++ b/lib/virtual-fs/src/arc_fs.rs +@@ -58,6 +58,10 @@ impl FileSystem for ArcFileSystem { + self.fs.remove_file(path) + } + ++ fn open_dir(&self, path: &Path) -> Result> { ++ self.fs.open_dir(path) ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_> { + self.fs.new_open_options() + } +diff --git a/lib/virtual-fs/src/host_fs.rs b/lib/virtual-fs/src/host_fs.rs +index 687f6ce..38062d1 100644 +--- a/lib/virtual-fs/src/host_fs.rs ++++ b/lib/virtual-fs/src/host_fs.rs +@@ -1,6 +1,6 @@ + use crate::{ +- DirEntry, FileType, FsError, Metadata, OpenOptions, OpenOptionsConfig, ReadDir, Result, +- VirtualFile, ++ DirEntry, FileAdvice, FileType, FileWritebackFlags, FsError, Metadata, OpenOptions, ++ OpenOptionsConfig, ReadDir, Result, VirtualDirectory, VirtualFile, + }; + use bytes::{Buf, Bytes}; + use futures::future::BoxFuture; +@@ -9,9 +9,13 @@ use serde::{Deserialize, Serialize, de}; + use std::convert::TryInto; + use std::fs; + use std::io::{self, Seek}; ++#[cfg(not(any(unix, windows)))] ++use std::io::{Read, Write}; + use std::path::{Component, Path, PathBuf}; + use std::pin::Pin; + use std::sync::Arc; ++#[cfg(unix)] ++use std::sync::OnceLock; + use std::task::{Context, Poll}; + use std::time::{SystemTime, UNIX_EPOCH}; + use tokio::fs as tfs; +@@ -221,6 +225,28 @@ impl crate::FileSystem for FileSystem { + fs::remove_file(path).map_err(Into::into) + } + ++ fn open_dir(&self, path: &Path) -> Result> { ++ #[cfg(unix)] ++ { ++ use std::os::unix::fs::OpenOptionsExt; ++ ++ let host_path = self.prepare_path(path)?; ++ let file = fs::OpenOptions::new() ++ .read(true) ++ // O_DIRECTORY rejects non-directories while retaining a real ++ // read descriptor that supports fsync (unlike Linux O_PATH). ++ .custom_flags(libc::O_DIRECTORY) ++ .open(&host_path)?; ++ Ok(Box::new(Directory { file })) ++ } ++ ++ #[cfg(not(unix))] ++ { ++ let _ = path; ++ Err(FsError::Unsupported) ++ } ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_> { + OpenOptions::new(self) + } +@@ -242,6 +268,27 @@ impl crate::FileSystem for FileSystem { + } + } + ++#[cfg(unix)] ++#[derive(Debug)] ++struct Directory { ++ file: fs::File, ++} ++ ++#[cfg(unix)] ++impl VirtualDirectory for Directory { ++ fn sync_all_to_disk(&self) -> BoxFuture<'_, io::Result<()>> { ++ Box::pin(async move { host_sync_all(&self.file) }) ++ } ++ ++ fn has_blocking_sync_all_to_disk(&self) -> bool { ++ true ++ } ++ ++ fn sync_all_to_disk_blocking(&self) -> io::Result<()> { ++ host_sync_all(&self.file) ++ } ++} ++ + impl TryInto for std::fs::Metadata { + type Error = io::Error; + +@@ -311,6 +358,7 @@ impl crate::FileOpener for FileSystem { + let append = if conf.truncate { false } else { conf.append() }; + + let mut oo = fs::OpenOptions::new(); ++ let (sync, data_sync) = apply_sync_open_options(&mut oo, conf.sync(), conf.data_sync()); + oo.read(conf.read()) + .write(conf.write()) + .create_new(conf.create_new()) +@@ -327,11 +375,42 @@ impl crate::FileOpener for FileSystem { + read, + write, + append, ++ sync, ++ data_sync, + )) as Box + }) + } + } + ++#[cfg(unix)] ++fn apply_sync_open_options( ++ options: &mut fs::OpenOptions, ++ sync: bool, ++ data_sync: bool, ++) -> (bool, bool) { ++ use std::os::unix::fs::OpenOptionsExt; ++ ++ let mut flags = 0; ++ if sync { ++ flags |= libc::O_SYNC; ++ } else if data_sync { ++ flags |= libc::O_DSYNC; ++ } ++ if flags != 0 { ++ options.custom_flags(flags); ++ } ++ (sync, !sync && data_sync) ++} ++ ++#[cfg(not(unix))] ++fn apply_sync_open_options( ++ _options: &mut fs::OpenOptions, ++ _sync: bool, ++ _data_sync: bool, ++) -> (bool, bool) { ++ (false, false) ++} ++ + /// A thin wrapper around `std::fs::File` + #[derive(Debug)] + #[cfg_attr(feature = "enable-serde", derive(Serialize))] +@@ -339,11 +418,12 @@ pub struct File { + #[cfg_attr(feature = "enable-serde", serde(skip, default = "default_handle"))] + handle: Handle, + #[cfg_attr(feature = "enable-serde", serde(skip))] +- inner: tfs::File, ++ inner: Option, + #[cfg_attr(feature = "enable-serde", serde(skip_serializing))] + inner_std: fs::File, + pub host_path: PathBuf, +- #[cfg(feature = "enable-serde")] ++ sync: bool, ++ data_sync: bool, + flags: u16, + } + +@@ -379,17 +459,26 @@ impl<'de> Deserialize<'de> for File { + let flags = seq + .next_element()? + .ok_or_else(|| de::Error::invalid_length(1, &self))?; +- let inner = fs::OpenOptions::new() +- .read(flags & File::READ != 0) +- .write(flags & File::WRITE != 0) +- .append(flags & File::APPEND != 0) ++ let read = flags & File::READ != 0; ++ let write = flags & File::WRITE != 0; ++ let append = flags & File::APPEND != 0; ++ let sync = flags & File::SYNC != 0; ++ let data_sync = flags & File::DATA_SYNC != 0; ++ let mut open_options = fs::OpenOptions::new(); ++ let (sync, data_sync) = apply_sync_open_options(&mut open_options, sync, data_sync); ++ let inner = open_options ++ .read(read) ++ .write(write) ++ .append(append) + .open(&host_path) + .map_err(|_| de::Error::custom("Could not open file on this system"))?; + Ok(File { + handle: Handle::current(), +- inner: tokio::fs::File::from_std(inner.try_clone().unwrap()), ++ inner: None, + inner_std: inner, + host_path, ++ sync, ++ data_sync, + flags, + }) + } +@@ -418,17 +507,26 @@ impl<'de> Deserialize<'de> for File { + } + let host_path = host_path.ok_or_else(|| de::Error::missing_field("host_path"))?; + let flags = flags.ok_or_else(|| de::Error::missing_field("flags"))?; +- let inner = fs::OpenOptions::new() +- .read(flags & File::READ != 0) +- .write(flags & File::WRITE != 0) +- .append(flags & File::APPEND != 0) ++ let read = flags & File::READ != 0; ++ let write = flags & File::WRITE != 0; ++ let append = flags & File::APPEND != 0; ++ let sync = flags & File::SYNC != 0; ++ let data_sync = flags & File::DATA_SYNC != 0; ++ let mut open_options = fs::OpenOptions::new(); ++ let (sync, data_sync) = apply_sync_open_options(&mut open_options, sync, data_sync); ++ let inner = open_options ++ .read(read) ++ .write(write) ++ .append(append) + .open(&host_path) + .map_err(|_| de::Error::custom("Could not open file on this system"))?; + Ok(File { + handle: Handle::current(), +- inner: tokio::fs::File::from_std(inner.try_clone().unwrap()), ++ inner: None, + inner_std: inner, + host_path, ++ sync, ++ data_sync, + flags, + }) + } +@@ -443,6 +541,8 @@ impl File { + const READ: u16 = 1; + const WRITE: u16 = 2; + const APPEND: u16 = 4; ++ const SYNC: u16 = 8; ++ const DATA_SYNC: u16 = 16; + + /// creates a new host file from a `std::fs::File` and a path + pub fn new( +@@ -452,9 +552,10 @@ impl File { + read: bool, + write: bool, + append: bool, ++ sync: bool, ++ data_sync: bool, + ) -> Self { + let mut _flags = 0; +- + if read { + _flags |= Self::READ; + } +@@ -467,13 +568,21 @@ impl File { + _flags |= Self::APPEND; + } + +- let async_file = tfs::File::from_std(file.try_clone().unwrap()); ++ if sync { ++ _flags |= Self::SYNC; ++ } ++ ++ if data_sync { ++ _flags |= Self::DATA_SYNC; ++ } ++ + Self { + handle, + inner_std: file, +- inner: async_file, ++ inner: None, + host_path, +- #[cfg(feature = "enable-serde")] ++ sync, ++ data_sync, + flags: _flags, + } + } +@@ -482,6 +591,370 @@ impl File { + // FIXME: no unwrap! + self.inner_std.metadata().unwrap() + } ++ ++ pub fn try_clone_std_file(&self) -> io::Result { ++ self.inner_std.try_clone() ++ } ++ ++ /// Materializes the Tokio view only for callers that actually use the ++ /// asynchronous I/O traits. Most WASIX host-file operations use the ++ /// blocking/positioned fast paths and therefore need only one host fd. ++ fn async_file(&mut self) -> io::Result<&mut tfs::File> { ++ if self.inner.is_none() { ++ self.inner = Some(tfs::File::from_std(self.inner_std.try_clone()?)); ++ } ++ Ok(self.inner.as_mut().unwrap()) ++ } ++} ++ ++#[cfg(target_os = "linux")] ++fn host_file_advise(file: &fs::File, offset: u64, len: u64, advice: FileAdvice) -> io::Result<()> { ++ use std::os::fd::AsRawFd; ++ ++ let range_error = || io::Error::from_raw_os_error(libc::EOVERFLOW); ++ let _end = offset.checked_add(len).ok_or_else(range_error)?; ++ let offset: libc::off_t = offset.try_into().map_err(|_| range_error())?; ++ let len: libc::off_t = len.try_into().map_err(|_| range_error())?; ++ let advice = match advice { ++ FileAdvice::Normal => libc::POSIX_FADV_NORMAL, ++ FileAdvice::Sequential => libc::POSIX_FADV_SEQUENTIAL, ++ FileAdvice::Random => libc::POSIX_FADV_RANDOM, ++ FileAdvice::WillNeed => libc::POSIX_FADV_WILLNEED, ++ FileAdvice::DontNeed => libc::POSIX_FADV_DONTNEED, ++ FileAdvice::NoReuse => libc::POSIX_FADV_NOREUSE, ++ }; ++ ++ // Unlike most libc calls, posix_fadvise returns the errno value directly ++ // and does not communicate failures through the thread-local errno slot. ++ let result = unsafe { libc::posix_fadvise(file.as_raw_fd(), offset, len, advice) }; ++ if result == 0 { ++ Ok(()) ++ } else { ++ Err(io::Error::from_raw_os_error(result)) ++ } ++} ++ ++#[cfg(not(target_os = "linux"))] ++fn host_file_advise( ++ _file: &fs::File, ++ _offset: u64, ++ _len: u64, ++ _advice: FileAdvice, ++) -> io::Result<()> { ++ Err(io::ErrorKind::Unsupported.into()) ++} ++ ++#[cfg(target_os = "linux")] ++fn host_file_writeback_range( ++ file: &fs::File, ++ offset: u64, ++ len: u64, ++ flags: FileWritebackFlags, ++) -> io::Result<()> { ++ use std::os::fd::AsRawFd; ++ ++ let representation_error = || io::Error::from_raw_os_error(libc::EOVERFLOW); ++ let invalid_range = || io::Error::from_raw_os_error(libc::EINVAL); ++ let offset: libc::off64_t = offset.try_into().map_err(|_| representation_error())?; ++ let len: libc::off64_t = len.try_into().map_err(|_| representation_error())?; ++ ++ // Linux requires a signed, representable exclusive end. A zero length ++ // means from offset through EOF and therefore has no finite end to check. ++ if len > 0 { ++ let end = offset.checked_add(len).ok_or_else(invalid_range)?; ++ let _end: libc::off64_t = end; ++ } ++ ++ let mut host_flags = 0; ++ if flags.contains(FileWritebackFlags::WAIT_BEFORE) { ++ host_flags |= libc::SYNC_FILE_RANGE_WAIT_BEFORE; ++ } ++ if flags.contains(FileWritebackFlags::WRITE) { ++ host_flags |= libc::SYNC_FILE_RANGE_WRITE; ++ } ++ if flags.contains(FileWritebackFlags::WAIT_AFTER) { ++ host_flags |= libc::SYNC_FILE_RANGE_WAIT_AFTER; ++ } ++ ++ let result = unsafe { libc::sync_file_range(file.as_raw_fd(), offset, len, host_flags) }; ++ if result == 0 { ++ Ok(()) ++ } else { ++ Err(io::Error::last_os_error()) ++ } ++} ++ ++#[cfg(not(target_os = "linux"))] ++fn host_file_writeback_range( ++ _file: &fs::File, ++ _offset: u64, ++ _len: u64, ++ _flags: FileWritebackFlags, ++) -> io::Result<()> { ++ Err(io::ErrorKind::Unsupported.into()) ++} ++ ++#[cfg(unix)] ++fn positioned_read(file: &fs::File, buf: &mut [u8], offset: u64) -> io::Result { ++ use std::os::unix::fs::FileExt; ++ file.read_at(buf, offset) ++} ++ ++#[cfg(unix)] ++fn positioned_write(file: &fs::File, buf: &[u8], offset: u64) -> io::Result { ++ use std::os::unix::fs::FileExt; ++ file.write_at(buf, offset) ++} ++ ++#[cfg(unix)] ++fn positioned_write_vectored( ++ file: &fs::File, ++ bufs: &[io::IoSlice<'_>], ++ offset: u64, ++) -> io::Result { ++ use std::os::unix::io::AsRawFd; ++ ++ let offset: libc::off_t = offset ++ .try_into() ++ .map_err(|_| io::Error::from(io::ErrorKind::InvalidInput))?; ++ let iovcnt: libc::c_int = bufs ++ .len() ++ .try_into() ++ .map_err(|_| io::Error::from(io::ErrorKind::InvalidInput))?; ++ let raw_iov = bufs ++ .iter() ++ .map(|buf| { ++ let bytes: &[u8] = buf; ++ libc::iovec { ++ iov_base: bytes.as_ptr() as *mut libc::c_void, ++ iov_len: bytes.len(), ++ } ++ }) ++ .collect::>(); ++ ++ loop { ++ let rc = unsafe { libc::pwritev(file.as_raw_fd(), raw_iov.as_ptr(), iovcnt, offset) }; ++ if rc >= 0 { ++ return Ok(rc as usize); ++ } ++ let err = io::Error::last_os_error(); ++ if err.raw_os_error() == Some(libc::EINTR) { ++ continue; ++ } ++ return Err(err); ++ } ++} ++ ++const ZERO_WRITE_CHUNK_LEN: usize = 8192; ++static ZERO_WRITE_CHUNK: [u8; ZERO_WRITE_CHUNK_LEN] = [0; ZERO_WRITE_CHUNK_LEN]; ++ ++fn try_extend_with_zeroes(file: &fs::File, len: u64, offset: u64) -> io::Result> { ++ let written = usize::try_from(len).map_err(|_| io::Error::from(io::ErrorKind::InvalidInput))?; ++ let end = offset ++ .checked_add(len) ++ .ok_or_else(|| io::Error::from(io::ErrorKind::InvalidInput))?; ++ ++ if len == 0 { ++ return Ok(Some(0)); ++ } ++ ++ if offset >= file.metadata()?.len() { ++ file.set_len(end)?; ++ return Ok(Some(written)); ++ } ++ ++ Ok(None) ++} ++ ++#[cfg(unix)] ++fn host_iov_max() -> usize { ++ static IOV_MAX: OnceLock = OnceLock::new(); ++ *IOV_MAX.get_or_init(|| { ++ let max = unsafe { libc::sysconf(libc::_SC_IOV_MAX) }; ++ if max > 0 { max as usize } else { 1024 } ++ }) ++} ++ ++#[cfg(unix)] ++fn positioned_write_zeroes(file: &fs::File, len: u64, offset: u64) -> io::Result { ++ if let Some(written) = try_extend_with_zeroes(file, len, offset)? { ++ return Ok(written); ++ } ++ ++ let mut written = 0usize; ++ let mut remaining = len; ++ let max_batch = host_iov_max().max(1) * ZERO_WRITE_CHUNK_LEN; ++ ++ while remaining > 0 { ++ let batch_len = (remaining.min(max_batch as u64)) as usize; ++ let full_chunks = batch_len / ZERO_WRITE_CHUNK_LEN; ++ let tail = batch_len % ZERO_WRITE_CHUNK_LEN; ++ let mut bufs = Vec::with_capacity(full_chunks + usize::from(tail > 0)); ++ for _ in 0..full_chunks { ++ bufs.push(io::IoSlice::new(&ZERO_WRITE_CHUNK)); ++ } ++ if tail > 0 { ++ bufs.push(io::IoSlice::new(&ZERO_WRITE_CHUNK[..tail])); ++ } ++ ++ let local = positioned_write_vectored(file, &bufs, offset + written as u64)?; ++ if local == 0 { ++ break; ++ } ++ written = written ++ .checked_add(local) ++ .ok_or_else(|| io::Error::from(io::ErrorKind::InvalidData))?; ++ if local < batch_len { ++ break; ++ } ++ remaining -= local as u64; ++ } ++ ++ Ok(written) ++} ++ ++#[cfg(windows)] ++fn positioned_read(file: &fs::File, buf: &mut [u8], offset: u64) -> io::Result { ++ use std::os::windows::fs::FileExt; ++ file.seek_read(buf, offset) ++} ++ ++#[cfg(windows)] ++fn positioned_write(file: &fs::File, buf: &[u8], offset: u64) -> io::Result { ++ use std::os::windows::fs::FileExt; ++ file.seek_write(buf, offset) ++} ++ ++#[cfg(not(any(unix, windows)))] ++fn positioned_read(file: &mut fs::File, buf: &mut [u8], offset: u64) -> io::Result { ++ let cursor = file.stream_position()?; ++ file.seek(io::SeekFrom::Start(offset))?; ++ let read_result = file.read(buf); ++ let restore_result = file.seek(io::SeekFrom::Start(cursor)); ++ ++ match (read_result, restore_result) { ++ (Ok(read), Ok(_)) => Ok(read), ++ (Err(err), _) => Err(err), ++ (Ok(_), Err(err)) => Err(err), ++ } ++} ++ ++#[cfg(not(any(unix, windows)))] ++fn positioned_write(file: &mut fs::File, buf: &[u8], offset: u64) -> io::Result { ++ let cursor = file.stream_position()?; ++ file.seek(io::SeekFrom::Start(offset))?; ++ let write_result = file.write(buf); ++ let restore_result = file.seek(io::SeekFrom::Start(cursor)); ++ ++ match (write_result, restore_result) { ++ (Ok(written), Ok(_)) => Ok(written), ++ (Err(err), _) => Err(err), ++ (Ok(_), Err(err)) => Err(err), ++ } ++} ++ ++#[cfg(not(unix))] ++fn positioned_write_zeroes(file: &mut fs::File, len: u64, offset: u64) -> io::Result { ++ if let Some(written) = try_extend_with_zeroes(file, len, offset)? { ++ return Ok(written); ++ } ++ ++ let mut written = 0usize; ++ while (written as u64) < len { ++ let remaining = (len - written as u64) as usize; ++ let local_len = remaining.min(ZERO_WRITE_CHUNK_LEN); ++ let local = positioned_write( ++ file, ++ &ZERO_WRITE_CHUNK[..local_len], ++ offset + written as u64, ++ )?; ++ if local == 0 { ++ break; ++ } ++ written = written ++ .checked_add(local) ++ .ok_or_else(|| io::Error::from(io::ErrorKind::InvalidData))?; ++ if local < local_len { ++ break; ++ } ++ } ++ Ok(written) ++} ++ ++#[cfg(unix)] ++fn cvt_sync_result(mut sync: impl FnMut() -> libc::c_int) -> io::Result<()> { ++ loop { ++ if sync() == 0 { ++ return Ok(()); ++ } ++ let err = io::Error::last_os_error(); ++ if err.raw_os_error() == Some(libc::EINTR) { ++ continue; ++ } ++ return Err(err); ++ } ++} ++ ++#[cfg(unix)] ++fn host_sync_all(file: &fs::File) -> io::Result<()> { ++ use std::os::unix::io::AsRawFd; ++ ++ cvt_sync_result(|| unsafe { libc::fsync(file.as_raw_fd()) }) ++} ++ ++#[cfg(any( ++ target_os = "android", ++ target_os = "freebsd", ++ target_os = "fuchsia", ++ target_os = "hurd", ++ target_os = "linux", ++ target_os = "netbsd", ++ target_os = "nto", ++ target_os = "openbsd", ++))] ++fn host_sync_data(file: &fs::File) -> io::Result<()> { ++ use std::os::unix::io::AsRawFd; ++ ++ cvt_sync_result(|| unsafe { libc::fdatasync(file.as_raw_fd()) }) ++} ++ ++#[cfg(target_vendor = "apple")] ++fn host_sync_data(file: &fs::File) -> io::Result<()> { ++ use std::os::unix::io::AsRawFd; ++ ++ unsafe extern "C" { ++ fn fdatasync(fd: libc::c_int) -> libc::c_int; ++ } ++ ++ cvt_sync_result(|| unsafe { fdatasync(file.as_raw_fd()) }) ++} ++ ++#[cfg(all( ++ unix, ++ not(any( ++ target_os = "android", ++ target_os = "freebsd", ++ target_os = "fuchsia", ++ target_os = "hurd", ++ target_os = "linux", ++ target_os = "netbsd", ++ target_os = "nto", ++ target_os = "openbsd", ++ target_vendor = "apple", ++ )) ++))] ++fn host_sync_data(file: &fs::File) -> io::Result<()> { ++ host_sync_all(file) ++} ++ ++#[cfg(not(unix))] ++fn host_sync_all(file: &fs::File) -> io::Result<()> { ++ file.sync_all() ++} ++ ++#[cfg(not(unix))] ++fn host_sync_data(file: &fs::File) -> io::Result<()> { ++ file.sync_data() + } + + //#[cfg_attr(feature = "enable-serde", typetag::serde)] +@@ -538,6 +1011,14 @@ impl VirtualFile for File { + None + } + ++ fn open_read(&self) -> Option { ++ Some(self.flags & Self::READ != 0) ++ } ++ ++ fn open_write(&self) -> Option { ++ Some(self.flags & (Self::WRITE | Self::APPEND) != 0) ++ } ++ + fn poll_read_ready(mut self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll> { + let cursor = match self.inner_std.stream_position() { + Ok(a) => a, +@@ -556,6 +1037,221 @@ impl VirtualFile for File { + fn poll_write_ready(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll> { + Poll::Ready(Ok(8192)) + } ++ ++ fn has_blocking_seek(&self) -> bool { ++ true ++ } ++ ++ fn seek_blocking(&mut self, pos: io::SeekFrom) -> io::Result { ++ std::io::Seek::seek(&mut self.inner_std, pos) ++ } ++ ++ fn has_blocking_read(&self) -> bool { ++ true ++ } ++ ++ fn read_blocking(&mut self, buf: &mut [u8]) -> io::Result { ++ std::io::Read::read(&mut self.inner_std, buf) ++ } ++ ++ fn has_blocking_read_at(&self) -> bool { ++ true ++ } ++ ++ fn read_at_blocking(&mut self, buf: &mut [u8], offset: u64) -> io::Result { ++ #[cfg(any(unix, windows))] ++ { ++ positioned_read(&self.inner_std, buf, offset) ++ } ++ #[cfg(not(any(unix, windows)))] ++ { ++ positioned_read(&mut self.inner_std, buf, offset) ++ } ++ } ++ ++ fn has_blocking_read_at_shared(&self) -> bool { ++ cfg!(any(unix, windows)) ++ } ++ ++ fn read_at_blocking_shared(&self, buf: &mut [u8], offset: u64) -> io::Result { ++ #[cfg(any(unix, windows))] ++ { ++ positioned_read(&self.inner_std, buf, offset) ++ } ++ #[cfg(not(any(unix, windows)))] ++ { ++ let _ = buf; ++ let _ = offset; ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ } ++ ++ fn has_blocking_write(&self) -> bool { ++ true ++ } ++ ++ fn write_blocking(&mut self, buf: &[u8]) -> io::Result { ++ std::io::Write::write(&mut self.inner_std, buf) ++ } ++ ++ fn write_at<'a>(&'a mut self, buf: &'a [u8], offset: u64) -> BoxFuture<'a, io::Result> { ++ Box::pin(async move { ++ #[cfg(any(unix, windows))] ++ { ++ positioned_write(&self.inner_std, buf, offset) ++ } ++ #[cfg(not(any(unix, windows)))] ++ { ++ positioned_write(&mut self.inner_std, buf, offset) ++ } ++ }) ++ } ++ ++ fn has_blocking_write_at(&self) -> bool { ++ true ++ } ++ ++ fn write_at_blocking(&mut self, buf: &[u8], offset: u64) -> io::Result { ++ #[cfg(any(unix, windows))] ++ { ++ positioned_write(&self.inner_std, buf, offset) ++ } ++ #[cfg(not(any(unix, windows)))] ++ { ++ positioned_write(&mut self.inner_std, buf, offset) ++ } ++ } ++ ++ fn has_blocking_write_at_shared(&self) -> bool { ++ cfg!(any(unix, windows)) ++ } ++ ++ fn write_at_blocking_shared(&self, buf: &[u8], offset: u64) -> io::Result { ++ #[cfg(any(unix, windows))] ++ { ++ positioned_write(&self.inner_std, buf, offset) ++ } ++ #[cfg(not(any(unix, windows)))] ++ { ++ let _ = buf; ++ let _ = offset; ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ } ++ ++ fn has_blocking_write_vectored_at(&self) -> bool { ++ cfg!(unix) ++ } ++ ++ fn write_vectored_at_blocking( ++ &mut self, ++ bufs: &[io::IoSlice<'_>], ++ offset: u64, ++ ) -> io::Result { ++ #[cfg(unix)] ++ { ++ positioned_write_vectored(&self.inner_std, bufs, offset) ++ } ++ #[cfg(not(unix))] ++ { ++ let _ = bufs; ++ let _ = offset; ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ } ++ ++ fn has_blocking_write_vectored_at_shared(&self) -> bool { ++ cfg!(unix) ++ } ++ ++ fn write_vectored_at_blocking_shared( ++ &self, ++ bufs: &[io::IoSlice<'_>], ++ offset: u64, ++ ) -> io::Result { ++ #[cfg(unix)] ++ { ++ positioned_write_vectored(&self.inner_std, bufs, offset) ++ } ++ #[cfg(not(unix))] ++ { ++ let _ = bufs; ++ let _ = offset; ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ } ++ ++ fn has_blocking_write_zeroes_at(&self) -> bool { ++ true ++ } ++ ++ fn write_zeroes_at_blocking(&mut self, len: u64, offset: u64) -> io::Result { ++ #[cfg(unix)] ++ { ++ positioned_write_zeroes(&self.inner_std, len, offset) ++ } ++ #[cfg(not(unix))] ++ { ++ positioned_write_zeroes(&mut self.inner_std, len, offset) ++ } ++ } ++ ++ fn has_blocking_write_zeroes_at_shared(&self) -> bool { ++ cfg!(unix) ++ } ++ ++ fn write_zeroes_at_blocking_shared(&self, len: u64, offset: u64) -> io::Result { ++ #[cfg(unix)] ++ { ++ positioned_write_zeroes(&self.inner_std, len, offset) ++ } ++ #[cfg(not(unix))] ++ { ++ let _ = len; ++ let _ = offset; ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ } ++ ++ fn sync_data_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { ++ Box::pin(async move { host_sync_data(&self.inner_std) }) ++ } ++ ++ fn sync_all_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { ++ Box::pin(async move { host_sync_all(&self.inner_std) }) ++ } ++ ++ fn has_blocking_sync_data_to_disk(&self) -> bool { ++ true ++ } ++ ++ fn sync_data_to_disk_blocking(&mut self) -> io::Result<()> { ++ host_sync_data(&self.inner_std) ++ } ++ ++ fn has_blocking_sync_all_to_disk(&self) -> bool { ++ true ++ } ++ ++ fn sync_all_to_disk_blocking(&mut self) -> io::Result<()> { ++ host_sync_all(&self.inner_std) ++ } ++ ++ fn is_sync_on_write(&self, sync_metadata: bool) -> bool { ++ if sync_metadata { ++ self.sync ++ } else { ++ self.sync || self.data_sync ++ } ++ } ++ ++ fn advise(&self, offset: u64, len: u64, advice: FileAdvice) -> io::Result<()> { ++ host_file_advise(&self.inner_std, offset, len, advice) ++ } ++ ++ fn writeback_range(&self, offset: u64, len: u64, flags: FileWritebackFlags) -> io::Result<()> { ++ host_file_writeback_range(&self.inner_std, offset, len, flags) ++ } + } + + impl AsyncRead for File { +@@ -564,8 +1260,13 @@ impl AsyncRead for File { + cx: &mut Context<'_>, + buf: &mut tokio::io::ReadBuf<'_>, + ) -> Poll> { +- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); +- let inner = Pin::new(&mut self.inner); ++ let this = self.as_mut().get_mut(); ++ let handle = this.handle.clone(); ++ let _guard = Handle::try_current().map_err(|_| handle.enter()); ++ let inner = match this.async_file() { ++ Ok(inner) => Pin::new(inner), ++ Err(err) => return Poll::Ready(Err(err)), ++ }; + inner.poll_read(cx, buf) + } + } +@@ -576,20 +1277,35 @@ impl AsyncWrite for File { + cx: &mut Context<'_>, + buf: &[u8], + ) -> Poll> { +- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); +- let inner = Pin::new(&mut self.inner); ++ let this = self.as_mut().get_mut(); ++ let handle = this.handle.clone(); ++ let _guard = Handle::try_current().map_err(|_| handle.enter()); ++ let inner = match this.async_file() { ++ Ok(inner) => Pin::new(inner), ++ Err(err) => return Poll::Ready(Err(err)), ++ }; + inner.poll_write(cx, buf) + } + + fn poll_flush(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { +- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); +- let inner = Pin::new(&mut self.inner); ++ let this = self.as_mut().get_mut(); ++ let handle = this.handle.clone(); ++ let _guard = Handle::try_current().map_err(|_| handle.enter()); ++ let inner = match this.async_file() { ++ Ok(inner) => Pin::new(inner), ++ Err(err) => return Poll::Ready(Err(err)), ++ }; + inner.poll_flush(cx) + } + + fn poll_shutdown(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { +- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); +- let inner = Pin::new(&mut self.inner); ++ let this = self.as_mut().get_mut(); ++ let handle = this.handle.clone(); ++ let _guard = Handle::try_current().map_err(|_| handle.enter()); ++ let inner = match this.async_file() { ++ Ok(inner) => Pin::new(inner), ++ Err(err) => return Poll::Ready(Err(err)), ++ }; + inner.poll_shutdown(cx) + } + +@@ -598,26 +1314,41 @@ impl AsyncWrite for File { + cx: &mut Context<'_>, + bufs: &[io::IoSlice<'_>], + ) -> Poll> { +- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); +- let inner = Pin::new(&mut self.inner); ++ let this = self.as_mut().get_mut(); ++ let handle = this.handle.clone(); ++ let _guard = Handle::try_current().map_err(|_| handle.enter()); ++ let inner = match this.async_file() { ++ Ok(inner) => Pin::new(inner), ++ Err(err) => return Poll::Ready(Err(err)), ++ }; + inner.poll_write_vectored(cx, bufs) + } + + fn is_write_vectored(&self) -> bool { +- self.inner.is_write_vectored() ++ self.inner ++ .as_ref() ++ .map(AsyncWrite::is_write_vectored) ++ .unwrap_or(true) + } + } + + impl AsyncSeek for File { + fn start_seek(mut self: Pin<&mut Self>, position: io::SeekFrom) -> io::Result<()> { +- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); +- let inner = Pin::new(&mut self.inner); ++ let this = self.as_mut().get_mut(); ++ let handle = this.handle.clone(); ++ let _guard = Handle::try_current().map_err(|_| handle.enter()); ++ let inner = Pin::new(this.async_file()?); + inner.start_seek(position) + } + + fn poll_complete(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { +- let _guard = Handle::try_current().map_err(|_| self.handle.enter()); +- let inner = Pin::new(&mut self.inner); ++ let this = self.as_mut().get_mut(); ++ let handle = this.handle.clone(); ++ let _guard = Handle::try_current().map_err(|_| handle.enter()); ++ let inner = match this.async_file() { ++ Ok(inner) => Pin::new(inner), ++ Err(err) => return Poll::Ready(Err(err)), ++ }; + inner.poll_complete(cx) + } + } +@@ -1026,10 +1757,220 @@ mod tests { + use tempfile::TempDir; + use tokio::runtime::Handle; + +- use super::FileSystem; ++ use super::{File, FileSystem}; ++ use crate::AsyncSeekExt; + use crate::FileSystem as FileSystemTrait; + use crate::FsError; ++ use crate::VirtualFile; ++ use crate::{FileAdvice, FileWritebackFlags}; ++ use std::io::{IoSlice, SeekFrom}; + use std::path::Path; ++ use std::sync::{Arc, Barrier}; ++ ++ #[tokio::test] ++ async fn host_file_defers_async_descriptor_until_async_io_is_requested() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("lazy-async-file.txt"); ++ std::fs::write(&path, b"contents").unwrap(); ++ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); ++ ++ let mut file = File::new( ++ Handle::current(), ++ std_file, ++ path, ++ true, ++ false, ++ false, ++ false, ++ false, ++ ); ++ ++ assert!(file.inner.is_none()); ++ file.async_file().unwrap(); ++ assert!(file.inner.is_some()); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[tokio::test] ++ async fn host_file_advice_uses_existing_descriptor_without_async_clone() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("advice.txt"); ++ std::fs::write(&path, vec![0u8; 8192]).unwrap(); ++ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); ++ ++ let file = File::new( ++ Handle::current(), ++ std_file, ++ path, ++ true, ++ false, ++ false, ++ false, ++ false, ++ ); ++ ++ assert!(file.inner.is_none()); ++ file.advise(0, 4096, FileAdvice::WillNeed).unwrap(); ++ file.advise(0, 4096, FileAdvice::DontNeed).unwrap(); ++ assert!(file.inner.is_none()); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[tokio::test] ++ async fn host_file_advice_rejects_unrepresentable_range_without_async_clone() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("advice-overflow.txt"); ++ std::fs::write(&path, b"contents").unwrap(); ++ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); ++ ++ let file = File::new( ++ Handle::current(), ++ std_file, ++ path, ++ true, ++ false, ++ false, ++ false, ++ false, ++ ); ++ ++ let beyond_off_t = (libc::off_t::MAX as u64) + 1; ++ for (offset, len) in [(u64::MAX, 1), (beyond_off_t, 0), (0, beyond_off_t)] { ++ let error = file.advise(offset, len, FileAdvice::WillNeed).unwrap_err(); ++ assert_eq!(error.raw_os_error(), Some(libc::EOVERFLOW)); ++ } ++ assert!(file.inner.is_none()); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[tokio::test] ++ async fn host_file_advice_accepts_individually_representable_boundary_ranges() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("advice-boundary.txt"); ++ std::fs::write(&path, b"contents").unwrap(); ++ let std_file = std::fs::OpenOptions::new().read(true).open(&path).unwrap(); ++ ++ let file = File::new( ++ Handle::current(), ++ std_file, ++ path, ++ true, ++ false, ++ false, ++ false, ++ false, ++ ); ++ ++ let off_max = libc::off_t::MAX as u64; ++ file.advise(off_max, 1, FileAdvice::WillNeed).unwrap(); ++ file.advise(off_max - 1, 3, FileAdvice::WillNeed).unwrap(); ++ assert!(file.inner.is_none()); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[tokio::test] ++ async fn host_file_range_writeback_uses_existing_descriptor_without_async_clone() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("writeback.txt"); ++ std::fs::write(&path, vec![0u8; 8192]).unwrap(); ++ let std_file = std::fs::OpenOptions::new() ++ .read(true) ++ .write(true) ++ .open(&path) ++ .unwrap(); ++ ++ let file = File::new( ++ Handle::current(), ++ std_file, ++ path, ++ true, ++ true, ++ false, ++ false, ++ false, ++ ); ++ ++ assert!(file.inner.is_none()); ++ file.writeback_range(0, 4096, FileWritebackFlags::WRITE) ++ .unwrap(); ++ file.writeback_range(0, 0, FileWritebackFlags::empty()) ++ .unwrap(); ++ assert!(file.inner.is_none()); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[tokio::test] ++ async fn host_file_range_writeback_rejects_unrepresentable_ranges() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("writeback-overflow.txt"); ++ std::fs::write(&path, b"contents").unwrap(); ++ let std_file = std::fs::OpenOptions::new() ++ .read(true) ++ .write(true) ++ .open(&path) ++ .unwrap(); ++ ++ let file = File::new( ++ Handle::current(), ++ std_file, ++ path, ++ true, ++ true, ++ false, ++ false, ++ false, ++ ); ++ ++ let beyond_off_t = (libc::off64_t::MAX as u64) + 1; ++ for (offset, len) in [(beyond_off_t, 0), (0, beyond_off_t)] { ++ let error = file ++ .writeback_range(offset, len, FileWritebackFlags::WRITE) ++ .unwrap_err(); ++ assert_eq!(error.raw_os_error(), Some(libc::EOVERFLOW)); ++ } ++ for (offset, len) in [ ++ (libc::off64_t::MAX as u64, 1), ++ (libc::off64_t::MAX as u64 - 1, 2), ++ ] { ++ let error = file ++ .writeback_range(offset, len, FileWritebackFlags::WRITE) ++ .unwrap_err(); ++ assert_eq!(error.raw_os_error(), Some(libc::EINVAL)); ++ } ++ assert!(file.inner.is_none()); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[tokio::test] ++ async fn host_file_range_writeback_accepts_maximum_finite_boundary() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("writeback-boundary.txt"); ++ std::fs::write(&path, b"contents").unwrap(); ++ let std_file = std::fs::OpenOptions::new() ++ .read(true) ++ .write(true) ++ .open(&path) ++ .unwrap(); ++ ++ let file = File::new( ++ Handle::current(), ++ std_file, ++ path, ++ true, ++ true, ++ false, ++ false, ++ false, ++ ); ++ ++ file.writeback_range( ++ libc::off64_t::MAX as u64 - 1, ++ 1, ++ FileWritebackFlags::empty(), ++ ) ++ .unwrap(); ++ assert!(file.inner.is_none()); ++ } + + #[tokio::test] + async fn test_new_filesystem() { +@@ -1050,6 +1991,172 @@ mod tests { + ); + } + ++ #[cfg(target_os = "linux")] ++ #[tokio::test] ++ async fn host_directory_sync_retains_identity_after_rename() { ++ let temp = TempDir::new().unwrap(); ++ let original = temp.path().join("original"); ++ let renamed = temp.path().join("renamed"); ++ std::fs::create_dir(&original).unwrap(); ++ ++ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); ++ let directory = fs.open_dir(Path::new("/original")).unwrap(); ++ assert!(directory.has_blocking_sync_all_to_disk()); ++ ++ std::fs::rename(&original, &renamed).unwrap(); ++ assert!(!original.exists()); ++ ++ // A path-reopen implementation would fail here. The retained host ++ // descriptor continues to refer to the directory opened above. ++ directory.sync_all_to_disk_blocking().unwrap(); ++ directory.sync_all_to_disk().await.unwrap(); ++ } ++ ++ #[tokio::test] ++ async fn host_file_write_at_preserves_cursor_and_syncs() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("foo.txt"); ++ std::fs::write(&path, b"abcdef").unwrap(); ++ ++ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); ++ let mut file = fs ++ .new_open_options() ++ .read(true) ++ .write(true) ++ .open(Path::new("/foo.txt")) ++ .expect("open file"); ++ ++ file.seek(SeekFrom::Start(3)).await.unwrap(); ++ assert_eq!(file.write_at(b"XY", 1).await.unwrap(), 2); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_read_at()); ++ let mut read_buf = [0; 2]; ++ assert_eq!(file.read_at_blocking(&mut read_buf, 1).unwrap(), 2); ++ assert_eq!(&read_buf, b"XY"); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_read_at_shared()); ++ let mut shared_read_buf = [0; 2]; ++ assert_eq!( ++ file.read_at_blocking_shared(&mut shared_read_buf, 1) ++ .unwrap(), ++ 2 ++ ); ++ assert_eq!(&shared_read_buf, b"XY"); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_read()); ++ file.seek_blocking(SeekFrom::Start(0)).unwrap(); ++ let mut read_buf = [0; 1]; ++ assert_eq!(file.read_blocking(&mut read_buf).unwrap(), 1); ++ assert_eq!(&read_buf, b"a"); ++ file.seek_blocking(SeekFrom::Start(3)).unwrap(); ++ assert!(file.has_blocking_write_at()); ++ assert_eq!(file.write_at_blocking(b"ZZ", 4).unwrap(), 2); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_write_at_shared()); ++ assert_eq!(file.write_at_blocking_shared(b"QQ", 4).unwrap(), 2); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_write_vectored_at()); ++ let bufs = [IoSlice::new(b"12"), IoSlice::new(b"34")]; ++ assert_eq!(file.write_vectored_at_blocking(&bufs, 1).unwrap(), 4); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_write_vectored_at_shared()); ++ let bufs = [IoSlice::new(b"56"), IoSlice::new(b"78")]; ++ assert_eq!(file.write_vectored_at_blocking_shared(&bufs, 1).unwrap(), 4); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_write_zeroes_at()); ++ assert_eq!(file.write_zeroes_at_blocking(2, 2).unwrap(), 2); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_write_zeroes_at_shared()); ++ assert_eq!(file.write_zeroes_at_blocking_shared(2, 2).unwrap(), 2); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 3); ++ assert!(file.has_blocking_seek()); ++ assert!(file.has_blocking_write()); ++ assert_eq!(file.seek_blocking(SeekFrom::Start(3)).unwrap(), 3); ++ assert_eq!(file.write_blocking(b"Y").unwrap(), 1); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 4); ++ assert!(file.has_blocking_sync_data_to_disk()); ++ assert!(file.has_blocking_sync_all_to_disk()); ++ file.sync_data_to_disk_blocking().unwrap(); ++ file.sync_all_to_disk_blocking().unwrap(); ++ file.sync_data_to_disk().await.unwrap(); ++ file.sync_all_to_disk().await.unwrap(); ++ assert_eq!(file.write_zeroes_at_blocking(3, 6).unwrap(), 3); ++ assert_eq!(file.seek(SeekFrom::Current(0)).await.unwrap(), 4); ++ ++ drop(file); ++ assert_eq!(std::fs::read(path).unwrap(), b"a5\0Y8Q\0\0\0"); ++ } ++ ++ #[cfg(any(unix, windows))] ++ #[tokio::test] ++ async fn host_file_shared_positioned_reads_are_concurrent_and_cursor_invariant() { ++ let temp = TempDir::new().unwrap(); ++ let path = temp.path().join("shared-positioned-read.txt"); ++ let contents = b"0123456789abcdefghijklmnopqrstuv"; ++ std::fs::write(&path, contents).unwrap(); ++ ++ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); ++ let mut file = fs ++ .new_open_options() ++ .read(true) ++ .open(Path::new("/shared-positioned-read.txt")) ++ .expect("open file"); ++ file.seek_blocking(SeekFrom::Start(7)).unwrap(); ++ assert!(file.has_blocking_read_at_shared()); ++ ++ let file = Arc::new(file); ++ let barrier = Arc::new(Barrier::new(4)); ++ let readers = [(0_u64, b"0123"), (8, b"89ab"), (16, b"ghij"), (24, b"opqr")] ++ .into_iter() ++ .map(|(offset, expected)| { ++ let file = file.clone(); ++ let barrier = barrier.clone(); ++ std::thread::spawn(move || { ++ barrier.wait(); ++ for _ in 0..256 { ++ let mut buf = [0_u8; 4]; ++ assert_eq!( ++ file.read_at_blocking_shared(&mut buf, offset).unwrap(), ++ buf.len() ++ ); ++ assert_eq!(&buf, expected); ++ } ++ }) ++ }) ++ .collect::>(); ++ ++ for reader in readers { ++ reader.join().unwrap(); ++ } ++ let mut file = Arc::try_unwrap(file).expect("reader retained the host file"); ++ assert_eq!(file.seek_blocking(SeekFrom::Current(0)).unwrap(), 7); ++ } ++ ++ #[tokio::test] ++ async fn host_file_reports_native_sync_open_modes() { ++ let temp = TempDir::new().unwrap(); ++ std::fs::write(temp.path().join("foo.txt"), b"").unwrap(); ++ ++ let fs = FileSystem::new(Handle::current(), temp.path()).expect("get filesystem"); ++ let data_sync_file = fs ++ .new_open_options() ++ .write(true) ++ .data_sync(true) ++ .open(Path::new("/foo.txt")) ++ .expect("open data-sync file"); ++ assert!(data_sync_file.is_sync_on_write(false)); ++ assert!(!data_sync_file.is_sync_on_write(true)); ++ ++ let sync_file = fs ++ .new_open_options() ++ .write(true) ++ .sync(true) ++ .open(Path::new("/foo.txt")) ++ .expect("open sync file"); ++ assert!(sync_file.is_sync_on_write(false)); ++ assert!(sync_file.is_sync_on_write(true)); ++ } ++ + #[tokio::test] + async fn test_create_dir() { + let temp: TempDir = TempDir::new().unwrap(); +diff --git a/lib/virtual-fs/src/lib.rs b/lib/virtual-fs/src/lib.rs +index 4fc7c5b..4fc28c2 100644 +--- a/lib/virtual-fs/src/lib.rs ++++ b/lib/virtual-fs/src/lib.rs +@@ -9,7 +9,7 @@ use shared_buffer::OwnedBuffer; + use std::any::Any; + use std::ffi::OsString; + use std::fmt; +-use std::io; ++use std::io::{self, SeekFrom}; + use std::ops::Deref; + use std::path::{Path, PathBuf}; + use std::pin::Pin; +@@ -85,6 +85,68 @@ pub trait ClonableVirtualFile: VirtualFile + Clone {} + + pub use ops::{copy_reference, copy_reference_ext, create_dir_all, walk}; + ++/// Backend-neutral file access advice. ++/// ++/// These variants correspond one-for-one with the six defined WASI ++/// `advice` values. Backends must treat advice as a range-scoped hint; a ++/// backend that cannot preserve those semantics should return ++/// [`io::ErrorKind::Unsupported`] instead of approximating them with a ++/// process-wide or persistent cache policy. ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub enum FileAdvice { ++ Normal, ++ Sequential, ++ Random, ++ WillNeed, ++ DontNeed, ++ NoReuse, ++} ++ ++/// Backend-neutral controls for range-scoped file writeback. ++/// ++/// These bits intentionally mirror Linux `sync_file_range(2)`, but this type ++/// does not imply durability: writeback does not flush file metadata or a ++/// device's volatile write cache. Backends that cannot preserve these exact ++/// advisory semantics must report [`io::ErrorKind::Unsupported`]. ++#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] ++pub struct FileWritebackFlags(u8); ++ ++impl FileWritebackFlags { ++ pub const WAIT_BEFORE: Self = Self(1); ++ pub const WRITE: Self = Self(2); ++ pub const WAIT_AFTER: Self = Self(4); ++ ++ const VALID_BITS: u32 = Self::WAIT_BEFORE.bits() | Self::WRITE.bits() | Self::WAIT_AFTER.bits(); ++ ++ pub const fn empty() -> Self { ++ Self(0) ++ } ++ ++ pub const fn from_bits(bits: u32) -> Option { ++ if bits & !Self::VALID_BITS == 0 { ++ Some(Self(bits as u8)) ++ } else { ++ None ++ } ++ } ++ ++ pub const fn bits(self) -> u32 { ++ self.0 as u32 ++ } ++ ++ pub const fn contains(self, other: Self) -> bool { ++ self.bits() & other.bits() == other.bits() ++ } ++} ++ ++impl std::ops::BitOr for FileWritebackFlags { ++ type Output = Self; ++ ++ fn bitor(self, rhs: Self) -> Self::Output { ++ Self(self.0 | rhs.0) ++ } ++} ++ + pub trait FileSystem: fmt::Debug + Send + Sync + 'static + Upcastable { + fn readlink(&self, path: &Path) -> Result; + fn read_dir(&self, path: &Path) -> Result; +@@ -101,6 +163,17 @@ pub trait FileSystem: fmt::Debug + Send + Sync + 'static + Upcastable { + fn symlink_metadata(&self, path: &Path) -> Result; + fn remove_file(&self, path: &Path) -> Result<()>; + ++ /// Open a directory as a stable object that can be synchronized later. ++ /// ++ /// This is deliberately separate from [`FileOpener`]: a directory handle ++ /// must retain the identity of the opened directory across renames and ++ /// unlinks, while reopening a path at sync time can target a different ++ /// directory. Backends without a real directory-sync primitive must fail ++ /// closed instead of reporting a successful no-op. ++ fn open_dir(&self, _path: &Path) -> Result> { ++ Err(FsError::Unsupported) ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_>; + } + +@@ -157,11 +230,34 @@ where + (**self).remove_file(path) + } + ++ fn open_dir(&self, path: &Path) -> Result> { ++ (**self).open_dir(path) ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_> { + (**self).new_open_options() + } + } + ++/// An opened directory whose identity is independent of its current path. ++/// ++/// Directory entries are metadata on POSIX filesystems, so both WASI ++/// `fd_sync` and `fd_datasync` use the full-directory sync operation exposed ++/// here. Implementations must not substitute a flush/no-op for durable sync. ++pub trait VirtualDirectory: fmt::Debug + Send + Sync + 'static { ++ fn sync_all_to_disk(&self) -> BoxFuture<'_, io::Result<()>> { ++ Box::pin(async { Err(io::ErrorKind::Unsupported.into()) }) ++ } ++ ++ fn has_blocking_sync_all_to_disk(&self) -> bool { ++ false ++ } ++ ++ fn sync_all_to_disk_blocking(&self) -> io::Result<()> { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++} ++ + pub trait FileOpener { + fn open( + &self, +@@ -178,6 +274,8 @@ pub struct OpenOptionsConfig { + pub create: bool, + pub append: bool, + pub truncate: bool, ++ pub sync: bool, ++ pub data_sync: bool, + } + + impl OpenOptionsConfig { +@@ -190,6 +288,8 @@ impl OpenOptionsConfig { + create: parent_rights.create && self.create, + append: parent_rights.append && self.append, + truncate: parent_rights.truncate && self.truncate, ++ sync: self.sync, ++ data_sync: self.data_sync, + } + } + +@@ -217,6 +317,14 @@ impl OpenOptionsConfig { + self.truncate + } + ++ pub const fn sync(&self) -> bool { ++ self.sync ++ } ++ ++ pub const fn data_sync(&self) -> bool { ++ self.data_sync ++ } ++ + /// Would a file opened with this [`OpenOptionsConfig`] change files on the + /// filesystem. + pub const fn would_mutate(&self) -> bool { +@@ -227,6 +335,8 @@ impl OpenOptionsConfig { + create, + append, + truncate, ++ sync: _, ++ data_sync: _, + } = *self; + append || write || create || create_new || truncate + } +@@ -254,6 +364,8 @@ impl<'a> OpenOptions<'a> { + create: false, + append: false, + truncate: false, ++ sync: false, ++ data_sync: false, + }, + } + } +@@ -323,6 +435,18 @@ impl<'a> OpenOptions<'a> { + self + } + ++ /// Sets synchronous write-through semantics for file data and metadata. ++ pub fn sync(&mut self, sync: bool) -> &mut Self { ++ self.conf.sync = sync; ++ self ++ } ++ ++ /// Sets synchronous write-through semantics for file data. ++ pub fn data_sync(&mut self, data_sync: bool) -> &mut Self { ++ self.conf.data_sync = data_sync; ++ self ++ } ++ + pub fn open>( + &mut self, + path: P, +@@ -374,12 +498,282 @@ pub trait VirtualFile: + None + } + ++ /// Returns whether this handle was opened with read access when the backend ++ /// can report that cheaply. ++ fn open_read(&self) -> Option { ++ None ++ } ++ ++ /// Returns whether this handle was opened with write access when the backend ++ /// can report that cheaply. ++ fn open_write(&self) -> Option { ++ None ++ } ++ + /// Writes to this file using an mmap offset and reference + /// (this method only works for mmap optimized file systems) + fn write_from_mmap(&mut self, _offset: u64, _len: u64) -> std::io::Result<()> { + Err(std::io::ErrorKind::Unsupported.into()) + } + ++ /// Advise the backing filesystem about access to one byte range. ++ /// ++ /// The default is deliberately unsupported: silently accepting advice ++ /// would prevent callers from distinguishing a real range-scoped hint ++ /// from a no-op. ++ fn advise(&self, _offset: u64, _len: u64, _advice: FileAdvice) -> io::Result<()> { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Initiate and/or wait for writeback of one file range. ++ /// ++ /// This is an advisory writeback operation, not a durability primitive. ++ /// In particular, callers must continue using data/full sync operations ++ /// wherever crash durability is required. ++ fn writeback_range( ++ &self, ++ _offset: u64, ++ _len: u64, ++ _flags: FileWritebackFlags, ++ ) -> io::Result<()> { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking seek path. ++ /// ++ /// Runtime users can use this to avoid async scheduling overhead for host ++ /// regular files while preserving the async fallback for virtual files. ++ fn has_blocking_seek(&self) -> bool { ++ false ++ } ++ ++ /// Seek using a backend-native blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_seek`] returns true. ++ fn seek_blocking(&mut self, _pos: SeekFrom) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking cursor-read path. ++ fn has_blocking_read(&self) -> bool { ++ false ++ } ++ ++ /// Read from the current file cursor using a backend-native blocking ++ /// primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_read`] returns true. ++ fn read_blocking(&mut self, _buf: &mut [u8]) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking positioned-read path. ++ fn has_blocking_read_at(&self) -> bool { ++ false ++ } ++ ++ /// Read at a file offset using a backend-native blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_read_at`] returns true. ++ fn read_at_blocking(&mut self, _buf: &mut [u8], _offset: u64) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking positioned-read path ++ /// that can be called through a shared reference. ++ /// ++ /// Backends should only return true when the operation does not mutate ++ /// backend cursor state and the backend is safe to use concurrently. ++ fn has_blocking_read_at_shared(&self) -> bool { ++ false ++ } ++ ++ /// Read at a file offset through a shared reference using a backend-native ++ /// blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_read_at_shared`] ++ /// returns true. ++ fn read_at_blocking_shared(&self, _buf: &mut [u8], _offset: u64) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking cursor-write path. ++ fn has_blocking_write(&self) -> bool { ++ false ++ } ++ ++ /// Write at the current file cursor using a backend-native blocking ++ /// primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_write`] returns true. ++ fn write_blocking(&mut self, _buf: &[u8]) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Write at a file offset without changing the file cursor. ++ /// ++ /// File-system backends should override this when they have a native ++ /// positioned-write primitive. The fallback preserves POSIX `pwrite` ++ /// cursor semantics for backends that only expose seek/write. ++ fn write_at<'a>(&'a mut self, buf: &'a [u8], offset: u64) -> BoxFuture<'a, io::Result> { ++ Box::pin(async move { ++ let cursor = self.seek(SeekFrom::Current(0)).await?; ++ self.seek(SeekFrom::Start(offset)).await?; ++ let write_result = self.write(buf).await; ++ let restore_result = self.seek(SeekFrom::Start(cursor)).await; ++ ++ match (write_result, restore_result) { ++ (Ok(written), Ok(_)) => Ok(written), ++ (Err(err), _) => Err(err), ++ (Ok(_), Err(err)) => Err(err), ++ } ++ }) ++ } ++ ++ /// Returns whether this file has a native blocking positioned-write path. ++ /// ++ /// Runtime users can use this to avoid async scheduling overhead for host ++ /// regular files while preserving the async fallback for virtual files. ++ fn has_blocking_write_at(&self) -> bool { ++ false ++ } ++ ++ /// Write at a file offset using a backend-native blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_write_at`] returns true. ++ fn write_at_blocking(&mut self, _buf: &[u8], _offset: u64) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking positioned-write path ++ /// that can be called through a shared reference. ++ /// ++ /// Backends should only return true when the operation does not mutate ++ /// backend cursor state and the backend is safe to use concurrently. ++ fn has_blocking_write_at_shared(&self) -> bool { ++ false ++ } ++ ++ /// Write at a file offset through a shared reference using a backend-native ++ /// blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_write_at_shared`] ++ /// returns true. ++ fn write_at_blocking_shared(&self, _buf: &[u8], _offset: u64) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking vectored ++ /// positioned-write path. ++ fn has_blocking_write_vectored_at(&self) -> bool { ++ false ++ } ++ ++ /// Vectored write at a file offset using a backend-native blocking ++ /// primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_write_vectored_at`] ++ /// returns true. ++ fn write_vectored_at_blocking( ++ &mut self, ++ _bufs: &[io::IoSlice<'_>], ++ _offset: u64, ++ ) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking vectored ++ /// positioned-write path that can be called through a shared reference. ++ fn has_blocking_write_vectored_at_shared(&self) -> bool { ++ false ++ } ++ ++ /// Vectored write at a file offset through a shared reference using a ++ /// backend-native blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_write_vectored_at_shared`] ++ /// returns true. ++ fn write_vectored_at_blocking_shared( ++ &self, ++ _bufs: &[io::IoSlice<'_>], ++ _offset: u64, ++ ) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking positioned zero-write ++ /// path. ++ fn has_blocking_write_zeroes_at(&self) -> bool { ++ false ++ } ++ ++ /// Write zero bytes at a file offset using a backend-native blocking ++ /// primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_write_zeroes_at`] ++ /// returns true. ++ fn write_zeroes_at_blocking(&mut self, _len: u64, _offset: u64) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking positioned zero-write ++ /// path that can be called through a shared reference. ++ fn has_blocking_write_zeroes_at_shared(&self) -> bool { ++ false ++ } ++ ++ /// Write zero bytes at a file offset through a shared reference using a ++ /// backend-native blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_write_zeroes_at_shared`] ++ /// returns true. ++ fn write_zeroes_at_blocking_shared(&self, _len: u64, _offset: u64) -> io::Result { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Synchronize file data to durable storage. ++ fn sync_data_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { ++ Box::pin(async move { self.flush().await }) ++ } ++ ++ /// Synchronize file data and metadata to durable storage. ++ fn sync_all_to_disk(&mut self) -> BoxFuture<'_, io::Result<()>> { ++ Box::pin(async move { self.flush().await }) ++ } ++ ++ /// Returns whether this file has a native blocking data-only sync path. ++ fn has_blocking_sync_data_to_disk(&self) -> bool { ++ false ++ } ++ ++ /// Synchronize file data using a backend-native blocking primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_sync_data_to_disk`] ++ /// returns true. ++ fn sync_data_to_disk_blocking(&mut self) -> io::Result<()> { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file has a native blocking full sync path. ++ fn has_blocking_sync_all_to_disk(&self) -> bool { ++ false ++ } ++ ++ /// Synchronize file data and metadata using a backend-native blocking ++ /// primitive. ++ /// ++ /// Only call this when [`VirtualFile::has_blocking_sync_all_to_disk`] ++ /// returns true. ++ fn sync_all_to_disk_blocking(&mut self) -> io::Result<()> { ++ Err(io::ErrorKind::Unsupported.into()) ++ } ++ ++ /// Returns whether this file handle was opened with host-native ++ /// synchronous write-through semantics. ++ fn is_sync_on_write(&self, _sync_metadata: bool) -> bool { ++ false ++ } ++ + /// This method will copy a file from a source to this destination where + /// the default is to do a straight byte copy however file system implementors + /// may optimize this to do a zero copy +@@ -765,3 +1159,53 @@ impl Iterator for ReadDir { + None + } + } ++ ++#[cfg(test)] ++mod advice_tests { ++ use super::{FileAdvice, FileWritebackFlags, NullFile, VirtualFile}; ++ ++ #[test] ++ fn virtual_file_advice_defaults_to_unsupported() { ++ let file = NullFile::default(); ++ let error = file.advise(0, 4096, FileAdvice::WillNeed).unwrap_err(); ++ ++ assert_eq!(error.kind(), std::io::ErrorKind::Unsupported); ++ } ++ ++ #[test] ++ fn virtual_file_shared_positioned_read_defaults_to_unsupported() { ++ let file = NullFile::default(); ++ let mut buf = [0_u8; 1]; ++ ++ assert!(!file.has_blocking_read_at_shared()); ++ let error = file.read_at_blocking_shared(&mut buf, 0).unwrap_err(); ++ assert_eq!(error.kind(), std::io::ErrorKind::Unsupported); ++ } ++ ++ #[test] ++ fn file_writeback_flags_preserve_the_linux_abi_bits() { ++ assert_eq!(FileWritebackFlags::WAIT_BEFORE.bits(), 1); ++ assert_eq!(FileWritebackFlags::WRITE.bits(), 2); ++ assert_eq!(FileWritebackFlags::WAIT_AFTER.bits(), 4); ++ assert_eq!(FileWritebackFlags::empty().bits(), 0); ++ assert_eq!( ++ (FileWritebackFlags::WAIT_BEFORE ++ | FileWritebackFlags::WRITE ++ | FileWritebackFlags::WAIT_AFTER) ++ .bits(), ++ 7 ++ ); ++ assert!(FileWritebackFlags::from_bits(7).is_some()); ++ assert!(FileWritebackFlags::from_bits(8).is_none()); ++ } ++ ++ #[test] ++ fn virtual_file_writeback_defaults_to_unsupported() { ++ let file = NullFile::default(); ++ let error = file ++ .writeback_range(0, 4096, FileWritebackFlags::WRITE) ++ .unwrap_err(); ++ ++ assert_eq!(error.kind(), std::io::ErrorKind::Unsupported); ++ } ++} +diff --git a/lib/virtual-fs/src/mount_fs.rs b/lib/virtual-fs/src/mount_fs.rs +index 25b9e23..f921671 100644 +--- a/lib/virtual-fs/src/mount_fs.rs ++++ b/lib/virtual-fs/src/mount_fs.rs +@@ -695,6 +695,22 @@ impl FileSystem for MountFileSystem { + } + } + ++ fn open_dir(&self, path: &Path) -> Result> { ++ let path = self.prepare_path(path)?; ++ ++ if let Some(node) = self.exact_node(&path) ++ && node.fs.is_none() ++ { ++ // A synthetic mount-tree directory has no durable backing object. ++ return Err(FsError::Unsupported); ++ } ++ ++ match self.resolve_mount(path) { ++ Some(resolved) => resolved.fs.open_dir(&resolved.delegated_path), ++ None => Err(FsError::EntryNotFound), ++ } ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_> { + OpenOptions::new(self) + } +@@ -988,6 +1004,8 @@ mod tests { + create: true, + append: false, + truncate: false, ++ sync: false, ++ data_sync: false, + }, + ) + .unwrap(); +@@ -1001,6 +1019,8 @@ mod tests { + create: true, + append: false, + truncate: false, ++ sync: false, ++ data_sync: false, + }, + ) + .unwrap(); +diff --git a/lib/virtual-fs/src/overlay_fs.rs b/lib/virtual-fs/src/overlay_fs.rs +index 846bdf7..bd608f6 100644 +--- a/lib/virtual-fs/src/overlay_fs.rs ++++ b/lib/virtual-fs/src/overlay_fs.rs +@@ -427,6 +427,44 @@ where + self.permission_error_or_not_found(path) + } + ++ fn open_dir( ++ &self, ++ path: &Path, ++ ) -> crate::Result> { ++ if ops::is_white_out(path).is_some() { ++ return Err(FsError::EntryNotFound); ++ } ++ ++ // Sync the highest-precedence directory that is actually visible. ++ // In particular, do not bypass an unsupported writable upper layer ++ // and report success after syncing an unrelated lower directory. ++ match self.primary.metadata(path) { ++ Ok(metadata) if metadata.is_dir() => return self.primary.open_dir(path), ++ Ok(_) => return Err(FsError::NotAFile), ++ Err(error) if should_continue(error) => {} ++ Err(error) => return Err(error), ++ } ++ ++ if ops::has_white_out(&self.primary, path) { ++ return Err(FsError::EntryNotFound); ++ } ++ ++ for fs in self.secondaries.filesystems() { ++ match fs.metadata(path) { ++ // A lower-only directory can be copied up after this open. A ++ // handle to the lower inode could then return successful fsync ++ // without making the visible upper-layer mutation durable. ++ // Fail closed until the directory exists in the mutable primary. ++ Ok(metadata) if metadata.is_dir() => return Err(FsError::Unsupported), ++ Ok(_) => return Err(FsError::NotAFile), ++ Err(error) if should_continue(error) => continue, ++ Err(error) => return Err(error), ++ } ++ } ++ ++ Err(FsError::EntryNotFound) ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_> { + OpenOptions::new(self) + } +@@ -1176,6 +1214,20 @@ mod tests { + ); + } + ++ #[test] ++ fn lower_only_directory_handle_fails_closed_before_copy_up() { ++ let primary = MemFS::default(); ++ let secondary = MemFS::default(); ++ ops::create_dir_all(&secondary, "/lower-only").unwrap(); ++ let overlay = OverlayFileSystem::new(primary, [secondary]); ++ ++ assert!(overlay.metadata(Path::new("/lower-only")).unwrap().is_dir()); ++ assert!(matches!( ++ overlay.open_dir(Path::new("/lower-only")), ++ Err(FsError::Unsupported) ++ )); ++ } ++ + #[tokio::test] + async fn remove_directory() { + let primary = MemFS::default(); +diff --git a/lib/virtual-fs/src/passthru_fs.rs b/lib/virtual-fs/src/passthru_fs.rs +index a32f9da..5fe117b 100644 +--- a/lib/virtual-fs/src/passthru_fs.rs ++++ b/lib/virtual-fs/src/passthru_fs.rs +@@ -61,6 +61,10 @@ impl FileSystem for PassthruFileSystem { + self.fs.remove_file(path) + } + ++ fn open_dir(&self, path: &Path) -> Result> { ++ self.fs.open_dir(path) ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_> { + self.fs.new_open_options() + } +diff --git a/lib/virtual-fs/src/trace_fs.rs b/lib/virtual-fs/src/trace_fs.rs +index 667bc84..139034e 100644 +--- a/lib/virtual-fs/src/trace_fs.rs ++++ b/lib/virtual-fs/src/trace_fs.rs +@@ -83,6 +83,14 @@ where + self.0.remove_file(path) + } + ++ #[tracing::instrument(level = "trace", skip(self), err)] ++ fn open_dir( ++ &self, ++ path: &std::path::Path, ++ ) -> crate::Result> { ++ self.0.open_dir(path) ++ } ++ + #[tracing::instrument(level = "trace", skip(self))] + fn new_open_options(&self) -> crate::OpenOptions<'_> { + crate::OpenOptions::new(self) diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0003-network-readiness-and-interest-lifetimes.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0003-network-readiness-and-interest-lifetimes.patch new file mode 100644 index 000000000..15d557666 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0003-network-readiness-and-interest-lifetimes.patch @@ -0,0 +1,1596 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 3/9] virtual-io/net: preserve readiness and interest ownership + +Keep selector interest registration and host socket readiness/lifetime changes +together. PostgreSQL's listener and accepted sockets cross fork/exec and epoll +boundaries; readiness must not disappear when a temporary registration is +dropped. The WASIX epoll adapter consumes these lower-level changes. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/lib/virtual-io/src/guard.rs b/lib/virtual-io/src/guard.rs +index 60d5596..f062bd9 100644 +--- a/lib/virtual-io/src/guard.rs ++++ b/lib/virtual-io/src/guard.rs +@@ -5,7 +5,9 @@ use std::{ + + use mio::Token; + +-use crate::{InterestHandler, InterestType, InterestWakerMap, Selector}; ++use crate::{ ++ InterestHandler, InterestType, InterestWakerMap, MultiplexedInterestHandler, Selector, ++}; + + #[derive(Debug)] + #[must_use = "Leaking token guards will break the IO subsystem"] +@@ -58,6 +60,12 @@ impl InterestGuard { + } + } + ++ pub fn attach_waker_map(&mut self, waker_map: InterestWakerMap) { ++ if let Some(selector) = self.selector.upgrade() { ++ selector.attach_waker_map(self.token, waker_map); ++ } ++ } ++ + fn drop_internal(&mut self) { + if let Some(selector) = self.selector.upgrade() { + selector.remove(self.token, None).ok(); +@@ -70,6 +78,125 @@ pub enum HandlerGuardState { + None, + ExternalHandler(InterestGuard), + WakerMap(InterestGuard, InterestWakerMap), ++ ExternalHandlerWithWakerMap(InterestGuard, InterestWakerMap), ++} ++ ++impl HandlerGuardState { ++ pub fn set_external_handler( ++ &mut self, ++ selector: &Arc, ++ source: &mut dyn mio::event::Source, ++ interest: mio::Interest, ++ handler: Box, ++ ) -> io::Result<()> { ++ let mut current = HandlerGuardState::None; ++ std::mem::swap(self, &mut current); ++ ++ match current { ++ HandlerGuardState::None => { ++ *self = HandlerGuardState::ExternalHandler(InterestGuard::new( ++ selector, handler, source, interest, ++ )?); ++ Ok(()) ++ } ++ HandlerGuardState::ExternalHandler(mut guard) => match guard.replace_handler(handler) { ++ Ok(()) => { ++ *self = HandlerGuardState::ExternalHandler(guard); ++ Ok(()) ++ } ++ Err(handler) => { ++ guard.unregister(source).ok(); ++ *self = HandlerGuardState::ExternalHandler(InterestGuard::new( ++ selector, handler, source, interest, ++ )?); ++ Ok(()) ++ } ++ }, ++ HandlerGuardState::WakerMap(mut guard, waker_map) ++ | HandlerGuardState::ExternalHandlerWithWakerMap(mut guard, waker_map) => { ++ let handler = MultiplexedInterestHandler::new(handler, waker_map.clone()); ++ match guard.replace_handler(handler) { ++ Ok(()) => { ++ *self = HandlerGuardState::ExternalHandlerWithWakerMap(guard, waker_map); ++ Ok(()) ++ } ++ Err(handler) => { ++ guard.unregister(source).ok(); ++ *self = HandlerGuardState::ExternalHandlerWithWakerMap( ++ InterestGuard::new(selector, handler, source, interest)?, ++ waker_map, ++ ); ++ Ok(()) ++ } ++ } ++ } ++ } ++ } ++ ++ pub fn remove_external_handler( ++ &mut self, ++ source: &mut dyn mio::event::Source, ++ ) -> io::Result<()> { ++ let mut current = HandlerGuardState::None; ++ std::mem::swap(self, &mut current); ++ match current { ++ HandlerGuardState::None => Ok(()), ++ HandlerGuardState::ExternalHandler(mut guard) => guard.unregister(source), ++ HandlerGuardState::WakerMap(guard, waker_map) => { ++ *self = HandlerGuardState::WakerMap(guard, waker_map); ++ Ok(()) ++ } ++ HandlerGuardState::ExternalHandlerWithWakerMap(mut guard, waker_map) => { ++ let res = guard.replace_handler(Box::new(waker_map.clone())); ++ match res { ++ Ok(()) => { ++ *self = HandlerGuardState::WakerMap(guard, waker_map); ++ Ok(()) ++ } ++ Err(_) => { ++ guard.unregister(source).ok(); ++ Ok(()) ++ } ++ } ++ } ++ } ++ } ++ ++ pub fn remove_all_handlers(&mut self, source: &mut dyn mio::event::Source) -> io::Result<()> { ++ let mut current = HandlerGuardState::None; ++ std::mem::swap(self, &mut current); ++ match current { ++ HandlerGuardState::None => Ok(()), ++ HandlerGuardState::ExternalHandler(mut guard) ++ | HandlerGuardState::WakerMap(mut guard, _) ++ | HandlerGuardState::ExternalHandlerWithWakerMap(mut guard, _) => { ++ guard.unregister(source) ++ } ++ } ++ } ++ ++ pub fn push_interest(&mut self, interest: InterestType) { ++ match self { ++ HandlerGuardState::ExternalHandler(guard) ++ | HandlerGuardState::ExternalHandlerWithWakerMap(guard, _) => { ++ guard.interest(interest); ++ } ++ HandlerGuardState::WakerMap(_, waker_map) => { ++ waker_map.push_interest(interest); ++ } ++ HandlerGuardState::None => {} ++ } ++ } ++ ++ pub fn pop_waker_interest(&mut self, interest: InterestType) -> bool { ++ match self { ++ HandlerGuardState::WakerMap(_, waker_map) ++ | HandlerGuardState::ExternalHandlerWithWakerMap(_, waker_map) => { ++ waker_map.pop(interest) ++ } ++ HandlerGuardState::ExternalHandler(_) | HandlerGuardState::None => false, ++ } ++ } + } + + pub fn state_as_waker_map<'a>( +@@ -77,20 +204,35 @@ pub fn state_as_waker_map<'a>( + selector: &'_ Arc, + source: &'_ mut dyn mio::event::Source, + ) -> io::Result<&'a mut InterestWakerMap> { +- if !matches!(state, HandlerGuardState::WakerMap(_, _)) { +- let waker_map = InterestWakerMap::default(); +- *state = HandlerGuardState::WakerMap( +- InterestGuard::new( +- selector, +- Box::new(waker_map.clone()), +- source, +- mio::Interest::READABLE | mio::Interest::WRITABLE, +- )?, +- waker_map, +- ); ++ match state { ++ HandlerGuardState::None => { ++ let waker_map = InterestWakerMap::default(); ++ *state = HandlerGuardState::WakerMap( ++ InterestGuard::new( ++ selector, ++ Box::new(waker_map.clone()), ++ source, ++ mio::Interest::READABLE | mio::Interest::WRITABLE, ++ )?, ++ waker_map, ++ ); ++ } ++ HandlerGuardState::ExternalHandler(guard) => { ++ let waker_map = InterestWakerMap::default(); ++ guard.attach_waker_map(waker_map.clone()); ++ ++ let mut current = HandlerGuardState::None; ++ std::mem::swap(state, &mut current); ++ if let HandlerGuardState::ExternalHandler(guard) = current { ++ *state = HandlerGuardState::ExternalHandlerWithWakerMap(guard, waker_map); ++ } ++ } ++ HandlerGuardState::WakerMap(_, _) ++ | HandlerGuardState::ExternalHandlerWithWakerMap(_, _) => {} + } + Ok(match state { +- HandlerGuardState::WakerMap(_, map) => map, ++ HandlerGuardState::WakerMap(_, map) ++ | HandlerGuardState::ExternalHandlerWithWakerMap(_, map) => map, + _ => unreachable!(), + }) + } +diff --git a/lib/virtual-io/src/interest.rs b/lib/virtual-io/src/interest.rs +index 7cf0529..24ad072 100644 +--- a/lib/virtual-io/src/interest.rs ++++ b/lib/virtual-io/src/interest.rs +@@ -1,7 +1,10 @@ + use serde::{Deserialize, Serialize}; + use std::{ + collections::{HashMap, HashSet}, +- sync::{Arc, Mutex}, ++ sync::{ ++ Arc, Mutex, Weak, ++ atomic::{AtomicU64, Ordering}, ++ }, + task::{Context, RawWaker, RawWakerVTable, Waker}, + }; + +@@ -77,6 +80,151 @@ pub trait InterestHandler: Send + Sync + std::fmt::Debug { + fn has_interest(&self, interest: InterestType) -> bool; + } + ++type SharedInterestHandler = Arc>>; ++ ++#[derive(Debug, Default)] ++struct InterestHandlerFanoutInner { ++ next_id: AtomicU64, ++ handlers: Mutex>, ++} ++ ++/// A cloneable interest handler that fans each callback out to independently ++/// registered consumers. The handler map is only held while taking a snapshot; ++/// callbacks execute outside it so consumers may safely unregister themselves. ++#[derive(Debug, Clone, Default)] ++pub struct InterestHandlerFanout { ++ inner: Arc, ++} ++ ++impl InterestHandlerFanout { ++ pub fn register( ++ &self, ++ handler: Box, ++ ) -> InterestHandlerRegistration { ++ let id = self ++ .inner ++ .next_id ++ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { ++ current.checked_add(1) ++ }) ++ .expect("interest-handler fanout identity space exhausted"); ++ self.inner ++ .handlers ++ .lock() ++ .unwrap() ++ .insert(id, Arc::new(Mutex::new(handler))); ++ InterestHandlerRegistration { ++ inner: Arc::downgrade(&self.inner), ++ id, ++ } ++ } ++ ++ pub fn handler_count(&self) -> usize { ++ self.inner.handlers.lock().unwrap().len() ++ } ++ ++ fn snapshot(&self) -> InterestHandlerSnapshot { ++ let handlers = self.inner.handlers.lock().unwrap(); ++ match handlers.len() { ++ 0 => InterestHandlerSnapshot::Empty, ++ 1 => InterestHandlerSnapshot::One(handlers.values().next().unwrap().clone()), ++ _ => InterestHandlerSnapshot::Many(handlers.values().cloned().collect()), ++ } ++ } ++} ++ ++enum InterestHandlerSnapshot { ++ Empty, ++ One(SharedInterestHandler), ++ Many(Vec), ++} ++ ++impl InterestHandler for InterestHandlerFanout { ++ fn push_interest(&mut self, interest: InterestType) { ++ match self.snapshot() { ++ InterestHandlerSnapshot::Empty => {} ++ InterestHandlerSnapshot::One(handler) => { ++ handler.lock().unwrap().push_interest(interest); ++ } ++ InterestHandlerSnapshot::Many(handlers) => { ++ for handler in handlers { ++ handler.lock().unwrap().push_interest(interest); ++ } ++ } ++ } ++ } ++ ++ fn pop_interest(&mut self, interest: InterestType) -> bool { ++ match self.snapshot() { ++ InterestHandlerSnapshot::Empty => false, ++ InterestHandlerSnapshot::One(handler) => handler.lock().unwrap().pop_interest(interest), ++ InterestHandlerSnapshot::Many(handlers) => { ++ handlers.into_iter().fold(false, |seen, handler| { ++ handler.lock().unwrap().pop_interest(interest) || seen ++ }) ++ } ++ } ++ } ++ ++ fn has_interest(&self, interest: InterestType) -> bool { ++ match self.snapshot() { ++ InterestHandlerSnapshot::Empty => false, ++ InterestHandlerSnapshot::One(handler) => handler.lock().unwrap().has_interest(interest), ++ InterestHandlerSnapshot::Many(handlers) => handlers ++ .into_iter() ++ .any(|handler| handler.lock().unwrap().has_interest(interest)), ++ } ++ } ++} ++ ++#[derive(Debug)] ++pub struct InterestHandlerRegistration { ++ inner: Weak, ++ id: u64, ++} ++ ++impl Drop for InterestHandlerRegistration { ++ fn drop(&mut self) { ++ let Some(inner) = self.inner.upgrade() else { ++ return; ++ }; ++ let removed = inner.handlers.lock().unwrap().remove(&self.id); ++ drop(removed); ++ } ++} ++ ++#[derive(Debug)] ++pub struct MultiplexedInterestHandler { ++ primary: Box, ++ waker_map: InterestWakerMap, ++} ++ ++impl MultiplexedInterestHandler { ++ pub fn new( ++ primary: Box, ++ waker_map: InterestWakerMap, ++ ) -> Box { ++ Box::new(Self { primary, waker_map }) ++ } ++} ++ ++impl InterestHandler for MultiplexedInterestHandler { ++ fn push_interest(&mut self, interest: InterestType) { ++ self.primary.push_interest(interest); ++ self.waker_map.push_interest(interest); ++ } ++ ++ fn pop_interest(&mut self, interest: InterestType) -> bool { ++ let primary = self.primary.pop_interest(interest); ++ let waker_map = self.waker_map.pop_interest(interest); ++ primary || waker_map ++ } ++ ++ fn has_interest(&self, interest: InterestType) -> bool { ++ self.primary.has_interest(interest) || self.waker_map.has_interest(interest) ++ } ++} ++ + impl From<&Waker> for Box { + fn from(waker: &Waker) -> Self { + WakerInterestHandler::new(waker) +@@ -194,3 +342,99 @@ impl InterestHandler for InterestWakerMap { + state.triggered.contains(&interest) + } + } ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use std::sync::atomic::{AtomicUsize, Ordering}; ++ use std::sync::mpsc; ++ use std::time::Duration; ++ ++ #[derive(Debug)] ++ struct CountingHandler(Arc); ++ ++ impl InterestHandler for CountingHandler { ++ fn push_interest(&mut self, _interest: InterestType) { ++ self.0.fetch_add(1, Ordering::SeqCst); ++ } ++ ++ fn pop_interest(&mut self, _interest: InterestType) -> bool { ++ false ++ } ++ ++ fn has_interest(&self, _interest: InterestType) -> bool { ++ false ++ } ++ } ++ ++ #[test] ++ fn fanout_removes_only_the_dropped_registration() { ++ let mut fanout = InterestHandlerFanout::default(); ++ let first = Arc::new(AtomicUsize::new(0)); ++ let second = Arc::new(AtomicUsize::new(0)); ++ let first_registration = fanout.register(Box::new(CountingHandler(first.clone()))); ++ let _second_registration = fanout.register(Box::new(CountingHandler(second.clone()))); ++ ++ fanout.push_interest(InterestType::Readable); ++ assert_eq!(first.load(Ordering::SeqCst), 1); ++ assert_eq!(second.load(Ordering::SeqCst), 1); ++ ++ drop(first_registration); ++ fanout.push_interest(InterestType::Readable); ++ assert_eq!(first.load(Ordering::SeqCst), 1); ++ assert_eq!(second.load(Ordering::SeqCst), 2); ++ assert_eq!(fanout.handler_count(), 1); ++ } ++ ++ #[test] ++ fn singleton_fanout_uses_allocation_free_snapshot_shape() { ++ let fanout = InterestHandlerFanout::default(); ++ let _registration = ++ fanout.register(Box::new(CountingHandler(Arc::new(AtomicUsize::new(0))))); ++ assert!(matches!(fanout.snapshot(), InterestHandlerSnapshot::One(_))); ++ } ++ ++ #[derive(Debug)] ++ struct ReentrantDropHandler { ++ fanout: InterestHandlerFanout, ++ dropped: mpsc::Sender<()>, ++ } ++ ++ impl InterestHandler for ReentrantDropHandler { ++ fn push_interest(&mut self, _interest: InterestType) {} ++ ++ fn pop_interest(&mut self, _interest: InterestType) -> bool { ++ false ++ } ++ ++ fn has_interest(&self, _interest: InterestType) -> bool { ++ false ++ } ++ } ++ ++ impl Drop for ReentrantDropHandler { ++ fn drop(&mut self) { ++ let transient = self ++ .fanout ++ .register(Box::new(CountingHandler(Arc::new(AtomicUsize::new(0))))); ++ drop(transient); ++ self.dropped.send(()).unwrap(); ++ } ++ } ++ ++ #[test] ++ fn registration_drop_releases_map_lock_before_handler_drop() { ++ let fanout = InterestHandlerFanout::default(); ++ let (dropped_tx, dropped_rx) = mpsc::channel(); ++ let registration = fanout.register(Box::new(ReentrantDropHandler { ++ fanout: fanout.clone(), ++ dropped: dropped_tx, ++ })); ++ ++ std::thread::spawn(move || drop(registration)); ++ dropped_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("handler drop should re-enter the fanout without deadlocking"); ++ assert_eq!(fanout.handler_count(), 0); ++ } ++} +diff --git a/lib/virtual-io/src/selector.rs b/lib/virtual-io/src/selector.rs +index b53e018..21399a8 100644 +--- a/lib/virtual-io/src/selector.rs ++++ b/lib/virtual-io/src/selector.rs +@@ -8,7 +8,7 @@ use std::{ + }, + }; + +-use crate::{InterestHandler, InterestType}; ++use crate::{InterestHandler, InterestType, InterestWakerMap, MultiplexedInterestHandler}; + + pub enum SelectorModification { + Add { +@@ -22,6 +22,10 @@ pub enum SelectorModification { + token: Token, + handler: Box, + }, ++ AttachWakerMap { ++ token: Token, ++ waker_map: InterestWakerMap, ++ }, + PushInterest { + token: Token, + interest: InterestType, +@@ -60,6 +64,13 @@ impl SelectorModification { + + lookup.insert(token, handler); + } ++ SelectorModification::AttachWakerMap { token, waker_map } => { ++ if let Some(last) = lookup.remove(&token) { ++ lookup.insert(token, MultiplexedInterestHandler::new(last, waker_map)); ++ } else { ++ lookup.insert(token, Box::new(waker_map)); ++ } ++ } + SelectorModification::PushInterest { token, interest } => { + if let Some(handler) = lookup.get_mut(&token) { + handler.push_interest(interest); +@@ -80,6 +91,10 @@ impl std::fmt::Debug for SelectorModification { + SelectorModification::Replace { token, .. } => { + f.debug_struct("Replace").field("token", token).finish() + } ++ SelectorModification::AttachWakerMap { token, .. } => f ++ .debug_struct("AttachWakerMap") ++ .field("token", token) ++ .finish(), + SelectorModification::PushInterest { token, interest } => f + .debug_struct("PushInterest") + .field("token", token) +@@ -151,14 +166,21 @@ impl Selector { + + // CONCURRENCY: This should never result in a deadlock, as long as source.deregister does not call remove or add again. + let inner_registry = self.registry.lock().unwrap(); +- match source.register(&inner_registry, token, interests) { +- Ok(()) => {} ++ let registration = match source.register(&inner_registry, token, interests) { ++ Ok(()) => Ok(()), + Err(err) if err.kind() == io::ErrorKind::AlreadyExists => { + source.deregister(&inner_registry).ok(); +- source.register(&inner_registry, token, interests)?; ++ source.register(&inner_registry, token, interests) + } +- Err(err) => return Err(err), ++ Err(err) => Err(err), + }; ++ drop(inner_registry); ++ if let Err(err) = registration { ++ // Balance the already queued Add. Remove wakes the idle poll loop, ++ // so the rejected handler is not retained waiting for another edge. ++ self.queue_modification(SelectorModification::Remove { token }); ++ return Err(err); ++ } + + Ok(token) + } +@@ -185,6 +207,10 @@ impl Selector { + self.queue_modification(SelectorModification::Replace { token, handler }); + } + ++ pub fn attach_waker_map(&self, token: Token, waker_map: InterestWakerMap) { ++ self.queue_modification(SelectorModification::AttachWakerMap { token, waker_map }); ++ } ++ + /// Generate a new unique token + #[must_use = "the token must be consumed"] + fn new_token(&self) -> Token { +@@ -193,10 +219,16 @@ impl Selector { + + /// Try to process a modification immediately, otherwise queue it up + fn queue_modification(&self, modification: SelectorModification) { +- // Replace and PushInterest can cause external code to be called so it is a good idea to process them asap so they don't get delayed too long ++ // Process selector modifications promptly. Add must wake the poll loop ++ // so the handler is installed even if the registered source is already ++ // ready and no later edge arrives. + let needs_wakeup = matches!( + &modification, +- SelectorModification::PushInterest { .. } | SelectorModification::Replace { .. } ++ SelectorModification::Add { .. } ++ | SelectorModification::Remove { .. } ++ | SelectorModification::PushInterest { .. } ++ | SelectorModification::AttachWakerMap { .. } ++ | SelectorModification::Replace { .. } + ); + + // CONCURRENCY: This will never deadlock as queued_modifications is always the innermost lock and we don't call any potentially blocking functions while holding the lock. +@@ -286,7 +318,10 @@ impl Selector { + #[cfg(all(unix, test))] + mod tests { + use super::*; ++ use crate::{HandlerGuardState, InterestGuard, state_as_waker_map}; ++ use futures::task::{ArcWake, waker}; + use std::io::Write; ++ use std::sync::Arc; + use std::sync::mpsc; + use std::thread; + use std::time::Duration; +@@ -316,6 +351,55 @@ mod tests { + token: Arc>>, + success_sender: mpsc::Sender<()>, + } ++ ++ #[derive(Debug)] ++ struct DropSignalHandler { ++ dropped: mpsc::Sender<()>, ++ } ++ ++ impl Drop for DropSignalHandler { ++ fn drop(&mut self) { ++ self.dropped.send(()).unwrap(); ++ } ++ } ++ ++ impl InterestHandler for DropSignalHandler { ++ fn push_interest(&mut self, _interest: InterestType) {} ++ ++ fn pop_interest(&mut self, _interest: InterestType) -> bool { ++ false ++ } ++ ++ fn has_interest(&self, _interest: InterestType) -> bool { ++ false ++ } ++ } ++ ++ struct FailingSource; ++ ++ impl mio::event::Source for FailingSource { ++ fn register( ++ &mut self, ++ _registry: &mio::Registry, ++ _token: Token, ++ _interests: mio::Interest, ++ ) -> io::Result<()> { ++ Err(io::Error::other("injected registration failure")) ++ } ++ ++ fn reregister( ++ &mut self, ++ _registry: &mio::Registry, ++ _token: Token, ++ _interests: mio::Interest, ++ ) -> io::Result<()> { ++ Err(io::Error::other("injected registration failure")) ++ } ++ ++ fn deregister(&mut self, _registry: &mio::Registry) -> io::Result<()> { ++ Ok(()) ++ } ++ } + impl InterestHandler for DeadlockingHandler { + fn push_interest(&mut self, _interest: InterestType) { + // This would deadlock without a queue +@@ -334,6 +418,16 @@ mod tests { + } + } + ++ struct ChannelWake { ++ sender: mpsc::Sender<()>, ++ } ++ ++ impl ArcWake for ChannelWake { ++ fn wake_by_ref(arc_self: &Arc) { ++ arc_self.sender.send(()).unwrap(); ++ } ++ } ++ + #[test] + fn test_push_interest() { + let (mut sender, mut receiver) = mio::unix::pipe::new().unwrap(); +@@ -396,6 +490,87 @@ mod tests { + selector.shutdown(); + } + ++ #[test] ++ fn selector_remove_wakes_idle_poll_and_drops_handler_promptly() { ++ let (_sender, mut receiver) = mio::unix::pipe::new().unwrap(); ++ let (dropped_tx, dropped_rx) = mpsc::channel(); ++ let selector = Selector::new(); ++ let token = selector ++ .add( ++ Box::new(DropSignalHandler { ++ dropped: dropped_tx, ++ }), ++ &mut receiver, ++ mio::Interest::READABLE, ++ ) ++ .unwrap(); ++ ++ selector.remove(token, Some(&mut receiver)).unwrap(); ++ ++ dropped_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("Remove must wake an idle selector and promptly drop its handler"); ++ selector.shutdown(); ++ } ++ ++ #[test] ++ fn selector_add_failure_balances_queued_handler_promptly() { ++ let (dropped_tx, dropped_rx) = mpsc::channel(); ++ let selector = Selector::new(); ++ let mut source = FailingSource; ++ ++ let result = selector.add( ++ Box::new(DropSignalHandler { ++ dropped: dropped_tx, ++ }), ++ &mut source, ++ mio::Interest::READABLE, ++ ); ++ assert!(result.is_err()); ++ dropped_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("failed Add must be balanced by a prompt queued Remove"); ++ selector.shutdown(); ++ } ++ ++ #[test] ++ fn external_handler_and_waker_map_both_receive_interest() { ++ let (mut sender, mut receiver) = mio::unix::pipe::new().unwrap(); ++ let (handler_sender, handler_receiver) = mpsc::channel(); ++ let (wake_sender, wake_receiver) = mpsc::channel(); ++ ++ let selector = Selector::new(); ++ let guard = InterestGuard::new( ++ &selector, ++ Box::new(TestHandler { ++ success_sender: handler_sender, ++ }), ++ &mut receiver, ++ mio::Interest::READABLE, ++ ) ++ .unwrap(); ++ let mut state = HandlerGuardState::ExternalHandler(guard); ++ let wake = waker(Arc::new(ChannelWake { ++ sender: wake_sender, ++ })); ++ ++ state_as_waker_map(&mut state, &selector, &mut receiver) ++ .unwrap() ++ .add(InterestType::Readable, &wake); ++ ++ thread::sleep(Duration::from_millis(10)); ++ sender.write_all(&[1]).unwrap(); ++ ++ handler_receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("external handler should receive readiness"); ++ wake_receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("waker map waiter should receive readiness"); ++ ++ state.remove_all_handlers(&mut receiver).unwrap(); ++ } ++ + #[test] + fn test_selector_no_deadlock_when_modifying_the_selector_from_push_interest() { + let (mut sender, mut receiver) = mio::unix::pipe::new().unwrap(); +diff --git a/lib/virtual-net/src/host.rs b/lib/virtual-net/src/host.rs +index 3f1af0f..90a1818 100644 +--- a/lib/virtual-net/src/host.rs ++++ b/lib/virtual-net/src/host.rs +@@ -25,9 +25,13 @@ use std::time::Duration; + use tokio::runtime::Handle; + #[allow(unused_imports, dead_code)] + use tracing::{debug, error, info, trace, warn}; +-use virtual_mio::{ +- HandlerGuardState, InterestGuard, InterestHandler, InterestType, Selector, state_as_waker_map, +-}; ++use virtual_mio::{HandlerGuardState, InterestHandler, InterestType, Selector, state_as_waker_map}; ++ ++const TCP_ACCEPT_BACKLOG_DRAIN_HARD_LIMIT: usize = 128; ++ ++fn normalize_accept_backlog_limit(backlog: usize) -> usize { ++ backlog.clamp(1, TCP_ACCEPT_BACKLOG_DRAIN_HARD_LIMIT) ++} + + #[derive(Debug)] + pub struct LocalNetworking { +@@ -75,6 +79,24 @@ impl VirtualNetworking for LocalNetworking { + only_v6: bool, + reuse_port: bool, + reuse_addr: bool, ++ ) -> Result> { ++ self.listen_tcp_with_backlog( ++ addr, ++ only_v6, ++ reuse_port, ++ reuse_addr, ++ TCP_ACCEPT_BACKLOG_DRAIN_HARD_LIMIT, ++ ) ++ .await ++ } ++ ++ async fn listen_tcp_with_backlog( ++ &self, ++ addr: SocketAddr, ++ only_v6: bool, ++ reuse_port: bool, ++ reuse_addr: bool, ++ backlog: usize, + ) -> Result> { + if let Some(ruleset) = self.ruleset.as_ref() + && !ruleset.allows_socket(addr, Direction::Inbound) +@@ -93,6 +115,7 @@ impl VirtualNetworking for LocalNetworking { + no_delay: None, + keep_alive: None, + backlog: Default::default(), ++ accept_backlog_limit: normalize_accept_backlog_limit(backlog), + ruleset: self.ruleset.clone(), + }) + }) +@@ -230,10 +253,30 @@ pub struct LocalTcpListener { + no_delay: Option, + keep_alive: Option, + backlog: VecDeque<(Box, SocketAddr)>, ++ accept_backlog_limit: usize, + ruleset: Option, + } + + impl LocalTcpListener { ++ fn prime_accept_backlog(&mut self) { ++ // Preserve level-triggered listener readiness for guests even when the ++ // host selector only delivered one edge for a burst of queued accepts. ++ while self.backlog.len() < self.accept_backlog_limit { ++ match self.try_accept_internal() { ++ Ok(child) => self.backlog.push_back(child), ++ Err(NetworkError::WouldBlock) => break, ++ Err(err) => { ++ tracing::debug!(error = ?err, "failed to prime TCP listener accept backlog"); ++ break; ++ } ++ } ++ } ++ ++ if !self.backlog.is_empty() { ++ self.handler_guard.push_interest(InterestType::Readable); ++ } ++ } ++ + fn try_accept_internal(&mut self) -> Result<(Box, SocketAddr)> { + match self.stream.accept().map_err(io_err_into_net_error) { + Ok((stream, addr)) => { +@@ -254,10 +297,10 @@ impl LocalTcpListener { + Ok((Box::new(socket), addr)) + } + Err(NetworkError::WouldBlock) => { +- if let HandlerGuardState::WakerMap(_, map) = &mut self.handler_guard { +- map.pop(InterestType::Readable); +- map.pop(InterestType::Writable); +- } ++ self.handler_guard ++ .pop_waker_interest(InterestType::Readable); ++ self.handler_guard ++ .pop_waker_interest(InterestType::Writable); + Err(NetworkError::WouldBlock) + } + Err(err) => Err(err), +@@ -267,34 +310,27 @@ impl LocalTcpListener { + + impl VirtualTcpListener for LocalTcpListener { + fn try_accept(&mut self) -> Result<(Box, SocketAddr)> { +- if let Some(child) = self.backlog.pop_front() { +- return Ok(child); +- } +- self.try_accept_internal() +- } ++ let child = match self.backlog.pop_front() { ++ Some(child) => child, ++ None => self.try_accept_internal()?, ++ }; + +- fn set_handler(&mut self, mut handler: Box) -> Result<()> { +- if let HandlerGuardState::ExternalHandler(guard) = &mut self.handler_guard { +- match guard.replace_handler(handler) { +- Ok(()) => return Ok(()), +- Err(h) => handler = h, +- } ++ self.prime_accept_backlog(); + +- // the handler could not be replaced so we need to build a new handler instead +- if let Err(err) = guard.unregister(&mut self.stream) { +- tracing::debug!("failed to unregister previous token - {}", err); +- } +- } ++ Ok(child) ++ } + +- let guard = InterestGuard::new( +- &self.selector, +- handler, +- &mut self.stream, +- mio::Interest::READABLE.add(mio::Interest::WRITABLE), +- ) +- .map_err(io_err_into_net_error)?; ++ fn set_handler(&mut self, handler: Box) -> Result<()> { ++ self.handler_guard ++ .set_external_handler( ++ &self.selector, ++ &mut self.stream, ++ mio::Interest::READABLE.add(mio::Interest::WRITABLE), ++ handler, ++ ) ++ .map_err(io_err_into_net_error)?; + +- self.handler_guard = HandlerGuardState::ExternalHandler(guard); ++ self.prime_accept_backlog(); + + Ok(()) + } +@@ -331,17 +367,9 @@ impl LocalTcpListener { + + impl VirtualIoSource for LocalTcpListener { + fn remove_handler(&mut self) { +- let mut guard = HandlerGuardState::None; +- std::mem::swap(&mut guard, &mut self.handler_guard); +- match guard { +- HandlerGuardState::ExternalHandler(mut guard) => { +- guard.unregister(&mut self.stream).ok(); +- } +- HandlerGuardState::WakerMap(mut guard, _) => { +- guard.unregister(&mut self.stream).ok(); +- } +- HandlerGuardState::None => {} +- } ++ self.handler_guard ++ .remove_external_handler(&mut self.stream) ++ .ok(); + } + + fn poll_read_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { +@@ -353,9 +381,14 @@ impl VirtualIoSource for LocalTcpListener { + let map = state_as_waker_map(state, selector, source).map_err(io_err_into_net_error)?; + map.add(InterestType::Readable, cx.waker()); + +- if let Ok(child) = self.try_accept_internal() { +- self.backlog.push_back(child); +- return Poll::Ready(Ok(1)); ++ match self.try_accept_internal() { ++ Ok(child) => { ++ self.backlog.push_back(child); ++ self.prime_accept_backlog(); ++ return Poll::Ready(Ok(self.backlog.len())); ++ } ++ Err(NetworkError::WouldBlock) => {} ++ Err(err) => return Poll::Ready(Err(err)), + } + Poll::Pending + } +@@ -369,9 +402,14 @@ impl VirtualIoSource for LocalTcpListener { + let map = state_as_waker_map(state, selector, source).map_err(io_err_into_net_error)?; + map.add(InterestType::Writable, cx.waker()); + +- if let Ok(child) = self.try_accept_internal() { +- self.backlog.push_back(child); +- return Poll::Ready(Ok(1)); ++ match self.try_accept_internal() { ++ Ok(child) => { ++ self.backlog.push_back(child); ++ self.prime_accept_backlog(); ++ return Poll::Ready(Ok(self.backlog.len())); ++ } ++ Err(NetworkError::WouldBlock) => {} ++ Err(err) => return Poll::Ready(Err(err)), + } + Poll::Pending + } +@@ -563,9 +601,8 @@ impl VirtualConnectedSocket for LocalTcpStream { + let ret = self.stream.write(data).map_err(io_err_into_net_error); + match &ret { + Ok(0) | Err(NetworkError::WouldBlock) => { +- if let HandlerGuardState::WakerMap(_, map) = &mut self.handler_guard { +- map.pop(InterestType::Writable); +- } ++ self.handler_guard ++ .pop_waker_interest(InterestType::Writable); + } + _ => {} + } +@@ -588,15 +625,37 @@ impl VirtualConnectedSocket for LocalTcpStream { + if !peek { + self.buffer.advance(amt); + } ++ if peek || !self.buffer.is_empty() { ++ self.handler_guard.push_interest(InterestType::Readable); ++ } + return Ok(amt); + } + +- if peek { ++ let ret = if peek { + self.stream.peek(buf) + } else { + self.stream.read(buf) + } +- .map_err(io_err_into_net_error) ++ .map_err(io_err_into_net_error); ++ ++ match &ret { ++ Ok(0) => self.handler_guard.push_interest(InterestType::Closed), ++ Ok(_) if peek => self.handler_guard.push_interest(InterestType::Readable), ++ Ok(_) => { ++ #[cfg(not(target_os = "windows"))] ++ prime_socket_read_readiness(&mut self.handler_guard, self.stream.as_raw_fd()); ++ } ++ Err(NetworkError::WouldBlock) => { ++ self.handler_guard ++ .pop_waker_interest(InterestType::Readable); ++ } ++ Err(NetworkError::ConnectionAborted) | Err(NetworkError::ConnectionReset) => { ++ self.handler_guard.push_interest(InterestType::Closed); ++ } ++ Err(_) => self.handler_guard.push_interest(InterestType::Error), ++ } ++ ++ ret + } + } + +@@ -651,29 +710,17 @@ impl VirtualSocket for LocalTcpStream { + } + } + +- fn set_handler(&mut self, mut handler: Box) -> Result<()> { +- if let HandlerGuardState::ExternalHandler(guard) = &mut self.handler_guard { +- match guard.replace_handler(handler) { +- Ok(()) => return Ok(()), +- Err(h) => handler = h, +- } +- +- // the handler could not be replaced so we need to build a new handler instead +- if let Err(err) = guard.unregister(&mut self.stream) { +- tracing::debug!("failed to unregister previous token - {}", err); +- } +- } +- +- let guard = InterestGuard::new( +- &self.selector, +- handler, +- &mut self.stream, +- mio::Interest::READABLE.add(mio::Interest::WRITABLE), +- ) +- .map_err(io_err_into_net_error)?; +- +- self.handler_guard = HandlerGuardState::ExternalHandler(guard); +- ++ fn set_handler(&mut self, handler: Box) -> Result<()> { ++ self.handler_guard ++ .set_external_handler( ++ &self.selector, ++ &mut self.stream, ++ mio::Interest::READABLE.add(mio::Interest::WRITABLE), ++ handler, ++ ) ++ .map_err(io_err_into_net_error)?; ++ #[cfg(not(target_os = "windows"))] ++ prime_socket_readiness(&mut self.handler_guard, self.stream.as_raw_fd()); + Ok(()) + } + } +@@ -698,17 +745,9 @@ impl LocalTcpStream { + + impl VirtualIoSource for LocalTcpStream { + fn remove_handler(&mut self) { +- let mut guard = HandlerGuardState::None; +- std::mem::swap(&mut guard, &mut self.handler_guard); +- match guard { +- HandlerGuardState::ExternalHandler(mut guard) => { +- guard.unregister(&mut self.stream).ok(); +- } +- HandlerGuardState::WakerMap(mut guard, _) => { +- guard.unregister(&mut self.stream).ok(); +- } +- HandlerGuardState::None => {} +- } ++ self.handler_guard ++ .remove_external_handler(&mut self.stream) ++ .ok(); + } + + fn poll_read_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { +@@ -740,6 +779,39 @@ impl VirtualIoSource for LocalTcpStream { + } + } + ++ fn poll_read_ready_direct(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { ++ #[cfg(target_os = "windows")] ++ { ++ return self.poll_read_ready(cx); ++ } ++ ++ if !self.buffer.is_empty() { ++ return Poll::Ready(Ok(self.buffer.len())); ++ } ++ ++ let (state, selector, stream, _) = self.split_borrow(); ++ let map = state_as_waker_map(state, selector, stream).map_err(io_err_into_net_error)?; ++ map.pop(InterestType::Readable); ++ map.add(InterestType::Readable, cx.waker()); ++ map.add(InterestType::Closed, cx.waker()); ++ ++ if map.has_interest(InterestType::Closed) { ++ return Poll::Ready(Ok(0)); ++ } ++ ++ #[cfg(not(target_os = "windows"))] ++ if let Some(revents) = libc_poll( ++ stream.as_raw_fd(), ++ libc::POLLIN | libc::POLLHUP | libc::POLLERR, ++ ) { ++ if (revents & (libc::POLLIN | libc::POLLHUP | libc::POLLERR)) != 0 { ++ return Poll::Ready(Ok(1)); ++ } ++ } ++ ++ Poll::Pending ++ } ++ + fn poll_write_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { + let (state, selector, stream, _) = self.split_borrow(); + let map = state_as_waker_map(state, selector, stream).map_err(io_err_into_net_error)?; +@@ -788,6 +860,46 @@ fn libc_poll(fd: RawFd, events: libc::c_short) -> Option { + } + } + ++#[cfg(not(target_os = "windows"))] ++fn prime_socket_readiness(handler_guard: &mut HandlerGuardState, fd: RawFd) { ++ let Some(revents) = libc_poll( ++ fd, ++ libc::POLLIN | libc::POLLOUT | libc::POLLHUP | libc::POLLERR, ++ ) else { ++ return; ++ }; ++ ++ if (revents & libc::POLLIN) != 0 { ++ handler_guard.push_interest(InterestType::Readable); ++ } ++ if (revents & libc::POLLOUT) != 0 { ++ handler_guard.push_interest(InterestType::Writable); ++ } ++ if (revents & libc::POLLHUP) != 0 { ++ handler_guard.push_interest(InterestType::Closed); ++ } ++ if (revents & libc::POLLERR) != 0 { ++ handler_guard.push_interest(InterestType::Error); ++ } ++} ++ ++#[cfg(not(target_os = "windows"))] ++fn prime_socket_read_readiness(handler_guard: &mut HandlerGuardState, fd: RawFd) { ++ let Some(revents) = libc_poll(fd, libc::POLLIN | libc::POLLHUP | libc::POLLERR) else { ++ return; ++ }; ++ ++ if (revents & libc::POLLIN) != 0 { ++ handler_guard.push_interest(InterestType::Readable); ++ } ++ if (revents & libc::POLLHUP) != 0 { ++ handler_guard.push_interest(InterestType::Closed); ++ } ++ if (revents & libc::POLLERR) != 0 { ++ handler_guard.push_interest(InterestType::Error); ++ } ++} ++ + #[derive(Debug)] + pub struct LocalUdpSocket { + socket: mio::net::UdpSocket, +@@ -910,9 +1022,8 @@ impl VirtualConnectionlessSocket for LocalUdpSocket { + .map_err(io_err_into_net_error); + match &ret { + Ok(0) | Err(NetworkError::WouldBlock) => { +- if let HandlerGuardState::WakerMap(_, map) = &mut self.handler_guard { +- map.pop(InterestType::Writable); +- } ++ self.handler_guard ++ .pop_waker_interest(InterestType::Writable); + } + _ => {} + } +@@ -925,12 +1036,30 @@ impl VirtualConnectionlessSocket for LocalUdpSocket { + peek: bool, + ) -> Result<(usize, SocketAddr)> { + let buf: &mut [u8] = unsafe { std::mem::transmute(buf) }; +- if peek { ++ let ret = if peek { + self.socket.peek_from(buf) + } else { + self.socket.recv_from(buf) + } +- .map_err(io_err_into_net_error) ++ .map_err(io_err_into_net_error); ++ ++ match &ret { ++ Ok(_) if peek => self.handler_guard.push_interest(InterestType::Readable), ++ Ok(_) => { ++ #[cfg(not(target_os = "windows"))] ++ prime_socket_read_readiness(&mut self.handler_guard, self.socket.as_raw_fd()); ++ } ++ Err(NetworkError::WouldBlock) => { ++ self.handler_guard ++ .pop_waker_interest(InterestType::Readable); ++ } ++ Err(NetworkError::ConnectionAborted) | Err(NetworkError::ConnectionReset) => { ++ self.handler_guard.push_interest(InterestType::Closed); ++ } ++ Err(_) => self.handler_guard.push_interest(InterestType::Error), ++ } ++ ++ ret + } + } + +@@ -951,31 +1080,17 @@ impl VirtualSocket for LocalUdpSocket { + Ok(SocketStatus::Opened) + } + +- fn set_handler(&mut self, mut handler: Box) -> Result<()> { +- if let HandlerGuardState::ExternalHandler(guard) = &mut self.handler_guard { +- match guard.replace_handler(handler) { +- Ok(()) => { +- return Ok(()); +- } +- Err(h) => handler = h, +- } +- +- // the handler could not be replaced so we need to build a new handler instead +- if let Err(err) = guard.unregister(&mut self.socket) { +- tracing::debug!("failed to unregister previous token - {}", err); +- } +- } +- +- let guard = InterestGuard::new( +- &self.selector, +- handler, +- &mut self.socket, +- mio::Interest::READABLE.add(mio::Interest::WRITABLE), +- ) +- .map_err(io_err_into_net_error)?; +- +- self.handler_guard = HandlerGuardState::ExternalHandler(guard); +- ++ fn set_handler(&mut self, handler: Box) -> Result<()> { ++ self.handler_guard ++ .set_external_handler( ++ &self.selector, ++ &mut self.socket, ++ mio::Interest::READABLE.add(mio::Interest::WRITABLE), ++ handler, ++ ) ++ .map_err(io_err_into_net_error)?; ++ #[cfg(not(target_os = "windows"))] ++ prime_socket_readiness(&mut self.handler_guard, self.socket.as_raw_fd()); + Ok(()) + } + } +@@ -994,17 +1109,9 @@ impl LocalUdpSocket { + + impl VirtualIoSource for LocalUdpSocket { + fn remove_handler(&mut self) { +- let mut guard = HandlerGuardState::None; +- std::mem::swap(&mut guard, &mut self.handler_guard); +- match guard { +- HandlerGuardState::ExternalHandler(mut guard) => { +- guard.unregister(&mut self.socket).ok(); +- } +- HandlerGuardState::WakerMap(mut guard, _) => { +- guard.unregister(&mut self.socket).ok(); +- } +- HandlerGuardState::None => {} +- } ++ self.handler_guard ++ .remove_external_handler(&mut self.socket) ++ .ok(); + } + + fn poll_read_ready(&mut self, cx: &mut std::task::Context<'_>) -> Poll> { +diff --git a/lib/virtual-net/src/lib.rs b/lib/virtual-net/src/lib.rs +index 22ae000..58cbbf4 100644 +--- a/lib/virtual-net/src/lib.rs ++++ b/lib/virtual-net/src/lib.rs +@@ -79,6 +79,11 @@ pub trait VirtualIoSource: fmt::Debug + Send + Sync + 'static { + /// Polls the source to see if there is data waiting + fn poll_read_ready(&mut self, cx: &mut Context<'_>) -> Poll>; + ++ /// Polls read readiness without eagerly buffering data when a backend can support it. ++ fn poll_read_ready_direct(&mut self, cx: &mut Context<'_>) -> Poll> { ++ self.poll_read_ready(cx) ++ } ++ + /// Polls the source to see if data can be sent + fn poll_write_ready(&mut self, cx: &mut Context<'_>) -> Poll>; + } +@@ -183,6 +188,19 @@ pub trait VirtualNetworking: fmt::Debug + Send + Sync + 'static { + Err(NetworkError::Unsupported) + } + ++ /// Listens for TCP connections and preserves the guest-requested accept ++ /// backlog when the backend can apply it. ++ async fn listen_tcp_with_backlog( ++ &self, ++ addr: SocketAddr, ++ only_v6: bool, ++ reuse_port: bool, ++ reuse_addr: bool, ++ backlog: usize, ++ ) -> Result> { ++ self.listen_tcp(addr, only_v6, reuse_port, reuse_addr).await ++ } ++ + /// Opens a UDP socket that listens on a specific IP and Port combination + /// Multiple servers (processes or threads) can bind to the same port if they each set + /// the reuse-port and-or reuse-addr flags +diff --git a/lib/virtual-net/src/tests.rs b/lib/virtual-net/src/tests.rs +index 9ed1a46..f1ce408 100644 +--- a/lib/virtual-net/src/tests.rs ++++ b/lib/virtual-net/src/tests.rs +@@ -1,7 +1,10 @@ + #![allow(unused)] + use std::{ + net::{Ipv4Addr, SocketAddrV4}, +- sync::atomic::{AtomicU16, Ordering}, ++ sync::{ ++ atomic::{AtomicU16, Ordering}, ++ mpsc, ++ }, + }; + + use tracing_test::traced_test; +@@ -13,6 +16,7 @@ use crate::{ + meta::FrameSerializationFormat, + }; + use tokio::io::{AsyncReadExt, AsyncWriteExt}; ++use virtual_mio::{InterestHandler, InterestType}; + + use super::*; + +@@ -513,6 +517,257 @@ async fn test_connect_tcp_returns_immediately_for_in_progress_connect() { + } + } + ++#[cfg(not(target_os = "windows"))] ++#[traced_test] ++#[tokio::test] ++#[serial_test::serial] ++async fn test_local_tcp_listener_reports_readable_while_accept_backlog_remains() { ++ use std::net::TcpStream; ++ use std::time::Duration; ++ ++ #[derive(Debug)] ++ struct ReadableHandler { ++ sender: mpsc::Sender, ++ } ++ ++ impl InterestHandler for ReadableHandler { ++ fn push_interest(&mut self, interest: InterestType) { ++ self.sender.send(interest).ok(); ++ } ++ ++ fn pop_interest(&mut self, _interest: InterestType) -> bool { ++ false ++ } ++ ++ fn has_interest(&self, _interest: InterestType) -> bool { ++ false ++ } ++ } ++ ++ let networking = LocalNetworking::new(); ++ let mut listener = networking ++ .listen_tcp( ++ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), ++ false, ++ false, ++ false, ++ ) ++ .await ++ .unwrap(); ++ let addr = listener.addr_local().unwrap(); ++ ++ let (sender, receiver) = mpsc::channel(); ++ listener ++ .set_handler(Box::new(ReadableHandler { sender })) ++ .unwrap(); ++ ++ let clients: Vec<_> = (0..20).map(|_| TcpStream::connect(addr).unwrap()).collect(); ++ receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("listener should report first readable accept"); ++ ++ let mut accepted = Vec::new(); ++ for idx in 0..clients.len() { ++ accepted.push(listener.try_accept().unwrap()); ++ if idx + 1 < clients.len() { ++ receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("listener should keep reporting readable across a queued burst"); ++ } ++ } ++ drop(accepted); ++ drop(clients); ++} ++ ++#[cfg(not(target_os = "windows"))] ++#[traced_test] ++#[tokio::test] ++#[serial_test::serial] ++async fn test_local_tcp_listener_reports_existing_accept_readiness_on_handler_attach() { ++ use std::net::TcpStream; ++ use std::time::Duration; ++ ++ #[derive(Debug)] ++ struct ReadableHandler { ++ sender: mpsc::Sender, ++ } ++ ++ impl InterestHandler for ReadableHandler { ++ fn push_interest(&mut self, interest: InterestType) { ++ self.sender.send(interest).ok(); ++ } ++ ++ fn pop_interest(&mut self, _interest: InterestType) -> bool { ++ false ++ } ++ ++ fn has_interest(&self, _interest: InterestType) -> bool { ++ false ++ } ++ } ++ ++ let networking = LocalNetworking::new(); ++ let mut listener = networking ++ .listen_tcp( ++ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), ++ false, ++ false, ++ false, ++ ) ++ .await ++ .unwrap(); ++ let addr = listener.addr_local().unwrap(); ++ let client = TcpStream::connect(addr).unwrap(); ++ ++ let (sender, receiver) = mpsc::channel(); ++ listener ++ .set_handler(Box::new(ReadableHandler { sender })) ++ .unwrap(); ++ assert_eq!( ++ receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("listener should report queued accept readiness"), ++ InterestType::Readable ++ ); ++ drop(client); ++} ++ ++#[cfg(not(target_os = "windows"))] ++#[traced_test] ++#[tokio::test] ++#[serial_test::serial] ++async fn test_local_tcp_stream_reports_existing_readiness_on_handler_attach() { ++ use std::io::Write; ++ use std::net::TcpStream; ++ use std::time::Duration; ++ ++ #[derive(Debug)] ++ struct ReadableHandler { ++ sender: mpsc::Sender, ++ } ++ ++ impl InterestHandler for ReadableHandler { ++ fn push_interest(&mut self, interest: InterestType) { ++ self.sender.send(interest).ok(); ++ } ++ ++ fn pop_interest(&mut self, _interest: InterestType) -> bool { ++ false ++ } ++ ++ fn has_interest(&self, _interest: InterestType) -> bool { ++ false ++ } ++ } ++ ++ let networking = LocalNetworking::new(); ++ let mut listener = networking ++ .listen_tcp( ++ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), ++ false, ++ false, ++ false, ++ ) ++ .await ++ .unwrap(); ++ let addr = listener.addr_local().unwrap(); ++ let mut client = TcpStream::connect(addr).unwrap(); ++ let (mut server, _) = listener.accept().await.unwrap(); ++ ++ client.write_all(b"ready-before-handler").unwrap(); ++ std::thread::sleep(Duration::from_millis(50)); ++ ++ let (sender, receiver) = mpsc::channel(); ++ server ++ .set_handler(Box::new(ReadableHandler { sender })) ++ .unwrap(); ++ let mut saw_readable = false; ++ for _ in 0..2 { ++ if receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("stream should report buffered host readiness") ++ == InterestType::Readable ++ { ++ saw_readable = true; ++ break; ++ } ++ } ++ assert!( ++ saw_readable, ++ "stream did not report pre-existing readable data" ++ ); ++} ++ ++#[cfg(not(target_os = "windows"))] ++#[traced_test] ++#[tokio::test] ++#[serial_test::serial] ++async fn test_local_tcp_stream_reissues_readable_after_partial_recv() { ++ use std::io::Write; ++ use std::net::TcpStream; ++ use std::time::Duration; ++ ++ #[derive(Debug)] ++ struct ReadableHandler { ++ sender: mpsc::Sender, ++ } ++ ++ impl InterestHandler for ReadableHandler { ++ fn push_interest(&mut self, interest: InterestType) { ++ self.sender.send(interest).ok(); ++ } ++ ++ fn pop_interest(&mut self, _interest: InterestType) -> bool { ++ false ++ } ++ ++ fn has_interest(&self, _interest: InterestType) -> bool { ++ false ++ } ++ } ++ ++ let networking = LocalNetworking::new(); ++ let mut listener = networking ++ .listen_tcp( ++ SocketAddr::from((Ipv4Addr::LOCALHOST, 0)), ++ false, ++ false, ++ false, ++ ) ++ .await ++ .unwrap(); ++ let addr = listener.addr_local().unwrap(); ++ let mut client = TcpStream::connect(addr).unwrap(); ++ let (mut server, _) = listener.accept().await.unwrap(); ++ ++ let (sender, receiver) = mpsc::channel(); ++ server ++ .set_handler(Box::new(ReadableHandler { sender })) ++ .unwrap(); ++ ++ client.write_all(&vec![7_u8; 64 * 1024]).unwrap(); ++ loop { ++ if receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("stream should report readable data") ++ == InterestType::Readable ++ { ++ break; ++ } ++ } ++ std::thread::sleep(Duration::from_millis(50)); ++ while receiver.try_recv().is_ok() {} ++ ++ let mut buf = [MaybeUninit::::uninit(); 8]; ++ assert_eq!(server.try_recv(&mut buf, false).unwrap(), 8); ++ assert_eq!( ++ receiver ++ .recv_timeout(Duration::from_secs(1)) ++ .expect("partial recv should leave the stream readable"), ++ InterestType::Readable ++ ); ++} ++ + #[cfg(not(target_os = "windows"))] + #[traced_test] + #[tokio::test] diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0004-wasix-open-file-description-operations.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0004-wasix-open-file-description-operations.patch new file mode 100644 index 000000000..540a43d53 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0004-wasix-open-file-description-operations.patch @@ -0,0 +1,5208 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 4/9] wasix: preserve descriptor identity and filesystem operations + +Apply filesystem operations to the original open-file description, including +positional reads/writes, seek, renumber, advice, mmap and explicit sync ranges. +These operations must retain shared offsets, flags and writeback ownership +across asynchronous suspension and descriptor reuse. Depends on virtual-fs +and the later instance/task integration; this is a review slice of a coupled +stack, not a standalone buildable or upstream submission. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/lib/wasix/src/fs/fd.rs b/lib/wasix/src/fs/fd.rs +index 362b933..6f0a94e 100644 +--- a/lib/wasix/src/fs/fd.rs ++++ b/lib/wasix/src/fs/fd.rs +@@ -2,19 +2,202 @@ use std::{ + borrow::Cow, + collections::HashMap, + path::PathBuf, +- sync::{Arc, RwLock, RwLockReadGuard, RwLockWriteGuard, atomic::AtomicU64}, ++ sync::{ ++ Arc, Mutex, OnceLock, RwLock, RwLockReadGuard, RwLockWriteGuard, Weak, ++ atomic::{AtomicU64, Ordering}, ++ }, + }; + + #[cfg(feature = "enable-serde")] + use serde_derive::{Deserialize, Serialize}; +-use virtual_fs::{Pipe, PipeRx, PipeTx, VirtualFile}; +-use wasmer_wasix_types::wasi::{Fd as WasiFd, Fdflags, Fdflagsext, Filestat, Rights}; ++use virtual_fs::{Pipe, PipeRx, PipeTx, VirtualDirectory, VirtualFile}; ++use virtual_mio::InterestHandlerFanout; ++use wasmer_wasix_types::wasi::{ ++ Errno, Fd as WasiFd, Fdflags, Fdflagsext, Filestat, Filetype, Rights, ++}; + + use crate::net::socket::InodeSocket; +-use crate::os::epoll::EpollState; ++use crate::os::epoll::{EpollState, EpollSubState, EpollSubscriptionKey}; + + use super::{InodeGuard, InodeWeakGuard, NotificationInner}; + ++pub type ReaddirSnapshot = Arc>; ++pub type ReaddirCache = Arc>>; ++pub(crate) type DirectorySyncHandle = Arc; ++ ++static NEXT_OPEN_FILE_DESCRIPTION_ID: AtomicU64 = AtomicU64::new(1); ++ ++#[derive(Debug)] ++struct EpollReverseRegistration { ++ state: Weak, ++ subscription: Weak, ++ key: EpollSubscriptionKey, ++} ++ ++#[derive(Debug, Default)] ++struct OpenFileDescriptionMutable { ++ descriptor_count: u32, ++ next_registration_id: u64, ++ registrations: HashMap, ++} ++ ++/// Lifecycle shared by descriptors produced by dup/fork from one open file ++/// description. Independent opens get distinct identities even when they ++/// reference the same inode. ++#[derive(Debug)] ++pub(crate) struct OpenFileDescription { ++ id: u64, ++ mutable: Mutex, ++ interest_fanout: OnceLock, ++ /// A real opened directory descriptor, or the error encountered while ++ /// acquiring one. `None` means this OFD is not a sync-capable directory. ++ /// Keeping the result on the OFD preserves identity across dup/fork and ++ /// ensures unsupported backends fail closed when sync is requested. ++ directory_sync: Option>, ++} ++ ++impl OpenFileDescription { ++ pub(crate) fn new() -> Arc { ++ Self::new_with_directory_sync(None) ++ } ++ ++ pub(crate) fn new_with_directory_sync( ++ directory_sync: Option>, ++ ) -> Arc { ++ let id = NEXT_OPEN_FILE_DESCRIPTION_ID ++ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { ++ current.checked_add(1) ++ }) ++ .expect("open-file-description identity space exhausted"); ++ Arc::new(Self { ++ id, ++ mutable: Mutex::new(OpenFileDescriptionMutable::default()), ++ interest_fanout: OnceLock::new(), ++ directory_sync, ++ }) ++ } ++ ++ pub(crate) fn directory_sync(&self) -> Option> { ++ self.directory_sync.clone() ++ } ++ ++ pub(crate) fn id(&self) -> u64 { ++ self.id ++ } ++ ++ #[cfg(test)] ++ pub(crate) fn descriptor_count(&self) -> u32 { ++ self.mutable.lock().unwrap().descriptor_count ++ } ++ ++ pub(crate) fn interest_fanout(&self) -> InterestHandlerFanout { ++ self.interest_fanout ++ .get_or_init(InterestHandlerFanout::default) ++ .clone() ++ } ++ ++ #[cfg(test)] ++ pub(crate) fn interest_fanout_initialized(&self) -> bool { ++ self.interest_fanout.get().is_some() ++ } ++ ++ #[cfg(test)] ++ pub(crate) fn epoll_registration_count(&self) -> usize { ++ self.mutable.lock().unwrap().registrations.len() ++ } ++ ++ pub(crate) fn acquire_descriptor(&self) { ++ let mut mutable = self.mutable.lock().unwrap(); ++ mutable.descriptor_count = mutable ++ .descriptor_count ++ .checked_add(1) ++ .expect("open-file-description descriptor count overflow"); ++ } ++ ++ /// Releases one user-visible descriptor. Reverse epoll records are drained ++ /// under the OFD lock and applied only after releasing it, avoiding an ++ /// OFD-registry -> epoll-state lock-order dependency. ++ pub(crate) fn release_descriptor(&self) -> bool { ++ let registrations = { ++ let mut mutable = self.mutable.lock().unwrap(); ++ mutable.descriptor_count = mutable ++ .descriptor_count ++ .checked_sub(1) ++ .expect("open-file-description descriptor dropped too many times"); ++ if mutable.descriptor_count != 0 { ++ return false; ++ } ++ std::mem::take(&mut mutable.registrations) ++ }; ++ ++ for registration in registrations.into_values() { ++ let Some(state) = registration.state.upgrade() else { ++ continue; ++ }; ++ let Some(subscription) = registration.subscription.upgrade() else { ++ continue; ++ }; ++ state.remove_if_same(registration.key, &subscription); ++ } ++ true ++ } ++ ++ pub(crate) fn register_epoll( ++ self: &Arc, ++ state: &Arc, ++ subscription: &Arc, ++ key: EpollSubscriptionKey, ++ ) -> Option { ++ let mut mutable = self.mutable.lock().unwrap(); ++ if mutable.descriptor_count == 0 { ++ return None; ++ } ++ ++ let registration_id = mutable.next_registration_id; ++ mutable.next_registration_id = mutable ++ .next_registration_id ++ .checked_add(1) ++ .expect("epoll reverse-registration identity space exhausted"); ++ mutable.registrations.insert( ++ registration_id, ++ EpollReverseRegistration { ++ state: Arc::downgrade(state), ++ subscription: Arc::downgrade(subscription), ++ key, ++ }, ++ ); ++ Some(EpollRegistrationGuard { ++ description: Arc::downgrade(self), ++ registration_id, ++ }) ++ } ++} ++ ++#[cfg(feature = "enable-serde")] ++fn default_open_file_description() -> Arc { ++ OpenFileDescription::new() ++} ++ ++#[derive(Debug)] ++pub(crate) struct EpollRegistrationGuard { ++ description: Weak, ++ registration_id: u64, ++} ++ ++impl Drop for EpollRegistrationGuard { ++ fn drop(&mut self) { ++ let Some(description) = self.description.upgrade() else { ++ return; ++ }; ++ description ++ .mutable ++ .lock() ++ .unwrap() ++ .registrations ++ .remove(&self.registration_id); ++ } ++} ++ + #[derive(Debug, Clone)] + #[cfg_attr(feature = "enable-serde", derive(Serialize, Deserialize))] + pub struct Fd { +@@ -38,7 +221,24 @@ pub struct FdInner { + pub rights_inheriting: Rights, + pub flags: Fdflags, // This is file table related flags, not fd flags + pub offset: Arc, // This also belongs in the file table +- pub fd_flags: Fdflagsext, // This is the actual FD flags that belongs here ++ /// Identity and lifecycle of the underlying open file description. This ++ /// is shared by dup/fork but not by independent opens of the same inode. ++ /// ++ /// `enable-serde` currently reconstructs this skipped field independently ++ /// for every deserialized `Fd`; it therefore does not preserve dup/fork ++ /// OFD topology, reverse epoll registrations, or live directory-sync ++ /// handles. A topology-aware snapshot representation is required before ++ /// restored descriptors can make the same lifecycle/durability guarantees ++ /// as live descriptors. The sealed PostgreSQL runtime does not enable ++ /// journaling; enabling it remains a release blocker for this path. ++ #[cfg_attr( ++ feature = "enable-serde", ++ serde(skip, default = "default_open_file_description") ++ )] ++ pub(crate) ofd: Arc, ++ #[cfg_attr(feature = "enable-serde", serde(skip, default))] ++ pub readdir_cache: ReaddirCache, ++ pub fd_flags: Fdflagsext, // This is the actual FD flags that belongs here + } + + impl Fd { +@@ -59,6 +259,19 @@ impl Fd { + /// + /// This permission is currently unused when deserializing. + pub const CREATE: u16 = 16; ++ ++ pub(crate) fn acquire_descriptor(&self) { ++ self.inner.ofd.acquire_descriptor(); ++ self.inode.acquire_handle(); ++ } ++ ++ pub(crate) fn release_descriptor(&self) { ++ let last_description_handle = self.inner.ofd.release_descriptor(); ++ if last_description_handle && let Kind::Epoll { state } = &*self.inode.read() { ++ state.close(); ++ } ++ self.inode.drop_one_handle(); ++ } + } + + /// A file that Wasi knows about that may or may not be open +diff --git a/lib/wasix/src/fs/fd_list.rs b/lib/wasix/src/fs/fd_list.rs +index 3540a55..f6d7c2b 100644 +--- a/lib/wasix/src/fs/fd_list.rs ++++ b/lib/wasix/src/fs/fd_list.rs +@@ -65,7 +65,7 @@ impl FdList { + } + + pub fn insert_first_free(&mut self, fd: Fd) -> WasiFd { +- fd.inode.acquire_handle(); ++ fd.acquire_descriptor(); + match self.first_free { + Some(free) => { + assert!(self.fds[free].is_none()); +@@ -107,7 +107,7 @@ impl FdList { + // If there's a hole but its index is too low, we need to search + Some(_) => { + // This is handled by insert or insert_first_free in every other case, but not this one +- fd.inode.acquire_handle(); ++ fd.acquire_descriptor(); + + match self.first_free_after(after_or_equal) { + // Found a suitable hole, and it's guaranteed to not be the first since +@@ -138,6 +138,25 @@ impl FdList { + } + + pub fn insert(&mut self, exclusive: bool, idx: WasiFd, fd: Fd) -> bool { ++ match self.insert_deferred(exclusive, idx, fd) { ++ Ok(displaced) => { ++ if let Some(displaced) = displaced { ++ displaced.release_descriptor(); ++ } ++ true ++ } ++ Err(_) => false, ++ } ++ } ++ ++ /// Inserts without finalizing a displaced descriptor. Callers holding the ++ /// outer fd-map lock must drop that lock before releasing the returned fd. ++ pub(crate) fn insert_deferred( ++ &mut self, ++ exclusive: bool, ++ idx: WasiFd, ++ fd: Fd, ++ ) -> Result, Fd> { + let idx = idx as usize; + + if self.fds.len() <= idx { +@@ -155,51 +174,92 @@ impl FdList { + self.fds.resize(idx + 1, None); + } + +- if let Some(ref prev_fd) = self.fds[idx] { +- if exclusive { +- return false; +- } else { +- prev_fd.inode.drop_one_handle(); +- } ++ if self.fds[idx].is_some() && exclusive { ++ return Err(fd); + } + +- fd.inode.acquire_handle(); ++ let displaced = self.fds[idx].take(); ++ fd.acquire_descriptor(); + self.fds[idx] = Some(fd); + + if self.first_free == Some(idx) { + self.first_free = self.first_free_after(idx as WasiFd + 1); + } + +- true ++ Ok(displaced) + } + + pub fn remove(&mut self, idx: WasiFd) -> Option { ++ let result = self.remove_deferred(idx); ++ if let Some(fd) = result.as_ref() { ++ fd.release_descriptor(); ++ } ++ result ++ } ++ ++ /// Removes without running last-descriptor callbacks under an outer ++ /// fd-map lock. ++ pub(crate) fn remove_deferred(&mut self, idx: WasiFd) -> Option { + let idx = idx as usize; + + let result = self.fds.get_mut(idx).and_then(|fd| fd.take()); + +- if let Some(fd) = result.as_ref() { ++ if result.is_some() { + match self.first_free { + None => self.first_free = Some(idx), + Some(x) if x > idx => self.first_free = Some(idx), + _ => (), + } +- +- fd.inode.drop_one_handle(); + } + + result + } + ++ /// Atomically moves `from` to `to` without changing the moved open-file ++ /// description's descriptor count. The displaced target is returned so ++ /// its final-close callbacks can run after the caller releases the outer ++ /// fd-map lock. ++ /// ++ /// Unlike remove followed by a normal insert, this deliberately does not ++ /// release and reacquire the source descriptor. That preserves both WASI ++ /// renumber semantics and OFD lifetime across the move. ++ pub(crate) fn renumber_deferred(&mut self, from: WasiFd, to: WasiFd) -> Result, ()> { ++ if from == to { ++ return self.get(from).map(|_| None).ok_or(()); ++ } ++ ++ let moved = self.remove_deferred(from).ok_or(())?; ++ let displaced = self.remove_deferred(to); ++ self.insert_moved(to, moved); ++ Ok(displaced) ++ } ++ ++ /// Inserts an fd whose descriptor ownership is already accounted for. ++ fn insert_moved(&mut self, idx: WasiFd, fd: Fd) { ++ let idx = idx as usize; ++ if self.fds.len() <= idx { ++ self.fds.resize(idx + 1, None); ++ } ++ debug_assert!(self.fds[idx].is_none()); ++ self.fds[idx] = Some(fd); ++ ++ if self.first_free == Some(idx) { ++ self.first_free = self.first_free_after(idx as WasiFd + 1); ++ } ++ } ++ + pub fn clear(&mut self) { +- for fd in &self.fds { +- if let Some(fd) = fd.as_ref() { +- fd.inode.drop_one_handle(); +- } ++ for fd in self.drain_deferred() { ++ fd.release_descriptor(); + } ++ } + ++ /// Drains the map without finalizing descriptor lifecycles. ++ pub(crate) fn drain_deferred(&mut self) -> Vec { ++ let drained = self.fds.iter_mut().filter_map(Option::take).collect(); + self.fds.clear(); + self.first_free = None; ++ drained + } + + pub fn iter(&self) -> FdListIterator<'_> { +@@ -225,7 +285,7 @@ impl Clone for FdList { + fn clone(&self) -> Self { + for fd in &self.fds { + if let Some(fd) = fd.as_ref() { +- fd.inode.acquire_handle(); ++ fd.acquire_descriptor(); + } + } + +@@ -292,27 +352,53 @@ impl<'a> Iterator for FdListIteratorMut<'a> { + mod tests { + use std::{ + borrow::Cow, ++ io, + sync::{ + Arc, RwLock, +- atomic::{AtomicI32, AtomicU64}, ++ atomic::{AtomicI32, AtomicU64, AtomicUsize, Ordering}, + }, + }; + + use assert_panic::assert_panic; +- use wasmer_wasix_types::wasi::{Fdflags, Fdflagsext, Rights}; ++ use wasmer_wasix_types::wasi::{EpollEventCtl, EpollType, Fdflags, Fdflagsext, Rights}; + +- use crate::fs::{Inode, InodeGuard, InodeVal, Kind, fd::FdInner}; ++ use crate::fs::{ ++ Inode, InodeGuard, InodeVal, Kind, ++ fd::{DirectorySyncHandle, FdInner, OpenFileDescription}, ++ }; ++ use crate::os::epoll::{EpollState, EpollSubscriptionKey}; + + use super::{Fd, FdList, WasiFd}; ++ use virtual_fs::VirtualDirectory; ++ ++ #[derive(Debug)] ++ struct CountingDirectory { ++ syncs: Arc, ++ } ++ ++ impl VirtualDirectory for CountingDirectory { ++ fn has_blocking_sync_all_to_disk(&self) -> bool { ++ true ++ } ++ ++ fn sync_all_to_disk_blocking(&self) -> io::Result<()> { ++ self.syncs.fetch_add(1, Ordering::Relaxed); ++ Ok(()) ++ } ++ } + + fn useless_fd(n: u16) -> Fd { ++ fd_with_kind(n, Kind::Buffer { buffer: vec![] }) ++ } ++ ++ fn fd_with_kind(n: u16, kind: Kind) -> Fd { + Fd { + open_flags: 0, + inode: InodeGuard { + ino: Inode(0), + inner: Arc::new(InodeVal { + is_preopened: false, +- kind: RwLock::new(Kind::Buffer { buffer: vec![] }), ++ kind: RwLock::new(kind), + name: RwLock::new(Cow::Borrowed("")), + stat: RwLock::new(Default::default()), + }), +@@ -321,9 +407,11 @@ mod tests { + is_stdio: false, + inner: FdInner { + offset: Arc::new(AtomicU64::new(0)), ++ ofd: OpenFileDescription::new(), + rights: Rights::empty(), + rights_inheriting: Rights::empty(), + flags: Fdflags::from_bits_preserve(n), ++ readdir_cache: Default::default(), + fd_flags: Fdflagsext::empty(), + }, + } +@@ -681,13 +769,354 @@ mod tests { + l2.clear(); + assert_eq!(fd0.inode.handle_count(), 0); + ++ // The descriptor now retains a trait-object directory handle. That ++ // handle deliberately makes the full descriptor conservative across ++ // unwind boundaries; this test is explicitly exercising a poisoned ++ // accounting invariant, so acknowledge the boundary locally instead ++ // of asserting an unwind-safety contract for every filesystem. ++ let fd0 = std::panic::AssertUnwindSafe(fd0); + assert_panic!( +- fd0.inode.drop_one_handle(), ++ fd0.0.inode.drop_one_handle(), + &str, + "InodeGuard handle dropped too many times" + ); + +- assert_panic!(drop(fd0.inode.write()), String, contains "PoisonError"); ++ assert_panic!(drop(fd0.0.inode.write()), String, contains "PoisonError"); ++ } ++ ++ #[test] ++ fn untouched_open_file_description_does_not_initialize_interest_fanout() { ++ let mut list = FdList::new(); ++ let fd = useless_fd(0); ++ let description = fd.inner.ofd.clone(); ++ ++ assert!(!description.interest_fanout_initialized()); ++ list.insert_first_free(fd); ++ assert!(!description.interest_fanout_initialized()); ++ list.remove(0).unwrap(); ++ assert!(!description.interest_fanout_initialized()); ++ } ++ ++ #[test] ++ fn remove_deferred_postpones_descriptor_finalization() { ++ let mut list = FdList::new(); ++ let fd = useless_fd(0); ++ let description = fd.inner.ofd.clone(); ++ list.insert_first_free(fd); ++ assert_eq!(description.descriptor_count(), 1); ++ ++ let removed = list.remove_deferred(0).unwrap(); ++ assert_eq!( ++ description.descriptor_count(), ++ 1, ++ "fd-map mutation must not run lifecycle callbacks while its outer lock is held" ++ ); ++ removed.release_descriptor(); ++ assert_eq!(description.descriptor_count(), 0); ++ } ++ ++ #[test] ++ fn renumber_is_an_atomic_move_and_preserves_source_descriptor_ownership() { ++ let mut list = FdList::new(); ++ let source = useless_fd(10); ++ let source_description = source.inner.ofd.clone(); ++ let target = useless_fd(20); ++ let target_description = target.inner.ofd.clone(); ++ assert!(list.insert(true, 3, source)); ++ assert!(list.insert(true, 8, target)); ++ assert_eq!(source_description.descriptor_count(), 1); ++ assert_eq!(target_description.descriptor_count(), 1); ++ ++ let displaced = list.renumber_deferred(3, 8).unwrap().unwrap(); ++ ++ assert!(list.get(3).is_none()); ++ assert!(is_useless_fd(list.get(8).unwrap(), 10)); ++ assert_eq!(source_description.descriptor_count(), 1); ++ assert_eq!(target_description.descriptor_count(), 1); ++ displaced.release_descriptor(); ++ assert_eq!(target_description.descriptor_count(), 0); ++ } ++ ++ #[test] ++ fn renumber_invalid_and_same_fd_leave_the_table_unchanged() { ++ let mut list = FdList::new(); ++ let fd = useless_fd(30); ++ let description = fd.inner.ofd.clone(); ++ assert!(list.insert(true, 4, fd)); ++ ++ assert!(list.renumber_deferred(99, 4).is_err()); ++ assert!(is_useless_fd(list.get(4).unwrap(), 30)); ++ assert_eq!(description.descriptor_count(), 1); ++ ++ assert!(list.renumber_deferred(4, 4).unwrap().is_none()); ++ assert!(is_useless_fd(list.get(4).unwrap(), 30)); ++ assert_eq!(description.descriptor_count(), 1); ++ assert!(list.renumber_deferred(99, 99).is_err()); ++ } ++ ++ #[test] ++ fn directory_sync_handle_survives_fork_clone_renumber_and_nonfinal_close() { ++ let syncs = Arc::new(AtomicUsize::new(0)); ++ let directory = Arc::new(CountingDirectory { ++ syncs: syncs.clone(), ++ }); ++ let weak_directory = Arc::downgrade(&directory); ++ let handle: DirectorySyncHandle = directory; ++ ++ let mut fd = useless_fd(50); ++ fd.inner.ofd = OpenFileDescription::new_with_directory_sync(Some(Ok(handle))); ++ let description = fd.inner.ofd.clone(); ++ let mut parent = FdList::new(); ++ assert!(parent.insert(true, 4, fd)); ++ let mut child = parent.clone(); ++ assert_eq!(description.descriptor_count(), 2); ++ ++ { ++ let parent_handle = parent ++ .get(4) ++ .unwrap() ++ .inner ++ .ofd ++ .directory_sync() ++ .unwrap() ++ .unwrap(); ++ parent_handle.sync_all_to_disk_blocking().unwrap(); ++ } ++ assert!(child.renumber_deferred(4, 9).unwrap().is_none()); ++ parent.remove(4).unwrap(); ++ assert_eq!(description.descriptor_count(), 1); ++ ++ { ++ let moved_handle = child ++ .get(9) ++ .unwrap() ++ .inner ++ .ofd ++ .directory_sync() ++ .unwrap() ++ .unwrap(); ++ moved_handle.sync_all_to_disk_blocking().unwrap(); ++ } ++ assert_eq!(syncs.load(Ordering::Relaxed), 2); ++ ++ child.remove(9).unwrap(); ++ assert_eq!(description.descriptor_count(), 0); ++ assert!(weak_directory.upgrade().is_some()); ++ drop(description); ++ assert!(weak_directory.upgrade().is_none()); ++ } ++ ++ #[test] ++ fn renumber_keeps_epoll_watch_until_moved_descriptor_final_close() { ++ let mut list = FdList::new(); ++ let fd = useless_fd(40); ++ let description = fd.inner.ofd.clone(); ++ assert!(list.insert(true, 5, fd)); ++ ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(5, description.id()); ++ let (_, subscription) = state.prepare_add(key, &readable_event(5)).unwrap(); ++ let registration = description ++ .register_epoll(&state, &subscription, key) ++ .unwrap(); ++ subscription ++ .attach_close_registration(registration) ++ .unwrap(); ++ ++ assert!(list.renumber_deferred(5, 9).unwrap().is_none()); ++ assert!(list.get(5).is_none()); ++ assert!(list.get(9).is_some()); ++ assert_eq!(description.descriptor_count(), 1); ++ assert!(state.contains_exact_subscription(key, &subscription)); ++ ++ list.remove(9).unwrap(); ++ assert_eq!(description.descriptor_count(), 0); ++ assert!(!state.contains_exact_subscription(key, &subscription)); ++ assert!(!subscription.is_active()); ++ } ++ ++ fn readable_event(fd: WasiFd) -> EpollEventCtl { ++ EpollEventCtl { ++ events: EpollType::EPOLLIN, ++ ptr: 0, ++ fd, ++ data1: 0, ++ data2: 0, ++ } ++ } ++ ++ #[test] ++ fn epoll_watch_survives_nonfinal_dup_close_and_detaches_on_final_close() { ++ let mut parent = FdList::new(); ++ let fd = useless_fd(0); ++ let description = fd.inner.ofd.clone(); ++ parent.insert_first_free(fd); ++ let mut child = parent.clone(); ++ assert_eq!(description.descriptor_count(), 2); ++ ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(0, description.id()); ++ let (_, subscription) = state.prepare_add(key, &readable_event(0)).unwrap(); ++ let registration = description ++ .register_epoll(&state, &subscription, key) ++ .unwrap(); ++ subscription ++ .attach_close_registration(registration) ++ .unwrap(); ++ ++ parent.remove(0).unwrap(); ++ assert_eq!(description.descriptor_count(), 1); ++ assert!(state.contains_exact_subscription(key, &subscription)); ++ assert!(subscription.is_active()); ++ ++ child.remove(0).unwrap(); ++ assert_eq!(description.descriptor_count(), 0); ++ assert!(!state.contains_exact_subscription(key, &subscription)); ++ assert!(!subscription.is_active()); ++ assert_eq!(description.epoll_registration_count(), 0); ++ } ++ ++ #[test] ++ fn final_close_between_reverse_registration_and_attach_rejects_add() { ++ let mut list = FdList::new(); ++ let fd = useless_fd(0); ++ let description = fd.inner.ofd.clone(); ++ list.insert_first_free(fd); ++ ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(0, description.id()); ++ let (_, subscription) = state.prepare_add(key, &readable_event(0)).unwrap(); ++ let registration = description ++ .register_epoll(&state, &subscription, key) ++ .unwrap(); ++ ++ // This is the ADD/final-close race window: reverse registration won ++ // the OFD mutex, then the last descriptor closes before ADD can attach ++ // the ownership guard to the subscription. ++ list.remove(0).unwrap(); ++ assert_eq!(description.descriptor_count(), 0); ++ assert!(!state.contains_exact_subscription(key, &subscription)); ++ assert!(!subscription.is_active()); ++ assert_eq!(description.epoll_registration_count(), 0); ++ ++ assert!( ++ subscription ++ .attach_close_registration(registration) ++ .is_err(), ++ "a final-close winner must make a late ADD attachment fail" ++ ); ++ assert_eq!(description.epoll_registration_count(), 0); ++ } ++ ++ #[test] ++ fn reused_fd_final_old_alias_close_removes_only_old_ofd_watch() { ++ let mut old_primary = FdList::new(); ++ let old_fd = useless_fd(0); ++ let old_description = old_fd.inner.ofd.clone(); ++ assert!(old_primary.insert(true, 5, old_fd)); ++ let mut old_alias = old_primary.clone(); ++ ++ let state = Arc::new(EpollState::new()); ++ let old_key = EpollSubscriptionKey::new(5, old_description.id()); ++ let (_, old_subscription) = state.prepare_add(old_key, &readable_event(5)).unwrap(); ++ let old_registration = old_description ++ .register_epoll(&state, &old_subscription, old_key) ++ .unwrap(); ++ old_subscription ++ .attach_close_registration(old_registration) ++ .unwrap(); ++ ++ old_primary.remove(5).unwrap(); ++ assert_eq!(old_description.descriptor_count(), 1); ++ ++ let mut reused_table = FdList::new(); ++ let new_fd = useless_fd(1); ++ let new_description = new_fd.inner.ofd.clone(); ++ assert!(reused_table.insert(true, 5, new_fd)); ++ let new_key = EpollSubscriptionKey::new(5, new_description.id()); ++ let (_, new_subscription) = state.prepare_add(new_key, &readable_event(5)).unwrap(); ++ let new_registration = new_description ++ .register_epoll(&state, &new_subscription, new_key) ++ .unwrap(); ++ new_subscription ++ .attach_close_registration(new_registration) ++ .unwrap(); ++ ++ assert_eq!(state.subscription_count(), 2); ++ old_alias.remove(5).unwrap(); ++ ++ assert_eq!(old_description.descriptor_count(), 0); ++ assert!(!state.contains_exact_subscription(old_key, &old_subscription)); ++ assert!(!old_subscription.is_active()); ++ assert!(state.contains_exact_subscription(new_key, &new_subscription)); ++ assert!(new_subscription.is_active()); ++ assert_eq!(new_description.epoll_registration_count(), 1); ++ ++ state.apply_del(new_key).unwrap(); ++ assert_eq!(new_description.epoll_registration_count(), 0); ++ assert!(!new_subscription.is_active()); ++ assert_eq!(new_description.descriptor_count(), 1); ++ ++ reused_table.remove(5).unwrap(); ++ assert_eq!(new_description.descriptor_count(), 0); ++ } ++ ++ #[test] ++ fn epoll_ofd_close_is_idempotent_and_waits_for_final_alias() { ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(44, 4044); ++ let (_, subscription) = state.prepare_add(key, &readable_event(44)).unwrap(); ++ let mut first = FdList::new(); ++ first.insert_first_free(fd_with_kind( ++ 0, ++ Kind::Epoll { ++ state: state.clone(), ++ }, ++ )); ++ let mut alias = first.clone(); ++ ++ first.remove(0).unwrap(); ++ assert!(!state.is_closed()); ++ assert!(state.contains_exact_subscription(key, &subscription)); ++ ++ alias.remove(0).unwrap(); ++ assert!(state.is_closed()); ++ assert_eq!(state.subscription_count(), 0); ++ assert!(!subscription.is_active()); ++ state.close(); ++ assert!(state.is_closed()); ++ } ++ ++ #[test] ++ fn closing_epoll_does_not_close_watched_open_file_description() { ++ let mut watched = FdList::new(); ++ let watched_fd = useless_fd(0); ++ let watched_description = watched_fd.inner.ofd.clone(); ++ watched.insert_first_free(watched_fd); ++ ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(0, watched_description.id()); ++ let (_, subscription) = state.prepare_add(key, &readable_event(0)).unwrap(); ++ let registration = watched_description ++ .register_epoll(&state, &subscription, key) ++ .unwrap(); ++ subscription ++ .attach_close_registration(registration) ++ .unwrap(); ++ ++ let mut epoll_fds = FdList::new(); ++ epoll_fds.insert_first_free(fd_with_kind( ++ 1, ++ Kind::Epoll { ++ state: state.clone(), ++ }, ++ )); ++ epoll_fds.remove(0).unwrap(); ++ ++ assert!(state.is_closed()); ++ assert_eq!(watched_description.descriptor_count(), 1); ++ assert_eq!(watched_description.epoll_registration_count(), 0); ++ assert!(watched.get(0).is_some()); + } + + #[test] +@@ -700,6 +1129,9 @@ mod tests { + let fd = l.get(0).unwrap(); + fd.inode.drop_one_handle(); + ++ // See `open_handles_are_updated_correctly`: only this deliberate ++ // invariant-panic test crosses the unwind boundary. ++ let l = std::panic::AssertUnwindSafe(l); + assert_panic!(drop(l), &str, "InodeGuard handle dropped too many times"); + } + } +diff --git a/lib/wasix/src/fs/inode_guard.rs b/lib/wasix/src/fs/inode_guard.rs +index 61ec5f6..9130b6b 100644 +--- a/lib/wasix/src/fs/inode_guard.rs ++++ b/lib/wasix/src/fs/inode_guard.rs +@@ -78,6 +78,26 @@ impl InodeValFilePollGuard { + subscription, + }) + } ++ ++ pub(crate) fn poll_immediate_ready(&self) -> heapless::Vec { ++ let mut join = InodeValFilePollGuardJoin { ++ mode: self.mode.clone(), ++ fd: self.fd, ++ peb: self.peb, ++ subscription: self.subscription, ++ }; ++ let waker = futures::task::noop_waker(); ++ let mut cx = Context::from_waker(&waker); ++ let mut ret = heapless::Vec::new(); ++ ++ if let Poll::Ready(events) = Future::poll(Pin::new(&mut join), &mut cx) { ++ for (_, readiness) in events { ++ ret.push(readiness).ok(); ++ } ++ } ++ ++ ret ++ } + } + + impl std::fmt::Debug for InodeValFilePollGuard { +@@ -165,7 +185,6 @@ impl Future for InodeValFilePollGuardJoin { + let mut has_write = false; + let mut has_close = false; + let mut has_hangup = false; +- + let mut ret = heapless::Vec::new(); + for in_event in iterate_poll_events(self.peb) { + match in_event { +diff --git a/lib/wasix/src/fs/mod.rs b/lib/wasix/src/fs/mod.rs +index 508dbef..35ad3b6 100644 +--- a/lib/wasix/src/fs/mod.rs ++++ b/lib/wasix/src/fs/mod.rs +@@ -46,6 +46,7 @@ use wasmer_wasix_types::{ + }, + }; + ++pub(crate) use self::fd::{DirectorySyncHandle, EpollRegistrationGuard, OpenFileDescription}; + pub use self::fd::{Fd, FdInner, InodeVal, Kind}; + pub(crate) use self::inode_guard::{ + InodeValFilePollGuard, InodeValFilePollGuardJoin, InodeValFilePollGuardMode, +@@ -134,6 +135,18 @@ pub struct InodeGuard { + // in the backing file (which may be a host file) getting closed. + open_handles: Arc, + } ++ ++#[derive(Debug)] ++pub(crate) struct InodeHandleReservation { ++ inode: InodeGuard, ++} ++ ++impl Drop for InodeHandleReservation { ++ fn drop(&mut self) { ++ self.inode.drop_one_handle(); ++ } ++} ++ + impl InodeGuard { + pub fn ino(&self) -> Inode { + self.ino +@@ -160,6 +173,17 @@ impl InodeGuard { + trace!(ino = %self.ino.0, new_count = %(prev_handles + 1), "acquiring handle for InodeGuard"); + } + ++ /// Keeps an inode's backing handle alive while a new descriptor is being ++ /// published. Call this while holding the inode lock that observed or ++ /// installed the handle so a concurrent final close cannot clear it in the ++ /// gap before `FdList` acquires the descriptor's permanent reference. ++ pub(crate) fn reserve_handle(&self) -> InodeHandleReservation { ++ self.acquire_handle(); ++ InodeHandleReservation { ++ inode: self.clone(), ++ } ++ } ++ + pub fn drop_one_handle(&self) { + let prev_handles = self.open_handles.fetch_sub(1, Ordering::SeqCst); + +@@ -516,6 +540,13 @@ impl FileSystem for WasiFsRoot { + self.root.remove_file(path) + } + ++ fn open_dir( ++ &self, ++ path: &Path, ++ ) -> virtual_fs::Result> { ++ self.root.open_dir(path) ++ } ++ + fn new_open_options(&self) -> OpenOptions<'_> { + self.root.new_open_options() + } +@@ -547,6 +578,12 @@ pub struct WasiFs { + pub(crate) init_vfs_preopens: Vec, + } + ++enum SyncTarget { ++ File(Arc>>), ++ Directory(DirectorySyncHandle), ++ Buffer, ++} ++ + impl WasiFs { + fn writable_package_mount( + fs: Arc, +@@ -674,10 +711,16 @@ impl WasiFs { + } + }); + +- if let Ok(mut map) = self.fd_map.write() { +- for fd in &to_close { +- map.remove(*fd); +- } ++ let removed = if let Ok(mut map) = self.fd_map.write() { ++ to_close ++ .iter() ++ .filter_map(|fd| map.remove_deferred(*fd)) ++ .collect::>() ++ } else { ++ Vec::new() ++ }; ++ for fd in removed { ++ fd.release_descriptor(); + } + } + +@@ -700,8 +743,13 @@ impl WasiFs { + } + }); + +- if let Ok(mut map) = self.fd_map.write() { +- map.clear(); ++ let removed = if let Ok(mut map) = self.fd_map.write() { ++ map.drain_deferred() ++ } else { ++ Vec::new() ++ }; ++ for fd in removed { ++ fd.release_descriptor(); + } + } + +@@ -1134,6 +1182,69 @@ impl WasiFs { + // loading inodes as necessary + 'symlink_resolution: while symlink_count < MAX_SYMLINKS { + let processing_cur_inode = cur_inode.clone(); ++ let component_name = component.as_os_str().to_string_lossy(); ++ ++ // The common path is walking directories already cached in the ++ // WASIX inode tree. Keep those lookups on shared locks and only ++ // fall through to the write-locked lazy-load path on misses. ++ { ++ let guard = processing_cur_inode.read(); ++ match guard.deref() { ++ Kind::Dir { ++ entries, ++ path, ++ parent, ++ .. ++ } => { ++ match component_name.borrow() { ++ ".." => { ++ if let Some(p) = parent.upgrade() { ++ cur_inode = p; ++ continue 'path_iter; ++ } else { ++ return Err(Errno::Access); ++ } ++ } ++ "." => continue 'path_iter, ++ _ => (), ++ } ++ ++ if let Some(entry) = entries.get(component_name.as_ref()) { ++ cur_inode = entry.clone(); ++ break 'symlink_resolution; ++ } ++ ++ let file = { ++ let mut cd = path.clone(); ++ cd.push(component); ++ cd ++ }; ++ if self.ephemeral_symlink_at(&file).is_none() ++ && self.root_fs.symlink_metadata(&file).is_err() ++ { ++ return Err(Errno::Noent); ++ } ++ } ++ Kind::Root { entries } => { ++ match component { ++ Component::ParentDir | Component::CurDir => continue 'path_iter, ++ _ => {} ++ } ++ ++ if let Some(entry) = entries.get(component_name.as_ref()) { ++ cur_inode = entry.clone(); ++ break 'symlink_resolution; ++ } else if let Some(root) = entries.get(&"/".to_string()) { ++ cur_inode = root.clone(); ++ continue 'symlink_resolution; ++ } else { ++ return Err(Errno::Notcapable); ++ } ++ } ++ _ => {} ++ } ++ } ++ + let mut guard = processing_cur_inode.write(); + match guard.deref_mut() { + Kind::Buffer { .. } => unimplemented!("state::get_inode_at_path for buffers"), +@@ -1170,6 +1281,7 @@ impl WasiFs { + // we want to insert newly opened dirs and files, but not transient symlinks + // TODO: explain why (think about this deeply when well rested) + let should_insert; ++ let stat; + + let kind = if let Some((base_po_dir, path_to_symlink, relative_path)) = + self.ephemeral_symlink_at(&file) +@@ -1179,6 +1291,7 @@ impl WasiFs { + should_insert = false; + loop_for_symlink = true; + symlink_count += 1; ++ stat = Filestat::default(); + Kind::Symlink { + base_po_dir, + path_to_symlink, +@@ -1191,6 +1304,16 @@ impl WasiFs { + .ok() + .ok_or(Errno::Noent)?; + let file_type = metadata.file_type(); ++ stat = Filestat { ++ st_filetype: virtual_file_type_to_wasi_file_type( ++ file_type.clone(), ++ ), ++ st_size: metadata.len(), ++ st_ctim: metadata.created(), ++ st_mtim: metadata.modified(), ++ st_atim: metadata.accessed(), ++ ..Filestat::default() ++ }; + if file_type.is_dir() { + should_insert = true; + // load DIR +@@ -1286,12 +1409,13 @@ impl WasiFs { + }; + drop(guard); + +- let new_inode = self.create_inode( ++ let new_inode = self.create_inode_with_stat( + inodes, + kind, + false, +- file.to_string_lossy().to_string(), +- )?; ++ file.to_string_lossy().to_string().into(), ++ stat, ++ ); + if should_insert { + let mut guard = processing_cur_inode.write(); + if let Kind::Dir { entries, .. } = guard.deref_mut() { +@@ -1532,6 +1656,8 @@ impl WasiFs { + rights_inheriting: ALL_RIGHTS, + flags: Fdflags::empty(), + offset: Arc::new(AtomicU64::new(0)), ++ ofd: OpenFileDescription::new(), ++ readdir_cache: Default::default(), + fd_flags: Fdflagsext::empty(), + }, + open_flags: 0, +@@ -1556,9 +1682,32 @@ impl WasiFs { + .map(|a| a.inode.clone()) + } + ++ /// Return the current backing-file length and refresh the inode cache. ++ /// Host-mounted files can be extended by another WASIX process, so the ++ /// per-inode snapshot is not authoritative for SEEK_END or filestat. ++ pub(crate) fn authoritative_fd_size(fd: &Fd) -> u64 { ++ let handle = { ++ let guard = fd.inode.read(); ++ match guard.deref() { ++ Kind::File { ++ handle: Some(handle), ++ .. ++ } => Some(handle.clone()), ++ _ => None, ++ } ++ }; ++ let Some(handle) = handle else { ++ return fd.inode.stat.read().unwrap().st_size; ++ }; ++ let size = handle.read().unwrap().size(); ++ fd.inode.stat.write().unwrap().st_size = size; ++ size ++ } ++ + pub fn filestat_fd(&self, fd: WasiFd) -> Result { +- let inode = self.get_fd_inode(fd)?; +- let guard = inode.stat.read().unwrap(); ++ let fd = self.get_fd(fd)?; ++ Self::authoritative_fd_size(&fd); ++ let guard = fd.inode.stat.read().unwrap(); + Ok(*guard.deref()) + } + +@@ -1650,6 +1799,24 @@ impl WasiFs { + } + } + ++ fn sync_target(fd: &Fd) -> Result { ++ let guard = fd.inode.read(); ++ match guard.deref() { ++ Kind::File { ++ handle: Some(file), .. ++ } => Ok(SyncTarget::File(file.clone())), ++ Kind::Dir { .. } => fd ++ .inner ++ .ofd ++ .directory_sync() ++ .ok_or(Errno::Notsup)? ++ .map(SyncTarget::Directory), ++ Kind::Root { .. } => Err(Errno::Notsup), ++ Kind::Buffer { .. } => Ok(SyncTarget::Buffer), ++ _ => Err(Errno::Io), ++ } ++ } ++ + #[allow(clippy::await_holding_lock)] + pub async fn flush(&self, fd: WasiFd) -> Result<(), Errno> { + match fd { +@@ -1702,6 +1869,108 @@ impl WasiFs { + Ok(()) + } + ++ #[allow(clippy::await_holding_lock)] ++ pub async fn sync(&self, fd: WasiFd, sync_metadata: bool) -> Result<(), Errno> { ++ match fd { ++ __WASI_STDIN_FILENO => (), ++ __WASI_STDOUT_FILENO => { ++ let mut file = ++ WasiInodes::stdout_mut(&self.fd_map).map_err(fs_error_into_wasi_err)?; ++ if sync_metadata { ++ file.sync_all_to_disk().await.map_err(map_io_err)?; ++ } else { ++ file.sync_data_to_disk().await.map_err(map_io_err)?; ++ } ++ } ++ __WASI_STDERR_FILENO => { ++ let mut file = ++ WasiInodes::stderr_mut(&self.fd_map).map_err(fs_error_into_wasi_err)?; ++ if sync_metadata { ++ file.sync_all_to_disk().await.map_err(map_io_err)?; ++ } else { ++ file.sync_data_to_disk().await.map_err(map_io_err)?; ++ } ++ } ++ _ => { ++ let fd = self.get_fd(fd)?; ++ let required_right = if sync_metadata { ++ Rights::FD_SYNC ++ } else { ++ Rights::FD_DATASYNC ++ }; ++ if !fd.inner.rights.contains(required_right) { ++ return Err(Errno::Access); ++ } ++ ++ let target = Self::sync_target(&fd)?; ++ drop(fd); ++ ++ match target { ++ SyncTarget::File(file) => { ++ let mut file = file.write().unwrap(); ++ if sync_metadata { ++ file.sync_all_to_disk().await.map_err(map_io_err)?; ++ } else { ++ file.sync_data_to_disk().await.map_err(map_io_err)?; ++ } ++ } ++ // A directory's entries are metadata, so fd_datasync must ++ // use the same full durability barrier as fd_sync. ++ SyncTarget::Directory(directory) => { ++ directory.sync_all_to_disk().await.map_err(map_io_err)?; ++ } ++ SyncTarget::Buffer => {} ++ } ++ } ++ } ++ Ok(()) ++ } ++ ++ pub fn sync_blocking(&self, fd: WasiFd, sync_metadata: bool) -> Result, Errno> { ++ match fd { ++ __WASI_STDIN_FILENO | __WASI_STDOUT_FILENO | __WASI_STDERR_FILENO => Ok(None), ++ _ => { ++ let fd = self.get_fd(fd)?; ++ let required_right = if sync_metadata { ++ Rights::FD_SYNC ++ } else { ++ Rights::FD_DATASYNC ++ }; ++ if !fd.inner.rights.contains(required_right) { ++ return Err(Errno::Access); ++ } ++ ++ let target = Self::sync_target(&fd)?; ++ drop(fd); ++ ++ match target { ++ SyncTarget::File(file) => { ++ let mut file = file.write().unwrap(); ++ if sync_metadata { ++ if !file.has_blocking_sync_all_to_disk() { ++ return Ok(None); ++ } ++ file.sync_all_to_disk_blocking().map_err(map_io_err)?; ++ } else { ++ if !file.has_blocking_sync_data_to_disk() { ++ return Ok(None); ++ } ++ file.sync_data_to_disk_blocking().map_err(map_io_err)?; ++ } ++ } ++ SyncTarget::Directory(directory) => { ++ if !directory.has_blocking_sync_all_to_disk() { ++ return Ok(None); ++ } ++ directory.sync_all_to_disk_blocking().map_err(map_io_err)?; ++ } ++ SyncTarget::Buffer => {} ++ } ++ Ok(Some(())) ++ } ++ } ++ } ++ + /// Creates an inode and inserts it given a Kind and some extra data + pub(crate) fn create_inode( + &self, +@@ -1841,6 +2110,21 @@ impl WasiFs { + idx: Option, + exclusive: bool, + ) -> Result { ++ let directory_sync: Option> = ++ if rights.contains(Rights::FD_SYNC) || rights.contains(Rights::FD_DATASYNC) { ++ let guard = inode.read(); ++ match guard.deref() { ++ Kind::Dir { path, .. } => Some( ++ self.root_fs ++ .open_dir(path) ++ .map(Arc::from) ++ .map_err(fs_error_into_wasi_err), ++ ), ++ _ => None, ++ } ++ } else { ++ None ++ }; + let is_stdio = matches!( + idx, + Some(__WASI_STDIN_FILENO) | Some(__WASI_STDOUT_FILENO) | Some(__WASI_STDERR_FILENO) +@@ -1851,6 +2135,8 @@ impl WasiFs { + rights_inheriting, + flags: fs_flags, + offset: Arc::new(AtomicU64::new(0)), ++ ofd: OpenFileDescription::new_with_directory_sync(directory_sync), ++ readdir_cache: Default::default(), + fd_flags, + }, + open_flags, +@@ -1858,17 +2144,20 @@ impl WasiFs { + is_stdio, + }; + +- let mut guard = self.fd_map.write().unwrap(); +- + match idx { + Some(idx) => { +- if guard.insert(exclusive, idx, fd) { +- Ok(idx) +- } else { +- Err(Errno::Exist) ++ let displaced = { ++ let mut guard = self.fd_map.write().unwrap(); ++ guard ++ .insert_deferred(exclusive, idx, fd) ++ .map_err(|_| Errno::Exist)? ++ }; ++ if let Some(displaced) = displaced { ++ displaced.release_descriptor(); + } ++ Ok(idx) + } +- None => Ok(guard.insert_first_free(fd)), ++ None => Ok(self.fd_map.write().unwrap().insert_first_free(fd)), + } + } + +@@ -1890,6 +2179,8 @@ impl WasiFs { + rights_inheriting: fd.inner.rights_inheriting, + flags: fd.inner.flags, + offset: fd.inner.offset.clone(), ++ ofd: fd.inner.ofd.clone(), ++ readdir_cache: fd.inner.readdir_cache.clone(), + fd_flags: match cloexec { + None => fd.inner.fd_flags, + Some(cloexec) => { +@@ -2201,23 +2492,30 @@ impl WasiFs { + kind: RwLock::new(kind), + }) + }; +- self.fd_map.write().unwrap().insert( +- false, +- raw_fd, +- Fd { +- inner: FdInner { +- rights, +- rights_inheriting: Rights::empty(), +- flags: fd_flags, +- offset: Arc::new(AtomicU64::new(0)), +- fd_flags: Fdflagsext::empty(), +- }, +- // since we're not calling open on this, we don't need open flags +- open_flags: 0, +- inode, +- is_stdio: true, ++ let new_fd = Fd { ++ inner: FdInner { ++ rights, ++ rights_inheriting: Rights::empty(), ++ flags: fd_flags, ++ offset: Arc::new(AtomicU64::new(0)), ++ ofd: OpenFileDescription::new(), ++ readdir_cache: Default::default(), ++ fd_flags: Fdflagsext::empty(), + }, +- ); ++ // since we're not calling open on this, we don't need open flags ++ open_flags: 0, ++ inode, ++ is_stdio: true, ++ }; ++ let displaced = self ++ .fd_map ++ .write() ++ .unwrap() ++ .insert_deferred(false, raw_fd, new_fd) ++ .expect("non-exclusive stdio insertion cannot fail"); ++ if let Some(displaced) = displaced { ++ displaced.release_descriptor(); ++ } + } + + pub fn get_stat_for_kind(&self, kind: &Kind) -> Result { +@@ -2291,23 +2589,22 @@ impl WasiFs { + + /// Closes an open FD, handling all details such as FD being preopen + pub(crate) fn close_fd(&self, fd: WasiFd) -> Result<(), Errno> { +- let mut fd_map = self.fd_map.write().unwrap(); +- +- let pfd = fd_map.remove(fd).ok_or(Errno::Badf); +- match pfd { +- Ok(fd_ref) => { +- let inode = fd_ref.inode.ino().as_u64(); +- let ref_cnt = fd_ref.inode.ref_cnt(); +- if ref_cnt == 1 { +- trace!(%fd, %inode, %ref_cnt, "closing file descriptor"); +- } else { +- trace!(%fd, %inode, %ref_cnt, "weakening file descriptor"); +- } +- } +- Err(err) => { +- trace!(%fd, "closing file descriptor failed - {}", err); +- } ++ let fd_ref = self ++ .fd_map ++ .write() ++ .unwrap() ++ .remove_deferred(fd) ++ .ok_or(Errno::Badf)?; ++ let inode = fd_ref.inode.ino().as_u64(); ++ let ref_cnt = fd_ref.inode.ref_cnt(); ++ if ref_cnt == 1 { ++ trace!(%fd, %inode, %ref_cnt, "closing file descriptor"); ++ } else { ++ trace!(%fd, %inode, %ref_cnt, "weakening file descriptor"); + } ++ // Finalization may acquire OFD, epoll, socket, and selector locks. The ++ // fd-map write guard above is deliberately gone before this call. ++ fd_ref.release_descriptor(); + Ok(()) + } + } +@@ -2461,6 +2758,7 @@ pub fn fs_error_into_wasi_err(fs_error: FsError) -> Errno { + mod tests { + use super::*; + use once_cell::sync::OnceCell; ++ #[cfg(feature = "host-fs")] + use tempfile::tempdir; + use virtual_fs::{RootFileSystemBuilder, TmpFileSystem}; + use wasmer::Engine; +@@ -2556,6 +2854,243 @@ mod tests { + ); + } + ++ #[cfg(all(target_os = "linux", feature = "host-fs"))] ++ #[tokio::test] ++ async fn host_directory_sync_uses_open_file_description_after_rename() { ++ let root_dir = tempdir().unwrap(); ++ let original = root_dir.path().join("original"); ++ let renamed = root_dir.path().join("renamed"); ++ std::fs::create_dir(&original).unwrap(); ++ ++ let host_fs = virtual_fs::host_fs::FileSystem::new( ++ tokio::runtime::Handle::current(), ++ root_dir.path(), ++ ) ++ .unwrap(); ++ let inodes = WasiInodes::new(); ++ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(host_fs)); ++ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); ++ let inode = wasi_fs ++ .create_inode( ++ &inodes, ++ Kind::Dir { ++ parent: wasi_fs.root_inode.downgrade(), ++ path: PathBuf::from("/original"), ++ entries: HashMap::new(), ++ }, ++ false, ++ "original".to_string(), ++ ) ++ .unwrap(); ++ let rights = Rights::FD_SYNC | Rights::FD_DATASYNC; ++ let fd = wasi_fs ++ .create_fd( ++ rights, ++ Rights::empty(), ++ Fdflags::empty(), ++ Fdflagsext::empty(), ++ 0, ++ inode, ++ ) ++ .unwrap(); ++ let duplicate = wasi_fs.clone_fd(fd).unwrap(); ++ ++ std::fs::rename(&original, &renamed).unwrap(); ++ assert!(!original.exists()); ++ wasi_fs.close_fd(fd).unwrap(); ++ ++ assert_eq!(wasi_fs.sync_blocking(duplicate, true), Ok(Some(()))); ++ assert_eq!(wasi_fs.sync_blocking(duplicate, false), Ok(Some(()))); ++ assert_eq!(wasi_fs.sync(duplicate, true).await, Ok(())); ++ } ++ ++ #[cfg(feature = "host-fs")] ++ #[tokio::test] ++ async fn host_file_size_refresh_observes_another_process_extension() { ++ let root_dir = tempdir().unwrap(); ++ let relation = root_dir.path().join("relation"); ++ std::fs::write(&relation, b"old").unwrap(); ++ ++ let host_fs = virtual_fs::host_fs::FileSystem::new( ++ tokio::runtime::Handle::current(), ++ root_dir.path(), ++ ) ++ .unwrap(); ++ let inodes = WasiInodes::new(); ++ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(host_fs)); ++ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); ++ let handle = wasi_fs ++ .root_fs ++ .new_open_options() ++ .read(true) ++ .write(true) ++ .open(Path::new("/relation")) ++ .unwrap(); ++ let inode = wasi_fs ++ .create_inode( ++ &inodes, ++ Kind::File { ++ handle: Some(Arc::new(RwLock::new(handle))), ++ path: PathBuf::from("/relation"), ++ fd: None, ++ }, ++ false, ++ "relation".to_string(), ++ ) ++ .unwrap(); ++ let fd = wasi_fs ++ .create_fd( ++ Rights::FD_FILESTAT_GET | Rights::FD_SEEK, ++ Rights::empty(), ++ Fdflags::empty(), ++ Fdflagsext::empty(), ++ 0, ++ inode, ++ ) ++ .unwrap(); ++ let fd_entry = wasi_fs.get_fd(fd).unwrap(); ++ assert_eq!(fd_entry.inode.stat.read().unwrap().st_size, 3); ++ ++ std::fs::OpenOptions::new() ++ .write(true) ++ .open(&relation) ++ .unwrap() ++ .set_len(8192) ++ .unwrap(); ++ assert_eq!(fd_entry.inode.stat.read().unwrap().st_size, 3); ++ assert_eq!(WasiFs::authoritative_fd_size(&fd_entry), 8192); ++ assert_eq!(wasi_fs.filestat_fd(fd).unwrap().st_size, 8192); ++ } ++ ++ #[cfg(feature = "host-fs")] ++ #[tokio::test] ++ async fn open_handle_reservation_bridges_final_close_to_fd_publication() { ++ let root_dir = tempdir().unwrap(); ++ std::fs::write(root_dir.path().join("relation"), b"data").unwrap(); ++ ++ let host_fs = virtual_fs::host_fs::FileSystem::new( ++ tokio::runtime::Handle::current(), ++ root_dir.path(), ++ ) ++ .unwrap(); ++ let inodes = WasiInodes::new(); ++ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(host_fs)); ++ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); ++ let handle = wasi_fs ++ .root_fs ++ .new_open_options() ++ .read(true) ++ .write(true) ++ .open(Path::new("/relation")) ++ .unwrap(); ++ let inode = wasi_fs ++ .create_inode( ++ &inodes, ++ Kind::File { ++ handle: Some(Arc::new(RwLock::new(handle))), ++ path: PathBuf::from("/relation"), ++ fd: None, ++ }, ++ false, ++ "relation".to_string(), ++ ) ++ .unwrap(); ++ let rights = Rights::FD_READ | Rights::FD_WRITE; ++ let first = wasi_fs ++ .create_fd( ++ rights, ++ Rights::empty(), ++ Fdflags::empty(), ++ Fdflagsext::empty(), ++ 0, ++ inode.clone(), ++ ) ++ .unwrap(); ++ ++ let reservation = { ++ let guard = inode.read(); ++ assert!(matches!( ++ guard.deref(), ++ Kind::File { ++ handle: Some(_), ++ .. ++ } ++ )); ++ inode.reserve_handle() ++ }; ++ wasi_fs.close_fd(first).unwrap(); ++ assert_eq!(inode.handle_count(), 1); ++ assert!(matches!( ++ inode.read().deref(), ++ Kind::File { ++ handle: Some(_), ++ .. ++ } ++ )); ++ ++ let second = wasi_fs ++ .create_fd( ++ rights, ++ Rights::empty(), ++ Fdflags::empty(), ++ Fdflagsext::empty(), ++ 0, ++ inode.clone(), ++ ) ++ .unwrap(); ++ drop(reservation); ++ assert_eq!(inode.handle_count(), 1); ++ assert!(matches!( ++ inode.read().deref(), ++ Kind::File { ++ handle: Some(_), ++ .. ++ } ++ )); ++ ++ wasi_fs.close_fd(second).unwrap(); ++ assert_eq!(inode.handle_count(), 0); ++ assert!(matches!( ++ inode.read().deref(), ++ Kind::File { handle: None, .. } ++ )); ++ } ++ ++ #[tokio::test] ++ async fn unsupported_directory_sync_fails_closed() { ++ let inodes = WasiInodes::new(); ++ let tmp = TmpFileSystem::new(); ++ tmp.create_dir(Path::new("/directory")).unwrap(); ++ let fs_backing = WasiFsRoot::from_filesystem(Arc::new(tmp)); ++ let wasi_fs = WasiFs::new_init(fs_backing, &inodes, FS_ROOT_INO).unwrap(); ++ let inode = wasi_fs ++ .create_inode( ++ &inodes, ++ Kind::Dir { ++ parent: wasi_fs.root_inode.downgrade(), ++ path: PathBuf::from("/directory"), ++ entries: HashMap::new(), ++ }, ++ false, ++ "directory".to_string(), ++ ) ++ .unwrap(); ++ let rights = Rights::FD_SYNC | Rights::FD_DATASYNC; ++ let fd = wasi_fs ++ .create_fd( ++ rights, ++ Rights::empty(), ++ Fdflags::empty(), ++ Fdflagsext::empty(), ++ 0, ++ inode, ++ ) ++ .unwrap(); ++ ++ assert_eq!(wasi_fs.sync_blocking(fd, true), Err(Errno::Notsup)); ++ assert_eq!(wasi_fs.sync(fd, false).await, Err(Errno::Notsup)); ++ } ++ + #[cfg(feature = "host-fs")] + #[tokio::test] + async fn mapped_preopen_inode_paths_should_stay_in_guest_space() { +diff --git a/lib/wasix/src/syscalls/wasi/fd_advise.rs b/lib/wasix/src/syscalls/wasi/fd_advise.rs +index 288a84a..0e5299c 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_advise.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_advise.rs +@@ -1,5 +1,6 @@ + use super::*; + use crate::syscalls::*; ++use virtual_fs::FileAdvice; + + /// ### `fd_advise()` + /// Advise the system about how a file will be used +@@ -22,7 +23,9 @@ pub fn fd_advise( + ) -> Result { + WasiEnv::do_pending_operations(&mut ctx)?; + +- wasi_try_ok!(fd_advise_internal(&mut ctx, fd, offset, len, advice)); ++ if let Err(error) = fd_advise_internal(&mut ctx, fd, offset, len, advice) { ++ return Ok(error); ++ } + let env = ctx.data(); + + #[cfg(feature = "journal")] +@@ -43,19 +46,206 @@ pub(crate) fn fd_advise_internal( + len: Filesize, + advice: Advice, + ) -> Result<(), Errno> { +- // Instead of unconditionally returning OK. This barebones implementation +- // only performs basic fd and rights checks. +- + let env = ctx.data(); + let (_, mut state) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; + let fd_entry = state.fs.get_fd(fd)?; + let inode = fd_entry.inode; + +- if !fd_entry.inner.rights.contains(Rights::FD_ADVISE) { +- return Err(Errno::Access); ++ require_fd_advise_right(fd_entry.inner.rights)?; ++ ++ delegate_fd_advice(offset, len, advice, |offset, len, advice| { ++ let guard = inode.read(); ++ advise_inode_kind(guard.deref(), offset, len, advice) ++ }) ++} ++ ++fn require_fd_advise_right(rights: Rights) -> Result<(), Errno> { ++ if rights.contains(Rights::FD_ADVISE) { ++ Ok(()) ++ } else { ++ Err(Errno::Notcapable) + } ++} + +- let _end = offset.checked_add(len).ok_or(Errno::Inval)?; ++fn advise_inode_kind( ++ kind: &Kind, ++ offset: Filesize, ++ len: Filesize, ++ advice: FileAdvice, ++) -> Result<(), Errno> { ++ match kind { ++ Kind::File { ++ handle: Some(handle), ++ .. ++ } => handle ++ .read() ++ .map_err(|_| Errno::Fault)? ++ .advise(offset, len, advice) ++ .map_err(fd_advise_io_error), ++ Kind::File { handle: None, .. } => Err(Errno::Badf), ++ Kind::Buffer { .. } => Err(Errno::Notsup), ++ Kind::PipeRx { .. } | Kind::PipeTx { .. } | Kind::DuplexPipe { .. } => Err(Errno::Spipe), ++ // Preview1 fd_advise classifies directories as invalid descriptors ++ // even though Linux posix_fadvise happens to accept directory fds. ++ Kind::Dir { .. } | Kind::Root { .. } => Err(Errno::Badf), ++ Kind::Socket { .. } ++ | Kind::Symlink { .. } ++ | Kind::EventNotifications { .. } ++ | Kind::Epoll { .. } => Err(Errno::Badf), ++ } ++} + +- Ok(()) ++fn delegate_fd_advice( ++ offset: Filesize, ++ len: Filesize, ++ advice: Advice, ++ delegate: impl FnOnce(Filesize, Filesize, FileAdvice) -> Result<(), Errno>, ++) -> Result<(), Errno> { ++ offset.checked_add(len).ok_or(Errno::Inval)?; ++ let advice = match advice { ++ Advice::Normal => FileAdvice::Normal, ++ Advice::Sequential => FileAdvice::Sequential, ++ Advice::Random => FileAdvice::Random, ++ Advice::Willneed => FileAdvice::WillNeed, ++ Advice::Dontneed => FileAdvice::DontNeed, ++ Advice::Noreuse => FileAdvice::NoReuse, ++ Advice::Unknown => return Err(Errno::Inval), ++ _ => return Err(Errno::Inval), ++ }; ++ delegate(offset, len, advice) ++} ++ ++fn fd_advise_io_error(error: io::Error) -> Errno { ++ #[cfg(target_os = "linux")] ++ if let Some(error) = error.raw_os_error() { ++ return match error { ++ libc::EACCES => Errno::Access, ++ libc::EBADF => Errno::Badf, ++ libc::EINVAL => Errno::Inval, ++ libc::EIO => Errno::Io, ++ libc::ENOSYS => Errno::Nosys, ++ libc::EOPNOTSUPP => Errno::Notsup, ++ libc::EOVERFLOW => Errno::Overflow, ++ libc::EPERM => Errno::Perm, ++ libc::ESPIPE => Errno::Spipe, ++ _ => return map_io_err(io::Error::from_raw_os_error(error)), ++ }; ++ } ++ ++ match error.kind() { ++ io::ErrorKind::InvalidInput => Errno::Inval, ++ io::ErrorKind::Unsupported => Errno::Notsup, ++ _ => map_io_err(error), ++ } ++} ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn fd_advise_maps_every_wasi_advice_exactly() { ++ let cases = [ ++ (Advice::Normal, FileAdvice::Normal), ++ (Advice::Sequential, FileAdvice::Sequential), ++ (Advice::Random, FileAdvice::Random), ++ (Advice::Willneed, FileAdvice::WillNeed), ++ (Advice::Dontneed, FileAdvice::DontNeed), ++ (Advice::Noreuse, FileAdvice::NoReuse), ++ ]; ++ ++ for (wasi_advice, expected) in cases { ++ let mut observed = None; ++ delegate_fd_advice(17, 23, wasi_advice, |offset, len, advice| { ++ observed = Some((offset, len, advice)); ++ Ok(()) ++ }) ++ .unwrap(); ++ assert_eq!(observed, Some((17, 23, expected))); ++ } ++ } ++ ++ #[test] ++ fn fd_advise_delegates_willneed_and_dontneed() { ++ let mut observed = Vec::new(); ++ for advice in [Advice::Willneed, Advice::Dontneed] { ++ delegate_fd_advice(4096, 8192, advice, |offset, len, advice| { ++ observed.push((offset, len, advice)); ++ Ok(()) ++ }) ++ .unwrap(); ++ } ++ ++ assert_eq!( ++ observed, ++ vec![ ++ (4096, 8192, FileAdvice::WillNeed), ++ (4096, 8192, FileAdvice::DontNeed), ++ ] ++ ); ++ } ++ ++ #[test] ++ fn fd_advise_rejects_invalid_ranges_and_unknown_advice_without_delegating() { ++ let mut calls = 0; ++ let overflow = delegate_fd_advice(u64::MAX, 1, Advice::Willneed, |_, _, _| { ++ calls += 1; ++ Ok(()) ++ }); ++ assert_eq!(overflow, Err(Errno::Inval)); ++ ++ let unknown = delegate_fd_advice(0, 0, Advice::Unknown, |_, _, _| { ++ calls += 1; ++ Ok(()) ++ }); ++ assert_eq!(unknown, Err(Errno::Inval)); ++ assert_eq!(calls, 0); ++ } ++ ++ #[test] ++ fn fd_advise_maps_unsupported_backend_to_notsup() { ++ let result = delegate_fd_advice(0, 4096, Advice::Dontneed, |_, _, _| { ++ Err(fd_advise_io_error(io::ErrorKind::Unsupported.into())) ++ }); ++ ++ assert_eq!(result, Err(Errno::Notsup)); ++ } ++ ++ #[test] ++ fn fd_advise_missing_right_is_notcapable() { ++ assert_eq!( ++ require_fd_advise_right(Rights::empty()), ++ Err(Errno::Notcapable) ++ ); ++ assert_eq!(require_fd_advise_right(Rights::FD_ADVISE), Ok(())); ++ } ++ ++ #[test] ++ fn fd_advise_directory_is_badf() { ++ let directory = Kind::Root { ++ entries: Default::default(), ++ }; ++ assert_eq!( ++ advise_inode_kind(&directory, 0, 4096, FileAdvice::WillNeed), ++ Err(Errno::Badf) ++ ); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[test] ++ fn fd_advise_preserves_posix_fadvise_errnos() { ++ let cases = [ ++ (libc::EBADF, Errno::Badf), ++ (libc::EINVAL, Errno::Inval), ++ (libc::EOVERFLOW, Errno::Overflow), ++ (libc::ESPIPE, Errno::Spipe), ++ ]; ++ ++ for (host_errno, expected) in cases { ++ assert_eq!( ++ fd_advise_io_error(io::Error::from_raw_os_error(host_errno)), ++ expected ++ ); ++ } ++ } + } +diff --git a/lib/wasix/src/syscalls/wasi/fd_datasync.rs b/lib/wasix/src/syscalls/wasi/fd_datasync.rs +index f799d90..6187bf6 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_datasync.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_datasync.rs +@@ -16,9 +16,11 @@ pub fn fd_datasync(mut ctx: FunctionEnvMut<'_, WasiEnv>, fd: WasiFd) -> Result { + if let Some(handle) = handle { + let mut handle = handle.write().unwrap(); ++ #[cfg(feature = "host-fs")] ++ let _shared_mapping_guard = { ++ let host_file = handle ++ .upcast_any_ref() ++ .downcast_ref::() ++ .map(|file| file.try_clone_std_file()) ++ .transpose() ++ .map_err(crate::utils::map_io_err)?; ++ host_file ++ .as_ref() ++ .map(|file| state.guard_shared_mapping_file_shrink(file, st_size)) ++ .transpose()? ++ .flatten() ++ }; + handle.set_len(st_size).map_err(fs_error_into_wasi_err)?; + } else { + return Err(Errno::Badf); +diff --git a/lib/wasix/src/syscalls/wasi/fd_read.rs b/lib/wasix/src/syscalls/wasi/fd_read.rs +index f5de216..400441b 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_read.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_read.rs +@@ -1,6 +1,6 @@ + use std::{collections::VecDeque, task::Waker}; + +-use virtual_fs::{AsyncReadExt, DeviceFile, ReadBuf}; ++use virtual_fs::{AsyncReadExt, DeviceFile, ReadBuf, VirtualFile}; + + use super::*; + use crate::{ +@@ -131,6 +131,88 @@ pub(crate) fn fd_read_internal_handler( + Ok(ret) + } + ++fn read_file_iovs_at( ++ memory: &wasmer::MemoryView<'_>, ++ iovs: WasmPtr<__wasi_iovec_t, M>, ++ iovs_len: M::Offset, ++ offset: usize, ++ mut read_at: impl FnMut(&mut [u8], u64) -> std::io::Result, ++) -> Result { ++ let mut total_read = 0usize; ++ let mut read_offset = offset as u64; ++ let iovs_arr = iovs.slice(memory, iovs_len).map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ for iovs in iovs_arr.iter() { ++ let mut buf = WasmPtr::::new(iovs.buf) ++ .slice(memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ if buf.as_ref().is_empty() { ++ continue; ++ } ++ let local_read = match read_at(buf.as_mut(), read_offset) { ++ Ok(s) => s, ++ Err(_) if total_read > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ total_read += local_read; ++ read_offset += local_read as u64; ++ if local_read != buf.len() { ++ break; ++ } ++ } ++ ++ Ok(total_read) ++} ++ ++fn try_read_file_positioned_shared_blocking( ++ handle: &(dyn VirtualFile + Send + Sync), ++ memory: &wasmer::MemoryView<'_>, ++ iovs: WasmPtr<__wasi_iovec_t, M>, ++ iovs_len: M::Offset, ++ offset: usize, ++) -> Result, Errno> { ++ if !handle.has_blocking_read_at_shared() { ++ return Ok(None); ++ } ++ ++ read_file_iovs_at(memory, iovs, iovs_len, offset, |buf, read_offset| { ++ handle.read_at_blocking_shared(buf, read_offset) ++ }) ++ .map(Some) ++} ++ ++fn try_read_file_blocking( ++ handle: &mut (dyn VirtualFile + Send + Sync), ++ memory: &wasmer::MemoryView<'_>, ++ iovs: WasmPtr<__wasi_iovec_t, M>, ++ iovs_len: M::Offset, ++ offset: usize, ++ positioned_read: bool, ++) -> Result, Errno> { ++ if positioned_read { ++ if !handle.has_blocking_read_at() { ++ return Ok(None); ++ } ++ return read_file_iovs_at(memory, iovs, iovs_len, offset, |buf, read_offset| { ++ handle.read_at_blocking(buf, read_offset) ++ }) ++ .map(Some); ++ } ++ ++ if !handle.has_blocking_seek() || !handle.has_blocking_read() { ++ return Ok(None); ++ } ++ handle ++ .seek_blocking(std::io::SeekFrom::Start(offset as u64)) ++ .map_err(map_io_err)?; ++ read_file_iovs_at(memory, iovs, iovs_len, offset, |buf, _| { ++ handle.read_blocking(buf) ++ }) ++ .map(Some) ++} ++ + #[allow(clippy::await_holding_lock)] + pub(crate) fn fd_read_internal( + ctx: &mut FunctionEnvMut<'_, WasiEnv>, +@@ -157,8 +239,8 @@ pub(crate) fn fd_read_internal( + let fd_flags = fd_entry.inner.flags; + + let (bytes_read, can_update_cursor) = { +- let mut guard = inode.write(); +- match guard.deref_mut() { ++ let guard = inode.read(); ++ match guard.deref() { + Kind::File { handle, .. } => { + let Some(handle) = handle else { + tracing::warn!("fd_read: file handle is None"); +@@ -168,66 +250,126 @@ pub(crate) fn fd_read_internal( + + drop(guard); + +- let res = __asyncify_light( +- env, +- if fd_flags.contains(Fdflags::NONBLOCK) { +- Some(Duration::ZERO) ++ let positioned_read = !is_stdio; ++ let blocking_read = if !is_stdio { ++ if positioned_read { ++ let handle_guard = match handle.read() { ++ Ok(a) => a, ++ Err(_) => return Ok(Err(Errno::Fault)), ++ }; ++ match try_read_file_positioned_shared_blocking::( ++ handle_guard.as_ref(), ++ &memory, ++ iovs, ++ iovs_len, ++ offset, ++ ) { ++ Ok(Some(read)) => Some(read), ++ Ok(None) => { ++ drop(handle_guard); ++ let mut handle = match handle.write() { ++ Ok(a) => a, ++ Err(_) => return Ok(Err(Errno::Fault)), ++ }; ++ match try_read_file_blocking::( ++ handle.as_mut(), ++ &memory, ++ iovs, ++ iovs_len, ++ offset, ++ true, ++ ) { ++ Ok(read) => read, ++ Err(err) => return Ok(Err(err)), ++ } ++ } ++ Err(err) => return Ok(Err(err)), ++ } + } else { +- None +- }, +- async move { + let mut handle = match handle.write() { + Ok(a) => a, +- Err(_) => return Err(Errno::Fault), ++ Err(_) => return Ok(Err(Errno::Fault)), + }; +- if !is_stdio { +- handle +- .seek(std::io::SeekFrom::Start(offset as u64)) +- .await +- .map_err(map_io_err)?; ++ match try_read_file_blocking::( ++ handle.as_mut(), ++ &memory, ++ iovs, ++ iovs_len, ++ offset, ++ false, ++ ) { ++ Ok(read) => read, ++ Err(err) => return Ok(Err(err)), + } ++ } ++ } else { ++ None ++ }; + +- let mut total_read = 0usize; ++ let read = if let Some(read) = blocking_read { ++ read ++ } else { ++ let res = __asyncify_light( ++ env, ++ if fd_flags.contains(Fdflags::NONBLOCK) { ++ Some(Duration::ZERO) ++ } else { ++ None ++ }, ++ async move { ++ let mut handle = match handle.write() { ++ Ok(a) => a, ++ Err(_) => return Err(Errno::Fault), ++ }; ++ if !is_stdio { ++ handle ++ .seek(std::io::SeekFrom::Start(offset as u64)) ++ .await ++ .map_err(map_io_err)?; ++ } + +- let iovs_arr = +- iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; +- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; +- for iovs in iovs_arr.iter() { +- let mut buf = WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi)? +- .access() +- .map_err(mem_error_to_wasi)?; +- let r = handle.read(buf.as_mut()).await.map_err(|err| { +- let err = From::::from(err); +- match err { +- Errno::Again => { +- if is_stdio { +- Errno::Badf +- } else { +- Errno::Again ++ let mut total_read = 0usize; ++ ++ let iovs_arr = ++ iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ for iovs in iovs_arr.iter() { ++ let mut buf = WasmPtr::::new(iovs.buf) ++ .slice(&memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ let r = handle.read(buf.as_mut()).await.map_err(|err| { ++ let err = From::::from(err); ++ match err { ++ Errno::Again => { ++ if is_stdio { ++ Errno::Badf ++ } else { ++ Errno::Again ++ } + } ++ a => a, + } +- a => a, ++ }); ++ let local_read = match r { ++ Ok(s) => s, ++ Err(_) if total_read > 0 => break, ++ Err(err) => return Err(err), ++ }; ++ total_read += local_read; ++ if local_read != buf.len() { ++ break; + } +- }); +- let local_read = match r { +- Ok(s) => s, +- Err(_) if total_read > 0 => break, +- Err(err) => return Err(err), +- }; +- total_read += local_read; +- if local_read != buf.len() { +- break; + } +- } +- Ok(total_read) +- }, +- ); +- let read = wasi_try_ok_ok!(res?.map_err(|err| match err { +- Errno::Timedout => Errno::Again, +- a => a, +- })); ++ Ok(total_read) ++ }, ++ ); ++ wasi_try_ok_ok!(res?.map_err(|err| match err { ++ Errno::Timedout => Errno::Again, ++ a => a, ++ })) ++ }; + (read, true) + } + Kind::Socket { socket } => { +diff --git a/lib/wasix/src/syscalls/wasi/fd_readdir.rs b/lib/wasix/src/syscalls/wasi/fd_readdir.rs +index 5672ab5..fba763a 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_readdir.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_readdir.rs +@@ -37,83 +37,105 @@ pub fn fd_readdir( + let working_dir = wasi_try_ok!(state.fs.get_fd(fd)); + let mut buf_idx = 0usize; + +- let entries: Vec<(String, Filetype, u64)> = { +- let guard = working_dir.inode.read(); +- match guard.deref() { +- Kind::Dir { path, entries, .. } => { +- trace!("reading dir {:?}", path); +- // TODO: refactor this code +- // we need to support multiple calls, +- // simple and obviously correct implementation for now: +- // maintain consistent order via lexacographic sorting +- let fs_info = wasi_try_ok!( +- wasi_try_ok!(state.fs_read_dir(path)) +- .collect::, _>>() +- .map_err(fs_error_into_wasi_err) +- ); +- let mut entry_vec = wasi_try_ok!( +- fs_info +- .into_iter() +- .map(|entry| { +- let filename = entry.file_name().to_string_lossy().to_string(); +- trace!("getting file: {:?}", filename); +- let filetype = virtual_file_type_to_wasi_file_type( +- entry.file_type().map_err(fs_error_into_wasi_err)?, +- ); +- Ok(( +- filename, filetype, 0, // TODO: inode +- )) +- }) +- .collect::, _>>() +- ); +- entry_vec.extend(entries.iter().filter(|(_, inode)| inode.is_preopened).map( +- |(name, inode)| { +- let stat = inode.stat.read().unwrap(); +- ( +- inode.name.read().unwrap().to_string(), +- stat.st_filetype, +- stat.st_ino, +- ) +- }, +- )); +- // adding . and .. special folders +- // TODO: inode +- entry_vec.push((".".to_string(), Filetype::Directory, 0)); +- entry_vec.push(("..".to_string(), Filetype::Directory, 0)); +- entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); +- entry_vec +- } +- Kind::Root { entries } => { +- trace!("reading root"); +- let sorted_entries = { +- let mut entry_vec: Vec<(String, InodeGuard)> = entries +- .iter() +- .map(|(a, b)| (a.clone(), b.clone())) +- .collect(); +- entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); +- entry_vec +- }; +- sorted_entries +- .into_iter() +- .map(|(name, inode)| { +- let stat = inode.stat.read().unwrap(); +- ( +- format!("/{}", inode.name.read().unwrap().as_ref()), +- stat.st_filetype, +- stat.st_ino, +- ) +- }) +- .collect() +- } +- Kind::File { .. } +- | Kind::Symlink { .. } +- | Kind::Buffer { .. } +- | Kind::Socket { .. } +- | Kind::PipeRx { .. } +- | Kind::PipeTx { .. } +- | Kind::DuplexPipe { .. } +- | Kind::EventNotifications { .. } +- | Kind::Epoll { .. } => return Ok(Errno::Notdir), ++ let entries = { ++ let cached = if cookie == 0 { ++ None ++ } else { ++ working_dir ++ .inner ++ .readdir_cache ++ .read() ++ .unwrap() ++ .as_ref() ++ .cloned() ++ }; ++ if let Some(entries) = cached { ++ entries ++ } else { ++ let entries = { ++ let guard = working_dir.inode.read(); ++ match guard.deref() { ++ Kind::Dir { path, entries, .. } => { ++ trace!("reading dir {:?}", path); ++ // TODO: refactor this code ++ // we need to support multiple calls, ++ // simple and obviously correct implementation for now: ++ // maintain consistent order via lexacographic sorting ++ let fs_info = wasi_try_ok!( ++ wasi_try_ok!(state.fs_read_dir(path)) ++ .collect::, _>>() ++ .map_err(fs_error_into_wasi_err) ++ ); ++ let mut entry_vec = wasi_try_ok!( ++ fs_info ++ .into_iter() ++ .map(|entry| { ++ let filename = entry.file_name().to_string_lossy().to_string(); ++ trace!("getting file: {:?}", filename); ++ let filetype = virtual_file_type_to_wasi_file_type( ++ entry.file_type().map_err(fs_error_into_wasi_err)?, ++ ); ++ Ok(( ++ filename, filetype, 0, // TODO: inode ++ )) ++ }) ++ .collect::, _>>() ++ ); ++ entry_vec.extend( ++ entries.iter().filter(|(_, inode)| inode.is_preopened).map( ++ |(name, inode)| { ++ let stat = inode.stat.read().unwrap(); ++ ( ++ inode.name.read().unwrap().to_string(), ++ stat.st_filetype, ++ stat.st_ino, ++ ) ++ }, ++ ), ++ ); ++ // adding . and .. special folders ++ // TODO: inode ++ entry_vec.push((".".to_string(), Filetype::Directory, 0)); ++ entry_vec.push(("..".to_string(), Filetype::Directory, 0)); ++ entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); ++ entry_vec ++ } ++ Kind::Root { entries } => { ++ trace!("reading root"); ++ let sorted_entries = { ++ let mut entry_vec: Vec<(String, InodeGuard)> = entries ++ .iter() ++ .map(|(a, b)| (a.clone(), b.clone())) ++ .collect(); ++ entry_vec.sort_by(|a, b| a.0.cmp(&b.0)); ++ entry_vec ++ }; ++ sorted_entries ++ .into_iter() ++ .map(|(name, inode)| { ++ let stat = inode.stat.read().unwrap(); ++ ( ++ format!("/{}", inode.name.read().unwrap().as_ref()), ++ stat.st_filetype, ++ stat.st_ino, ++ ) ++ }) ++ .collect() ++ } ++ Kind::File { .. } ++ | Kind::Symlink { .. } ++ | Kind::Buffer { .. } ++ | Kind::Socket { .. } ++ | Kind::PipeRx { .. } ++ | Kind::PipeTx { .. } ++ | Kind::DuplexPipe { .. } ++ | Kind::EventNotifications { .. } ++ | Kind::Epoll { .. } => return Ok(Errno::Notdir), ++ } ++ }; ++ let entries = std::sync::Arc::new(entries); ++ *working_dir.inner.readdir_cache.write().unwrap() = Some(entries.clone()); ++ entries + } + }; + +diff --git a/lib/wasix/src/syscalls/wasi/fd_renumber.rs b/lib/wasix/src/syscalls/wasi/fd_renumber.rs +index 4dc700f..c7253f5 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_renumber.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_renumber.rs +@@ -54,9 +54,6 @@ pub(crate) fn fd_renumber_internal( + from: WasiFd, + to: WasiFd, + ) -> Result { +- if from == to { +- return Ok(Errno::Success); +- } + let env = ctx.data(); + let (_, mut state) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; + +@@ -68,7 +65,11 @@ pub(crate) fn fd_renumber_internal( + let mut fd_map = state.fs.fd_map.write().unwrap(); + + // Validate the source first. If `from` is invalid we must not mutate `to`. +- let fd_entry = wasi_try_ok!(fd_map.get(from).ok_or(Errno::Badf)); ++ wasi_try_ok!(fd_map.get(from).ok_or(Errno::Badf)); ++ ++ if from == to { ++ return Ok(Errno::Success); ++ } + + // Never allow renumbering over preopens. + if let Some(target_fd) = fd_map.get(to) +@@ -79,28 +80,11 @@ pub(crate) fn fd_renumber_internal( + return Ok(Errno::Notsup); + } + +- let new_fd_entry = Fd { +- inner: FdInner { +- offset: fd_entry.inner.offset.clone(), +- rights: fd_entry.inner.rights_inheriting, +- fd_flags: { +- let mut f = fd_entry.inner.fd_flags; +- f.set(Fdflagsext::CLOEXEC, false); +- f +- }, +- ..fd_entry.inner +- }, +- inode: fd_entry.inode.clone(), +- ..*fd_entry +- }; +- +- // Remove the target FD under the same lock (replaces the separate +- // close_fd call which would acquire its own lock). +- old_fd = fd_map.remove(to); +- +- if !fd_map.insert(true, to, new_fd_entry) { +- panic!("Internal error: expected FD {to} to be free after closing in fd_renumber"); +- } ++ // This is a move, not dup-to: source ownership remains accounted for ++ // and the source slot ceases to exist in the same critical section. ++ old_fd = fd_map ++ .renumber_deferred(from, to) ++ .expect("source was validated while holding the fd-map write lock"); + } + // Flush and drop the old FD outside the lock. The flush is best-effort: + // failures are intentionally ignored so fd_renumber result depends only on +@@ -114,6 +98,9 @@ pub(crate) fn fd_renumber_internal( + _ => None, + } + }); ++ if let Some(old_fd) = old_fd.as_ref() { ++ old_fd.release_descriptor(); ++ } + drop(old_fd); + + if let Some(file) = flush_target { +diff --git a/lib/wasix/src/syscalls/wasi/fd_seek.rs b/lib/wasix/src/syscalls/wasi/fd_seek.rs +index 35fdf7a..38b3cd5 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_seek.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_seek.rs +@@ -1,4 +1,5 @@ + use super::*; ++use crate::WasiFs; + use crate::syscalls::*; + + /// ### `fd_seek()` +@@ -55,7 +56,6 @@ pub(crate) fn fd_seek_internal( + ) -> Result, WasiError> { + let env = ctx.data(); + let state = env.state.clone(); +- let (memory, _) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; + let fd_entry = wasi_try_ok_ok!(state.fs.get_fd(fd)); + + if !fd_entry.inner.rights.contains(Rights::FD_SEEK) { +@@ -85,55 +85,15 @@ pub(crate) fn fd_seek_internal( + } + } + Whence::End => { +- use std::io::SeekFrom; +- let mut guard = fd_entry.inode.write(); +- let deref_mut = guard.deref_mut(); +- match deref_mut { +- Kind::File { handle, .. } => { +- // TODO: remove allow once inodes are refactored (see comments on [`WasiState`]) +- #[allow(clippy::await_holding_lock)] +- if let Some(handle) = handle { +- let handle = handle.clone(); +- let fd_offset = fd_entry.inner.offset.clone(); +- drop(guard); +- +- wasi_try_ok_ok!(__asyncify(ctx, None, async move { +- let mut handle = handle.write().unwrap(); +- let end = handle +- .seek(SeekFrom::End(offset)) +- .await +- .map_err(map_io_err)?; +- +- // Keep updating the original file description offset even if this +- // numeric FD was concurrently closed or reused. +- fd_offset.store(end, Ordering::Release); +- Ok(()) +- })?); +- } else { +- return Ok(Err(Errno::Inval)); +- } +- } +- Kind::Symlink { .. } => { +- unimplemented!("wasi::fd_seek not implemented for symlinks") +- } +- Kind::Dir { .. } +- | Kind::Root { .. } +- | Kind::Socket { .. } +- | Kind::PipeRx { .. } +- | Kind::PipeTx { .. } +- | Kind::DuplexPipe { .. } +- | Kind::EventNotifications { .. } +- | Kind::Epoll { .. } => { +- // TODO: check this +- return Ok(Err(Errno::Inval)); +- } +- Kind::Buffer { .. } => { +- // seeking buffers probably makes sense +- // FIXME: implement this +- return Ok(Err(Errno::Inval)); +- } +- } +- fd_entry.inner.offset.load(Ordering::Acquire) ++ let end = WasiFs::authoritative_fd_size(&fd_entry); ++ let new_offset = if offset >= 0 { ++ end.checked_add(offset as u64).ok_or(Errno::Overflow) ++ } else { ++ end.checked_sub(offset.unsigned_abs()).ok_or(Errno::Inval) ++ }; ++ let new_offset = wasi_try_ok_ok!(new_offset); ++ fd_entry.inner.offset.store(new_offset, Ordering::Release); ++ new_offset + } + Whence::Set => { + let offset: u64 = wasi_try_ok_ok!(u64::try_from(offset).map_err(|_| Errno::Inval)); +diff --git a/lib/wasix/src/syscalls/wasi/fd_sync.rs b/lib/wasix/src/syscalls/wasi/fd_sync.rs +index d16db5a..3e460d2 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_sync.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_sync.rs +@@ -15,55 +15,15 @@ pub fn fd_sync(mut ctx: FunctionEnvMut<'_, WasiEnv>, fd: WasiFd) -> Result { +- if let Some(handle) = handle { +- let handle = handle.clone(); +- drop(guard); +- +- // TODO: remove allow once inodes are refactored (see comments on [`WasiState`]) +- #[allow(clippy::await_holding_lock)] +- let size = { +- wasi_try_ok!(__asyncify(&mut ctx, None, async move { +- // TODO: remove allow once inodes are refactored (see comments on [`WasiState`]) +- #[allow(clippy::await_holding_lock)] +- let mut handle = handle.write().unwrap(); +- handle.flush().await.map_err(map_io_err)?; +- Ok(handle.size()) +- })?) +- }; +- +- // Update FileStat to reflect the correct current size. +- // TODO: don't lock twice - currently needed to not keep a lock on all inodes +- { +- let mut guard = inode.stat.write().unwrap(); +- guard.st_size = size; +- } +- } else { +- return Ok(Errno::Inval); +- } +- } +- Kind::Root { .. } | Kind::Dir { .. } => return Ok(Errno::Isdir), +- Kind::Buffer { .. } +- | Kind::Symlink { .. } +- | Kind::Socket { .. } +- | Kind::PipeTx { .. } +- | Kind::PipeRx { .. } +- | Kind::DuplexPipe { .. } +- | Kind::EventNotifications { .. } +- | Kind::Epoll { .. } => return Ok(Errno::Inval), +- } ++ if let Some(()) = wasi_try_ok!(state.fs.sync_blocking(fd, true)) { ++ return Ok(Errno::Success); + } +- +- Ok(Errno::Success) ++ Ok(wasi_try_ok!(__asyncify(&mut ctx, None, async move { ++ state.fs.sync(fd, true).await.map(|_| Errno::Success) ++ })?)) + } +diff --git a/lib/wasix/src/syscalls/wasi/fd_write.rs b/lib/wasix/src/syscalls/wasi/fd_write.rs +index 68e50b5..362f917 100644 +--- a/lib/wasix/src/syscalls/wasi/fd_write.rs ++++ b/lib/wasix/src/syscalls/wasi/fd_write.rs +@@ -1,4 +1,8 @@ +-use std::task::Waker; ++use std::{ ++ sync::{Arc, RwLock}, ++ task::Waker, ++ time::Duration, ++}; + + use super::*; + #[cfg(feature = "journal")] +@@ -7,6 +11,7 @@ use crate::{ + utils::map_snapshot_err, + }; + use crate::{net::socket::TimeType, syscalls::*}; ++use virtual_fs::VirtualFile; + + /// ### `fd_write()` + /// Write data to the file descriptor +@@ -105,7 +110,6 @@ pub fn fd_pwrite( + )?); + + Span::current().record("nwritten", bytes_written); +- + let mut env = ctx.data(); + let memory = unsafe { env.memory_view(&ctx) }; + let nwritten_ref = nwritten.deref(&memory); +@@ -124,6 +128,605 @@ pub(crate) enum FdWriteSource<'a, M: MemorySize> { + Buffer(Cow<'a, [u8]>), + } + ++fn blocking_sync_available(handle: &(dyn VirtualFile + Send + Sync), fd_flags: Fdflags) -> bool { ++ if fd_flags.contains(Fdflags::SYNC) { ++ handle.is_sync_on_write(true) || handle.has_blocking_sync_all_to_disk() ++ } else if fd_flags.contains(Fdflags::DSYNC) { ++ handle.is_sync_on_write(false) || handle.has_blocking_sync_data_to_disk() ++ } else { ++ true ++ } ++} ++ ++fn sync_after_blocking_write( ++ handle: &mut (dyn VirtualFile + Send + Sync), ++ fd_flags: Fdflags, ++) -> Result<(), Errno> { ++ if fd_flags.contains(Fdflags::SYNC) { ++ if !handle.is_sync_on_write(true) { ++ handle.sync_all_to_disk_blocking().map_err(map_io_err)?; ++ } ++ } else if fd_flags.contains(Fdflags::DSYNC) && !handle.is_sync_on_write(false) { ++ handle.sync_data_to_disk_blocking().map_err(map_io_err)?; ++ } ++ Ok(()) ++} ++ ++fn positioned_write_needs_explicit_sync(fd_flags: Fdflags) -> bool { ++ fd_flags.intersects(Fdflags::SYNC | Fdflags::DSYNC) ++} ++ ++fn is_all_zero(buf: &[u8]) -> bool { ++ buf.iter().all(|byte| *byte == 0) ++} ++ ++fn try_write_file_positioned_shared_blocking( ++ handle: &(dyn VirtualFile + Send + Sync), ++ memory: &wasmer::MemoryView<'_>, ++ data: &FdWriteSource<'_, M>, ++ offset: u64, ++ fd_flags: Fdflags, ++) -> Result, Errno> { ++ if positioned_write_needs_explicit_sync(fd_flags) || !handle.has_blocking_write_at_shared() { ++ return Ok(None); ++ } ++ ++ let mut written = 0usize; ++ let mut write_offset = offset; ++ ++ match data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ if *iovs_len == M::ONE { ++ let iov = iovs.read(memory).map_err(mem_error_to_wasi)?; ++ let buf = WasmPtr::::new(iov.buf) ++ .slice(memory, iov.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ if buf.as_ref().is_empty() { ++ return Ok(Some(0)); ++ } ++ ++ if handle.has_blocking_write_zeroes_at_shared() && is_all_zero(buf.as_ref()) { ++ let zeroes_written = match handle ++ .write_zeroes_at_blocking_shared(buf.as_ref().len() as u64, offset) ++ { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ return Ok(Some(zeroes_written)); ++ } ++ ++ let local_written = match handle.write_at_blocking_shared(buf.as_ref(), offset) { ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ return Ok(Some(local_written)); ++ } ++ ++ if handle.has_blocking_write_vectored_at_shared() { ++ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ let mut non_empty_count = 0usize; ++ let mut repeated_iov = None; ++ let mut all_non_empty_iovs_match = true; ++ for iov in iovs_arr.iter() { ++ if iov.buf_len == M::ZERO { ++ continue; ++ } ++ non_empty_count += 1; ++ match repeated_iov { ++ Some((buf, buf_len)) if buf == iov.buf && buf_len == iov.buf_len => {} ++ Some(_) => all_non_empty_iovs_match = false, ++ None => repeated_iov = Some((iov.buf, iov.buf_len)), ++ } ++ } ++ ++ if non_empty_count == 0 { ++ return Ok(Some(0)); ++ } ++ ++ if non_empty_count > 1 { ++ if all_non_empty_iovs_match { ++ let (buf, buf_len) = repeated_iov.expect("non-empty iovec missing"); ++ let access_guard = WasmPtr::::new(buf) ++ .slice(memory, buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ if handle.has_blocking_write_zeroes_at_shared() ++ && is_all_zero(access_guard.as_ref()) ++ { ++ let buf_len = from_offset::(buf_len)?; ++ let total_len = buf_len ++ .checked_mul(non_empty_count) ++ .ok_or(Errno::Overflow)?; ++ let zeroes_written = match handle ++ .write_zeroes_at_blocking_shared(total_len as u64, offset) ++ { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ return Ok(Some(zeroes_written)); ++ } ++ ++ let repeated = std::io::IoSlice::new(access_guard.as_ref()); ++ let bufs = std::iter::repeat(repeated) ++ .take(non_empty_count) ++ .collect::>(); ++ let vectored_written = ++ match handle.write_vectored_at_blocking_shared(&bufs, offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ return Ok(Some(vectored_written)); ++ } ++ ++ let mut access_guards = Vec::with_capacity(non_empty_count); ++ for iov in iovs_arr.iter() { ++ if iov.buf_len == M::ZERO { ++ continue; ++ } ++ let buf = WasmPtr::::new(iov.buf) ++ .slice(memory, iov.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ access_guards.push(buf); ++ } ++ ++ let bufs = access_guards ++ .iter() ++ .map(|buf| std::io::IoSlice::new(buf.as_ref())) ++ .collect::>(); ++ let vectored_written = ++ match handle.write_vectored_at_blocking_shared(&bufs, offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ return Ok(Some(vectored_written)); ++ } ++ } ++ ++ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ for iovs in iovs_arr.iter() { ++ let buf = WasmPtr::::new(iovs.buf) ++ .slice(memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ let local_written = ++ match handle.write_at_blocking_shared(buf.as_ref(), write_offset) { ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ written += local_written; ++ write_offset += local_written as u64; ++ if local_written != buf.len() { ++ break; ++ } ++ } ++ } ++ FdWriteSource::Buffer(data) => { ++ while written < data.len() { ++ let local_written = ++ match handle.write_at_blocking_shared(&data[written..], write_offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ written += local_written; ++ write_offset += local_written as u64; ++ } ++ } ++ } ++ ++ Ok(Some(written)) ++} ++ ++fn try_write_file_blocking( ++ handle: &mut (dyn VirtualFile + Send + Sync), ++ memory: &wasmer::MemoryView<'_>, ++ data: &FdWriteSource<'_, M>, ++ offset: u64, ++ positioned_write: bool, ++ fd_flags: Fdflags, ++) -> Result, Errno> { ++ let has_blocking_write = if positioned_write { ++ handle.has_blocking_write_at() ++ } else { ++ handle.has_blocking_seek() && handle.has_blocking_write() ++ }; ++ ++ if !has_blocking_write || !blocking_sync_available(handle, fd_flags) { ++ return Ok(None); ++ } ++ ++ if !positioned_write { ++ handle ++ .seek_blocking(std::io::SeekFrom::Start(offset)) ++ .map_err(map_io_err)?; ++ } ++ ++ let mut written = 0usize; ++ let mut write_offset = offset; ++ ++ match data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ if *iovs_len == M::ONE { ++ let iov = iovs.read(memory).map_err(mem_error_to_wasi)?; ++ let buf = WasmPtr::::new(iov.buf) ++ .slice(memory, iov.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ if buf.as_ref().is_empty() { ++ return Ok(Some(0)); ++ } ++ ++ if positioned_write ++ && handle.has_blocking_write_zeroes_at() ++ && is_all_zero(buf.as_ref()) ++ { ++ let zeroes_written = ++ match handle.write_zeroes_at_blocking(buf.as_ref().len() as u64, offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ sync_after_blocking_write(handle, fd_flags)?; ++ return Ok(Some(zeroes_written)); ++ } ++ ++ let local_written = if positioned_write { ++ match handle.write_at_blocking(buf.as_ref(), offset) { ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ } else { ++ match handle.write_blocking(buf.as_ref()) { ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ }; ++ if local_written > 0 { ++ sync_after_blocking_write(handle, fd_flags)?; ++ } ++ return Ok(Some(local_written)); ++ } ++ ++ if positioned_write && handle.has_blocking_write_vectored_at() { ++ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ let mut non_empty_count = 0usize; ++ let mut repeated_iov = None; ++ let mut all_non_empty_iovs_match = true; ++ for iov in iovs_arr.iter() { ++ if iov.buf_len == M::ZERO { ++ continue; ++ } ++ non_empty_count += 1; ++ match repeated_iov { ++ Some((buf, buf_len)) if buf == iov.buf && buf_len == iov.buf_len => {} ++ Some(_) => all_non_empty_iovs_match = false, ++ None => repeated_iov = Some((iov.buf, iov.buf_len)), ++ } ++ } ++ ++ if non_empty_count == 0 { ++ return Ok(Some(0)); ++ } ++ ++ if non_empty_count > 1 { ++ if all_non_empty_iovs_match { ++ let (buf, buf_len) = repeated_iov.expect("non-empty iovec missing"); ++ let access_guard = WasmPtr::::new(buf) ++ .slice(memory, buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ if handle.has_blocking_write_zeroes_at() ++ && is_all_zero(access_guard.as_ref()) ++ { ++ let buf_len = from_offset::(buf_len)?; ++ let total_len = buf_len ++ .checked_mul(non_empty_count) ++ .ok_or(Errno::Overflow)?; ++ let zeroes_written = ++ match handle.write_zeroes_at_blocking(total_len as u64, offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ sync_after_blocking_write(handle, fd_flags)?; ++ return Ok(Some(zeroes_written)); ++ } ++ ++ let repeated = std::io::IoSlice::new(access_guard.as_ref()); ++ let bufs = std::iter::repeat(repeated) ++ .take(non_empty_count) ++ .collect::>(); ++ let vectored_written = ++ match handle.write_vectored_at_blocking(&bufs, offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ sync_after_blocking_write(handle, fd_flags)?; ++ return Ok(Some(vectored_written)); ++ } ++ ++ let mut access_guards = Vec::with_capacity(non_empty_count); ++ for iov in iovs_arr.iter() { ++ if iov.buf_len == M::ZERO { ++ continue; ++ } ++ let buf = WasmPtr::::new(iov.buf) ++ .slice(memory, iov.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ access_guards.push(buf); ++ } ++ ++ let bufs = access_guards ++ .iter() ++ .map(|buf| std::io::IoSlice::new(buf.as_ref())) ++ .collect::>(); ++ let vectored_written = match handle.write_vectored_at_blocking(&bufs, offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ sync_after_blocking_write(handle, fd_flags)?; ++ return Ok(Some(vectored_written)); ++ } ++ } ++ ++ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ for iovs in iovs_arr.iter() { ++ let buf = WasmPtr::::new(iovs.buf) ++ .slice(memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ let local_written = if positioned_write { ++ match handle.write_at_blocking(buf.as_ref(), write_offset) { ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ } else { ++ match handle.write_blocking(buf.as_ref()) { ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ }; ++ written += local_written; ++ write_offset += local_written as u64; ++ if local_written != buf.len() { ++ break; ++ } ++ } ++ } ++ FdWriteSource::Buffer(data) => { ++ while written < data.len() { ++ let local_written = if positioned_write { ++ match handle.write_at_blocking(&data[written..], write_offset) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ } else { ++ match handle.write_blocking(&data[written..]) { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ }; ++ written += local_written; ++ write_offset += local_written as u64; ++ } ++ } ++ } ++ ++ if written > 0 { ++ sync_after_blocking_write(handle, fd_flags)?; ++ } ++ ++ Ok(Some(written)) ++} ++ ++fn try_write_file_positioned_blocking( ++ handle: &mut (dyn VirtualFile + Send + Sync), ++ memory: &wasmer::MemoryView<'_>, ++ data: &FdWriteSource<'_, M>, ++ offset: u64, ++ fd_flags: Fdflags, ++) -> Result, Errno> { ++ try_write_file_blocking(handle, memory, data, offset, true, fd_flags) ++} ++ ++fn try_write_file_cursor_blocking( ++ handle: &mut (dyn VirtualFile + Send + Sync), ++ memory: &wasmer::MemoryView<'_>, ++ data: &FdWriteSource<'_, M>, ++ offset: u64, ++ fd_flags: Fdflags, ++) -> Result, Errno> { ++ try_write_file_blocking(handle, memory, data, offset, false, fd_flags) ++} ++ ++#[allow(clippy::too_many_arguments)] ++fn write_regular_file( ++ env: &WasiEnv, ++ fd_entry: &Fd, ++ is_stdio: bool, ++ fd_flags: Fdflags, ++ memory: &wasmer::MemoryView<'_>, ++ data: &FdWriteSource<'_, M>, ++ offset: &mut u64, ++ should_update_cursor: bool, ++ handle: Arc>>, ++) -> Result, WasiError> { ++ let positioned_write = !is_stdio && !should_update_cursor; ++ if !is_stdio && fd_flags.contains(Fdflags::APPEND) { ++ // `fdflags::append` means we need to seek to the end before writing. ++ *offset = fd_entry.inode.stat.read().unwrap().st_size; ++ fd_entry.inner.offset.store(*offset, Ordering::Release); ++ } ++ ++ let blocking_written = if !is_stdio { ++ if positioned_write { ++ let handle_guard = handle.read().unwrap(); ++ let shared_write = try_write_file_positioned_shared_blocking::( ++ handle_guard.as_ref(), ++ memory, ++ data, ++ *offset, ++ fd_flags, ++ ); ++ match shared_write { ++ Ok(Some(written)) => Some(written), ++ Ok(None) => { ++ drop(handle_guard); ++ let mut handle = handle.write().unwrap(); ++ match try_write_file_positioned_blocking::( ++ handle.as_mut(), ++ memory, ++ data, ++ *offset, ++ fd_flags, ++ ) { ++ Ok(written) => written, ++ Err(err) => return Ok(Err(err)), ++ } ++ } ++ Err(err) => return Ok(Err(err)), ++ } ++ } else { ++ let mut handle = handle.write().unwrap(); ++ match try_write_file_cursor_blocking::( ++ handle.as_mut(), ++ memory, ++ data, ++ *offset, ++ fd_flags, ++ ) { ++ Ok(written) => written, ++ Err(err) => return Ok(Err(err)), ++ } ++ } ++ } else { ++ None ++ }; ++ ++ if let Some(written) = blocking_written { ++ return Ok(Ok(written)); ++ } ++ ++ let res = __asyncify_light( ++ env, ++ if fd_entry.inner.flags.contains(Fdflags::NONBLOCK) { ++ Some(Duration::ZERO) ++ } else { ++ None ++ }, ++ async { ++ let mut handle = handle.write().unwrap(); ++ let positioned_write = !is_stdio && !should_update_cursor; ++ if !is_stdio && !positioned_write { ++ handle ++ .seek(std::io::SeekFrom::Start(*offset)) ++ .await ++ .map_err(map_io_err)?; ++ } ++ ++ let mut written = 0usize; ++ let mut write_offset = *offset; ++ ++ match data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ let iovs_arr = iovs.slice(memory, *iovs_len).map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ for iovs in iovs_arr.iter() { ++ let buf = WasmPtr::::new(iovs.buf) ++ .slice(memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi)?; ++ let local_written = if positioned_write { ++ match handle.write_at(buf.as_ref(), write_offset).await { ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ } else { ++ match handle.write(buf.as_ref()).await { ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ } ++ }; ++ written += local_written; ++ write_offset += local_written as u64; ++ if local_written != buf.len() { ++ break; ++ } ++ } ++ } ++ FdWriteSource::Buffer(data) => { ++ if positioned_write { ++ while written < data.len() { ++ let local_written = ++ match handle.write_at(&data[written..], write_offset).await { ++ Ok(0) => return Err(Errno::Io), ++ Ok(s) => s, ++ Err(_) if written > 0 => break, ++ Err(err) => return Err(map_io_err(err)), ++ }; ++ written += local_written; ++ write_offset += local_written as u64; ++ } ++ } else { ++ handle.write_all(data).await?; ++ written += data.len(); ++ } ++ } ++ } ++ ++ if is_stdio { ++ handle.flush().await.map_err(map_io_err)?; ++ } else if written > 0 { ++ if fd_flags.contains(Fdflags::SYNC) { ++ if !handle.is_sync_on_write(true) { ++ handle.sync_all_to_disk().await.map_err(map_io_err)?; ++ } ++ } else if fd_flags.contains(Fdflags::DSYNC) && !handle.is_sync_on_write(false) { ++ handle.sync_data_to_disk().await.map_err(map_io_err)?; ++ } ++ } ++ Ok(written) ++ }, ++ ); ++ let written = res?.map_err(|err| match err { ++ Errno::Timedout => Errno::Again, ++ a => a, ++ }); ++ Ok(written) ++} ++ + #[allow(clippy::await_holding_lock)] + pub(crate) fn fd_write_internal( + mut ctx: &mut FunctionEnvMut<'_, WasiEnv>, +@@ -149,352 +752,332 @@ pub(crate) fn fd_write_internal( + + let (bytes_written, is_file, can_snapshot) = { + let (mut memory, _) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; +- let mut guard = fd_entry.inode.write(); +- match guard.deref_mut() { +- Kind::File { handle, .. } => { +- if let Some(handle) = handle { +- let handle = handle.clone(); ++ if let Some(handle) = { ++ let guard = fd_entry.inode.read(); ++ match guard.deref() { ++ Kind::File { ++ handle: Some(handle), ++ .. ++ } => Some(handle.clone()), ++ Kind::File { handle: None, .. } => return Ok(Err(Errno::Inval)), ++ _ => None, ++ } ++ } { ++ let written = wasi_try_ok_ok!(write_regular_file::( ++ env, ++ &fd_entry, ++ is_stdio, ++ fd_flags, ++ &memory, ++ &data, ++ &mut offset, ++ should_update_cursor, ++ handle, ++ )?); ++ (written, true, true) ++ } else { ++ let mut guard = fd_entry.inode.write(); ++ match guard.deref_mut() { ++ Kind::File { handle, .. } => { ++ if let Some(handle) = handle { ++ let handle = handle.clone(); ++ drop(guard); ++ let written = wasi_try_ok_ok!(write_regular_file::( ++ env, ++ &fd_entry, ++ is_stdio, ++ fd_flags, ++ &memory, ++ &data, ++ &mut offset, ++ should_update_cursor, ++ handle, ++ )?); ++ (written, true, true) ++ } else { ++ return Ok(Err(Errno::Inval)); ++ } ++ } ++ Kind::Socket { socket } => { ++ let socket = socket.clone(); + drop(guard); + +- let res = __asyncify_light( +- env, +- if fd_entry.inner.flags.contains(Fdflags::NONBLOCK) { +- Some(Duration::ZERO) +- } else { +- None +- }, +- async { +- let mut handle = handle.write().unwrap(); +- if !is_stdio { +- if fd_entry.inner.flags.contains(Fdflags::APPEND) { +- // `fdflags::append` means we need to seek to the end before writing. +- offset = fd_entry.inode.stat.read().unwrap().st_size; +- fd_entry.inner.offset.store(offset, Ordering::Release); +- } ++ let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); ++ let timeout = socket ++ .opt_time(TimeType::WriteTimeout) ++ .ok() ++ .flatten() ++ .unwrap_or(Duration::from_secs(30)); + +- handle +- .seek(std::io::SeekFrom::Start(offset)) +- .await +- .map_err(map_io_err)?; +- } ++ let tasks = env.tasks().clone(); + +- let mut written = 0usize; ++ let res = __asyncify_light(env, None, async { ++ let mut sent = 0usize; + +- match &data { +- FdWriteSource::Iovs { iovs, iovs_len } => { +- let iovs_arr = iovs +- .slice(&memory, *iovs_len) ++ match &data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ let iovs_arr = iovs ++ .slice(&memory, *iovs_len) ++ .map_err(mem_error_to_wasi)?; ++ let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ for iovs in iovs_arr.iter() { ++ let buf = WasmPtr::::new(iovs.buf) ++ .slice(&memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() + .map_err(mem_error_to_wasi)?; +- let iovs_arr = +- iovs_arr.access().map_err(mem_error_to_wasi)?; +- for iovs in iovs_arr.iter() { +- let buf = WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi)? +- .access() +- .map_err(mem_error_to_wasi)?; +- let local_written = +- match handle.write(buf.as_ref()).await { +- Ok(s) => s, +- Err(_) if written > 0 => break, +- Err(err) => return Err(map_io_err(err)), +- }; +- written += local_written; +- if local_written != buf.len() { +- break; +- } ++ let local_sent = socket ++ .send( ++ tasks.deref(), ++ buf.as_ref(), ++ Some(timeout), ++ nonblocking, ++ ) ++ .await?; ++ sent += local_sent; ++ if local_sent != buf.len() { ++ break; + } + } +- FdWriteSource::Buffer(data) => { +- handle.write_all(data).await?; +- written += data.len(); +- } +- } +- +- if is_stdio { +- handle.flush().await.map_err(map_io_err)?; + } +- Ok(written) +- }, +- ); +- let written = wasi_try_ok_ok!(res?.map_err(|err| match err { +- Errno::Timedout => Errno::Again, +- a => a, +- })); +- +- (written, true, true) +- } else { +- return Ok(Err(Errno::Inval)); +- } +- } +- Kind::Socket { socket } => { +- let socket = socket.clone(); +- drop(guard); +- +- let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); +- let timeout = socket +- .opt_time(TimeType::WriteTimeout) +- .ok() +- .flatten() +- .unwrap_or(Duration::from_secs(30)); +- +- let tasks = env.tasks().clone(); +- +- let res = __asyncify_light(env, None, async { +- let mut sent = 0usize; +- +- match &data { +- FdWriteSource::Iovs { iovs, iovs_len } => { +- let iovs_arr = +- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi)?; +- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; +- for iovs in iovs_arr.iter() { +- let buf = WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi)? +- .access() +- .map_err(mem_error_to_wasi)?; +- let local_sent = socket ++ FdWriteSource::Buffer(data) => { ++ sent += socket + .send( + tasks.deref(), +- buf.as_ref(), ++ data.as_ref(), + Some(timeout), + nonblocking, + ) + .await?; +- sent += local_sent; +- if local_sent != buf.len() { +- break; +- } + } + } +- FdWriteSource::Buffer(data) => { +- sent += socket +- .send(tasks.deref(), data.as_ref(), Some(timeout), nonblocking) +- .await?; +- } +- } +- Ok(sent) +- }); +- let written = wasi_try_ok_ok!(res?); +- (written, false, false) +- } +- Kind::PipeRx { .. } => { +- return Ok(Err(Errno::Badf)); +- } +- Kind::PipeTx { tx } => { +- let mut written = 0usize; +- +- match &data { +- FdWriteSource::Iovs { iovs, iovs_len } => { +- let mut raise_sigpipe = false; +- let iovs_arr = wasi_try_ok_ok!( +- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) +- ); +- let iovs_arr = +- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); +- for iovs in iovs_arr.iter() { +- let buf = wasi_try_ok_ok!( +- WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi) ++ Ok(sent) ++ }); ++ let written = wasi_try_ok_ok!(res?); ++ (written, false, false) ++ } ++ Kind::PipeRx { .. } => { ++ return Ok(Err(Errno::Badf)); ++ } ++ Kind::PipeTx { tx } => { ++ let mut written = 0usize; ++ ++ match &data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ let mut raise_sigpipe = false; ++ let iovs_arr = wasi_try_ok_ok!( ++ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) + ); +- let buf = wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); +- let write_result = std::io::Write::write(tx, buf.as_ref()); +- let local_written = match write_result { +- Ok(w) => w, +- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { +- // Need to do this to avoid double borrow on ctx with iovs_arr +- raise_sigpipe = true; ++ let iovs_arr = ++ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); ++ for iovs in iovs_arr.iter() { ++ let buf = wasi_try_ok_ok!( ++ WasmPtr::::new(iovs.buf) ++ .slice(&memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi) ++ ); ++ let buf = ++ wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); ++ let write_result = std::io::Write::write(tx, buf.as_ref()); ++ let local_written = match write_result { ++ Ok(w) => w, ++ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { ++ // Need to do this to avoid double borrow on ctx with iovs_arr ++ raise_sigpipe = true; ++ break; ++ } ++ Err(e) => return Ok(Err(map_io_err(e))), ++ }; ++ ++ written += local_written; ++ if local_written != buf.len() { + break; + } +- Err(e) => return Ok(Err(map_io_err(e))), +- }; +- +- written += local_written; +- if local_written != buf.len() { +- break; + } +- } + +- drop(iovs_arr); ++ drop(iovs_arr); + +- if raise_sigpipe { +- env.process.signal_process(Signal::Sigpipe); +- wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); +- return Ok(Err(Errno::Pipe)); +- } +- } +- FdWriteSource::Buffer(data) => { +- match std::io::Write::write_all(tx, data) { +- Ok(()) => (), +- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { ++ if raise_sigpipe { + env.process.signal_process(Signal::Sigpipe); + wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); + return Ok(Err(Errno::Pipe)); + } +- Err(e) => return Ok(Err(map_io_err(e))), +- }; +- written += data.len(); ++ } ++ FdWriteSource::Buffer(data) => { ++ match std::io::Write::write_all(tx, data) { ++ Ok(()) => (), ++ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { ++ env.process.signal_process(Signal::Sigpipe); ++ wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); ++ return Ok(Err(Errno::Pipe)); ++ } ++ Err(e) => return Ok(Err(map_io_err(e))), ++ }; ++ written += data.len(); ++ } + } ++ ++ (written, false, true) + } ++ Kind::DuplexPipe { pipe } => { ++ let mut written = 0usize; + +- (written, false, true) +- } +- Kind::DuplexPipe { pipe } => { +- let mut written = 0usize; +- +- match &data { +- FdWriteSource::Iovs { iovs, iovs_len } => { +- let mut raise_sigpipe = false; +- let iovs_arr = wasi_try_ok_ok!( +- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) +- ); +- let iovs_arr = +- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); +- for iovs in iovs_arr.iter() { +- let buf = wasi_try_ok_ok!( +- WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi) ++ match &data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ let mut raise_sigpipe = false; ++ let iovs_arr = wasi_try_ok_ok!( ++ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) + ); +- let buf = wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); +- let write_result = std::io::Write::write(pipe, buf.as_ref()); +- let local_written = match write_result { +- Ok(w) => w, +- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { +- // Need to do this to avoid double borrow on ctx with iovs_arr +- raise_sigpipe = true; ++ let iovs_arr = ++ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); ++ for iovs in iovs_arr.iter() { ++ let buf = wasi_try_ok_ok!( ++ WasmPtr::::new(iovs.buf) ++ .slice(&memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi) ++ ); ++ let buf = ++ wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); ++ let write_result = std::io::Write::write(pipe, buf.as_ref()); ++ let local_written = match write_result { ++ Ok(w) => w, ++ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { ++ // Need to do this to avoid double borrow on ctx with iovs_arr ++ raise_sigpipe = true; ++ break; ++ } ++ Err(e) => return Ok(Err(map_io_err(e))), ++ }; ++ ++ written += local_written; ++ if local_written != buf.len() { + break; + } +- Err(e) => return Ok(Err(map_io_err(e))), +- }; +- +- written += local_written; +- if local_written != buf.len() { +- break; + } +- } + +- drop(iovs_arr); ++ drop(iovs_arr); + +- if raise_sigpipe { +- env.process.signal_process(Signal::Sigpipe); +- wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); +- return Ok(Err(Errno::Pipe)); +- } +- } +- FdWriteSource::Buffer(data) => { +- match std::io::Write::write_all(pipe, data) { +- Ok(()) => (), +- Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { ++ if raise_sigpipe { + env.process.signal_process(Signal::Sigpipe); + wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); + return Ok(Err(Errno::Pipe)); + } +- Err(e) => return Ok(Err(map_io_err(e))), +- }; +- written += data.len(); ++ } ++ FdWriteSource::Buffer(data) => { ++ match std::io::Write::write_all(pipe, data) { ++ Ok(()) => (), ++ Err(e) if e.kind() == std::io::ErrorKind::BrokenPipe => { ++ env.process.signal_process(Signal::Sigpipe); ++ wasi_try_ok_ok!(WasiEnv::process_signals_and_exit(ctx)?); ++ return Ok(Err(Errno::Pipe)); ++ } ++ Err(e) => return Ok(Err(map_io_err(e))), ++ }; ++ written += data.len(); ++ } + } ++ ++ (written, false, true) ++ } ++ Kind::Dir { .. } | Kind::Root { .. } => { ++ // TODO: verify ++ return Ok(Err(Errno::Isdir)); + } ++ Kind::EventNotifications { inner } => { ++ let mut written = 0usize; + +- (written, false, true) +- } +- Kind::Dir { .. } | Kind::Root { .. } => { +- // TODO: verify +- return Ok(Err(Errno::Isdir)); +- } +- Kind::EventNotifications { inner } => { +- let mut written = 0usize; +- +- match &data { +- FdWriteSource::Iovs { iovs, iovs_len } => { +- let iovs_arr = wasi_try_ok_ok!( +- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) +- ); +- let iovs_arr = +- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); +- for iovs in iovs_arr.iter() { +- let buf_len: usize = wasi_try_ok_ok!( +- iovs.buf_len.try_into().map_err(|_| Errno::Inval) ++ match &data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ let iovs_arr = wasi_try_ok_ok!( ++ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) + ); +- let will_be_written = buf_len; ++ let iovs_arr = ++ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); ++ for iovs in iovs_arr.iter() { ++ let buf_len: usize = wasi_try_ok_ok!( ++ iovs.buf_len.try_into().map_err(|_| Errno::Inval) ++ ); ++ let will_be_written = buf_len; + +- let val_cnt = buf_len / std::mem::size_of::(); +- let val_cnt: M::Offset = +- wasi_try_ok_ok!(val_cnt.try_into().map_err(|_| Errno::Inval)); ++ let val_cnt = buf_len / std::mem::size_of::(); ++ let val_cnt: M::Offset = wasi_try_ok_ok!( ++ val_cnt.try_into().map_err(|_| Errno::Inval) ++ ); + +- let vals = wasi_try_ok_ok!( +- WasmPtr::::new(iovs.buf) +- .slice(&memory, val_cnt as M::Offset) +- .map_err(mem_error_to_wasi) +- ); +- let vals = +- wasi_try_ok_ok!(vals.access().map_err(mem_error_to_wasi)); +- for val in vals.iter() { +- inner.write(*val); +- } ++ let vals = wasi_try_ok_ok!( ++ WasmPtr::::new(iovs.buf) ++ .slice(&memory, val_cnt as M::Offset) ++ .map_err(mem_error_to_wasi) ++ ); ++ let vals = ++ wasi_try_ok_ok!(vals.access().map_err(mem_error_to_wasi)); ++ for val in vals.iter() { ++ inner.write(*val); ++ } + +- written += will_be_written; ++ written += will_be_written; ++ } + } +- } +- FdWriteSource::Buffer(data) => { +- let cnt = data.len() / std::mem::size_of::(); +- for n in 0..cnt { +- let start = n * std::mem::size_of::(); +- let data = [ +- data[start], +- data[start + 1], +- data[start + 2], +- data[start + 3], +- data[start + 4], +- data[start + 5], +- data[start + 6], +- data[start + 7], +- ]; +- inner.write(u64::from_ne_bytes(data)); ++ FdWriteSource::Buffer(data) => { ++ let cnt = data.len() / std::mem::size_of::(); ++ for n in 0..cnt { ++ let start = n * std::mem::size_of::(); ++ let data = [ ++ data[start], ++ data[start + 1], ++ data[start + 2], ++ data[start + 3], ++ data[start + 4], ++ data[start + 5], ++ data[start + 6], ++ data[start + 7], ++ ]; ++ inner.write(u64::from_ne_bytes(data)); ++ } + } + } ++ ++ (written, false, true) + } ++ Kind::Symlink { .. } | Kind::Epoll { .. } => return Ok(Err(Errno::Inval)), ++ Kind::Buffer { buffer } => { ++ let mut written = 0usize; + +- (written, false, true) +- } +- Kind::Symlink { .. } | Kind::Epoll { .. } => return Ok(Err(Errno::Inval)), +- Kind::Buffer { buffer } => { +- let mut written = 0usize; +- +- match &data { +- FdWriteSource::Iovs { iovs, iovs_len } => { +- let iovs_arr = wasi_try_ok_ok!( +- iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) +- ); +- let iovs_arr = +- wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); +- for iovs in iovs_arr.iter() { +- let buf = wasi_try_ok_ok!( +- WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi) +- ); +- let buf = wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); +- let local_written = wasi_try_ok_ok!( +- std::io::Write::write(buffer, buf.as_ref()).map_err(map_io_err) ++ match &data { ++ FdWriteSource::Iovs { iovs, iovs_len } => { ++ let iovs_arr = wasi_try_ok_ok!( ++ iovs.slice(&memory, *iovs_len).map_err(mem_error_to_wasi) + ); +- written += local_written; +- if local_written != buf.len() { +- break; ++ let iovs_arr = ++ wasi_try_ok_ok!(iovs_arr.access().map_err(mem_error_to_wasi)); ++ for iovs in iovs_arr.iter() { ++ let buf = wasi_try_ok_ok!( ++ WasmPtr::::new(iovs.buf) ++ .slice(&memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi) ++ ); ++ let buf = ++ wasi_try_ok_ok!(buf.access().map_err(mem_error_to_wasi)); ++ let local_written = wasi_try_ok_ok!( ++ std::io::Write::write(buffer, buf.as_ref()) ++ .map_err(map_io_err) ++ ); ++ written += local_written; ++ if local_written != buf.len() { ++ break; ++ } + } + } ++ FdWriteSource::Buffer(data) => { ++ wasi_try_ok_ok!( ++ std::io::Write::write_all(buffer, data).map_err(map_io_err) ++ ); ++ written += data.len(); ++ } + } +- FdWriteSource::Buffer(data) => { +- wasi_try_ok_ok!( +- std::io::Write::write_all(buffer, data).map_err(map_io_err) +- ); +- written += data.len(); +- } +- } + +- (written, false, true) ++ (written, false, true) ++ } + } + } + }; +@@ -512,9 +1095,6 @@ pub(crate) fn fd_write_internal( + })?; + } + +- env = ctx.data(); +- memory = unsafe { env.memory_view(&ctx) }; +- + // reborrow and update the size + if !is_stdio { + let curr_offset = if is_file && should_update_cursor { +@@ -529,10 +1109,6 @@ pub(crate) fn fd_write_internal( + fd_entry.inner.offset.load(Ordering::Acquire) + }; + +- // we set the size but we don't return any errors if it fails as +- // pipes and sockets will not do anything with this +- let (mut memory, _, inodes) = +- unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; + if is_file { + let mut stat = fd_entry.inode.stat.write().unwrap(); + if should_update_cursor { +diff --git a/lib/wasix/src/syscalls/wasix/fd_sync_range.rs b/lib/wasix/src/syscalls/wasix/fd_sync_range.rs +new file mode 100644 +index 0000000..eb479ab +--- /dev/null ++++ b/lib/wasix/src/syscalls/wasix/fd_sync_range.rs +@@ -0,0 +1,252 @@ ++use super::*; ++use crate::syscalls::*; ++use virtual_fs::FileWritebackFlags; ++ ++const SYNC_FILE_RANGE_WAIT_BEFORE: u32 = 1; ++const SYNC_FILE_RANGE_WRITE: u32 = 2; ++const SYNC_FILE_RANGE_WAIT_AFTER: u32 = 4; ++const SYNC_FILE_RANGE_WAIT_BEFORE_WRITE: u32 = SYNC_FILE_RANGE_WAIT_BEFORE | SYNC_FILE_RANGE_WRITE; ++const SYNC_FILE_RANGE_WAIT_BEFORE_WAIT_AFTER: u32 = ++ SYNC_FILE_RANGE_WAIT_BEFORE | SYNC_FILE_RANGE_WAIT_AFTER; ++const SYNC_FILE_RANGE_WRITE_WAIT_AFTER: u32 = SYNC_FILE_RANGE_WRITE | SYNC_FILE_RANGE_WAIT_AFTER; ++const SYNC_FILE_RANGE_VALID_FLAGS: u32 = ++ SYNC_FILE_RANGE_WAIT_BEFORE | SYNC_FILE_RANGE_WRITE | SYNC_FILE_RANGE_WAIT_AFTER; ++ ++/// ### `fd_sync_range()` ++/// ++/// Starts and/or waits for writeback of a byte range without implying crash ++/// durability. This is the WASIX counterpart of Linux `sync_file_range(2)`; ++/// callers must still use `fd_datasync` or `fd_sync` at durability boundaries. ++#[instrument( ++ level = "trace", ++ skip_all, ++ fields(%fd, %offset, %len, flags = format_args!("{flags:#x}")), ++ ret ++)] ++pub fn fd_sync_range( ++ mut ctx: FunctionEnvMut<'_, WasiEnv>, ++ fd: WasiFd, ++ offset: i64, ++ len: i64, ++ flags: u32, ++) -> Result { ++ WasiEnv::do_pending_operations(&mut ctx)?; ++ ++ match fd_sync_range_internal(&mut ctx, fd, offset, len, flags) { ++ Ok(()) => Ok(Errno::Success), ++ Err(error) => Ok(error), ++ } ++} ++ ++pub(crate) fn fd_sync_range_internal( ++ ctx: &mut FunctionEnvMut<'_, WasiEnv>, ++ fd: WasiFd, ++ offset: i64, ++ len: i64, ++ flags: u32, ++) -> Result<(), Errno> { ++ let env = ctx.data(); ++ let (_, mut state) = unsafe { env.get_memory_and_wasi_state(&ctx, 0) }; ++ let fd_entry = state.fs.get_fd(fd)?; ++ let inode = fd_entry.inode; ++ ++ require_fd_sync_range_right(fd_entry.inner.rights)?; ++ ++ delegate_fd_sync_range(offset, len, flags, |offset, len, flags| { ++ let guard = inode.read(); ++ sync_range_inode_kind(guard.deref(), offset, len, flags) ++ }) ++} ++ ++fn sync_range_inode_kind( ++ kind: &Kind, ++ offset: u64, ++ len: u64, ++ flags: FileWritebackFlags, ++) -> Result<(), Errno> { ++ match kind { ++ Kind::File { ++ handle: Some(handle), ++ .. ++ } => handle ++ .read() ++ .map_err(|_| Errno::Fault)? ++ .writeback_range(offset, len, flags) ++ .map_err(fd_sync_range_io_error), ++ Kind::File { handle: None, .. } => Err(Errno::Badf), ++ Kind::Buffer { .. } => Err(Errno::Nosys), ++ Kind::PipeRx { .. } | Kind::PipeTx { .. } | Kind::DuplexPipe { .. } => Err(Errno::Spipe), ++ // The product ABI exposes range writeback only for regular ++ // VirtualFile handles. Linux may accept native directory fds, but that ++ // is outside PostgreSQL's use and is not emulated here. ++ Kind::Dir { .. } | Kind::Root { .. } => Err(Errno::Badf), ++ Kind::Socket { .. } => Err(Errno::Spipe), ++ Kind::Symlink { .. } | Kind::EventNotifications { .. } | Kind::Epoll { .. } => { ++ Err(Errno::Badf) ++ } ++ } ++} ++ ++fn require_fd_sync_range_right(rights: Rights) -> Result<(), Errno> { ++ // sync_file_range schedules or waits for writeback; it is not a ++ // durability boundary. PostgreSQL intentionally invokes it on O_RDONLY ++ // descriptors, so require the range-advice capability rather than a write ++ // or data-sync capability. ++ if rights.contains(Rights::FD_ADVISE) { ++ Ok(()) ++ } else { ++ Err(Errno::Notcapable) ++ } ++} ++ ++fn delegate_fd_sync_range( ++ offset: i64, ++ len: i64, ++ flags: u32, ++ delegate: impl FnOnce(u64, u64, FileWritebackFlags) -> Result<(), Errno>, ++) -> Result<(), Errno> { ++ if offset < 0 || len < 0 || flags & !SYNC_FILE_RANGE_VALID_FLAGS != 0 { ++ return Err(Errno::Inval); ++ } ++ ++ // Linux requires a representable signed exclusive end. len=0 is the ++ // through-EOF sentinel and therefore has no finite end to check. ++ if len > 0 { ++ offset.checked_add(len).ok_or(Errno::Inval)?; ++ } ++ ++ let flags = FileWritebackFlags::from_bits(flags).ok_or(Errno::Inval)?; ++ delegate(offset as u64, len as u64, flags) ++} ++ ++fn fd_sync_range_io_error(error: io::Error) -> Errno { ++ #[cfg(target_os = "linux")] ++ if let Some(error) = error.raw_os_error() { ++ return match error { ++ libc::EACCES => Errno::Access, ++ libc::EBADF => Errno::Badf, ++ libc::EINVAL => Errno::Inval, ++ libc::EIO => Errno::Io, ++ libc::ENOMEM => Errno::Nomem, ++ libc::ENOSPC => Errno::Nospc, ++ libc::ENOSYS => Errno::Nosys, ++ libc::EOPNOTSUPP => Errno::Notsup, ++ libc::EOVERFLOW => Errno::Overflow, ++ libc::EPERM => Errno::Perm, ++ libc::ESPIPE => Errno::Spipe, ++ _ => return map_io_err(io::Error::from_raw_os_error(error)), ++ }; ++ } ++ ++ match error.kind() { ++ io::ErrorKind::InvalidInput => Errno::Inval, ++ io::ErrorKind::Unsupported => Errno::Nosys, ++ _ => map_io_err(error), ++ } ++} ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn fd_sync_range_maps_all_exact_flag_combinations() { ++ for raw_flags in 0..=SYNC_FILE_RANGE_VALID_FLAGS { ++ let mut observed = None; ++ delegate_fd_sync_range(17, 23, raw_flags, |offset, len, flags| { ++ observed = Some((offset, len, flags.bits())); ++ Ok(()) ++ }) ++ .unwrap(); ++ ++ assert_eq!(observed, Some((17, 23, raw_flags))); ++ } ++ } ++ ++ #[test] ++ fn fd_sync_range_rejects_negative_overflowing_and_unknown_ranges() { ++ let cases = [ ++ (-1, 0, 0), ++ (0, -1, 0), ++ (i64::MAX, 1, 0), ++ (i64::MAX - 1, 2, 0), ++ (0, 0, SYNC_FILE_RANGE_VALID_FLAGS + 1), ++ ]; ++ ++ for (offset, len, flags) in cases { ++ let mut calls = 0; ++ let result = delegate_fd_sync_range(offset, len, flags, |_, _, _| { ++ calls += 1; ++ Ok(()) ++ }); ++ assert_eq!(result, Err(Errno::Inval)); ++ assert_eq!(calls, 0); ++ } ++ } ++ ++ #[test] ++ fn fd_sync_range_preserves_zero_length_and_maximum_finite_range() { ++ let mut observed = Vec::new(); ++ for (offset, len) in [(9, 0), (i64::MAX - 1, 1)] { ++ delegate_fd_sync_range(offset, len, 0, |offset, len, flags| { ++ observed.push((offset, len, flags.bits())); ++ Ok(()) ++ }) ++ .unwrap(); ++ } ++ ++ assert_eq!(observed, vec![(9, 0, 0), (i64::MAX as u64 - 1, 1, 0)]); ++ } ++ ++ #[test] ++ fn fd_sync_range_maps_unsupported_backends_to_nosys() { ++ let result = delegate_fd_sync_range(0, 4096, SYNC_FILE_RANGE_WRITE, |_, _, _| { ++ Err(fd_sync_range_io_error(io::ErrorKind::Unsupported.into())) ++ }); ++ ++ assert_eq!(result, Err(Errno::Nosys)); ++ } ++ ++ #[test] ++ fn fd_sync_range_read_only_advice_right_is_accepted() { ++ assert_eq!( ++ require_fd_sync_range_right(Rights::FD_READ), ++ Err(Errno::Notcapable) ++ ); ++ assert_eq!( ++ require_fd_sync_range_right(Rights::FD_READ | Rights::FD_ADVISE), ++ Ok(()) ++ ); ++ } ++ ++ #[test] ++ fn fd_sync_range_directory_is_badf() { ++ let directory = Kind::Root { ++ entries: Default::default(), ++ }; ++ assert_eq!( ++ sync_range_inode_kind(&directory, 0, 0, FileWritebackFlags::empty()), ++ Err(Errno::Badf) ++ ); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[test] ++ fn fd_sync_range_preserves_linux_writeback_errnos() { ++ let cases = [ ++ (libc::EBADF, Errno::Badf), ++ (libc::EINVAL, Errno::Inval), ++ (libc::EIO, Errno::Io), ++ (libc::ENOMEM, Errno::Nomem), ++ (libc::ENOSPC, Errno::Nospc), ++ (libc::ESPIPE, Errno::Spipe), ++ ]; ++ ++ for (host_errno, expected) in cases { ++ assert_eq!( ++ fd_sync_range_io_error(io::Error::from_raw_os_error(host_errno)), ++ expected ++ ); ++ } ++ } ++} +diff --git a/lib/wasix/src/syscalls/wasix/mem_mmap.rs b/lib/wasix/src/syscalls/wasix/mem_mmap.rs +new file mode 100644 +index 0000000..7a197de +--- /dev/null ++++ b/lib/wasix/src/syscalls/wasix/mem_mmap.rs +@@ -0,0 +1,453 @@ ++use super::*; ++use crate::state::{WasiSharedMemoryMapOrigin, WasiSharedMemoryRuntimePlacement}; ++use crate::syscalls::*; ++ ++const PROT_READ: u32 = 0x01; ++const PROT_WRITE: u32 = 0x02; ++const PROT_ALLOWED: u32 = PROT_READ | PROT_WRITE; ++ ++const MAP_SHARED: u32 = 0x01; ++const MAP_PRIVATE: u32 = 0x02; ++const MAP_TYPE: u32 = 0x0f; ++const MAP_FIXED: u32 = 0x10; ++const MAP_ANON: u32 = 0x20; ++const MAP_ALLOWED: u32 = MAP_TYPE | MAP_FIXED; ++const WASM_PAGE_SIZE: usize = 65_536; ++ ++const MS_ASYNC: u32 = 0x01; ++const MS_INVALIDATE: u32 = 0x02; ++const MS_SYNC: u32 = 0x04; ++ ++/// ### `mem_mmap()` ++/// ++/// Remaps a guest memory range to a host file-backed shared mapping. ++/// ++/// File-backed `MAP_SHARED` supports both runtime-selected and fixed mappings. ++/// A non-fixed request atomically claims new WebAssembly pages using the old ++/// size returned by memory.grow, so it cannot race a concurrent sbrk. A fixed ++/// request may replace only a fully tracked active mapping or an exact inherited ++/// exec reservation whose range, offset, and backing identity all match. ++#[instrument( ++ level = "trace", ++ skip_all, ++ fields(%addr, %len, %prot, %flags, %fd, %offset, ret_addr = field::Empty), ++ ret ++)] ++pub fn mem_mmap( ++ mut ctx: FunctionEnvMut<'_, WasiEnv>, ++ addr: M::Offset, ++ len: M::Offset, ++ prot: u32, ++ flags: u32, ++ fd: WasiFd, ++ offset: Filesize, ++ ret_addr: WasmPtr, ++) -> Result { ++ WasiEnv::do_pending_operations(&mut ctx)?; ++ ++ let requested_addr = wasi_try_ok!(from_offset::(addr)); ++ let len = wasi_try_ok!(from_offset::(len)); ++ let file_offset = wasi_try_ok!(offset.try_into().map_err(|_| Errno::Overflow)); ++ let page_size = host_page_size(); ++ let mapped_len = wasi_try_ok!(round_up_to_page(len, page_size)); ++ ++ // The sys backend currently installs a read/write host map. Enforce that ++ // exact contract at the ABI boundary too: callers may import mem_mmap ++ // directly and bypass wasix-libc's validation. ++ if len == 0 || prot != PROT_ALLOWED { ++ return Ok(Errno::Inval); ++ } ++ if file_offset & (page_size - 1) != 0 { ++ return Ok(Errno::Inval); ++ } ++ ++ if (flags & !MAP_ALLOWED) != 0 { ++ return Ok(Errno::Notsup); ++ } ++ ++ if (flags & MAP_TYPE) != MAP_SHARED || (flags & MAP_PRIVATE) != 0 { ++ return Ok(Errno::Notsup); ++ } ++ ++ if (flags & MAP_ANON) != 0 { ++ return Ok(Errno::Notsup); ++ } ++ let fixed = (flags & MAP_FIXED) != 0; ++ if fixed && requested_addr & (page_size - 1) != 0 { ++ return Ok(Errno::Inval); ++ } ++ ++ // Validate the result slot before memory.grow or a host remap. Linear ++ // memory never shrinks, so a pointer readable now remains writable at the ++ // final publication point; a bad pointer cannot strand a live mapping. ++ { ++ let memory_view = unsafe { ctx.data().memory_view(&ctx) }; ++ wasi_try_mem_ok!(ret_addr.read(&memory_view)); ++ } ++ ++ let memory = { ++ let env = ctx.data(); ++ env.inner().main_module_instance_handles().memory_clone() ++ }; ++ // Reject before fd/registry mutation and, critically, before the ++ // runtime-selected path uses memory.grow as its reservation primitive. ++ if !memory.supports_persistent_shared_fixed_remap(&ctx.as_store_ref()) { ++ return Ok(Errno::Notsup); ++ } ++ ++ let file = Arc::new(wasi_try_ok!(mappable_file(&ctx, fd, prot))); ++ let futexs = wasi_try_ok!( ++ ctx.data() ++ .state() ++ .futex_registry_for_shared_file(file.clone()) ++ ); ++ ++ let state = ctx.data().state.clone(); ++ let mapped_addr = if fixed { ++ let mapped_end = wasi_try_ok!( ++ requested_addr ++ .checked_add(mapped_len) ++ .ok_or(Errno::Overflow) ++ ); ++ let mapping = WasiSharedMemoryMapping { ++ start: requested_addr as u64, ++ len: mapped_len as u64, ++ file: file.clone(), ++ file_offset: offset, ++ futexs, ++ }; ++ wasi_try_ok!(state.install_shared_memory_mapping( ++ mapping, ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || { ++ // Validation happens before this closure. Rejected fixed ++ // requests therefore cannot grow or otherwise mutate memory. ++ memory ++ .grow_at_least(&mut ctx.as_store_mut(), mapped_end as u64) ++ .map_err(|err| { ++ tracing::warn!( ++ error = &err as &dyn std::error::Error, ++ mapped_end, ++ "failed to grow guest memory for fixed shared mapping" ++ ); ++ Errno::Nomem ++ })?; ++ // SAFETY: the process execution lease admits only this continuation for the ++ // image, every temporary MemoryView was dropped before entering the mapping ++ // registry lock, and the registry retains the backing file. WASIX truncation ++ // paths reject shrinking an inode while its shared registry is live; the sealed ++ // runtime additionally owns its HostFS mount for the mapping lifetime. ++ unsafe { ++ memory.remap_shared_file_fixed( ++ &mut ctx.as_store_mut(), ++ requested_addr, ++ mapped_len, ++ &file, ++ file_offset, ++ ) ++ } ++ .map_err(|err| { ++ tracing::warn!( ++ error = &err as &dyn std::error::Error, ++ "failed to remap guest memory to shared file backing" ++ ); ++ Errno::Inval ++ }) ++ } ++ )); ++ requested_addr ++ } else { ++ let delta_pages = wasi_try_ok!( ++ mapped_len ++ .checked_add(WASM_PAGE_SIZE - 1) ++ .ok_or(Errno::Overflow) ++ .and_then(|bytes| { ++ u32::try_from(bytes / WASM_PAGE_SIZE).map_err(|_| Errno::Overflow) ++ }) ++ ); ++ let selected = wasi_try_ok!(state.install_runtime_selected_shared_memory_mapping( ++ mapped_len as u64, ++ file.clone(), ++ offset, ++ futexs, ++ |placement| { ++ let (selected, selected_end) = match placement { ++ WasiSharedMemoryRuntimePlacement::Fresh { minimum_start } => { ++ // memory.grow is the reservation primitive: its returned old ++ // page count is unique with respect to concurrent guest growth. ++ let old_pages = ++ memory ++ .grow(&mut ctx.as_store_mut(), delta_pages) ++ .map_err(|err| { ++ tracing::warn!( ++ error = &err as &dyn std::error::Error, ++ delta_pages, ++ "failed to reserve guest pages for shared mapping" ++ ); ++ Errno::Nomem ++ })?; ++ let selected = u64::from(old_pages.0) ++ .checked_mul(WASM_PAGE_SIZE as u64) ++ .ok_or(Errno::Overflow)?; ++ let selected_end = selected ++ .checked_add(mapped_len as u64) ++ .ok_or(Errno::Overflow)?; ++ if selected < minimum_start { ++ tracing::error!( ++ selected, ++ selected_end, ++ minimum_start, ++ "WebAssembly memory end is below tracked shared mappings" ++ ); ++ return Err(Errno::Inval); ++ } ++ (selected, selected_end) ++ } ++ WasiSharedMemoryRuntimePlacement::Inherited { start } => { ++ let selected_end = start ++ .checked_add(mapped_len as u64) ++ .ok_or(Errno::Overflow)?; ++ memory ++ .grow_at_least(&mut ctx.as_store_mut(), selected_end) ++ .map_err(|err| { ++ tracing::warn!( ++ error = &err as &dyn std::error::Error, ++ selected_end, ++ "failed to cover inherited shared mapping reservation" ++ ); ++ Errno::Nomem ++ })?; ++ (start, selected_end) ++ } ++ }; ++ let selected_usize: usize = selected.try_into().map_err(|_| Errno::Overflow)?; ++ // SAFETY: memory.grow reserved this unique range for the currently leased ++ // continuation, no MemoryView is live, and the mapping registry retains the file ++ // and prevents guest-visible truncation while the mapping exists. ++ unsafe { ++ memory.remap_shared_file_fixed( ++ &mut ctx.as_store_mut(), ++ selected_usize, ++ mapped_len, ++ &file, ++ file_offset, ++ ) ++ } ++ .map_err(|err| { ++ tracing::warn!( ++ error = &err as &dyn std::error::Error, ++ selected, ++ selected_end, ++ "failed to install runtime-selected shared mapping" ++ ); ++ Errno::Inval ++ })?; ++ Ok(selected) ++ } ++ )); ++ wasi_try_ok!(selected.try_into().map_err(|_| Errno::Overflow)) ++ }; ++ ++ let memory_view = unsafe { ctx.data().memory_view(&ctx) }; ++ let addr_ret = wasi_try_ok!(to_offset::(mapped_addr)); ++ wasi_try_mem_ok!(ret_addr.write(&memory_view, addr_ret)); ++ Span::current().record("ret_addr", mapped_addr); ++ ++ Ok(Errno::Success) ++} ++ ++/// ### `mem_munmap()` ++/// ++/// Replaces a guest shared file mapping with private zero-filled memory and ++/// removes it from the fork replay registry. ++#[instrument( ++ level = "trace", ++ skip_all, ++ fields(%addr, %len), ++ ret ++)] ++pub fn mem_munmap( ++ mut ctx: FunctionEnvMut<'_, WasiEnv>, ++ addr: M::Offset, ++ len: M::Offset, ++) -> Result { ++ WasiEnv::do_pending_operations(&mut ctx)?; ++ ++ let addr = wasi_try_ok!(from_offset::(addr)); ++ let len = wasi_try_ok!(from_offset::(len)); ++ let page_size = host_page_size(); ++ let mapped_len = wasi_try_ok!(round_up_to_page(len, page_size)); ++ ++ if len == 0 { ++ return Ok(Errno::Inval); ++ } ++ ++ let memory = { ++ let env = ctx.data(); ++ env.inner().main_module_instance_handles().memory_clone() ++ }; ++ ++ let state = ctx.data().state.clone(); ++ wasi_try_ok!(state.remove_shared_memory_mapping( ++ addr as u64, ++ len as u64, ++ mapped_len as u64, ++ || { ++ // SAFETY: the process execution lease excludes another continuation, no MemoryView ++ // is live, and the mapping registry lock validates and owns the complete range until ++ // the fixed replacement succeeds. ++ unsafe { memory.remap_private_fixed(&mut ctx.as_store_mut(), addr, mapped_len) } ++ .map_err(|err| { ++ tracing::warn!( ++ error = &err as &dyn std::error::Error, ++ "failed to restore private guest memory over shared mapping" ++ ); ++ Errno::Inval ++ }) ++ } ++ )); ++ ++ Ok(Errno::Success) ++} ++ ++/// ### `mem_msync()` ++/// ++/// Synchronizes a shared guest mapping with its host file backing. ++#[instrument( ++ level = "trace", ++ skip_all, ++ fields(%addr, %len, %flags), ++ ret ++)] ++pub fn mem_msync( ++ mut ctx: FunctionEnvMut<'_, WasiEnv>, ++ addr: M::Offset, ++ len: M::Offset, ++ flags: u32, ++) -> Result { ++ WasiEnv::do_pending_operations(&mut ctx)?; ++ ++ if (flags & !(MS_ASYNC | MS_INVALIDATE | MS_SYNC)) != 0 ++ || (flags & MS_ASYNC) != 0 && (flags & MS_SYNC) != 0 ++ { ++ return Ok(Errno::Inval); ++ } ++ ++ let addr = wasi_try_ok!(from_offset::(addr)); ++ let len = wasi_try_ok!(from_offset::(len)); ++ let page_size = host_page_size(); ++ let mapped_len = wasi_try_ok!(round_up_to_page(len, page_size)); ++ ++ if len == 0 { ++ return Ok(Errno::Success); ++ } ++ ++ let host_flags = host_msync_flags(flags); ++ let memory = { ++ let env = ctx.data(); ++ env.inner().main_module_instance_handles().memory_clone() ++ }; ++ let state = ctx.data().state.clone(); ++ wasi_try_ok!(state.with_active_shared_memory_mapping( ++ addr as u64, ++ len as u64, ++ mapped_len as u64, ++ || { ++ memory ++ .msync(&ctx.as_store_ref(), addr, mapped_len, host_flags) ++ .map_err(|err| { ++ tracing::warn!( ++ error = &err as &dyn std::error::Error, ++ "failed to synchronize shared guest memory mapping" ++ ); ++ Errno::Inval ++ }) ++ }, ++ )); ++ ++ Ok(Errno::Success) ++} ++ ++fn round_up_to_page(len: usize, page_size: usize) -> Result { ++ debug_assert!(page_size.is_power_of_two()); ++ len.checked_add(page_size - 1) ++ .map(|value| value & !(page_size - 1)) ++ .ok_or(Errno::Overflow) ++} ++ ++fn host_page_size() -> usize { ++ #[cfg(not(target_os = "windows"))] ++ { ++ let page_size = unsafe { libc::sysconf(libc::_SC_PAGESIZE) }; ++ if page_size > 0 { ++ return page_size as usize; ++ } ++ } ++ ++ 65536 ++} ++ ++fn host_msync_flags(flags: u32) -> i32 { ++ #[cfg(not(target_os = "windows"))] ++ { ++ let mut host_flags = 0; ++ if (flags & MS_ASYNC) != 0 { ++ host_flags |= libc::MS_ASYNC; ++ } ++ if (flags & MS_INVALIDATE) != 0 { ++ host_flags |= libc::MS_INVALIDATE; ++ } ++ if (flags & MS_SYNC) != 0 { ++ host_flags |= libc::MS_SYNC; ++ } ++ host_flags ++ } ++ ++ #[cfg(target_os = "windows")] ++ { ++ let _ = flags; ++ 0 ++ } ++} ++ ++#[cfg(feature = "host-fs")] ++fn mappable_file( ++ ctx: &FunctionEnvMut<'_, WasiEnv>, ++ fd: WasiFd, ++ prot: u32, ++) -> Result { ++ let fd_entry = ctx.data().state().fs.get_fd(fd)?; ++ ++ if !fd_entry.inner.rights.contains(Rights::FD_READ) { ++ return Err(Errno::Access); ++ } ++ if (prot & PROT_WRITE) != 0 && !fd_entry.inner.rights.contains(Rights::FD_WRITE) { ++ return Err(Errno::Access); ++ } ++ ++ let guard = fd_entry.inode.read(); ++ let Kind::File { ++ handle: Some(handle), ++ .. ++ } = guard.deref() ++ else { ++ return Err(Errno::Badf); ++ }; ++ ++ let handle = handle.read().map_err(|_| Errno::Fault)?; ++ let host_file = handle ++ .upcast_any_ref() ++ .downcast_ref::() ++ .ok_or(Errno::Notsup)?; ++ ++ host_file.try_clone_std_file().map_err(map_io_err) ++} ++ ++#[cfg(not(feature = "host-fs"))] ++fn mappable_file( ++ _ctx: &FunctionEnvMut<'_, WasiEnv>, ++ _fd: WasiFd, ++ _prot: u32, ++) -> Result { ++ Err(Errno::Notsup) ++} +diff --git a/lib/wasix/src/syscalls/wasix/path_open2.rs b/lib/wasix/src/syscalls/wasix/path_open2.rs +index 83db5f2..eaa083d 100644 +--- a/lib/wasix/src/syscalls/wasix/path_open2.rs ++++ b/lib/wasix/src/syscalls/wasix/path_open2.rs +@@ -46,8 +46,7 @@ pub fn path_open2( + Span::current().record("follow_symlinks", true); + } + let env = ctx.data(); +- let (memory, mut state, mut inodes) = +- unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; ++ let memory = unsafe { env.memory_view(&ctx) }; + /* TODO: find actual upper bound on name size (also this is a path, not a name :think-fish:) */ + let path_len64: u64 = path_len.into(); + if path_len64 > 1024u64 * 1024u64 { +@@ -102,8 +101,7 @@ pub fn path_open2( + } + + let env = ctx.data(); +- let (memory, mut state, mut inodes) = +- unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; ++ let memory = unsafe { env.memory_view(&ctx) }; + + Span::current().record("ret_fd", out_fd); + +@@ -149,19 +147,20 @@ pub(crate) fn path_open_internal( + } + + let mut open_flags = 0; +- // TODO: traverse rights of dirs properly +- // COMMENTED OUT: WASI isn't giving appropriate rights here when opening +- // TODO: look into this; file a bug report if this is a bug +- // +- // Maximum rights: should be the working dir rights +- // Minimum rights: whatever rights are provided +- let adjusted_rights = /*fs_rights_base &*/ working_dir_rights_inheriting; ++ let sync_on_write = fs_flags.contains(Fdflags::SYNC); ++ let data_sync_on_write = fs_flags.contains(Fdflags::DSYNC); ++ // The directory's inheriting rights are an upper bound, not an access ++ // request. Promoting a read-only open to FD_WRITE makes immutable host ++ // mounts unusable and grants the resulting descriptor rights the guest ++ // never requested. ++ let adjusted_rights = constrain_requested_rights(fs_rights_base, working_dir_rights_inheriting); ++ let adjusted_inheriting = ++ constrain_requested_rights(fs_rights_inheriting, working_dir_rights_inheriting); + let mut open_options = state.fs_new_open_options(); ++ let write_permission = adjusted_rights.contains(Rights::FD_WRITE); + + let target_rights = match maybe_inode { + Ok(_) => { +- let write_permission = adjusted_rights.contains(Rights::FD_WRITE); +- + // append, truncate, and create all require the permission to write + let (append_permission, truncate_permission, create_permission) = if write_permission { + ( +@@ -174,21 +173,25 @@ pub(crate) fn path_open_internal( + }; + + virtual_fs::OpenOptionsConfig { +- read: fs_rights_base.contains(Rights::FD_READ), ++ read: adjusted_rights.contains(Rights::FD_READ), + write: write_permission, + create_new: create_permission && o_flags.contains(Oflags::EXCL), + create: create_permission, + append: append_permission, + truncate: truncate_permission, ++ sync: sync_on_write, ++ data_sync: data_sync_on_write, + } + } + Err(_) => virtual_fs::OpenOptionsConfig { + append: fs_flags.contains(Fdflags::APPEND), +- write: fs_rights_base.contains(Rights::FD_WRITE), +- read: fs_rights_base.contains(Rights::FD_READ), ++ write: write_permission, ++ read: adjusted_rights.contains(Rights::FD_READ), + create_new: o_flags.contains(Oflags::CREATE) && o_flags.contains(Oflags::EXCL), + create: o_flags.contains(Oflags::CREATE), + truncate: o_flags.contains(Oflags::TRUNC), ++ sync: sync_on_write, ++ data_sync: data_sync_on_write, + }, + }; + +@@ -201,133 +204,249 @@ pub(crate) fn path_open_internal( + create: true, + append: true, + truncate: true, ++ sync: true, ++ data_sync: true, + }; + + let minimum_rights = target_rights.minimum_rights(&parent_rights); + + open_options.options(minimum_rights.clone()); + ++ let handle_satisfies_open = ++ |handle: &(dyn virtual_fs::VirtualFile + Send + Sync), ++ requested_config: &virtual_fs::OpenOptionsConfig| { ++ let read_ok = !requested_config.read || handle.open_read().unwrap_or(true); ++ let write_requested = requested_config.write || requested_config.append; ++ let write_ok = !write_requested || handle.open_write().unwrap_or(false); ++ read_ok && write_ok ++ }; ++ + let orig_path = path; ++ let existing_file_requested_config = virtual_fs::OpenOptionsConfig { ++ append: false, ++ ..minimum_rights.clone() ++ }; ++ ++ let record_file_open_flags = |open_flags: &mut u16| { ++ if minimum_rights.read { ++ *open_flags |= Fd::READ; ++ } ++ if minimum_rights.write { ++ *open_flags |= Fd::WRITE; ++ } ++ if minimum_rights.create { ++ *open_flags |= Fd::CREATE; ++ } ++ if minimum_rights.truncate { ++ *open_flags |= Fd::TRUNCATE; ++ } ++ }; + ++ let mut handle_reservation = None; + let inode = if let Ok(inode) = maybe_inode { + // Happy path, we found the file we're trying to open + let processing_inode = inode.clone(); +- let mut guard = processing_inode.write(); +- +- let deref_mut = guard.deref_mut(); + + if o_flags.contains(Oflags::EXCL) && o_flags.contains(Oflags::CREATE) { + return Ok(Err(Errno::Exist)); + } + +- match deref_mut { +- Kind::File { +- handle, path, fd, .. +- } => { +- if let Some(special_fd) = fd { +- // short circuit if we're dealing with a special file +- assert!(handle.is_some()); +- return Ok(Ok(*special_fd)); +- } +- if o_flags.contains(Oflags::DIRECTORY) || orig_path.ends_with('/') { +- return Ok(Err(Errno::Notdir)); +- } +- +- let open_options = open_options +- .write(minimum_rights.write) +- .create(minimum_rights.create) +- .append(false) +- .truncate(minimum_rights.truncate); +- +- if minimum_rights.read { +- open_flags |= Fd::READ; +- } +- if minimum_rights.write { +- open_flags |= Fd::WRITE; +- } +- if minimum_rights.create { +- open_flags |= Fd::CREATE; +- } +- if minimum_rights.truncate { +- open_flags |= Fd::TRUNCATE; +- } +- // Keep a stable shared handle per inode whenever possible, but reopen it +- // when this open requires stronger rights than the existing handle may have. +- let requires_stronger_handle = +- minimum_rights.write || minimum_rights.truncate || minimum_rights.create; +- if handle.is_none() { +- *handle = Some(Arc::new(std::sync::RwLock::new(wasi_try_ok_ok!( +- open_options.open(&path).map_err(fs_error_into_wasi_err) +- )))); +- } else if requires_stronger_handle { +- let mut file = handle.as_ref().unwrap().write().unwrap(); +- *file = +- wasi_try_ok_ok!(open_options.open(&path).map_err(fs_error_into_wasi_err)); +- } ++ // If the cached regular-file handle already satisfies this open, ++ // creating a new WASIX fd is the only side effect. Keep that path on a ++ // shared inode lock; create/truncate/reopen/symlink handling still ++ // falls through to the write-locked path below. ++ let cached_inode = { ++ let guard = processing_inode.read(); ++ match guard.deref() { ++ Kind::File { ++ handle: Some(handle), ++ fd, ++ .. ++ } if !minimum_rights.truncate => { ++ if let Some(special_fd) = fd { ++ return Ok(Ok(*special_fd)); ++ } ++ if o_flags.contains(Oflags::DIRECTORY) || orig_path.ends_with('/') { ++ return Ok(Err(Errno::Notdir)); ++ } + +- if let Some(handle) = handle { + let handle = handle.read().unwrap(); + if let Some(fd) = handle.get_special_fd() { +- // We clone the file descriptor so that when its closed +- // nothing bad happens + let dup_fd = wasi_try_ok_ok!(state.fs.clone_fd(fd)); + trace!( + %dup_fd + ); +- +- // some special files will return a constant FD rather than +- // actually open the file (/dev/stdin, /dev/stdout, /dev/stderr) + return Ok(Ok(dup_fd)); + } ++ ++ handle_satisfies_open(handle.as_ref(), &existing_file_requested_config).then( ++ || { ++ record_file_open_flags(&mut open_flags); ++ handle_reservation = Some(processing_inode.reserve_handle()); ++ processing_inode.clone() ++ }, ++ ) + } ++ _ => None, + } +- Kind::Buffer { .. } => unimplemented!("wasi::path_open for Buffer type files"), +- Kind::Root { .. } => { +- if !o_flags.contains(Oflags::DIRECTORY) { +- return Ok(Err(Errno::Isdir)); ++ }; ++ if let Some(inode) = cached_inode { ++ inode ++ } else { ++ let mut guard = processing_inode.write(); ++ ++ let deref_mut = guard.deref_mut(); ++ ++ match deref_mut { ++ Kind::File { ++ handle, path, fd, .. ++ } => { ++ if let Some(special_fd) = fd { ++ // short circuit if we're dealing with a special file ++ assert!(handle.is_some()); ++ return Ok(Ok(*special_fd)); ++ } ++ if o_flags.contains(Oflags::DIRECTORY) || orig_path.ends_with('/') { ++ return Ok(Err(Errno::Notdir)); ++ } ++ ++ let requested_config = open_options ++ .write(minimum_rights.write) ++ .create(minimum_rights.create) ++ .append(false) ++ .truncate(minimum_rights.truncate) ++ .get_config(); ++ ++ record_file_open_flags(&mut open_flags); ++ // Keep a stable shared handle per inode whenever possible, but reopen when ++ // the existing handle cannot satisfy the requested access. Truncate opens the ++ // exact target without O_TRUNC, locks that opened inode's shared-mapping ++ // registry, and only then shrinks it before publication. ++ let should_reopen_handle = if handle.is_none() || minimum_rights.truncate { ++ true ++ } else { ++ let file = handle.as_ref().unwrap().read().unwrap(); ++ !handle_satisfies_open(file.as_ref(), &existing_file_requested_config) ++ }; ++ if should_reopen_handle { ++ let mut requested_open_options = state.fs_new_open_options(); ++ let mut shared_open_options = state.fs_new_open_options(); ++ let manual_truncate = requested_config.truncate; ++ let open_config = virtual_fs::OpenOptionsConfig { ++ truncate: false, ++ ..requested_config.clone() ++ }; ++ let shared_config = virtual_fs::OpenOptionsConfig { ++ read: true, ++ write: true, ++ ..open_config.clone() ++ }; ++ let new_handle = shared_open_options ++ .options(shared_config) ++ .open(path.as_path()) ++ .or_else(|_| { ++ requested_open_options ++ .options(open_config) ++ .open(path.as_path()) ++ }) ++ .map_err(fs_error_into_wasi_err); ++ let mut new_handle = wasi_try_ok_ok!(new_handle); ++ if manual_truncate { ++ #[cfg(feature = "host-fs")] ++ let truncation_file = new_handle ++ .upcast_any_ref() ++ .downcast_ref::() ++ .map(|file| file.try_clone_std_file()) ++ .transpose() ++ .map_err(crate::utils::map_io_err); ++ #[cfg(feature = "host-fs")] ++ let truncation_file = wasi_try_ok_ok!(truncation_file); ++ #[cfg(feature = "host-fs")] ++ let _shared_mapping_guard = truncation_file ++ .as_ref() ++ .map(|file| state.guard_shared_mapping_file_shrink(file, 0)) ++ .transpose(); ++ #[cfg(feature = "host-fs")] ++ let _shared_mapping_guard = ++ wasi_try_ok_ok!(_shared_mapping_guard).flatten(); ++ ++ wasi_try_ok_ok!(new_handle.set_len(0).map_err(fs_error_into_wasi_err)); ++ } ++ match handle { ++ Some(handle) => { ++ *handle.write().unwrap() = new_handle; ++ } ++ None => { ++ *handle = Some(Arc::new(std::sync::RwLock::new(new_handle))); ++ } ++ } ++ } ++ ++ if let Some(handle) = handle { ++ let handle = handle.read().unwrap(); ++ if let Some(fd) = handle.get_special_fd() { ++ // We clone the file descriptor so that when its closed ++ // nothing bad happens ++ let dup_fd = wasi_try_ok_ok!(state.fs.clone_fd(fd)); ++ trace!( ++ %dup_fd ++ ); ++ ++ // some special files will return a constant FD rather than ++ // actually open the file (/dev/stdin, /dev/stdout, /dev/stderr) ++ return Ok(Ok(dup_fd)); ++ } ++ } ++ handle_reservation = Some(processing_inode.reserve_handle()); + } +- } +- Kind::Dir { .. } => { +- if fs_rights_base.contains(Rights::FD_WRITE) { +- return Ok(Err(Errno::Isdir)); ++ Kind::Buffer { .. } => unimplemented!("wasi::path_open for Buffer type files"), ++ Kind::Root { .. } => { ++ if !o_flags.contains(Oflags::DIRECTORY) { ++ return Ok(Err(Errno::Isdir)); ++ } ++ } ++ Kind::Dir { .. } => { ++ if fs_rights_base.contains(Rights::FD_WRITE) { ++ return Ok(Err(Errno::Isdir)); ++ } ++ } ++ Kind::Socket { .. } ++ | Kind::PipeTx { .. } ++ | Kind::PipeRx { .. } ++ | Kind::DuplexPipe { .. } ++ | Kind::EventNotifications { .. } ++ | Kind::Epoll { .. } => {} ++ Kind::Symlink { ++ base_po_dir, ++ path_to_symlink, ++ relative_path, ++ } => { ++ // Resolve the symlink via the existing path traversal logic and restart ++ // path_open with lookup-follow semantics for this resolved path. ++ let (resolved_base_fd, resolved_path) = if relative_path.is_absolute() { ++ (VIRTUAL_ROOT_FD, relative_path.clone()) ++ } else { ++ let mut resolved_path = path_to_symlink.clone(); ++ resolved_path.pop(); ++ resolved_path.push(relative_path); ++ (*base_po_dir, resolved_path) ++ }; ++ return path_open_internal( ++ env, ++ resolved_base_fd, ++ __WASI_LOOKUP_SYMLINK_FOLLOW, ++ &resolved_path.to_string_lossy(), ++ o_flags, ++ fs_rights_base, ++ fs_rights_inheriting, ++ fs_flags, ++ fd_flags, ++ with_fd, ++ ); + } + } +- Kind::Socket { .. } +- | Kind::PipeTx { .. } +- | Kind::PipeRx { .. } +- | Kind::DuplexPipe { .. } +- | Kind::EventNotifications { .. } +- | Kind::Epoll { .. } => {} +- Kind::Symlink { +- base_po_dir, +- path_to_symlink, +- relative_path, +- } => { +- // Resolve the symlink via the existing path traversal logic and restart +- // path_open with lookup-follow semantics for this resolved path. +- let (resolved_base_fd, resolved_path) = if relative_path.is_absolute() { +- (VIRTUAL_ROOT_FD, relative_path.clone()) +- } else { +- let mut resolved_path = path_to_symlink.clone(); +- resolved_path.pop(); +- resolved_path.push(relative_path); +- (*base_po_dir, resolved_path) +- }; +- return path_open_internal( +- env, +- resolved_base_fd, +- __WASI_LOOKUP_SYMLINK_FOLLOW, +- &resolved_path.to_string_lossy(), +- o_flags, +- fs_rights_base, +- fs_rights_inheriting, +- fs_flags, +- fd_flags, +- with_fd, +- ); +- } ++ inode + } +- inode + } else { + // less-happy path, we have to try to create the file + if o_flags.contains(Oflags::CREATE) { +@@ -417,6 +536,8 @@ pub(crate) fn path_open_internal( + ) + }; + ++ handle_reservation = Some(new_inode.reserve_handle()); ++ + { + let mut guard = parent_inode.write(); + if let Kind::Dir { entries, .. } = guard.deref_mut() { +@@ -432,12 +553,12 @@ pub(crate) fn path_open_internal( + + // TODO: check and reduce these + // TODO: ensure a mutable fd to root can never be opened +- let out_fd = wasi_try_ok_ok!(if let Some(fd) = with_fd { ++ let out_fd = if let Some(fd) = with_fd { + state + .fs + .with_fd( + adjusted_rights, +- fs_rights_inheriting, ++ adjusted_inheriting, + fs_flags, + fd_flags, + open_flags, +@@ -448,13 +569,53 @@ pub(crate) fn path_open_internal( + } else { + state.fs.create_fd( + adjusted_rights, +- fs_rights_inheriting, ++ adjusted_inheriting, + fs_flags, + fd_flags, + open_flags, + inode, + ) +- }); ++ }; ++ drop(handle_reservation); ++ let out_fd = wasi_try_ok_ok!(out_fd); + + Ok(Ok(out_fd)) + } ++ ++fn constrain_requested_rights(requested: Rights, parent_inheriting: Rights) -> Rights { ++ requested & parent_inheriting ++} ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn read_only_open_never_inherits_parent_write_access() { ++ let parent = Rights::FD_READ | Rights::FD_WRITE | Rights::FD_SEEK; ++ let adjusted = constrain_requested_rights(Rights::FD_READ | Rights::FD_SEEK, parent); ++ ++ assert!(adjusted.contains(Rights::FD_READ)); ++ assert!(adjusted.contains(Rights::FD_SEEK)); ++ assert!(!adjusted.contains(Rights::FD_WRITE)); ++ } ++ ++ #[test] ++ fn parent_capabilities_remain_an_upper_bound() { ++ let requested = Rights::FD_READ | Rights::FD_WRITE; ++ let adjusted = constrain_requested_rights(requested, Rights::FD_READ); ++ ++ assert_eq!(adjusted, Rights::FD_READ); ++ } ++ ++ #[test] ++ fn descriptor_inheriting_rights_are_capped_independently() { ++ let parent = Rights::FD_READ | Rights::FD_SEEK; ++ let adjusted = constrain_requested_rights( ++ Rights::FD_READ | Rights::FD_WRITE | Rights::FD_SEEK, ++ parent, ++ ); ++ ++ assert_eq!(adjusted, parent); ++ } ++} diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0005-wasix-process-thread-and-join-lifetimes.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0005-wasix-process-thread-and-join-lifetimes.patch new file mode 100644 index 000000000..4a36b8998 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0005-wasix-process-thread-and-join-lifetimes.patch @@ -0,0 +1,6747 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 5/9] wasix: own process, thread and join lifetimes + +Keep process creation/exec, child wait/reap, signals, futex wakeups and task +budgets with their control-plane ownership changes. PostgreSQL Postmaster +requires concurrent EXEC_BACKEND children and correct nonblocking wait/SIGCHLD +behavior. This is the concurrency review boundary; do not infer correctness +from single-program benchmarks or substitute scalar synchronization. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/lib/wasix/src/bin_factory/exec.rs b/lib/wasix/src/bin_factory/exec.rs +index 4e60ee0..637be9e 100644 +--- a/lib/wasix/src/bin_factory/exec.rs ++++ b/lib/wasix/src/bin_factory/exec.rs +@@ -10,9 +10,11 @@ use crate::{ + ModuleInput, TaintReason, + module_cache::HashedModuleData, + task_manager::{ +- TaskWasm, TaskWasmRecycle, TaskWasmRecycleProperties, TaskWasmRunProperties, ++ TaskWasm, TaskWasmAcceptedExecutionGuard, TaskWasmRecycle, TaskWasmRecycleProperties, ++ TaskWasmRunProperties, + }, + }, ++ state::PreinitializedMemoryImageMode, + state::context_switching::ContextSwitchingEnvironment, + syscalls::rewind_ext, + }; +@@ -128,6 +130,29 @@ pub fn spawn_exec_module( + env: WasiEnv, + runtime: &Arc, + ) -> Result { ++ spawn_exec_module_with_preinitialized_memory_image(module, env, runtime, None) ++} ++ ++pub fn spawn_exec_module_with_preinitialized_memory_image( ++ module: Module, ++ env: WasiEnv, ++ runtime: &Arc, ++ preinitialized_memory_image: Option, ++) -> Result { ++ // A fresh executable image must not see inherited mappings as active in ++ // its new linear memory. Convert them transactionally into address/backing ++ // reservations before Wasmer can instantiate (and run a module start ++ // function). If task admission fails, the guard restores the old image's ++ // registry so exec failure remains non-destructive. ++ let shared_memory_exec = env ++ .state ++ .prepare_shared_memory_for_exec() ++ .map_err(|errno| { ++ SpawnError::Other(Box::new(std::io::Error::other(format!( ++ "failed to prepare shared-memory exec reservations: {errno}" ++ )))) ++ })?; ++ + // Create a new task manager + let tasks = runtime.task_manager(); + +@@ -139,20 +164,23 @@ pub fn spawn_exec_module( + // Create a thread that will run this process + let tasks_outer = tasks.clone(); + +- tasks_outer +- .task_wasm( +- TaskWasm::new(Box::new(run_exec), env, module, true, true).with_pre_run(Box::new( +- |ctx, store| { +- Box::pin(async move { +- ctx.data(store).state.fs.close_cloexec_fds().await; +- }) +- }, +- )), +- ) +- .map_err(|err| { +- error!("wasi[{}]::failed to launch module - {}", pid, err); +- SpawnError::Other(Box::new(err)) +- })? ++ let accepted_run_exec = move |props| { ++ shared_memory_exec.commit(); ++ run_exec(props); ++ }; ++ let mut task = TaskWasm::new(Box::new(accepted_run_exec), env, module, true, true) ++ .with_pre_run(Box::new(|ctx, store| { ++ Box::pin(async move { ++ ctx.data(store).state.fs.close_cloexec_fds().await; ++ }) ++ })); ++ if let Some(image) = preinitialized_memory_image { ++ task = task.with_preinitialized_memory_image(image); ++ } ++ tasks_outer.task_wasm(task).map_err(|err| { ++ error!("wasi[{}]::failed to launch module - {}", pid, err); ++ SpawnError::Other(Box::new(err)) ++ })?; + }; + + Ok(join_handle) +@@ -163,9 +191,15 @@ pub fn spawn_exec_module( + /// otherwise it will cause a panic + unsafe fn run_recycle( + callback: Option>, ++ execution_guard: Option, + ctx: WasiFunctionEnv, + mut store: Store, + ) { ++ // Terminal status must be visible before this point. Releasing the lease ++ // makes join/reinit observe real guest quiescence before the recyclable ++ // environment is published to another request. ++ ctx.data_mut(&mut store).clear_task_wasm_execution(); ++ drop(execution_guard); + if let Some(callback) = callback { + let env = ctx.data_mut(&mut store); + let memory = unsafe { env.memory() }.clone(); +@@ -186,6 +220,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { + // Create the WasiFunctionEnv + let thread = WasiThreadRunGuard::new(ctx.data(&store).thread.clone()); + let recycle = props.recycle; ++ let execution_guard = props.execution_guard; + + // Perform the initialization + // If this module exports an _initialize function, run that first. +@@ -206,7 +241,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { + thread.thread.set_status_finished(Err(err.into())); + ctx.data(&store) + .blocking_on_exit(Some(Errno::Noexec.into())); +- unsafe { run_recycle(recycle, ctx, store) }; ++ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; + return; + } + } +@@ -221,7 +256,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { + thread.thread.set_status_finished(Err(err)); + ctx.data(&store) + .blocking_on_exit(Some(Errno::Noexec.into())); +- unsafe { run_recycle(recycle, ctx, store) }; ++ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; + return; + } + }; +@@ -231,7 +266,7 @@ pub fn run_exec(props: TaskWasmRunProperties) { + // TODO: rewrite to use crate::run_wasi_func + + // Call the module +- call_module(ctx, store, thread, rewind_state, recycle); ++ call_module(ctx, store, thread, rewind_state, recycle, execution_guard); + } + + fn get_start(ctx: &WasiFunctionEnv, store: &Store) -> Option { +@@ -252,6 +287,7 @@ fn call_module( + handle: WasiThreadRunGuard, + rewind_state: Option<(RewindState, RewindResultType)>, + recycle: Option>, ++ execution_guard: Option, + ) { + let env = ctx.data(&store); + let pid = env.pid(); +@@ -272,7 +308,15 @@ fn call_module( + ); + if res != Errno::Success { + ctx.data().blocking_on_exit(Some(res.into())); +- unsafe { run_recycle(recycle, WasiFunctionEnv { env: ctx.as_ref() }, store) }; ++ handle.thread.set_status_finished(Ok(res.into())); ++ unsafe { ++ run_recycle( ++ recycle, ++ execution_guard, ++ WasiFunctionEnv { env: ctx.as_ref() }, ++ store, ++ ) ++ }; + return; + } + } else { +@@ -285,7 +329,15 @@ fn call_module( + ); + if res != Errno::Success { + ctx.data().blocking_on_exit(Some(res.into())); +- unsafe { run_recycle(recycle, WasiFunctionEnv { env: ctx.as_ref() }, store) }; ++ handle.thread.set_status_finished(Ok(res.into())); ++ unsafe { ++ run_recycle( ++ recycle, ++ execution_guard, ++ WasiFunctionEnv { env: ctx.as_ref() }, ++ store, ++ ) ++ }; + return; + } + }; +@@ -297,7 +349,8 @@ fn call_module( + debug!("wasi[{}]::exec-failed: missing _start function", pid); + ctx.data(&store) + .blocking_on_exit(Some(Errno::Noexec.into())); +- unsafe { run_recycle(recycle, ctx, store) }; ++ handle.thread.set_status_finished(Ok(Errno::Noexec.into())); ++ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; + return; + }; + +@@ -335,8 +388,9 @@ fn call_module( + Ok(WasiError::DeepSleep(deep)) => { + // Create the callback that will be invoked when the thread respawns after a deep sleep + let rewind = deep.rewind; ++ let terminal_thread = handle.thread.clone(); + let respawn = { +- move |ctx, store, rewind_result| { ++ move |ctx, store, rewind_result, execution_guard| { + // Call the thread + call_module( + ctx, +@@ -344,15 +398,26 @@ fn call_module( + handle, + Some((rewind, RewindResultType::RewindWithResult(rewind_result))), + recycle, ++ execution_guard, + ); + } + }; + + // Spawns the WASM process after a trigger + if let Err(err) = unsafe { +- tasks.resume_wasm_after_poller(Box::new(respawn), ctx, store, deep.trigger) ++ tasks.resume_wasm_after_poller( ++ Box::new(respawn), ++ ctx, ++ store, ++ deep.trigger, ++ execution_guard, ++ ) + } { + debug!("failed to go into deep sleep - {}", err); ++ terminal_thread.set_status_finished(Err(RuntimeError::new(format!( ++ "failed to resume process after deep sleep: {err}" ++ )) ++ .into())); + } + return; + } +@@ -366,6 +431,7 @@ fn call_module( + runtime.on_taint(TaintReason::DlSymbolResolutionFailed(symbol.clone())); + Err(WasiError::DlSymbolResolutionFailed(symbol).into()) + } ++ Ok(WasiError::StoreSnapshot(err)) => Err(WasiError::StoreSnapshot(err).into()), + Err(err) => { + runtime.on_taint(TaintReason::RuntimeError(err.clone())); + Err(WasiRuntimeError::from(err)) +@@ -391,10 +457,9 @@ fn call_module( + + // Cleanup the environment + ctx.data(&store).blocking_on_exit(Some(code)); +- unsafe { run_recycle(recycle, ctx, store) }; +- + debug!("wasi[{pid}]::main() has exited with {code}"); + handle.thread.set_status_finished(ret.map(|a| a.into())); ++ unsafe { run_recycle(recycle, execution_guard, ctx, store) }; + } + + #[allow(clippy::type_complexity)] +@@ -417,6 +482,7 @@ fn resume_vfork( + Some(WasiError::ThreadExit) => (None, wasmer_wasix_types::wasi::ExitCode::from(0u16)), + Some(WasiError::UnknownWasiVersion) => (None, Errno::Noexec.into()), + Some(WasiError::DlSymbolResolutionFailed(_)) => (None, Errno::Nolink.into()), ++ Some(WasiError::StoreSnapshot(_)) => (None, Errno::Noexec.into()), + None => ( + Some(WasiRuntimeError::from(err.clone())), + Errno::Unknown.into(), +@@ -447,10 +513,16 @@ fn resume_vfork( + vfork.env.swap_inner(ctx.data_mut(&mut store)); + std::mem::swap(vfork.env.as_mut(), ctx.data_mut(&mut store)); + let mut child_env = *vfork.env; ++ // A well-defined parent return is normally still owned by the ++ // original TaskWasm. Preserve one supplemental guard only if a ++ // continuation transferred that ownership while the child ran. ++ ctx.data_mut(&mut store) ++ .restore_parent_execution_guard(vfork.parent_execution); + child_env.owned_handles.push(vfork.handle); + +- // Terminate the child process +- child_env.process.terminate(code); ++ // Terminal status precedes release of the vfork child's execution ++ // lease; the restored parent guard remains with the active env. ++ vfork.child_execution.finish(Ok(code)); + + // If the vfork contained a context-switching environment, exit now + if ctx.data(&store).context_switching_environment.is_some() { +@@ -522,3 +594,85 @@ fn resume_vfork( + (store, Ok(None)) + } + } ++ ++#[cfg(all(test, feature = "sys-thread"))] ++mod lifecycle_tests { ++ use std::{ ++ sync::{Arc, mpsc}, ++ time::Duration, ++ }; ++ ++ use super::*; ++ use crate::{ ++ PluggableRuntime, ++ runtime::task_manager::tokio::TokioTaskManager, ++ runtime::task_manager::{TaskWasm, TaskWasmRunProperties}, ++ }; ++ use wasmer::Engine; ++ ++ #[test] ++ fn run_exec_panic_terminalizes_before_releasing_accepted_guard() { ++ let tokio_runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = tokio_runtime.enter(); ++ let engine = Engine::default(); ++ let mut runtime = PluggableRuntime::new(Arc::new(TokioTaskManager::new( ++ tokio_runtime.handle().clone(), ++ ))); ++ runtime.set_engine(engine.clone()); ++ let env = WasiEnv::builder("run-exec-panic-test") ++ .runtime(Arc::new(runtime)) ++ .build() ++ .unwrap(); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let module = Module::new( ++ &engine, ++ br#"(module (memory (export "memory") 1) (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let task = TaskWasm::new(Box::new(|_| {}), env.clone(), module, false, false); ++ let TaskWasm { ++ env: mut task_env, ++ execution_lease, ++ .. ++ } = task; ++ let accepted = execution_lease ++ .expect("test TaskWasm must acquire its pending execution guard") ++ .accept_callback(&mut task_env); ++ ++ // A raw, uninstantiated WasiFunctionEnv makes concrete run_exec panic ++ // while looking up its module handles, after both the run guard and ++ // accepted execution guard have been installed. ++ let mut store = task_env.runtime().new_store(); ++ let ctx = WasiFunctionEnv::new(&mut store, task_env); ++ let observer_process = process.clone(); ++ let observer_thread = thread.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = std::thread::spawn(move || { ++ observer_process.wait_for_execution_quiescence_blocking(); ++ let _ = observed_tx.send(observer_thread.try_join()); ++ }); ++ ++ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ run_exec(TaskWasmRunProperties { ++ ctx, ++ store, ++ trigger_result: None, ++ recycle: None, ++ execution_guard: Some(accepted), ++ }); ++ })); ++ ++ assert!(panic.is_err()); ++ let status = observed_rx ++ .recv_timeout(Duration::from_secs(2)) ++ .unwrap() ++ .expect("run_exec panic released quiescence before terminal status"); ++ assert!(status.is_err()); ++ assert_eq!(process.lock().execution_leases, 0); ++ observer.join().unwrap(); ++ } ++} +diff --git a/lib/wasix/src/bin_factory/mod.rs b/lib/wasix/src/bin_factory/mod.rs +index a290e6e..8de579c 100644 +--- a/lib/wasix/src/bin_factory/mod.rs ++++ b/lib/wasix/src/bin_factory/mod.rs +@@ -13,6 +13,7 @@ use shared_buffer::OwnedBuffer; + use virtual_fs::{AsyncReadExt, FileSystem}; + use wasmer::FunctionEnvMut; + use wasmer_package::utils::from_bytes; ++use wasmer_types::ModuleHash; + + mod binary_package; + mod exec; +@@ -21,7 +22,7 @@ pub use self::{ + binary_package::*, + exec::{ + import_package_mounts, package_command_by_name, run_exec, spawn_exec, spawn_exec_module, +- spawn_exec_wasm, spawn_load_module, ++ spawn_exec_module_with_preinitialized_memory_image, spawn_exec_wasm, spawn_load_module, + }, + }; + use crate::{ +@@ -31,6 +32,7 @@ use crate::{ + task::TaskJoinHandle, + }, + runtime::module_cache::HashedModuleData, ++ state::{PreinitializedMemoryImageHandle, PreinitializedMemoryImageMode}, + }; + + #[derive(Debug, Clone)] +@@ -38,14 +40,61 @@ pub struct BinFactory { + pub(crate) commands: Commands, + runtime: Arc, + pub(crate) local: Arc>>>>, ++ sealed_modules: Arc>, ++} ++ ++#[derive(Debug, Clone)] ++pub struct SealedExecutable { ++ module_hash: ModuleHash, ++ preinitialized_memory_image: Option, ++} ++ ++enum SealedAliasLookup<'a> { ++ Match(&'a SealedExecutable), ++ Denied, ++ Unsealed, ++} ++ ++fn lookup_sealed_alias<'a>( ++ sealed_modules: &'a HashMap, ++ name: &str, ++) -> SealedAliasLookup<'a> { ++ if let Some(executable) = sealed_modules.get(name) { ++ SealedAliasLookup::Match(executable) ++ } else if sealed_modules.is_empty() { ++ SealedAliasLookup::Unsealed ++ } else { ++ SealedAliasLookup::Denied ++ } + } + + impl BinFactory { + pub fn new(runtime: Arc) -> BinFactory { ++ Self::new_with_sealed_modules(runtime, HashMap::new()) ++ } ++ ++ pub(crate) fn new_with_sealed_modules( ++ runtime: Arc, ++ sealed_modules: HashMap)>, ++ ) -> BinFactory { + BinFactory { + commands: Commands::new_with_builtins(runtime.clone()), + runtime, + local: Arc::new(RwLock::new(HashMap::new())), ++ sealed_modules: Arc::new( ++ sealed_modules ++ .into_iter() ++ .map(|(path, (module_hash, preinitialized_memory_image))| { ++ ( ++ path, ++ SealedExecutable { ++ module_hash, ++ preinitialized_memory_image, ++ }, ++ ) ++ }) ++ .collect(), ++ ), + } + } + +@@ -101,7 +150,7 @@ impl BinFactory { + self.get_executable(name, fs) + .await + .and_then(|executable| match executable { +- Executable::Wasm(_) => None, ++ Executable::Wasm(_) | Executable::SealedModule(_) => None, + Executable::BinaryPackage(pkg) => Some(pkg), + }) + } +@@ -123,6 +172,27 @@ impl BinFactory { + + // Execute + match executable { ++ Executable::SealedModule(executable) => { ++ let mut env = env; ++ env.process.module_hash = executable.module_hash; ++ let module = self ++ .runtime ++ .module_cache() ++ .load(executable.module_hash, &self.runtime.engine()) ++ .await ++ .map_err(SpawnError::CacheError)?; ++ let image = executable ++ .preinitialized_memory_image ++ .map(|image| image.load().map(PreinitializedMemoryImageMode::Apply)) ++ .transpose() ++ .map_err(|message| SpawnError::ModuleLoad { message })?; ++ spawn_exec_module_with_preinitialized_memory_image( ++ module, ++ env, ++ &self.runtime, ++ image, ++ ) ++ } + Executable::Wasm(bytes) => { + let data = HashedModuleData::new(bytes.clone()); + spawn_exec_wasm(data, name.as_str(), env, &self.runtime).await +@@ -145,6 +215,13 @@ impl BinFactory { + parent_ctx: Option<&FunctionEnvMut<'_, WasiEnv>>, + builder: &mut Option, + ) -> Result { ++ // Syscall paths consult built-ins before `spawn`, so the sealed policy ++ // must close this resolver too. Otherwise an undeclared host command ++ // (currently `/bin/wasmer`) could bypass the exact manifest closure. ++ if !self.sealed_modules.is_empty() { ++ return Err(SpawnError::BinaryNotFound { binary: name }); ++ } ++ + // We check for built in commands + if let Some(parent_ctx) = parent_ctx { + if self.commands.exists(name.as_str()) { +@@ -164,6 +241,17 @@ impl BinFactory { + name: &str, + fs: Option<&dyn FileSystem>, + ) -> Option { ++ // A sealed registry is an exact immutable-executable policy. Once it is ++ // present, an unknown alias is denied before consulting mutable guest ++ // filesystem state. ++ match lookup_sealed_alias(&self.sealed_modules, name) { ++ SealedAliasLookup::Match(executable) => { ++ return Some(Executable::SealedModule(executable.clone())); ++ } ++ SealedAliasLookup::Denied => return None, ++ SealedAliasLookup::Unsealed => {} ++ } ++ + let name = name.to_string(); + + // Return early if the path is already cached +@@ -209,7 +297,44 @@ impl BinFactory { + } + } + ++#[cfg(test)] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn sealed_aliases_are_exact_and_close_the_executable_namespace() { ++ let module_hash = ModuleHash::from_bytes([0x5a; 32]); ++ let sealed_modules = HashMap::from([( ++ "/bin/postgres".to_string(), ++ SealedExecutable { ++ module_hash, ++ preinitialized_memory_image: None, ++ }, ++ )]); ++ ++ match lookup_sealed_alias(&sealed_modules, "/bin/postgres") { ++ SealedAliasLookup::Match(executable) => { ++ assert_eq!(executable.module_hash, module_hash) ++ } ++ _ => panic!("exact sealed alias was not resolved"), ++ } ++ assert!(matches!( ++ lookup_sealed_alias(&sealed_modules, "postgres"), ++ SealedAliasLookup::Denied ++ )); ++ assert!(matches!( ++ lookup_sealed_alias(&sealed_modules, "/bin/Postgres"), ++ SealedAliasLookup::Denied ++ )); ++ assert!(matches!( ++ lookup_sealed_alias(&HashMap::new(), "/bin/postgres"), ++ SealedAliasLookup::Unsealed ++ )); ++ } ++} ++ + pub enum Executable { ++ SealedModule(SealedExecutable), + Wasm(OwnedBuffer), + BinaryPackage(Arc), + } +diff --git a/lib/wasix/src/os/task/control_plane.rs b/lib/wasix/src/os/task/control_plane.rs +index 18e8ada..3468f5b 100644 +--- a/lib/wasix/src/os/task/control_plane.rs ++++ b/lib/wasix/src/os/task/control_plane.rs +@@ -7,8 +7,14 @@ use std::{ + time::Duration, + }; + +-use crate::{WasiProcess, WasiProcessId}; ++#[cfg(test)] ++use std::sync::atomic::AtomicBool; ++ ++use crate::os::task::process::WasiChildPublicationGuard; ++use crate::os::task::thread::WasiMemoryLayout; ++use crate::{WasiProcess, WasiProcessId, WasiThreadHandle}; + use wasmer_types::ModuleHash; ++use wasmer_wasix_types::wasix::ThreadStartType; + + #[derive(Debug, Clone)] + pub struct WasiControlPlane { +@@ -73,6 +79,10 @@ struct State { + /// Total number of active tasks (threads) across all processes. + task_count: Arc, + ++ /// Deterministic failure injection for post-seal transaction tests. ++ #[cfg(test)] ++ fail_next_task_admission: AtomicBool, ++ + /// Mutable state. + mutable: RwLock, + } +@@ -92,6 +102,8 @@ impl WasiControlPlane { + state: Arc::new(State { + config, + task_count: Arc::new(AtomicUsize::new(0)), ++ #[cfg(test)] ++ fail_next_task_admission: AtomicBool::new(false), + mutable: RwLock::new(MutableState { + process_seed: 0, + processes: Default::default(), +@@ -105,8 +117,8 @@ impl WasiControlPlane { + } + + /// Get the current count of active tasks (threads). +- fn active_task_count(&self) -> usize { +- self.state.task_count.load(Ordering::SeqCst) ++ pub(crate) fn active_task_count(&self) -> usize { ++ self.state.task_count.load(Ordering::Acquire) + } + + /// Returns the configuration for this control plane +@@ -116,22 +128,73 @@ impl WasiControlPlane { + + /// Register a new task. + /// +- // Currently just increments the task counter. ++ // The CAS is the authoritative admission decision. An advisory process ++ // precheck cannot enforce a global limit when threads start concurrently. + pub(crate) fn register_task(&self) -> Result { +- let count = self.state.task_count.fetch_add(1, Ordering::SeqCst); +- if let Some(max) = self.state.config.max_task_count +- && count > max ++ #[cfg(test)] ++ if self ++ .state ++ .fail_next_task_admission ++ .swap(false, Ordering::AcqRel) + { +- self.state.task_count.fetch_sub(1, Ordering::SeqCst); +- return Err(ControlPlaneError::TaskLimitReached { max: count }); ++ return Err(ControlPlaneError::TaskLimitReached { ++ max: self.state.config.max_task_count.unwrap_or(usize::MAX), ++ }); ++ } ++ ++ let mut current = self.state.task_count.load(Ordering::Acquire); ++ loop { ++ if let Some(max) = self.state.config.max_task_count ++ && current >= max ++ { ++ return Err(ControlPlaneError::TaskLimitReached { max }); ++ } ++ let Some(next) = current.checked_add(1) else { ++ return Err(ControlPlaneError::TaskLimitReached { max: usize::MAX }); ++ }; ++ match self.state.task_count.compare_exchange_weak( ++ current, ++ next, ++ Ordering::AcqRel, ++ Ordering::Acquire, ++ ) { ++ Ok(_) => return Ok(TaskCountGuard(self.state.task_count.clone())), ++ Err(observed) => current = observed, ++ } + } +- Ok(TaskCountGuard(self.state.task_count.clone())) + } + +- /// Creates a new process +- // FIXME: De-register terminated processes! +- // Currently they just accumulate. +- pub fn new_process(&self, module_hash: ModuleHash) -> Result { ++ /// Creates and publishes a process together with its main thread. ++ /// ++ /// A process is never globally visible without a successfully admitted ++ /// main thread. Finished processes remain registered as zombies until a ++ /// successful join reaps them. ++ pub fn new_process_with_main_thread( ++ &self, ++ module_hash: ModuleHash, ++ layout: WasiMemoryLayout, ++ ) -> Result<(WasiProcess, WasiThreadHandle), ControlPlaneError> { ++ let (process, handle, mut registration) = ++ self.new_process_with_main_thread_guarded(module_hash, layout)?; ++ registration.commit(); ++ Ok((process, handle)) ++ } ++ ++ /// Reserves a unique PID and constructs a process off-map. Until the ++ /// returned guard is committed, registry lookups, signals, joins, and ++ /// process enumeration cannot observe or retain the tentative process. ++ pub(crate) fn new_process_guarded( ++ &self, ++ module_hash: ModuleHash, ++ ) -> Result<(WasiProcess, WasiProcessRegistrationGuard), ControlPlaneError> { ++ self.new_process_guarded_with_lifecycle(module_hash, None) ++ } ++ ++ fn new_process_guarded_with_lifecycle( ++ &self, ++ module_hash: ModuleHash, ++ child_parent: Option<&WasiProcess>, ++ ) -> Result<(WasiProcess, WasiProcessRegistrationGuard), ControlPlaneError> { + if let Some(max) = self.state.config.max_task_count + && self.active_task_count() >= max + { +@@ -140,15 +203,73 @@ impl WasiControlPlane { + return Err(ControlPlaneError::TaskLimitReached { max }); + } + +- // Create the process first to do all the allocations before locking. +- let mut proc = WasiProcess::new(WasiProcessId::from(0), module_hash, self.handle()); ++ let pid = self.state.mutable.write().unwrap().next_process_id()?; ++ let process = match child_parent { ++ Some(parent) => WasiProcess::new_pending_child( ++ pid, ++ module_hash, ++ self.handle(), ++ parent, ++ parent.tree_epoch(), ++ ), ++ None => WasiProcess::new(pid, module_hash, self.handle()), ++ }; ++ let guard = WasiProcessRegistrationGuard { ++ control_plane: self.clone(), ++ process: process.clone(), ++ parent: None, ++ parent_publication: None, ++ committed: false, ++ launch_complete: false, ++ }; ++ Ok((process, guard)) ++ } + ++ /// Atomically publishes a PID reserved by `new_process_guarded`. Reserved ++ /// IDs are monotonic and never reused, so occupancy indicates an internal ++ /// invariant violation rather than a recoverable runtime race. ++ fn publish_reserved_process(&self, process: &WasiProcess) { + let mut mutable = self.state.mutable.write().unwrap(); ++ match mutable.processes.entry(process.pid()) { ++ std::collections::hash_map::Entry::Vacant(entry) => { ++ entry.insert(process.clone()); ++ } ++ std::collections::hash_map::Entry::Occupied(_) => { ++ panic!("reserved process ID was published more than once"); ++ } ++ } ++ } + +- let pid = mutable.next_process_id()?; +- proc.set_pid(pid); +- mutable.processes.insert(pid, proc.clone()); +- Ok(proc) ++ /// Creates an off-map process and a registered main thread as one ++ /// failure-safe unit. Task admission failure drops the tentative object; ++ /// successful callers publish it by committing the returned guard. ++ pub(crate) fn new_process_with_main_thread_guarded( ++ &self, ++ module_hash: ModuleHash, ++ layout: WasiMemoryLayout, ++ ) -> Result<(WasiProcess, WasiThreadHandle, WasiProcessRegistrationGuard), ControlPlaneError> ++ { ++ let (process, guard) = self.new_process_guarded(module_hash)?; ++ let handle = process.new_thread(layout, ThreadStartType::MainThread)?; ++ Ok((process, handle, guard)) ++ } ++ ++ /// Reserves parent publication first, then creates the tentative child in ++ /// the parent's tree epoch with a pending guest-start gate. This is the ++ /// only supported constructor for a process that will be adopted. ++ pub(crate) fn new_child_process_with_main_thread_guarded( ++ &self, ++ parent: &WasiProcess, ++ module_hash: ModuleHash, ++ layout: WasiMemoryLayout, ++ ) -> Result<(WasiProcess, WasiThreadHandle, WasiProcessRegistrationGuard), ControlPlaneError> ++ { ++ let publication = parent.begin_child_publication()?; ++ let (process, guard) = ++ self.new_process_guarded_with_lifecycle(module_hash, Some(parent))?; ++ let handle = process.new_thread(layout, ThreadStartType::MainThread)?; ++ let guard = guard.with_parent(parent, publication); ++ Ok((process, handle, guard)) + } + + /// Generates a new process ID +@@ -167,6 +288,240 @@ impl WasiControlPlane { + .get(&pid) + .cloned() + } ++ ++ /// Removes a finished process from the global registry after it has been ++ /// joined. The identity check prevents a stale handle from removing a ++ /// future process if PID reuse is introduced. ++ pub(crate) fn reap_process(&self, process: &WasiProcess) -> bool { ++ if process.try_join().is_none() { ++ return false; ++ } ++ ++ let mut mutable = self.state.mutable.write().unwrap(); ++ let is_registered_process = mutable ++ .processes ++ .get(&process.pid()) ++ .is_some_and(|registered| Arc::ptr_eq(®istered.inner, &process.inner)); ++ if !is_registered_process { ++ return false; ++ } ++ mutable.processes.remove(&process.pid()); ++ true ++ } ++ ++ /// Retires a completed reusable execution epoch. An already-absent entry ++ /// is accepted (another legitimate join may have reaped it); a different ++ /// identity under the same PID is never removed. ++ pub(crate) fn retire_process_epoch( ++ &self, ++ process: &WasiProcess, ++ ) -> Result<(), ControlPlaneError> { ++ if process.try_join().is_none() { ++ return Err(ControlPlaneError::ProcessStillRunning { ++ pid: process.pid().raw(), ++ }); ++ } ++ ++ let mut mutable = self.state.mutable.write().unwrap(); ++ match mutable.processes.get(&process.pid()) { ++ None => Ok(()), ++ Some(registered) if registered.same_identity(process) => { ++ mutable.processes.remove(&process.pid()); ++ Ok(()) ++ } ++ Some(_) => Err(ControlPlaneError::ProcessIdentityChanged { ++ pid: process.pid().raw(), ++ }), ++ } ++ } ++ ++ /// Returns the number of process identities currently published in this ++ /// control plane. This is used by opt-in lifecycle diagnostics as well as ++ /// invariant tests; callers must not use it for admission decisions. ++ #[cfg(test)] ++ pub(crate) fn registered_process_count(&self) -> usize { ++ self.state.mutable.read().unwrap().processes.len() ++ } ++ ++ fn process_epoch_snapshot(&self, root: &WasiProcess) -> Vec { ++ let mut processes = vec![root.clone()]; ++ processes.extend( ++ self.state ++ .mutable ++ .read() ++ .unwrap() ++ .processes ++ .values() ++ .filter(|process| process.same_tree(root) && !process.same_identity(root)) ++ .cloned(), ++ ); ++ processes ++ } ++ ++ /// Wait for every published process in `root`'s epoch to publish terminal ++ /// status and release both execution and child-publication ownership. ++ /// ++ /// The final exclusive epoch pass linearizes against a child published ++ /// after the preceding registry snapshot. A process removed from the ++ /// registry is already join-claimed and quiescent, so it cannot hide live ++ /// guest work. Once this returns, terminal product evidence can take a ++ /// coherent process-tree snapshot without polling or timing assumptions. ++ pub(crate) async fn wait_for_process_tree_quiescence(&self, root: &WasiProcess) { ++ loop { ++ for process in self.process_epoch_snapshot(root) { ++ let _ = process.join().await; ++ } ++ ++ let epoch = root.tree_epoch(); ++ let _exclusive_epoch = epoch.write().unwrap(); ++ let stable = self.process_epoch_snapshot(root); ++ let all_quiescent = stable.iter().all(|process| { ++ if !process.finished.status().is_finished() { ++ return false; ++ } ++ let inner = process.inner.0.lock().unwrap(); ++ inner.execution_leases == 0 && inner.pending_child_publications == 0 ++ }); ++ if all_quiescent { ++ return; ++ } ++ } ++ } ++ ++ #[cfg(test)] ++ pub(crate) fn fail_next_task_admission(&self) { ++ self.state ++ .fail_next_task_admission ++ .store(true, Ordering::Release); ++ } ++} ++ ++/// Rollback token for a process registration that has not yet crossed its ++/// public success boundary. ++#[derive(Debug)] ++pub(crate) struct WasiProcessRegistrationGuard { ++ control_plane: WasiControlPlane, ++ process: WasiProcess, ++ parent: Option, ++ parent_publication: Option, ++ committed: bool, ++ launch_complete: bool, ++} ++ ++impl WasiProcessRegistrationGuard { ++ /// Records the potential parent so rollback also removes an adoption link ++ /// if a later construction step fails. ++ pub(crate) fn with_parent( ++ mut self, ++ parent: &WasiProcess, ++ publication: WasiChildPublicationGuard, ++ ) -> Self { ++ assert!( ++ self.process.same_tree(parent), ++ "child process must inherit its parent's tree epoch" ++ ); ++ self.parent = Some(parent.clone()); ++ self.parent_publication = Some(publication.bind_child(&self.process)); ++ self ++ } ++ ++ pub(crate) fn commit(&mut self) { ++ assert!( ++ self.parent.is_none() && self.parent_publication.is_none(), ++ "child process registration requires commit_child" ++ ); ++ assert!(!self.committed, "process registration committed twice"); ++ self.control_plane.publish_reserved_process(&self.process); ++ self.committed = true; ++ self.launch_complete = true; ++ } ++ ++ /// Atomically adopts and publishes a child while its exact parent-side ++ /// publication reservation prevents epoch retirement. No published child ++ /// can be observed without its parent link, and no permit can authorize a ++ /// different parent/child pair. ++ pub(crate) fn commit_child(&mut self) -> Result<(), ControlPlaneError> { ++ assert!(!self.committed, "process registration committed twice"); ++ let parent = self ++ .parent ++ .clone() ++ .ok_or(ControlPlaneError::InvalidChildPublication { ++ parent_pid: 0, ++ child_pid: self.process.pid().raw(), ++ })?; ++ let publication = ++ self.parent_publication ++ .take() ++ .ok_or(ControlPlaneError::InvalidChildPublication { ++ parent_pid: parent.pid().raw(), ++ child_pid: self.process.pid().raw(), ++ })?; ++ let control_plane = self.control_plane.clone(); ++ let process = self.process.clone(); ++ parent.adopt_child_process(process.clone(), publication, || { ++ control_plane.publish_reserved_process(&process); ++ })?; ++ self.committed = true; ++ process.commit_guest_start(); ++ Ok(()) ++ } ++ ++ /// Completes the child launch transaction after a task/command has been ++ /// accepted. Exit notification is deliberately delayed until this point, ++ /// so a launch rollback cannot emit a spurious SIGCHLD. ++ pub(crate) fn complete_child_launch( ++ &mut self, ++ tasks: &Arc, ++ ) { ++ assert!(self.committed, "child launch completed before publication"); ++ assert!(!self.launch_complete, "child launch completed twice"); ++ let parent = self ++ .parent ++ .clone() ++ .expect("child registration has no parent"); ++ parent.notify_on_child_exit(self.process.clone(), tasks); ++ self.launch_complete = true; ++ } ++ ++ /// Rolls back an already-published embryonic child whose launch failed. ++ /// The start gate was committed before instantiation, so termination and a ++ /// real lease-quiescence wait are required before exact topology/registry ++ /// removal. No polling or terminate-as-cancel shortcut is used. ++ pub(crate) fn rollback_child(&mut self, exit_code: wasmer_wasix_types::wasi::ExitCode) { ++ assert!(self.committed, "cannot roll back an unpublished child"); ++ assert!( ++ !self.launch_complete, ++ "cannot roll back a completed child launch" ++ ); ++ self.process.terminate(exit_code); ++ self.process.wait_for_execution_quiescence_blocking(); ++ if let Some(parent) = self.parent.as_ref() { ++ // A concurrent POSIX waiter may already have consumed the failed ++ // child's status and removed this exact link. ++ parent.remove_child_if_same(&self.process); ++ } ++ // Likewise, reap is idempotent with a concurrent exact-identity wait. ++ // PIDs are monotonic, so an absent entry cannot name a replacement. ++ self.control_plane.reap_process(&self.process); ++ self.committed = false; ++ self.launch_complete = true; ++ } ++} ++ ++impl Drop for WasiProcessRegistrationGuard { ++ fn drop(&mut self) { ++ if self.committed && !self.launch_complete { ++ self.rollback_child(wasmer_wasix_types::wasi::Errno::Canceled.into()); ++ return; ++ } ++ if self.committed || self.launch_complete { ++ return; ++ } ++ self.process.abort_guest_start(); ++ if let Some(parent) = self.parent.as_ref() { ++ parent.remove_child_if_same(&self.process); ++ } ++ } + } + + impl MutableState { +@@ -195,7 +550,11 @@ pub struct TaskCountGuard(Arc); + + impl Drop for TaskCountGuard { + fn drop(&mut self) { +- self.0.fetch_sub(1, Ordering::SeqCst); ++ self.0 ++ .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { ++ current.checked_sub(1) ++ }) ++ .expect("control-plane task counter underflow"); + } + } + +@@ -207,11 +566,47 @@ pub enum ControlPlaneError { + /// The maximum number of tasks. + max: usize, + }, ++ /// A live thread already owns the requested numeric thread ID. ++ #[error("Thread ID {tid} is already registered")] ++ DuplicateThreadId { tid: u32 }, ++ /// A reusable environment may only be reset after its current process exits. ++ #[error("Process {pid} is still running")] ++ ProcessStillRunning { pid: u32 }, ++ /// Terminal processes cannot accept a new thread or execution task. ++ #[error("Process {pid} has already finished")] ++ ProcessFinished { pid: u32 }, ++ /// Tentative children cannot instantiate guest code until their exact ++ /// parent adoption and registry publication transaction has committed. ++ #[error("Process {pid} has not been published for guest execution")] ++ ProcessNotPublished { pid: u32 }, ++ /// In-place process switching must begin inside an accepted TaskWasm so ++ /// its exact physical execution owner can authorize a later handoff. ++ #[error("Process {pid} thread {tid} has no accepted TaskWasm execution owner")] ++ ExecutionOwnerUnavailable { pid: u32, tid: u32 }, ++ /// The process registry no longer contains the expected process identity. ++ #[error("Process {pid} registration changed identity")] ++ ProcessIdentityChanged { pid: u32 }, ++ /// The process epoch has begun terminal retirement. ++ #[error("Process {pid} is retiring")] ++ ProcessRetiring { pid: u32 }, ++ /// Background thread handles/tasks still retain the process epoch. ++ #[error("Process {pid} still has {count} live non-main thread(s)")] ++ ProcessHasLiveThreads { pid: u32, count: usize }, ++ /// Child processes or in-flight child publications still retain the epoch. ++ #[error("Process {pid} still has {count} live child process(es)")] ++ ProcessHasLiveChildren { pid: u32, count: usize }, ++ /// A child publication token was used for a different process pair. ++ #[error("Invalid child publication for parent {parent_pid} and child {child_pid}")] ++ InvalidChildPublication { parent_pid: u32, child_pid: u32 }, ++ /// Fork must not snapshot address-space metadata while a fresh exec image ++ /// is taking ownership of inherited shared mappings. ++ #[error("shared-memory exec transition prevents a consistent fork snapshot")] ++ SharedMemoryForkUnavailable, + } + + #[cfg(test)] + mod tests { +- use wasmer_wasix_types::wasix::ThreadStartType; ++ use wasmer_wasix_types::{wasi::Errno, wasix::ThreadStartType}; + + use crate::os::task::thread::WasiMemoryLayout; + +@@ -226,16 +621,28 @@ mod tests { + enable_exponential_cpu_backoff: None, + }); + +- let p1 = p.new_process(ModuleHash::random()).unwrap(); +- let _t1 = p1 +- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ let (p1, _t1) = p ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) + .unwrap(); + let _t2 = p1 +- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) + .unwrap(); + + assert_eq!( +- p.new_process(ModuleHash::random()).unwrap_err(), ++ p1.new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) ++ .unwrap_err(), ++ ControlPlaneError::TaskLimitReached { max: 2 } ++ ); ++ ++ assert_eq!( ++ p.new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap_err(), + ControlPlaneError::TaskLimitReached { max: 2 } + ); + } +@@ -249,24 +656,1170 @@ mod tests { + enable_exponential_cpu_backoff: None, + }); + +- let p1 = p.new_process(ModuleHash::random()).unwrap(); ++ let (p1, _initial_main) = p ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); + + for _ in 0..10 { + let _thread = p1 +- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) + .unwrap(); + } + +- let _t1 = p1 +- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) +- .unwrap(); + let _t2 = p1 +- .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) + .unwrap(); + + assert_eq!( +- p.new_process(ModuleHash::random()).unwrap_err(), ++ p.new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap_err(), + ControlPlaneError::TaskLimitReached { max: 2 } + ); + } ++ ++ #[test] ++ fn finished_process_remains_until_join_reaps_it() { ++ let plane = WasiControlPlane::default(); ++ let (process, _main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ ++ assert!(!plane.reap_process(&process)); ++ assert!(plane.get_process(process.pid()).is_some()); ++ process.terminate(0u16.into()); ++ assert!(process.try_join().is_some()); ++ assert!(plane.get_process(process.pid()).is_some()); ++ ++ assert!(plane.reap_process(&process)); ++ assert!(plane.get_process(process.pid()).is_none()); ++ assert_eq!(plane.registered_process_count(), 0); ++ assert!(!plane.reap_process(&process)); ++ } ++ ++ #[test] ++ fn stale_process_identity_cannot_reap_registered_process() { ++ let plane = WasiControlPlane::default(); ++ let (registered, _registered_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let mut stale = WasiProcess::new(registered.pid(), ModuleHash::random(), plane.handle()); ++ stale.set_pid(registered.pid()); ++ let _main = stale ++ .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ .unwrap(); ++ stale.terminate(0u16.into()); ++ ++ assert!(!plane.reap_process(&stale)); ++ let current = plane.get_process(registered.pid()).unwrap(); ++ assert!(Arc::ptr_eq(¤t.inner, ®istered.inner)); ++ } ++ ++ #[test] ++ fn try_join_any_child_removes_parent_link_and_reaps_registry_entry() { ++ let plane = WasiControlPlane::default(); ++ let (mut parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, _main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let child_pid = child.pid(); ++ parent.lock().children.push(child.clone()); ++ ++ child.terminate(0u16.into()); ++ let joined = parent.try_join_any_child().unwrap().unwrap(); ++ ++ assert_eq!(joined.0, child_pid); ++ assert_eq!(joined.1, 0u16.into()); ++ assert!(parent.lock().children.is_empty()); ++ assert!(plane.get_process(child_pid).is_none()); ++ assert!(plane.get_process(parent.pid()).is_some()); ++ } ++ ++ #[test] ++ fn task_admission_honors_zero_and_one_as_exact_limits() { ++ let zero = WasiControlPlane::new(ControlPlaneConfig { ++ max_task_count: Some(0), ++ ..Default::default() ++ }); ++ assert_eq!( ++ zero.register_task().unwrap_err(), ++ ControlPlaneError::TaskLimitReached { max: 0 } ++ ); ++ assert_eq!(zero.active_task_count(), 0); ++ ++ let one = WasiControlPlane::new(ControlPlaneConfig { ++ max_task_count: Some(1), ++ ..Default::default() ++ }); ++ let (process, main) = one ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ assert_eq!(one.active_task_count(), 1); ++ assert_eq!( ++ process ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) ++ .unwrap_err(), ++ ControlPlaneError::TaskLimitReached { max: 1 } ++ ); ++ assert_eq!(one.active_task_count(), 1); ++ drop(main); ++ assert_eq!(one.active_task_count(), 0); ++ } ++ ++ #[test] ++ fn concurrent_task_reservations_never_oversubscribe_limit() { ++ use std::sync::{Barrier, mpsc}; ++ use std::thread; ++ ++ const LIMIT: usize = 4; ++ const CONTENDERS: usize = 24; ++ let plane = WasiControlPlane::new(ControlPlaneConfig { ++ max_task_count: Some(LIMIT), ++ ..Default::default() ++ }); ++ let start = Arc::new(Barrier::new(CONTENDERS + 1)); ++ let release = Arc::new(Barrier::new(LIMIT + 1)); ++ let (tx, rx) = mpsc::channel(); ++ let mut workers = Vec::new(); ++ ++ for _ in 0..CONTENDERS { ++ let plane = plane.clone(); ++ let start = start.clone(); ++ let release = release.clone(); ++ let tx = tx.clone(); ++ workers.push(thread::spawn(move || { ++ start.wait(); ++ match plane.register_task() { ++ Ok(guard) => { ++ tx.send(true).unwrap(); ++ release.wait(); ++ drop(guard); ++ } ++ Err(ControlPlaneError::TaskLimitReached { max: LIMIT }) => { ++ tx.send(false).unwrap(); ++ } ++ Err(other) => panic!("unexpected admission error: {other}"), ++ } ++ })); ++ } ++ drop(tx); ++ start.wait(); ++ let admitted = (0..CONTENDERS) ++ .map(|_| rx.recv().unwrap()) ++ .filter(|admitted| *admitted) ++ .count(); ++ assert_eq!(admitted, LIMIT); ++ assert_eq!(plane.active_task_count(), LIMIT); ++ release.wait(); ++ for worker in workers { ++ worker.join().unwrap(); ++ } ++ assert_eq!(plane.active_task_count(), 0); ++ } ++ ++ #[test] ++ fn duplicate_live_tid_is_rejected_without_count_or_slot_corruption() { ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let main_alias = main.clone(); ++ let observational_thread_clone = main.as_thread(); ++ let tid = main.id(); ++ ++ assert_eq!( ++ process ++ .new_thread_with_id( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ tid, ++ ) ++ .unwrap_err(), ++ ControlPlaneError::DuplicateThreadId { tid: tid.raw() } ++ ); ++ assert_eq!(process.active_threads(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ assert!(process.get_thread(&tid).unwrap().same_identity(&main)); ++ ++ drop(main); ++ assert_eq!(process.active_threads(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ assert_eq!( ++ process ++ .new_thread_with_id( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ tid, ++ ) ++ .unwrap_err(), ++ ControlPlaneError::DuplicateThreadId { tid: tid.raw() } ++ ); ++ drop(main_alias); ++ assert_eq!(process.active_threads(), 0); ++ assert_eq!(plane.active_task_count(), 0); ++ assert_eq!(observational_thread_clone.tid(), tid); ++ ++ assert_eq!( ++ process ++ .new_thread_with_id( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ tid, ++ ) ++ .unwrap_err(), ++ ControlPlaneError::ProcessFinished { ++ pid: process.pid().raw(), ++ } ++ ); ++ assert_eq!(process.active_threads(), 0); ++ assert_eq!(plane.active_task_count(), 0); ++ drop(observational_thread_clone); ++ } ++ ++ #[test] ++ fn unpublished_process_guard_rolls_back_registry_and_parent_link() { ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let baseline = plane.registered_process_count(); ++ let (child, child_main, guard) = plane ++ .new_child_process_with_main_thread_guarded( ++ &parent, ++ ModuleHash::random(), ++ WasiMemoryLayout::default(), ++ ) ++ .unwrap(); ++ parent.lock().children.push(child.clone()); ++ assert_eq!(plane.registered_process_count(), baseline); ++ assert!(plane.get_process(child.pid()).is_none()); ++ assert_eq!(parent.lock().children.len(), 1); ++ ++ drop(guard); ++ ++ assert_eq!(plane.registered_process_count(), baseline); ++ assert!(parent.lock().children.is_empty()); ++ assert!(plane.get_process(child.pid()).is_none()); ++ drop(child_main); ++ } ++ ++ #[test] ++ fn main_thread_admission_failure_cannot_leave_published_process() { ++ let plane = WasiControlPlane::new(ControlPlaneConfig { ++ max_task_count: Some(1), ++ ..Default::default() ++ }); ++ let (process, registration) = plane.new_process_guarded(ModuleHash::random()).unwrap(); ++ assert_eq!(plane.registered_process_count(), 0); ++ assert!(plane.get_process(process.pid()).is_none()); ++ ++ // Deterministically fill the only task slot after process publication, ++ // reproducing the race that an advisory precheck cannot prevent. ++ let competing_reservation = plane.register_task().unwrap(); ++ assert_eq!( ++ process ++ .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ .unwrap_err(), ++ ControlPlaneError::TaskLimitReached { max: 1 } ++ ); ++ drop(registration); ++ ++ assert_eq!(plane.registered_process_count(), 0); ++ assert!(plane.get_process(process.pid()).is_none()); ++ assert_eq!(plane.active_task_count(), 1); ++ drop(competing_reservation); ++ assert_eq!(plane.active_task_count(), 0); ++ } ++ ++ #[test] ++ fn tentative_process_is_invisible_and_abort_releases_last_object() { ++ use std::sync::Barrier; ++ use std::thread; ++ ++ let plane = WasiControlPlane::default(); ++ let (process, registration) = plane.new_process_guarded(ModuleHash::random()).unwrap(); ++ let pid = process.pid(); ++ let weak_process = Arc::downgrade(&process.inner); ++ let phases = Arc::new(Barrier::new(2)); ++ let observer_plane = plane.clone(); ++ let observer_phases = phases.clone(); ++ let observer = thread::spawn(move || { ++ observer_phases.wait(); ++ let visible_before_abort = observer_plane.get_process(pid).is_some(); ++ observer_phases.wait(); ++ observer_phases.wait(); ++ let visible_after_abort = observer_plane.get_process(pid).is_some(); ++ (visible_before_abort, visible_after_abort) ++ }); ++ ++ phases.wait(); ++ phases.wait(); ++ assert!(plane.get_process(pid).is_none()); ++ assert_eq!(plane.registered_process_count(), 0); ++ drop(registration); ++ drop(process); ++ assert!(weak_process.upgrade().is_none()); ++ phases.wait(); ++ ++ assert_eq!(observer.join().unwrap(), (false, false)); ++ assert_eq!(plane.registered_process_count(), 0); ++ } ++ ++ #[test] ++ fn process_wait_status_has_exactly_one_concurrent_consumer() { ++ use std::sync::Barrier; ++ use std::thread; ++ ++ let plane = WasiControlPlane::default(); ++ let (process, _main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ process.terminate(0u16.into()); ++ let start = Arc::new(Barrier::new(3)); ++ let mut waiters = Vec::new(); ++ for _ in 0..2 { ++ let process = process.clone(); ++ let start = start.clone(); ++ waiters.push(thread::spawn(move || { ++ start.wait(); ++ process.try_claim_join().is_some() ++ })); ++ } ++ start.wait(); ++ let winners = waiters ++ .into_iter() ++ .map(|waiter| waiter.join().unwrap()) ++ .filter(|won| *won) ++ .count(); ++ assert_eq!(winners, 1); ++ assert!( ++ process.try_join().is_some(), ++ "observation remains non-consuming" ++ ); ++ assert!(process.try_claim_join().is_none()); ++ } ++ ++ #[test] ++ fn concurrent_parent_waiters_return_one_child_status() { ++ use std::sync::Barrier; ++ use std::thread; ++ ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, _child_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let child_pid = child.pid(); ++ parent.lock().children.push(child.clone()); ++ child.terminate(0u16.into()); ++ ++ let start = Arc::new(Barrier::new(3)); ++ let mut waiters = Vec::new(); ++ for _ in 0..2 { ++ let mut parent = parent.clone(); ++ let start = start.clone(); ++ waiters.push(thread::spawn(move || { ++ start.wait(); ++ parent.try_join_any_child() ++ })); ++ } ++ start.wait(); ++ let results: Vec<_> = waiters ++ .into_iter() ++ .map(|waiter| waiter.join().unwrap()) ++ .collect(); ++ let joined: Vec<_> = results ++ .iter() ++ .filter_map(|result| result.as_ref().ok().and_then(|status| *status)) ++ .collect(); ++ ++ assert_eq!(joined, vec![(child_pid, 0u16.into())]); ++ assert!(parent.lock().children.is_empty()); ++ assert!(plane.get_process(child_pid).is_none()); ++ } ++ ++ #[test] ++ fn public_process_construction_never_publishes_without_a_main_thread() { ++ let plane = WasiControlPlane::new(ControlPlaneConfig { ++ max_task_count: Some(0), ++ ..Default::default() ++ }); ++ ++ assert_eq!( ++ plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default(),) ++ .unwrap_err(), ++ ControlPlaneError::TaskLimitReached { max: 0 } ++ ); ++ assert_eq!(plane.registered_process_count(), 0); ++ assert_eq!(plane.active_task_count(), 0); ++ } ++ ++ #[test] ++ fn retirement_seals_a_finished_process_against_new_threads() { ++ use std::sync::Barrier; ++ use std::thread; ++ ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ process.terminate(0u16.into()); ++ ++ let start = Arc::new(Barrier::new(3)); ++ let thread_process = process.clone(); ++ let thread_start = start.clone(); ++ let thread_racer = thread::spawn(move || { ++ thread_start.wait(); ++ thread_process.new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) ++ }); ++ let retirement_process = process.clone(); ++ let retirement_start = start.clone(); ++ let retirement_racer = thread::spawn(move || { ++ retirement_start.wait(); ++ retirement_process.begin_epoch_retirement() ++ }); ++ start.wait(); ++ ++ let thread_result = thread_racer.join().unwrap(); ++ let retirement_result = retirement_racer.join().unwrap(); ++ match (thread_result, retirement_result) { ++ ( ++ Err( ++ ControlPlaneError::ProcessFinished { .. } ++ | ControlPlaneError::ProcessRetiring { .. }, ++ ), ++ Ok(children), ++ ) => { ++ assert!(children.is_empty()); ++ assert!(process.lock().retiring); ++ } ++ (thread_result, retirement_result) => panic!( ++ "finished-process retirement admitted a new thread: thread={thread_result:?}, retirement={retirement_result:?}" ++ ), ++ } ++ ++ assert_eq!(process.active_threads(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ process.retire_epoch_task_registrations(); ++ plane.retire_process_epoch(&process).unwrap(); ++ drop(main); ++ assert_eq!(plane.active_task_count(), 0); ++ assert_eq!(plane.registered_process_count(), 0); ++ } ++ ++ #[test] ++ fn retirement_and_child_publication_linearize_without_leaking_permit() { ++ use std::sync::Barrier; ++ use std::thread; ++ ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ process.terminate(0u16.into()); ++ ++ let start = Arc::new(Barrier::new(3)); ++ let publication_process = process.clone(); ++ let publication_start = start.clone(); ++ let publication_racer = thread::spawn(move || { ++ publication_start.wait(); ++ publication_process.begin_child_publication() ++ }); ++ let retirement_process = process.clone(); ++ let retirement_start = start.clone(); ++ let retirement_racer = thread::spawn(move || { ++ retirement_start.wait(); ++ retirement_process.begin_epoch_retirement() ++ }); ++ start.wait(); ++ ++ let publication_result = publication_racer.join().unwrap(); ++ let retirement_result = retirement_racer.join().unwrap(); ++ match (publication_result, retirement_result) { ++ ( ++ Err( ++ ControlPlaneError::ProcessFinished { .. } ++ | ControlPlaneError::ProcessRetiring { .. }, ++ ), ++ Ok(children), ++ ) => { ++ assert!(children.is_empty()); ++ assert!(process.lock().retiring); ++ assert_eq!(process.lock().pending_child_publications, 0); ++ } ++ (publication_result, retirement_result) => panic!( ++ "race did not produce exactly one winner: publication={publication_result:?}, retirement={retirement_result:?}" ++ ), ++ } ++ ++ process.retire_epoch_task_registrations(); ++ plane.retire_process_epoch(&process).unwrap(); ++ drop(main); ++ assert_eq!(plane.active_task_count(), 0); ++ assert_eq!(plane.registered_process_count(), 0); ++ } ++ ++ #[test] ++ fn child_adoption_rejects_a_permit_for_a_different_parent_without_panic() { ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (other_parent, _other_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, child_handle, registration) = plane ++ .new_process_with_main_thread_guarded(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let publication = parent.begin_child_publication().unwrap().bind_child(&child); ++ ++ let error = other_parent ++ .adopt_child_process(child.clone(), publication, || { ++ panic!("invalid adoption must not publish the child") ++ }) ++ .unwrap_err(); ++ assert_eq!( ++ error, ++ ControlPlaneError::InvalidChildPublication { ++ parent_pid: other_parent.pid().raw(), ++ child_pid: child.pid().raw(), ++ } ++ ); ++ assert_eq!(parent.lock().pending_child_publications, 0); ++ assert!(other_parent.lock().children.is_empty()); ++ assert!(plane.get_process(child.pid()).is_none()); ++ drop(registration); ++ drop(child_handle); ++ } ++ ++ #[test] ++ fn child_execution_admission_requires_exact_publication() { ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, child_main, mut registration) = plane ++ .new_child_process_with_main_thread_guarded( ++ &parent, ++ ModuleHash::random(), ++ WasiMemoryLayout::default(), ++ ) ++ .unwrap(); ++ ++ assert_eq!( ++ child.acquire_execution_lease().unwrap_err(), ++ ControlPlaneError::ProcessNotPublished { ++ pid: child.pid().raw(), ++ } ++ ); ++ assert!(plane.get_process(child.pid()).is_none()); ++ assert!(parent.lock().children.is_empty()); ++ ++ registration.commit_child().unwrap(); ++ let lease = child.acquire_execution_lease().unwrap(); ++ assert_eq!(child.ppid(), parent.pid()); ++ assert!(child.has_exact_parent(&parent)); ++ assert!( ++ plane ++ .get_process(child.pid()) ++ .is_some_and(|registered| registered.same_identity(&child)) ++ ); ++ assert!( ++ parent ++ .lock() ++ .children ++ .iter() ++ .any(|candidate| candidate.same_identity(&child)) ++ ); ++ ++ child.terminate(Errno::Canceled.into()); ++ assert!( ++ child.try_join().is_none(), ++ "terminal status is not quiescence" ++ ); ++ assert!(!plane.reap_process(&child)); ++ drop(lease); ++ registration.rollback_child(Errno::Canceled.into()); ++ assert!(plane.get_process(child.pid()).is_none()); ++ assert!(parent.lock().children.is_empty()); ++ drop(child_main); ++ } ++ ++ #[test] ++ fn failed_launch_rollback_waits_for_real_execution_quiescence() { ++ use std::{sync::mpsc, thread, time::Duration}; ++ ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, child_main, mut registration) = plane ++ .new_child_process_with_main_thread_guarded( ++ &parent, ++ ModuleHash::random(), ++ WasiMemoryLayout::default(), ++ ) ++ .unwrap(); ++ registration.commit_child().unwrap(); ++ let lease = child.acquire_execution_lease().unwrap(); ++ ++ let (entered_tx, entered_rx) = mpsc::channel(); ++ let (done_tx, done_rx) = mpsc::channel(); ++ let rollback = thread::spawn(move || { ++ entered_tx.send(()).unwrap(); ++ registration.rollback_child(Errno::Canceled.into()); ++ done_tx.send(()).unwrap(); ++ }); ++ ++ entered_rx.recv().unwrap(); ++ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ assert!(plane.get_process(child.pid()).is_some()); ++ assert!( ++ parent ++ .lock() ++ .children ++ .iter() ++ .any(|candidate| candidate.same_identity(&child)) ++ ); ++ ++ drop(lease); ++ done_rx.recv_timeout(Duration::from_secs(1)).unwrap(); ++ rollback.join().unwrap(); ++ assert!(plane.get_process(child.pid()).is_none()); ++ assert!(parent.lock().children.is_empty()); ++ drop(child_main); ++ } ++ ++ #[test] ++ fn process_tree_barrier_waits_for_late_descendant_execution_quiescence() { ++ use std::{sync::mpsc, thread, time::Duration}; ++ ++ let plane = WasiControlPlane::default(); ++ let (root, _root_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let root_lease = root.acquire_execution_lease().unwrap(); ++ let (child, _child_main, mut child_registration) = plane ++ .new_child_process_with_main_thread_guarded( ++ &root, ++ ModuleHash::random(), ++ WasiMemoryLayout::default(), ++ ) ++ .unwrap(); ++ child_registration.commit_child().unwrap(); ++ let child_lease = child.acquire_execution_lease().unwrap(); ++ ++ root.terminate(Errno::Success.into()); ++ drop(root_lease); ++ ++ let waiting_plane = plane.clone(); ++ let waiting_root = root.clone(); ++ let (done_tx, done_rx) = mpsc::channel(); ++ let waiter = thread::spawn(move || { ++ futures::executor::block_on( ++ waiting_plane.wait_for_process_tree_quiescence(&waiting_root), ++ ); ++ done_tx.send(()).unwrap(); ++ }); ++ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ ++ // Publish a grandchild after the barrier's first registry snapshot. ++ // The terminal child cannot hide it: the exclusive epoch recheck must ++ // discover the new identity and wait for its execution lease too. ++ let (grandchild, _grandchild_main, mut grandchild_registration) = plane ++ .new_child_process_with_main_thread_guarded( ++ &child, ++ ModuleHash::random(), ++ WasiMemoryLayout::default(), ++ ) ++ .unwrap(); ++ grandchild_registration.commit_child().unwrap(); ++ let grandchild_lease = grandchild.acquire_execution_lease().unwrap(); ++ ++ child.terminate(Errno::Success.into()); ++ drop(child_lease); ++ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ ++ grandchild.terminate(Errno::Success.into()); ++ drop(grandchild_lease); ++ done_rx.recv_timeout(Duration::from_secs(1)).unwrap(); ++ waiter.join().unwrap(); ++ ++ grandchild_registration.rollback_child(Errno::Success.into()); ++ child_registration.rollback_child(Errno::Success.into()); ++ } ++ ++ #[test] ++ fn process_join_waits_for_pending_child_publication_ownership() { ++ use std::{sync::mpsc, thread, time::Duration}; ++ ++ let plane = WasiControlPlane::default(); ++ let (process, _main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let publication = process.begin_child_publication().unwrap(); ++ process.terminate(Errno::Success.into()); ++ assert!(process.try_join().is_none()); ++ ++ let waiting_process = process.clone(); ++ let (done_tx, done_rx) = mpsc::channel(); ++ let waiter = thread::spawn(move || { ++ futures::executor::block_on(waiting_process.join()).unwrap(); ++ done_tx.send(()).unwrap(); ++ }); ++ assert!(done_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ ++ drop(publication); ++ assert!(process.try_join().is_some()); ++ done_rx.recv_timeout(Duration::from_secs(1)).unwrap(); ++ waiter.join().unwrap(); ++ assert_eq!(process.lock().pending_child_publications, 0); ++ } ++ ++ #[test] ++ fn execution_guard_publishes_terminal_before_quiescence() { ++ use std::{sync::mpsc, thread, time::Duration}; ++ ++ let plane = WasiControlPlane::default(); ++ let (process, _main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let guard = process.acquire_execution_guard().unwrap(); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ let observer = process.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let waiter = thread::spawn(move || { ++ observer.wait_for_execution_quiescence_blocking(); ++ observed_tx.send(observer.try_join()).unwrap(); ++ }); ++ ++ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ guard.finish(Ok(7u16.into())); ++ ++ let observed = observed_rx.recv_timeout(Duration::from_secs(1)).unwrap(); ++ assert_eq!( ++ observed.expect("process never became joinable").unwrap(), ++ 7u16.into() ++ ); ++ waiter.join().unwrap(); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn panicking_monitor_manager_terminates_and_reaps_before_quiescence() { ++ use std::{future::Future, pin::Pin, sync::mpsc, thread, time::Duration}; ++ ++ use crate::{ ++ WasiThreadError, ++ os::task::{OwnedTaskStatus, signal::SignalHandlerAbi}, ++ runtime::task_manager::{TaskWasm, VirtualTaskManager}, ++ }; ++ use wasmer::FromToNativeWasmType; ++ use wasmer_wasix_types::wasi::Signal; ++ ++ #[derive(Debug)] ++ struct FinishingSignalHandler { ++ sender: mpsc::Sender, ++ } ++ ++ impl SignalHandlerAbi for FinishingSignalHandler { ++ fn signal( ++ &self, ++ signal: u8, ++ ) -> Result<(), crate::os::task::signal::SignalDeliveryError> { ++ self.sender ++ .send(signal) ++ .map_err(|_| crate::os::task::signal::SignalDeliveryError) ++ } ++ } ++ ++ #[derive(Debug)] ++ struct PanickingMonitorTaskManager; ++ ++ impl VirtualTaskManager for PanickingMonitorTaskManager { ++ fn sleep_now( ++ &self, ++ _time: Duration, ++ ) -> Pin + Send + Sync + 'static>> { ++ Box::pin(async {}) ++ } ++ ++ fn task_shared( ++ &self, ++ task: Box futures::future::BoxFuture<'static, ()> + Send + 'static>, ++ ) -> Result<(), WasiThreadError> { ++ // Model a custom manager that constructs, then cancels, the ++ // admitted watcher before panicking out of task admission. ++ drop(task()); ++ panic!("synthetic process-monitor admission panic") ++ } ++ ++ fn task_wasm(&self, _task: TaskWasm) -> Result<(), WasiThreadError> { ++ unreachable!("process monitoring does not schedule Wasm") ++ } ++ ++ fn task_dedicated( ++ &self, ++ _task: Box, ++ ) -> Result<(), WasiThreadError> { ++ unreachable!("process monitoring does not use a dedicated task") ++ } ++ ++ fn thread_parallelism(&self) -> Result { ++ Ok(1) ++ } ++ } ++ ++ let plane = WasiControlPlane::default(); ++ let (process, _main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let execution = process.acquire_execution_guard().unwrap(); ++ ++ let (signal_tx, signal_rx) = mpsc::channel(); ++ let mut command = OwnedTaskStatus::default(); ++ command.set_signal_handler(Arc::new(FinishingSignalHandler { sender: signal_tx })); ++ let command = Arc::new(command); ++ let command_handle = command.handle(); ++ let command_finisher = command.clone(); ++ let finisher = thread::spawn(move || { ++ assert_eq!( ++ signal_rx.recv_timeout(Duration::from_secs(1)).unwrap(), ++ Signal::Sigkill.to_native() as u8 ++ ); ++ command_finisher.set_finished(Ok(9u16.into())); ++ }); ++ ++ let observed_process = process.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = thread::spawn(move || { ++ observed_process.wait_for_execution_quiescence_blocking(); ++ let _ = observed_tx.send(observed_process.try_join()); ++ }); ++ ++ let tasks: Arc = Arc::new(PanickingMonitorTaskManager); ++ assert!(matches!( ++ execution.monitor(command_handle, &tasks), ++ Err(WasiThreadError::InvalidWasmContext) ++ )); ++ ++ finisher.join().unwrap(); ++ let status = observed_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .unwrap() ++ .expect("monitor released quiescence before publishing terminal status"); ++ assert_eq!(status.unwrap(), Errno::Canceled.into()); ++ assert_eq!( ++ command ++ .status() ++ .into_finished() ++ .expect("authoritative command handle was not reaped") ++ .unwrap(), ++ 9u16.into() ++ ); ++ assert_eq!(process.lock().execution_leases, 0); ++ observer.join().unwrap(); ++ } ++ ++ #[test] ++ fn abandoned_host_execution_fails_closed_only_after_last_guard_clone() { ++ let plane = WasiControlPlane::default(); ++ let (process, _main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let guard = process.acquire_execution_guard().unwrap(); ++ let successor = guard.clone(); ++ ++ drop(guard); ++ assert!(process.try_join().is_none()); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ drop(successor); ++ assert_eq!(process.lock().execution_leases, 0); ++ assert_eq!( ++ process ++ .try_join() ++ .expect("abandoned execution never became joinable") ++ .unwrap(), ++ Errno::Canceled.into() ++ ); ++ } ++ ++ #[test] ++ fn supplemental_parent_guard_requires_an_accepted_successor() { ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let thread = main.as_thread(); ++ let owner = Arc::new(()); ++ let owner_weak = Arc::downgrade(&owner); ++ let task_wasm_lease = process.acquire_execution_lease().unwrap(); ++ let parent_guard = process ++ .acquire_supplemental_execution_guard(thread.clone(), owner_weak.clone()) ++ .unwrap(); ++ assert_eq!(process.lock().execution_leases, 2); ++ ++ assert!(parent_guard.try_handoff_to_current_task_wasm(&process, &thread, &owner_weak)); ++ assert_eq!(process.lock().execution_leases, 1); ++ assert!(process.finished.status().into_finished().is_none()); ++ ++ drop(task_wasm_lease); ++ assert_eq!(process.lock().execution_leases, 0); ++ assert!(process.finished.status().into_finished().is_none()); ++ } ++ ++ #[test] ++ fn repeated_parent_switch_guards_remain_bounded_without_a_task_successor() { ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let thread = main.as_thread(); ++ let owner = Arc::new(()); ++ let owner_weak = Arc::downgrade(&owner); ++ let task_wasm_lease = process.acquire_execution_lease().unwrap(); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ for _ in 0..2_000 { ++ let supplemental = process ++ .acquire_supplemental_execution_guard(thread.clone(), owner_weak.clone()) ++ .unwrap(); ++ assert!(supplemental.try_handoff_to_current_task_wasm(&process, &thread, &owner_weak)); ++ assert_eq!(process.lock().execution_leases, 1); ++ } ++ drop(task_wasm_lease); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn concurrent_guard_clones_have_one_linearizable_handoff_winner() { ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let thread = main.as_thread(); ++ let owner = Arc::new(()); ++ let owner_weak = Arc::downgrade(&owner); ++ let successor = process.acquire_execution_lease().unwrap(); ++ let guard = process ++ .acquire_supplemental_execution_guard(thread.clone(), owner_weak.clone()) ++ .unwrap(); ++ let barrier = Arc::new(std::sync::Barrier::new(3)); ++ let (result_tx, result_rx) = std::sync::mpsc::channel(); ++ let mut workers = Vec::new(); ++ ++ for candidate in [guard.clone(), guard] { ++ let process = process.clone(); ++ let thread = thread.clone(); ++ let owner_weak = owner_weak.clone(); ++ let barrier = barrier.clone(); ++ let result_tx = result_tx.clone(); ++ workers.push(std::thread::spawn(move || { ++ barrier.wait(); ++ result_tx ++ .send(candidate.try_handoff_to_current_task_wasm( ++ &process, ++ &thread, ++ &owner_weak, ++ )) ++ .unwrap(); ++ })); ++ } ++ barrier.wait(); ++ let outcomes = [result_rx.recv().unwrap(), result_rx.recv().unwrap()]; ++ assert_eq!(outcomes.into_iter().filter(|won| *won).count(), 1); ++ for worker in workers { ++ worker.join().unwrap(); ++ } ++ assert_eq!(process.lock().execution_leases, 1); ++ drop(successor); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn unrelated_task_or_thread_cannot_authorize_parent_guard_handoff() { ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let main_thread = main.as_thread(); ++ let other_thread = process ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) ++ .unwrap(); ++ let owner = Arc::new(()); ++ let wrong_owner = Arc::new(()); ++ let owner_weak = Arc::downgrade(&owner); ++ let guard = process ++ .acquire_supplemental_execution_guard(main_thread.clone(), owner_weak.clone()) ++ .unwrap(); ++ ++ assert!(!guard.try_handoff_to_current_task_wasm( ++ &process, ++ &main_thread, ++ &Arc::downgrade(&wrong_owner) ++ )); ++ assert!(!guard.try_handoff_to_accepted_task_wasm(&process, &other_thread)); ++ assert_eq!(process.lock().execution_leases, 1); ++ assert!(guard.try_handoff_to_current_task_wasm(&process, &main_thread, &owner_weak)); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn committed_vfork_parent_guard_fails_closed_before_quiescence() { ++ use std::{sync::mpsc, thread, time::Duration}; ++ ++ let plane = WasiControlPlane::default(); ++ let (parent, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let owner = Arc::new(()); ++ let original_execution = parent.acquire_execution_lease().unwrap(); ++ let parent_vfork = parent ++ .acquire_supplemental_execution_guard(main.as_thread(), Arc::downgrade(&owner)) ++ .unwrap(); ++ parent_vfork.arm_fail_closed(); ++ drop(original_execution); ++ ++ let observer_process = parent.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = thread::spawn(move || { ++ observer_process.wait_for_execution_quiescence_blocking(); ++ let _ = observed_tx.send(observer_process.try_join()); ++ }); ++ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ ++ drop(parent_vfork); ++ let status = observed_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .unwrap() ++ .expect("committed vfork parent became quiescent before terminal status"); ++ assert_eq!(status.unwrap(), Errno::Canceled.into()); ++ assert_eq!(parent.lock().execution_leases, 0); ++ observer.join().unwrap(); ++ } ++ ++ #[test] ++ fn abandoned_non_main_vfork_owner_terminalizes_whole_process_before_quiescence() { ++ use std::{sync::mpsc, thread, time::Duration}; ++ ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let worker = process ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) ++ .unwrap(); ++ let owner = Arc::new(()); ++ let original_execution = process.acquire_execution_lease().unwrap(); ++ let vfork_owner = process ++ .acquire_supplemental_execution_guard(worker.as_thread(), Arc::downgrade(&owner)) ++ .unwrap(); ++ vfork_owner.arm_fail_closed(); ++ drop(original_execution); ++ ++ let observer_process = process.clone(); ++ let observer_main = main.as_thread(); ++ let observer_worker = worker.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = thread::spawn(move || { ++ observer_process.wait_for_execution_quiescence_blocking(); ++ observed_tx ++ .send((observer_main.try_join(), observer_worker.try_join())) ++ .unwrap(); ++ }); ++ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ ++ drop(vfork_owner); ++ let (main_status, worker_status) = ++ observed_rx.recv_timeout(Duration::from_secs(1)).unwrap(); ++ assert_eq!(main_status.unwrap().unwrap(), Errno::Canceled.into()); ++ assert_eq!(worker_status.unwrap().unwrap(), Errno::Canceled.into()); ++ assert_eq!(process.lock().execution_leases, 0); ++ observer.join().unwrap(); ++ } ++ ++ #[test] ++ fn vfork_parent_and_child_ownership_coexist_across_deep_sleep_handoffs() { ++ let plane = WasiControlPlane::default(); ++ let (parent, parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, child_main, mut registration) = plane ++ .new_child_process_with_main_thread_guarded( ++ &parent, ++ ModuleHash::random(), ++ WasiMemoryLayout::default(), ++ ) ++ .unwrap(); ++ registration.commit_child().unwrap(); ++ ++ // The currently executing TaskWasm and the supplemental vfork guard ++ // both own the suspended parent while the in-place child has its own ++ // fail-closed guard. ++ let parent_owner = Arc::new(()); ++ let parent_task = parent.acquire_execution_lease().unwrap(); ++ let parent_vfork = parent ++ .acquire_supplemental_execution_guard( ++ parent_main.as_thread(), ++ Arc::downgrade(&parent_owner), ++ ) ++ .unwrap(); ++ let child_vfork = child.acquire_execution_guard().unwrap(); ++ assert_eq!(parent.lock().execution_leases, 2); ++ assert_eq!(child.lock().execution_leases, 1); ++ ++ // A deep-sleep child successor acquires before the old callback gives ++ // up the parent TaskWasm lease. ++ let child_successor = child.acquire_execution_lease().unwrap(); ++ assert_eq!(child.lock().execution_leases, 2); ++ drop(parent_task); ++ assert_eq!(parent.lock().execution_leases, 1); ++ ++ // Restoring the parent terminalizes the child in-place ownership, but ++ // the accepted child successor remains until its callback returns. ++ child_vfork.finish(Ok(Errno::Success.into())); ++ assert_eq!(child.lock().execution_leases, 1); ++ assert!(child.try_join().is_none()); ++ ++ // Parent resume acquires its successor before the supplemental guard ++ // follows the restored environment to terminal cleanup. ++ let parent_successor = parent.acquire_execution_lease().unwrap(); ++ assert_eq!(parent.lock().execution_leases, 2); ++ drop(child_successor); ++ assert!(child.try_join().is_some()); ++ ++ parent.terminate(Errno::Success.into()); ++ drop(parent_vfork); ++ assert_eq!(parent.lock().execution_leases, 1); ++ assert!(parent.try_join().is_none()); ++ drop(parent_successor); ++ assert!(parent.try_join().is_some()); ++ ++ registration.rollback_child(Errno::Success.into()); ++ drop(child_main); ++ } + } +diff --git a/lib/wasix/src/os/task/mod.rs b/lib/wasix/src/os/task/mod.rs +index 755645f..e49fbc2 100644 +--- a/lib/wasix/src/os/task/mod.rs ++++ b/lib/wasix/src/os/task/mod.rs +@@ -9,6 +9,16 @@ pub mod thread; + + #[allow(unused_imports)] + pub(crate) use process::WasiProcessInner; ++pub(crate) use task_join_handle::terminate_and_reap_abandoned_task; ++#[cfg(all(feature = "ctrlc", unix))] + pub use task_join_handle::{ +- OwnedTaskStatus, TaskJoinHandle, TaskStatus, TaskTerminatedError, VirtualTaskHandle, ++ HostLifecycleSupervisor, UnixHostLifecycleError, UnixHostLifecycleSupervisor, ++}; ++#[cfg(all(feature = "ctrlc", windows))] ++pub use task_join_handle::{ ++ HostLifecycleSupervisor, WindowsHostLifecycleError, WindowsHostLifecycleSupervisor, ++}; ++pub use task_join_handle::{ ++ OwnedTaskStatus, TaskJoinHandle, TaskSignalController, TaskSignalError, TaskStatus, ++ TaskTerminatedError, VirtualTaskHandle, + }; +diff --git a/lib/wasix/src/os/task/process.rs b/lib/wasix/src/os/task/process.rs +index f1d0f12..f83b6b5 100644 +--- a/lib/wasix/src/os/task/process.rs ++++ b/lib/wasix/src/os/task/process.rs +@@ -1,18 +1,18 @@ +-use crate::{WasiEnv, WasiRuntimeError, journal::SnapshotTrigger}; ++use crate::{WasiEnv, WasiRuntimeError, journal::SnapshotTrigger, runtime::VirtualTaskManager}; + #[cfg(feature = "journal")] + use crate::{WasiResult, journal::JournalEffector, syscalls::do_checkpoint_from_outside, unwind}; + use serde::{Deserialize, Serialize}; +-#[cfg(feature = "journal")] +-use std::collections::HashSet; + use std::{ +- collections::HashMap, ++ collections::{HashMap, HashSet}, + convert::TryInto, ++ future::Future, + ops::Range, ++ pin::Pin, + sync::{ + Arc, Condvar, Mutex, MutexGuard, RwLock, Weak, +- atomic::{AtomicU32, Ordering}, ++ atomic::{AtomicBool, AtomicU32, Ordering}, + }, +- task::Waker, ++ task::{Context, Poll, Waker}, + time::Duration, + }; + use tracing::trace; +@@ -34,8 +34,11 @@ use super::{ + backoff::WasiProcessCpuBackoff, + control_plane::{ControlPlaneError, WasiControlPlaneHandle}, + signal::{SignalDeliveryError, SignalHandlerAbi}, +- task_join_handle::OwnedTaskStatus, +- thread::WasiMemoryLayout, ++ task_join_handle::{ ++ OwnedTaskStatus, TaskAbandonCallback, TaskCompletionCallback, TaskJoinHandle, ++ terminate_and_reap_abandoned_task, ++ }, ++ thread::{WasiMemoryLayout, WasiThreadError}, + }; + + /// Represents the ID of a sub-process +@@ -86,6 +89,121 @@ impl std::fmt::Debug for WasiProcessId { + + pub type LockableWasiProcessInner = Arc<(Mutex, Condvar)>; + ++/// Serializes process-tree admission and topology changes against reusable ++/// epoch retirement. Every descendant inherits the same gate. ++pub(crate) type WasiProcessTreeEpoch = Arc>; ++ ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++enum WasiProcessStartState { ++ Pending, ++ Committed, ++ Aborted, ++} ++ ++#[derive(Debug)] ++struct WasiProcessStartInner { ++ state: WasiProcessStartState, ++} ++ ++/// A one-shot publication barrier for a tentative child. Task admission is ++/// rejected while pending, because Wasmer instantiation itself may execute a ++/// guest start function. Adoption and registry publication commit this gate ++/// before any TaskWasm can be constructed. ++#[derive(Debug, Clone)] ++pub(crate) struct WasiProcessStartGate { ++ inner: Arc>, ++} ++ ++impl WasiProcessStartGate { ++ fn new(state: WasiProcessStartState) -> Self { ++ Self { ++ inner: Arc::new(Mutex::new(WasiProcessStartInner { state })), ++ } ++ } ++ ++ fn pending() -> Self { ++ Self::new(WasiProcessStartState::Pending) ++ } ++ ++ fn committed() -> Self { ++ Self::new(WasiProcessStartState::Committed) ++ } ++ ++ fn transition(&self, next: WasiProcessStartState) { ++ let mut inner = self.inner.lock().unwrap(); ++ if inner.state != WasiProcessStartState::Pending { ++ return; ++ } ++ inner.state = next; ++ } ++ ++ pub(crate) fn commit(&self) { ++ self.transition(WasiProcessStartState::Committed); ++ } ++ ++ pub(crate) fn abort(&self) { ++ self.transition(WasiProcessStartState::Aborted); ++ } ++ ++ pub(crate) fn is_committed(&self) -> bool { ++ self.inner.lock().unwrap().state == WasiProcessStartState::Committed ++ } ++} ++ ++/// Keeps a process non-reapable while accepted Wasm work can still execute. ++/// It deliberately outlives terminal status and is released only after the ++/// final callback (or cancellation/drop) has completed. ++#[derive(Debug)] ++pub struct WasiProcessExecutionLease { ++ inner: Weak<(Mutex, Condvar)>, ++} ++ ++struct WasiProcessQuiescenceFuture { ++ inner: LockableWasiProcessInner, ++} ++ ++impl Future for WasiProcessQuiescenceFuture { ++ type Output = (); ++ ++ fn poll(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll { ++ let mut inner = self.inner.0.lock().unwrap(); ++ if inner.execution_leases == 0 && inner.pending_child_publications == 0 { ++ return Poll::Ready(()); ++ } ++ if !inner ++ .quiescence_wakers ++ .iter() ++ .any(|waker| waker.will_wake(cx.waker())) ++ { ++ inner.quiescence_wakers.push(cx.waker().clone()); ++ } ++ Poll::Pending ++ } ++} ++ ++impl Drop for WasiProcessExecutionLease { ++ fn drop(&mut self) { ++ let Some(inner) = self.inner.upgrade() else { ++ return; ++ }; ++ let mut state = inner.0.lock().unwrap(); ++ state.execution_leases = state ++ .execution_leases ++ .checked_sub(1) ++ .expect("process execution lease underflow"); ++ notify_process_quiescence_if_ready(&mut state, &inner.1); ++ } ++} ++ ++fn notify_process_quiescence_if_ready(state: &mut WasiProcessInner, condvar: &Condvar) { ++ if state.execution_leases == 0 && state.pending_child_publications == 0 { ++ for waker in state.quiescence_wakers.drain(..) { ++ waker.wake(); ++ } ++ condvar.notify_all(); ++ } ++} ++ + /// Represents a process running within the compute state + /// TODO: fields should be private and only accessed via methods. + #[derive(Debug, Clone)] +@@ -95,16 +213,24 @@ pub struct WasiProcess { + /// Hash of the module that this process is using + pub(crate) module_hash: ModuleHash, + /// List of all the children spawned from this thread +- pub(crate) parent: Option>>, ++ pub(crate) parent: Option, Condvar)>>, + /// The inner protected region of the process with a conditional + /// variable that is used for coordination such as snapshots. + pub(crate) inner: LockableWasiProcessInner, ++ /// Shared by this process and every descendant. The lock order is tree ++ /// epoch, process inner, then the control-plane registry. ++ pub(crate) tree_epoch: WasiProcessTreeEpoch, ++ /// Child publication barrier inherited by every TaskWasm for this process. ++ pub(crate) start_gate: WasiProcessStartGate, + /// Reference back to the compute engine + // TODO: remove this reference, access should happen via separate state instead + // (we don't want cyclical references) + pub(crate) compute: WasiControlPlaneHandle, + /// Reference to the exit code for the main thread + pub(crate) finished: Arc, ++ /// POSIX wait status is consumable once, even when multiple parent waiters ++ /// race through clones of the same process handle. ++ pub(crate) join_claimed: Arc, + /// Number of threads waiting for children to exit + pub(crate) waiting: Arc, + /// Number of tokens that are currently active and thus +@@ -113,6 +239,231 @@ pub struct WasiProcess { + pub(crate) cpu_run_tokens: Arc, + } + ++/// Execution ownership for host commands and vfork continuations that do not ++/// naturally live inside one `TaskWasm`. Clones share one lease. Completing ++/// the associated task publishes terminal status before releasing that lease; ++/// abandoning the guard fails closed with `Canceled`. ++#[derive(Clone, Debug)] ++pub struct WasiProcessExecutionGuard { ++ inner: Arc, ++} ++ ++#[derive(Debug)] ++struct WasiProcessExecutionGuardInner { ++ process: WasiProcess, ++ lease: Mutex>, ++ fail_closed: AtomicBool, ++ handoff_owner: Option, ++} ++ ++#[derive(Debug)] ++struct WasiProcessExecutionHandoffOwner { ++ thread: WasiThread, ++ task_wasm_owner: Weak<()>, ++} ++ ++impl WasiProcessExecutionGuardInner { ++ fn finish(&self, status: Result>) { ++ let lease = self.lease.lock().unwrap().take(); ++ if lease.is_none() { ++ return; ++ } ++ self.process.finished.set_finished(status); ++ drop(lease); ++ } ++} ++ ++impl Drop for WasiProcessExecutionGuardInner { ++ fn drop(&mut self) { ++ let lease = self.lease.get_mut().unwrap().take(); ++ if lease.is_none() { ++ return; ++ } ++ if self.fail_closed.load(Ordering::Acquire) { ++ if let Some(owner) = &self.handoff_owner { ++ owner.thread.set_status_finished(Ok(Errno::Canceled.into())); ++ } ++ // Losing an armed vfork owner is process-terminal. Mark every ++ // registered thread before releasing the final quiescence lease; ++ // the exact recorded thread above also covers a concurrent ++ // registry removal. ++ self.process.terminate(Errno::Canceled.into()); ++ self.process ++ .finished ++ .set_finished(Ok(Errno::Canceled.into())); ++ } ++ drop(lease); ++ } ++} ++ ++impl WasiProcessExecutionGuard { ++ fn new( ++ process: WasiProcess, ++ lease: WasiProcessExecutionLease, ++ fail_closed: bool, ++ handoff_owner: Option, ++ ) -> Self { ++ Self { ++ inner: Arc::new(WasiProcessExecutionGuardInner { ++ process, ++ lease: Mutex::new(Some(lease)), ++ fail_closed: AtomicBool::new(fail_closed), ++ handoff_owner, ++ }), ++ } ++ } ++ ++ pub(crate) fn finish(&self, status: Result>) { ++ self.inner.finish(status); ++ } ++ ++ /// Converts a setup-only supplemental guard into authoritative execution ++ /// ownership once an in-place vfork commits. Before commit, rollback can ++ /// hand it back to the original TaskWasm; after commit, abandoning the ++ /// stored/restored parent environment must fail closed. ++ pub(crate) fn arm_fail_closed(&self) { ++ self.inner.fail_closed.store(true, Ordering::Release); ++ } ++ ++ pub(crate) fn matches_process_thread( ++ &self, ++ process: &WasiProcess, ++ thread: &WasiThread, ++ ) -> bool { ++ self.inner.process.same_identity(process) ++ && self ++ .inner ++ .handoff_owner ++ .as_ref() ++ .is_some_and(|owner| owner.thread.same_identity(thread)) ++ } ++ ++ fn release_handoff_lease(&self) -> bool { ++ let mut lease = self.inner.lease.lock().unwrap(); ++ let Some(current) = lease.take() else { ++ return false; ++ }; ++ drop(lease); ++ drop(current); ++ true ++ } ++ ++ /// Releases a supplemental owner only when the currently executing ++ /// physical TaskWasm is exactly the owner recorded before the process ++ /// switch. Process-wide lease counts cannot authorize this transition. ++ pub(crate) fn try_handoff_to_current_task_wasm( ++ &self, ++ process: &WasiProcess, ++ thread: &WasiThread, ++ current_owner: &Weak<()>, ++ ) -> bool { ++ let Some(expected) = self.inner.handoff_owner.as_ref() else { ++ return false; ++ }; ++ if !self.matches_process_thread(process, thread) ++ || !Weak::ptr_eq(&expected.task_wasm_owner, current_owner) ++ || current_owner.upgrade().is_none() ++ { ++ return false; ++ } ++ self.release_handoff_lease() ++ } ++ ++ /// An accepted successor may adopt deferred ownership for the exact same ++ /// guest thread. The caller holds the successor guard while invoking this ++ /// method, so releasing the predecessor is successor-before-predecessor. ++ pub(crate) fn try_handoff_to_accepted_task_wasm( ++ &self, ++ process: &WasiProcess, ++ thread: &WasiThread, ++ ) -> bool { ++ self.matches_process_thread(process, thread) && self.release_handoff_lease() ++ } ++ ++ /// Retains execution ownership until the command's authoritative task ++ /// status becomes terminal. The task-manager acceptance boundary is the ++ /// handoff: rejection, panic, or cancellation transfers the command and ++ /// its execution guard to an independent reaper. ++ pub(crate) fn monitor( ++ self, ++ handle: TaskJoinHandle, ++ tasks: &Arc, ++ ) -> Result<(), WasiThreadError> { ++ let watcher = ProcessMonitorLifecycleGuard::new(handle, self); ++ let admission = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ tasks.task_shared(Box::new(move || { ++ Box::pin(async move { watcher.wait_finished().await }) ++ })) ++ })); ++ match admission { ++ Ok(result) => result, ++ Err(_) => { ++ tracing::error!("task manager panicked while admitting a process monitor"); ++ Err(WasiThreadError::InvalidWasmContext) ++ } ++ } ++ } ++} ++ ++/// Owns both sides of an admitted command-monitor relationship. ++/// ++/// A task manager may reject or panic while consuming the closure, or cancel ++/// the returned future later. Dropping a `TaskJoinHandle` is only a detach, so ++/// every abnormal path must transfer the exact handle and execution lease to a ++/// stable host reaper. ++struct ProcessMonitorLifecycleGuard { ++ task: Option, ++ execution: Option, ++} ++ ++impl ProcessMonitorLifecycleGuard { ++ fn new(task: TaskJoinHandle, execution: WasiProcessExecutionGuard) -> Self { ++ Self { ++ task: Some(task), ++ execution: Some(execution), ++ } ++ } ++ ++ async fn wait_finished(mut self) { ++ let Some(task) = self.task.as_mut() else { ++ return; ++ }; ++ let status = task.wait_finished().await; ++ if let Some(execution) = self.execution.take() { ++ execution.finish(status); ++ } ++ // Normal completion observed the authoritative status and published ++ // process terminal state before releasing the lease. Disarm so the hot ++ // path never creates a reaper thread. ++ self.task.take(); ++ } ++} ++ ++impl Drop for ProcessMonitorLifecycleGuard { ++ fn drop(&mut self) { ++ let Some(task) = self.task.take() else { ++ return; ++ }; ++ let process = self ++ .execution ++ .as_ref() ++ .map(|execution| execution.inner.process.clone()); ++ let abandon = process.map(|process| { ++ Box::new(move || process.terminate(Errno::Canceled.into())) as TaskAbandonCallback ++ }); ++ let completion = self.execution.take().map(|execution| { ++ Box::new(move |status| execution.finish(status)) as TaskCompletionCallback ++ }); ++ terminate_and_reap_abandoned_task( ++ task, ++ "wasmer-process-monitor-reaper", ++ "process monitor", ++ abandon, ++ completion, ++ ); ++ } ++} ++ + /// Represents a freeze of all threads to perform some action + /// on the total state-machine. This is normally done for + /// things like snapshots which require the memory to remain +@@ -165,6 +516,16 @@ pub struct WasiProcessInner { + pub signal_intervals: HashMap, + /// List of all the children spawned from this thread + pub children: Vec, ++ /// Prevents new threads/children from entering an execution epoch once a ++ /// reusable environment has begun terminal retirement. ++ pub(crate) retiring: bool, ++ /// Fork/spawn operations that have reserved the right to publish a child ++ /// but have not yet completed parent adoption. ++ pub(crate) pending_child_publications: u32, ++ /// Accepted Wasm tasks which have not reached their final callback/drop. ++ pub(crate) execution_leases: u32, ++ /// Joiners waiting for terminal execution quiescence. ++ pub(crate) quiescence_wakers: Vec, + /// Represents a checkpoint which blocks all the threads + /// and then executes some maintenance action + pub checkpoint: WasiProcessCheckpoint, +@@ -242,9 +603,12 @@ impl WasiProcessInner { + let thread_layout = ctx.data().thread.memory_layout().clone(); + unwind::(ctx, move |mut ctx, memory_stack, rewind_stack| { + // Grab all the globals and serialize them +- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) +- .serialize() +- .unwrap(); ++ let snapshot = ++ match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { ++ Ok(snapshot) => snapshot, ++ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), ++ }; ++ let store_data = snapshot.serialize().unwrap(); + let memory_stack = memory_stack.freeze(); + let rewind_stack = rewind_stack.freeze(); + let store_data = Bytes::from(store_data); +@@ -414,8 +778,84 @@ impl Drop for WasiProcessWait { + } + } + ++/// Holds one parent-side child publication reservation across process ++/// construction, registry commit, and exact parent adoption. ++#[derive(Debug)] ++pub(crate) struct WasiChildPublicationGuard { ++ parent: WasiProcess, ++ child_identity: Option, ++ active: bool, ++} ++ ++impl Drop for WasiChildPublicationGuard { ++ fn drop(&mut self) { ++ if !self.active { ++ return; ++ } ++ let _epoch = self.parent.tree_epoch.read().unwrap(); ++ let mut inner = self.parent.inner.0.lock().unwrap(); ++ inner.pending_child_publications = inner ++ .pending_child_publications ++ .checked_sub(1) ++ .expect("pending child publication count underflow"); ++ notify_process_quiescence_if_ready(&mut inner, &self.parent.inner.1); ++ } ++} ++ ++impl WasiChildPublicationGuard { ++ pub(crate) fn bind_child(mut self, child: &WasiProcess) -> Self { ++ assert!( ++ self.child_identity.is_none(), ++ "child publication bound twice" ++ ); ++ self.child_identity = Some(Arc::as_ptr(&child.inner) as usize); ++ self ++ } ++ ++ fn matches(&self, parent: &WasiProcess, child: &WasiProcess) -> bool { ++ self.active ++ && self.parent.same_identity(parent) ++ && self.child_identity == Some(Arc::as_ptr(&child.inner) as usize) ++ } ++} ++ + impl WasiProcess { + pub fn new(pid: WasiProcessId, module_hash: ModuleHash, plane: WasiControlPlaneHandle) -> Self { ++ Self::new_with_lifecycle( ++ pid, ++ module_hash, ++ plane, ++ None, ++ Arc::new(RwLock::new(())), ++ WasiProcessStartGate::committed(), ++ ) ++ } ++ ++ pub(crate) fn new_pending_child( ++ pid: WasiProcessId, ++ module_hash: ModuleHash, ++ plane: WasiControlPlaneHandle, ++ parent: &WasiProcess, ++ tree_epoch: WasiProcessTreeEpoch, ++ ) -> Self { ++ Self::new_with_lifecycle( ++ pid, ++ module_hash, ++ plane, ++ Some(Arc::downgrade(&parent.inner)), ++ tree_epoch, ++ WasiProcessStartGate::pending(), ++ ) ++ } ++ ++ fn new_with_lifecycle( ++ pid: WasiProcessId, ++ module_hash: ModuleHash, ++ plane: WasiControlPlaneHandle, ++ parent: Option, Condvar)>>, ++ tree_epoch: WasiProcessTreeEpoch, ++ start_gate: WasiProcessStartGate, ++ ) -> Self { + let max_cpu_backoff_time = plane + .upgrade() + .and_then(|p| p.config().enable_exponential_cpu_backoff) +@@ -430,6 +870,10 @@ impl WasiProcess { + thread_count: Default::default(), + signal_intervals: Default::default(), + children: Default::default(), ++ retiring: false, ++ pending_child_publications: 0, ++ execution_leases: 0, ++ quiescence_wakers: Default::default(), + checkpoint: WasiProcessCheckpoint::Execute, + wakers: Default::default(), + waiting: waiting.clone(), +@@ -460,18 +904,22 @@ impl WasiProcess { + WasiProcess { + pid, + module_hash, +- parent: None, ++ parent, + compute: plane, + inner: inner.clone(), ++ tree_epoch, ++ start_gate, + finished: Arc::new( + OwnedTaskStatus::new(TaskStatus::Pending) + .with_signal_handler(Arc::new(SignalHandler(inner))), + ), ++ join_claimed: Arc::new(AtomicBool::new(false)), + waiting, + cpu_run_tokens: Arc::new(AtomicU32::new(0)), + } + } + ++ #[cfg(test)] + pub(super) fn set_pid(&mut self, pid: WasiProcessId) { + self.pid = pid; + } +@@ -481,12 +929,126 @@ impl WasiProcess { + self.pid + } + ++ pub(crate) fn same_identity(&self, other: &Self) -> bool { ++ Arc::ptr_eq(&self.inner, &other.inner) ++ } ++ ++ pub(crate) fn same_tree(&self, other: &Self) -> bool { ++ Arc::ptr_eq(&self.tree_epoch, &other.tree_epoch) ++ } ++ ++ pub(crate) fn has_exact_parent(&self, parent: &Self) -> bool { ++ self.parent ++ .as_ref() ++ .and_then(Weak::upgrade) ++ .is_some_and(|candidate| Arc::ptr_eq(&candidate, &parent.inner)) ++ } ++ ++ pub(crate) fn tree_epoch(&self) -> WasiProcessTreeEpoch { ++ self.tree_epoch.clone() ++ } ++ ++ pub(crate) fn start_gate(&self) -> WasiProcessStartGate { ++ self.start_gate.clone() ++ } ++ ++ pub(crate) fn guest_start_is_committed(&self) -> bool { ++ self.start_gate.is_committed() ++ } ++ ++ pub(crate) fn commit_guest_start(&self) { ++ self.start_gate.commit(); ++ } ++ ++ pub(crate) fn abort_guest_start(&self) { ++ self.start_gate.abort(); ++ } ++ ++ pub(crate) fn acquire_execution_lease( ++ &self, ++ ) -> Result { ++ if !self.guest_start_is_committed() { ++ return Err(ControlPlaneError::ProcessNotPublished { ++ pid: self.pid().raw(), ++ }); ++ } ++ let _epoch = self.tree_epoch.read().unwrap(); ++ let mut inner = self.inner.0.lock().unwrap(); ++ if inner.retiring { ++ return Err(ControlPlaneError::ProcessRetiring { ++ pid: self.pid().raw(), ++ }); ++ } ++ if self.finished.status().into_finished().is_some() { ++ return Err(ControlPlaneError::ProcessFinished { ++ pid: self.pid().raw(), ++ }); ++ } ++ inner.execution_leases = inner ++ .execution_leases ++ .checked_add(1) ++ .expect("process execution lease count exhausted"); ++ Ok(WasiProcessExecutionLease { ++ inner: Arc::downgrade(&self.inner), ++ }) ++ } ++ ++ pub(crate) fn acquire_execution_guard( ++ &self, ++ ) -> Result { ++ let lease = self.acquire_execution_lease()?; ++ Ok(WasiProcessExecutionGuard::new( ++ self.clone(), ++ lease, ++ true, ++ None, ++ )) ++ } ++ ++ pub(crate) fn acquire_supplemental_execution_guard( ++ &self, ++ thread: WasiThread, ++ task_wasm_owner: Weak<()>, ++ ) -> Result { ++ let owns_thread = self ++ .inner ++ .0 ++ .lock() ++ .unwrap() ++ .threads ++ .values() ++ .any(|candidate| candidate.same_identity(&thread)); ++ if !owns_thread || task_wasm_owner.upgrade().is_none() { ++ return Err(ControlPlaneError::ExecutionOwnerUnavailable { ++ pid: self.pid().raw(), ++ tid: thread.tid().raw(), ++ }); ++ } ++ let lease = self.acquire_execution_lease()?; ++ Ok(WasiProcessExecutionGuard::new( ++ self.clone(), ++ lease, ++ false, ++ Some(WasiProcessExecutionHandoffOwner { ++ thread, ++ task_wasm_owner, ++ }), ++ )) ++ } ++ ++ pub(crate) fn wait_for_execution_quiescence_blocking(&self) { ++ let mut inner = self.inner.0.lock().unwrap(); ++ while inner.execution_leases != 0 { ++ inner = self.inner.1.wait(inner).unwrap(); ++ } ++ } ++ + /// Gets the process ID of the parent process + pub fn ppid(&self) -> WasiProcessId { + self.parent + .iter() + .filter_map(|parent| parent.upgrade()) +- .map(|parent| parent.read().unwrap().pid) ++ .map(|parent| parent.0.lock().unwrap().pid) + .next() + .unwrap_or(WasiProcessId(0)) + } +@@ -528,12 +1090,30 @@ impl WasiProcess { + tid: WasiThreadId, + ) -> Result { + let control_plane = self.compute.must_upgrade(); +- let task_count_guard = control_plane.register_task()?; +- + let is_main = matches!(start, ThreadStartType::MainThread); + +- // The wait finished should be the process version if its the main thread ++ // Numeric TID ownership and insertion linearize under the shared tree ++ // admission gate followed by the process lock. ++ // Task admission itself is an atomic reservation and rolls back if any ++ // construction below returns before the handle is published. ++ let _epoch = self.tree_epoch.read().unwrap(); + let mut inner = self.inner.0.lock().unwrap(); ++ if inner.retiring { ++ return Err(ControlPlaneError::ProcessRetiring { ++ pid: self.pid().raw(), ++ }); ++ } ++ if self.finished.status().into_finished().is_some() { ++ return Err(ControlPlaneError::ProcessFinished { ++ pid: self.pid().raw(), ++ }); ++ } ++ if inner.threads.contains_key(&tid) { ++ return Err(ControlPlaneError::DuplicateThreadId { tid: tid.raw() }); ++ } ++ let task_count_guard = control_plane.register_task()?; ++ ++ // The wait finished should be the process version if its the main thread + let finished = if is_main { + self.finished.clone() + } else { +@@ -551,7 +1131,10 @@ impl WasiProcess { + start, + ); + inner.threads.insert(tid, ctrl.clone()); +- inner.thread_count += 1; ++ inner.thread_count = inner ++ .thread_count ++ .checked_add(1) ++ .expect("process thread count exhausted"); + + Ok(WasiThreadHandle::new(ctrl, &self.inner)) + } +@@ -597,6 +1180,248 @@ impl WasiProcess { + signal_process_internal(&self.inner, signal); + } + ++ /// Adds and publishes the exact child authorized by `publication` while ++ /// holding the parent epoch lock. Retirement therefore linearizes either ++ /// before the reservation or after both the parent link and registry entry. ++ pub(crate) fn adopt_child_process( ++ &self, ++ child: WasiProcess, ++ mut publication: WasiChildPublicationGuard, ++ publish: impl FnOnce(), ++ ) -> Result<(), ControlPlaneError> { ++ if !publication.matches(self, &child) { ++ return Err(ControlPlaneError::InvalidChildPublication { ++ parent_pid: self.pid().raw(), ++ child_pid: child.pid().raw(), ++ }); ++ } ++ if !self.same_tree(&child) || !child.has_exact_parent(self) { ++ return Err(ControlPlaneError::InvalidChildPublication { ++ parent_pid: self.pid().raw(), ++ child_pid: child.pid().raw(), ++ }); ++ } ++ ++ let _epoch = self.tree_epoch.read().unwrap(); ++ let mut inner = self.inner.0.lock().unwrap(); ++ if inner.retiring { ++ drop(inner); ++ return Err(ControlPlaneError::ProcessRetiring { ++ pid: self.pid().raw(), ++ }); ++ } ++ if self.finished.status().into_finished().is_some() { ++ drop(inner); ++ return Err(ControlPlaneError::ProcessFinished { ++ pid: self.pid().raw(), ++ }); ++ } ++ if inner.pending_child_publications == 0 ++ || inner ++ .children ++ .iter() ++ .any(|candidate| candidate.same_identity(&child)) ++ { ++ drop(inner); ++ return Err(ControlPlaneError::InvalidChildPublication { ++ parent_pid: self.pid().raw(), ++ child_pid: child.pid().raw(), ++ }); ++ } ++ ++ inner.pending_child_publications -= 1; ++ publication.active = false; ++ inner.children.push(child); ++ publish(); ++ notify_process_quiescence_if_ready(&mut inner, &self.inner.1); ++ Ok(()) ++ } ++ ++ /// Arranges POSIX-style child-exit notification after the child ++ /// publication transaction has become externally visible. ++ pub(crate) fn notify_on_child_exit( ++ &self, ++ child: WasiProcess, ++ tasks: &Arc, ++ ) { ++ let parent = self.clone(); ++ let parent_pid = parent.pid(); ++ let child_pid = child.pid(); ++ if let Err(err) = tasks.task_shared(Box::new(move || { ++ Box::pin(async move { ++ let _ = child.join().await; ++ tracing::trace!(%parent_pid, %child_pid, "signaling child exit"); ++ parent.signal_process(Signal::Sigchld); ++ }) ++ })) { ++ tracing::warn!( ++ %parent_pid, ++ %child_pid, ++ "failed to schedule child-exit signal delivery: {err}" ++ ); ++ } ++ } ++ ++ /// Reserves an in-flight child publication so epoch retirement cannot ++ /// pass quiescence while fork/spawn is between construction and adoption. ++ pub(crate) fn begin_child_publication( ++ &self, ++ ) -> Result { ++ let _epoch = self.tree_epoch.read().unwrap(); ++ let mut inner = self.inner.0.lock().unwrap(); ++ if inner.retiring { ++ return Err(ControlPlaneError::ProcessRetiring { ++ pid: self.pid().raw(), ++ }); ++ } ++ if self.finished.status().into_finished().is_some() { ++ return Err(ControlPlaneError::ProcessFinished { ++ pid: self.pid().raw(), ++ }); ++ } ++ inner.pending_child_publications = inner ++ .pending_child_publications ++ .checked_add(1) ++ .expect("pending child publication count exhausted"); ++ Ok(WasiChildPublicationGuard { ++ parent: self.clone(), ++ child_identity: None, ++ active: true, ++ }) ++ } ++ ++ /// Verifies and seals a complete process tree without changing any node on ++ /// rejection. The exclusive tree gate makes the validation and sealing ++ /// passes one atomic admission epoch: descendants cannot add work or ++ /// topology between the two passes. ++ /// ++ /// The returned descendants are in post-order. Every returned process and ++ /// `self` is already sealed when this method succeeds. ++ pub(crate) fn begin_epoch_retirement(&self) -> Result, ControlPlaneError> { ++ let _epoch = self.tree_epoch.write().unwrap(); ++ self.validate_retirement_node()?; ++ ++ let children = self.inner.0.lock().unwrap().children.clone(); ++ let mut visited = HashSet::from([Arc::as_ptr(&self.inner) as usize]); ++ let mut descendants = Vec::new(); ++ let mut live_children = 0; ++ for child in children { ++ if child.collect_retirement_candidates(&self.tree_epoch, &mut visited, &mut descendants) ++ { ++ descendants.push(child); ++ } else { ++ live_children += 1; ++ } ++ } ++ if live_children != 0 { ++ return Err(ControlPlaneError::ProcessHasLiveChildren { ++ pid: self.pid().raw(), ++ count: live_children, ++ }); ++ } ++ ++ // Pass two begins only after every node passed validation. No ++ // admission or topology mutation can interleave while `_epoch` lives. ++ for process in descendants.iter().chain(std::iter::once(self)) { ++ let mut inner = process.inner.0.lock().unwrap(); ++ debug_assert!(!inner.retiring); ++ inner.retiring = true; ++ } ++ Ok(descendants) ++ } ++ ++ fn validate_retirement_node(&self) -> Result<(), ControlPlaneError> { ++ if self.finished.status().into_finished().is_none() { ++ return Err(ControlPlaneError::ProcessStillRunning { ++ pid: self.pid().raw(), ++ }); ++ } ++ let inner = self.inner.0.lock().unwrap(); ++ if inner.retiring { ++ return Err(ControlPlaneError::ProcessRetiring { ++ pid: self.pid().raw(), ++ }); ++ } ++ if inner.execution_leases != 0 { ++ return Err(ControlPlaneError::ProcessStillRunning { ++ pid: self.pid().raw(), ++ }); ++ } ++ let live_non_main_threads = inner ++ .threads ++ .values() ++ .filter(|thread| !thread.is_main()) ++ .count(); ++ if live_non_main_threads != 0 { ++ return Err(ControlPlaneError::ProcessHasLiveThreads { ++ pid: self.pid().raw(), ++ count: live_non_main_threads, ++ }); ++ } ++ if inner.pending_child_publications != 0 { ++ return Err(ControlPlaneError::ProcessHasLiveChildren { ++ pid: self.pid().raw(), ++ count: inner.pending_child_publications as usize, ++ }); ++ } ++ Ok(()) ++ } ++ ++ fn collect_retirement_candidates( ++ &self, ++ expected_tree: &WasiProcessTreeEpoch, ++ visited: &mut HashSet, ++ descendants: &mut Vec, ++ ) -> bool { ++ let identity = Arc::as_ptr(&self.inner) as usize; ++ if !Arc::ptr_eq(&self.tree_epoch, expected_tree) || !visited.insert(identity) { ++ // Process children are a tree. A cycle is malformed retained state, ++ // and must block retirement rather than deadlock recursive locks. ++ return false; ++ } ++ if self.validate_retirement_node().is_err() { ++ return false; ++ } ++ let children = self.inner.0.lock().unwrap().children.clone(); ++ for child in children { ++ if !child.collect_retirement_candidates(expected_tree, visited, descendants) { ++ return false; ++ } ++ descendants.push(child); ++ } ++ true ++ } ++ ++ /// Releases task-count reservations for a sealed, quiescent epoch. The ++ /// thread objects may remain referenced by stale handles, but can no ++ /// longer consume embedded-runtime admission or execute new work. ++ pub(crate) fn retire_epoch_task_registrations(&self) { ++ let inner = self.inner.0.lock().unwrap(); ++ assert!(inner.retiring, "task retirement requires a sealed epoch"); ++ for thread in inner.threads.values() { ++ thread.retire_task_registration(); ++ } ++ } ++ ++ /// Removes only the exact child identity. PID equality alone is not safe ++ /// once stale handles and eventual PID reuse are considered. ++ pub(crate) fn remove_child_if_same(&self, child: &WasiProcess) -> bool { ++ let _epoch = self.tree_epoch.read().unwrap(); ++ let mut inner = self.inner.0.lock().unwrap(); ++ let before = inner.children.len(); ++ inner ++ .children ++ .retain(|candidate| !candidate.same_identity(child)); ++ inner.children.len() != before ++ } ++ ++ pub(crate) fn clear_retired_children(&self) { ++ let _epoch = self.tree_epoch.read().unwrap(); ++ let mut inner = self.inner.0.lock().unwrap(); ++ assert!(inner.retiring, "child cleanup requires a sealed epoch"); ++ inner.children.clear(); ++ } ++ + /// Takes a snapshot of the process and disables journaling returning + /// a future that can be waited on for the snapshot to complete + /// +@@ -765,12 +1590,78 @@ impl WasiProcess { + /// Waits until the process is finished. + pub async fn join(&self) -> Result> { + let _guard = WasiProcessWait::new(self); +- self.finished.await_termination().await ++ let status = self.finished.await_termination().await; ++ WasiProcessQuiescenceFuture { ++ inner: self.inner.clone(), ++ } ++ .await; ++ status + } + + /// Attempts to join on the process + pub fn try_join(&self) -> Option>> { +- self.finished.status().into_finished() ++ let status = self.finished.status().into_finished()?; ++ let inner = self.inner.0.lock().unwrap(); ++ (inner.execution_leases == 0 && inner.pending_child_publications == 0).then_some(status) ++ } ++ ++ /// Attempts to observe and atomically consume this process's wait status. ++ pub(crate) fn try_claim_join(&self) -> Option>> { ++ let status = self.try_join()?; ++ self.join_claimed ++ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) ++ .ok() ++ .map(|_| status) ++ } ++ ++ /// Waits for termination and atomically consumes the status. A competing ++ /// waiter that already claimed it causes `None` rather than a duplicate ++ /// successful wait result. ++ pub(crate) async fn claim_join(&self) -> Option>> { ++ let status = self.join().await; ++ self.join_claimed ++ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) ++ .ok() ++ .map(|_| status) ++ } ++ ++ /// Removes this process from the control-plane registry after its exit ++ /// status has been consumed by a join path. Process exit alone deliberately ++ /// does not call this so zombie/wait semantics remain intact. ++ pub(crate) fn reap(&self) -> bool { ++ self.compute ++ .upgrade() ++ .is_some_and(|control_plane| control_plane.reap_process(self)) ++ } ++ ++ /// Attempts to join any finished child without blocking. ++ pub fn try_join_any_child(&mut self) -> Result, Errno> { ++ let children: Vec<_> = { ++ let inner = self.inner.0.lock().unwrap(); ++ inner.children.clone() ++ }; ++ if children.is_empty() { ++ return Err(Errno::Child); ++ } ++ ++ for child in children { ++ let process = self ++ .compute ++ .must_upgrade() ++ .get_process(child.pid) ++ .unwrap_or_else(|| child.clone()); ++ let Some(status) = process.try_claim_join() else { ++ continue; ++ }; ++ ++ self.remove_child_if_same(&child); ++ let code = status ++ .unwrap_or_else(|e| e.as_exit_code().unwrap_or_else(|| Errno::Canceled.into())); ++ process.reap(); ++ return Ok(Some((child.pid, code))); ++ } ++ ++ Ok(None) + } + + /// Waits for all the children to be finished +@@ -785,17 +1676,26 @@ impl WasiProcess { + } + let mut waits = Vec::new(); + for child in children { +- if let Some(process) = self.compute.must_upgrade().get_process(child.pid) { +- let inner = self.inner.clone(); +- waits.push(async move { +- let join = process.join().await; +- let mut inner = inner.0.lock().unwrap(); +- inner.children.retain(|a| a.pid != child.pid); +- join +- }) +- } ++ let process = self ++ .compute ++ .must_upgrade() ++ .get_process(child.pid) ++ .unwrap_or_else(|| child.clone()); ++ let parent = self.clone(); ++ waits.push(async move { ++ let join = process.claim_join().await; ++ parent.remove_child_if_same(&child); ++ if join.is_some() { ++ process.reap(); ++ } ++ join ++ }) + } +- futures::future::join_all(waits).await.into_iter().next() ++ futures::future::join_all(waits) ++ .await ++ .into_iter() ++ .flatten() ++ .next() + } + + /// Waits for any of the children to finished +@@ -811,24 +1711,33 @@ impl WasiProcess { + + let mut waits = Vec::new(); + for child in children { +- if let Some(process) = self.compute.must_upgrade().get_process(child.pid) { +- let inner = self.inner.clone(); +- waits.push(async move { +- let join = process.join().await; +- let mut inner = inner.0.lock().unwrap(); +- inner.children.retain(|a| a.pid != child.pid); +- (child, join) +- }) +- } ++ let process = self ++ .compute ++ .must_upgrade() ++ .get_process(child.pid) ++ .unwrap_or_else(|| child.clone()); ++ let parent = self.clone(); ++ waits.push(async move { ++ let join = process.claim_join().await; ++ parent.remove_child_if_same(&child); ++ (child, process, join) ++ }) ++ } ++ let mut waits: Vec<_> = waits.into_iter().map(Box::pin).collect(); ++ while !waits.is_empty() { ++ let ((child, process, claimed), _, remaining) = ++ futures::future::select_all(waits).await; ++ waits = remaining; ++ let Some(res) = claimed else { ++ continue; ++ }; ++ process.reap(); ++ let code = ++ res.unwrap_or_else(|e| e.as_exit_code().unwrap_or_else(|| Errno::Canceled.into())); ++ return Ok(Some((child.pid, code))); + } +- let (child, res) = futures::future::select_all(waits.into_iter().map(Box::pin)) +- .await +- .0; +- +- let code = +- res.unwrap_or_else(|e| e.as_exit_code().unwrap_or_else(|| Errno::Canceled.into())); + +- Ok(Some((child.pid, code))) ++ Ok(None) + } + + /// Terminate the process and all its threads +diff --git a/lib/wasix/src/os/task/task_join_handle.rs b/lib/wasix/src/os/task/task_join_handle.rs +index d503817..c37169a 100644 +--- a/lib/wasix/src/os/task/task_join_handle.rs ++++ b/lib/wasix/src/os/task/task_join_handle.rs +@@ -1,14 +1,48 @@ + use std::{ + pin::Pin, +- sync::Arc, ++ sync::{Arc, Mutex}, + task::{Context, Poll}, + }; + ++#[cfg(all(feature = "ctrlc", windows))] ++use std::{ ++ ffi::c_void, ++ io, ++ sync::atomic::{AtomicBool, AtomicPtr, AtomicU32, AtomicUsize, Ordering}, ++ thread::JoinHandle, ++}; ++#[cfg(all(feature = "ctrlc", unix))] ++use std::{ ++ io, ++ mem::MaybeUninit, ++ os::fd::RawFd, ++ sync::atomic::{AtomicBool, AtomicI32, AtomicU32, AtomicUsize, Ordering}, ++ thread::JoinHandle, ++}; ++ ++#[cfg(all(feature = "ctrlc", windows))] ++use windows_sys::{ ++ Win32::{ ++ Foundation::{CloseHandle, HANDLE, WAIT_OBJECT_0}, ++ System::{ ++ Console::{ ++ CTRL_BREAK_EVENT, CTRL_C_EVENT, CTRL_CLOSE_EVENT, CTRL_LOGOFF_EVENT, ++ CTRL_SHUTDOWN_EVENT, SetConsoleCtrlHandler, ++ }, ++ Threading::{CreateEventW, INFINITE, SetEvent, WaitForSingleObject}, ++ }, ++ }, ++ core::BOOL, ++}; ++ + use wasmer_wasix_types::wasi::{Errno, ExitCode}; + ++use wasmer::FromToNativeWasmType; ++use wasmer_wasix_types::wasi::Signal; ++ + use crate::WasiRuntimeError; + +-use super::signal::{DynSignalHandlerAbi, default_signal_handler}; ++use super::signal::{DynSignalHandlerAbi, SignalDeliveryError, default_signal_handler}; + + #[derive(Clone, Debug)] + pub enum TaskStatus { +@@ -54,6 +88,44 @@ impl TaskStatus { + #[error("Task already terminated")] + pub struct TaskTerminatedError; + ++/// A platform-neutral, cloneable controller for delivering a WASI signal to ++/// one task. It does not install or mutate host operating-system signal ++/// dispositions. ++#[derive(Clone, Debug)] ++pub struct TaskSignalController { ++ signal_handler: Arc, ++ watch: tokio::sync::watch::Receiver, ++} ++ ++#[derive(thiserror::Error, Debug)] ++pub enum TaskSignalError { ++ #[error("task already terminated")] ++ Terminated, ++ #[error(transparent)] ++ Delivery(#[from] SignalDeliveryError), ++} ++ ++impl TaskSignalController { ++ /// Retrieve the current task status without changing it. ++ pub fn status(&self) -> TaskStatus { ++ self.watch.borrow().clone() ++ } ++ ++ /// Deliver a WASI signal directly to this task. ++ /// ++ /// The status check prevents knowingly targeting a completed task. Task ++ /// completion can still race signal delivery, exactly as it can for a ++ /// native process. ++ pub fn send_signal(&self, signal: Signal) -> Result<(), TaskSignalError> { ++ if self.status().is_finished() { ++ return Err(TaskSignalError::Terminated); ++ } ++ self.signal_handler ++ .signal(signal.to_native() as u8) ++ .map_err(TaskSignalError::from) ++ } ++} ++ + pub trait VirtualTaskHandle: std::fmt::Debug + Send + Sync + 'static { + fn status(&self) -> TaskStatus; + +@@ -176,7 +248,6 @@ impl Default for OwnedTaskStatus { + /// A handle that allows awaiting the termination of a task, and retrieving its exit code. + #[derive(Clone, Debug)] + pub struct TaskJoinHandle { +- #[allow(unused)] + signal_handler: Arc, + watch: tokio::sync::watch::Receiver, + } +@@ -187,22 +258,18 @@ impl TaskJoinHandle { + self.watch.borrow().clone() + } + +- #[cfg(feature = "ctrlc")] +- pub fn install_ctrlc_handler(&self) { +- use wasmer::FromToNativeWasmType; +- use wasmer_wasix_types::wasi::Signal; +- +- let signal_handler = self.signal_handler.clone(); ++ /// Create a platform-neutral controller that can deliver WASI signals to ++ /// this task without taking ownership of host OS signal policy. ++ pub fn signal_controller(&self) -> TaskSignalController { ++ TaskSignalController { ++ signal_handler: self.signal_handler.clone(), ++ watch: self.watch.clone(), ++ } ++ } + +- tokio::spawn(async move { +- // Loop sending ctrl-c presses as signals to the signal handler +- while tokio::signal::ctrl_c().await.is_ok() { +- if let Err(err) = signal_handler.signal(Signal::Sigint.to_native() as u8) { +- tracing::error!("failed to process signal - {}", err); +- std::process::exit(1); +- } +- } +- }); ++ /// Deliver a WASI signal directly to this task. ++ pub fn send_signal(&self, signal: Signal) -> Result<(), TaskSignalError> { ++ self.signal_controller().send_signal(signal) + } + + /// Wait until the task finishes. +@@ -221,3 +288,1230 @@ impl TaskJoinHandle { + } + } + } ++ ++pub(crate) type TaskCompletion = Result>; ++pub(crate) type TaskAbandonCallback = Box; ++pub(crate) type TaskCompletionCallback = Box; ++ ++type AbandonedTask = ( ++ TaskJoinHandle, ++ Option, ++ Option, ++); ++ ++fn take_abandoned_task(slot: &Mutex>) -> Option { ++ match slot.lock() { ++ Ok(mut slot) => slot.take(), ++ Err(poisoned) => poisoned.into_inner().take(), ++ } ++} ++ ++fn terminate_and_wait_for_abandoned_task( ++ mut task: TaskJoinHandle, ++ abandon: Option, ++ completion: Option, ++ task_description: &'static str, ++) { ++ if let Some(abandon) = abandon ++ && std::panic::catch_unwind(std::panic::AssertUnwindSafe(abandon)).is_err() ++ { ++ tracing::error!( ++ task = task_description, ++ "abandoned task termination callback panicked" ++ ); ++ } ++ ++ let signal = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ task.send_signal(Signal::Sigkill) ++ })); ++ match signal { ++ Ok(Ok(())) => {} ++ Ok(Err(error)) => { ++ tracing::debug!(%error, task = task_description, "abandoned task rejected SIGKILL") ++ } ++ Err(_) => { ++ tracing::error!( ++ task = task_description, ++ "abandoned task signal handler panicked" ++ ) ++ } ++ } ++ ++ let status = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ virtual_mio::block_on(task.wait_finished()) ++ })); ++ let Ok(status) = status else { ++ // Dropping the completion callback is intentional. Lifecycle callbacks ++ // own their fail-closed process guard, so even a broken task-status ++ // implementation cannot unwind this independent reaper or release a ++ // live process lease without first publishing terminal status. ++ tracing::error!( ++ task = task_description, ++ "abandoned task reaper panicked while waiting" ++ ); ++ return; ++ }; ++ ++ match &status { ++ Ok(code) => { ++ tracing::debug!( ++ exit_code = code.raw(), ++ task = task_description, ++ "abandoned task reaped" ++ ) ++ } ++ Err(error) => { ++ tracing::debug!(%error, task = task_description, "abandoned task reaped with error") ++ } ++ } ++ ++ if let Some(completion) = completion ++ && std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| completion(status))).is_err() ++ { ++ tracing::error!( ++ task = task_description, ++ "abandoned task completion callback panicked" ++ ); ++ } ++} ++ ++/// Transfers an abandoned accepted task to a stable host reaper. ++/// ++/// `Builder::spawn` consumes its closure even on failure. The shared slot lets ++/// the caller recover both the authoritative task handle and its terminal ++/// callback for a synchronous fallback, so neither cancellation nor unwinding ++/// can silently detach accepted guest work. ++pub(crate) fn terminate_and_reap_abandoned_task( ++ task: TaskJoinHandle, ++ reaper_name: &'static str, ++ task_description: &'static str, ++ abandon: Option, ++ completion: Option, ++) { ++ let abandoned = Arc::new(Mutex::new(Some((task, abandon, completion)))); ++ let reaper_task = abandoned.clone(); ++ let spawn = std::thread::Builder::new() ++ .name(reaper_name.to_string()) ++ .spawn(move || { ++ if let Some((task, abandon, completion)) = take_abandoned_task(&reaper_task) { ++ terminate_and_wait_for_abandoned_task(task, abandon, completion, task_description); ++ } ++ }); ++ if let Err(error) = spawn { ++ tracing::error!(%error, task = task_description, "failed to spawn task reaper; waiting synchronously"); ++ if let Some((task, abandon, completion)) = take_abandoned_task(&abandoned) { ++ terminate_and_wait_for_abandoned_task(task, abandon, completion, task_description); ++ } ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++const UNIX_HOST_SIGNALS: [(libc::c_int, u32, Signal, &str); 4] = [ ++ (libc::SIGINT, 1 << 0, Signal::Sigint, "SIGINT"), ++ (libc::SIGTERM, 1 << 1, Signal::Sigterm, "SIGTERM"), ++ (libc::SIGQUIT, 1 << 2, Signal::Sigquit, "SIGQUIT"), ++ (libc::SIGHUP, 1 << 3, Signal::Sighup, "SIGHUP"), ++]; ++ ++#[cfg(all(feature = "ctrlc", unix))] ++static UNIX_HOST_SIGNAL_OWNER_ACTIVE: AtomicBool = AtomicBool::new(false); ++#[cfg(all(feature = "ctrlc", unix))] ++static UNIX_HOST_SIGNAL_WRITE_FD: AtomicI32 = AtomicI32::new(-1); ++#[cfg(all(feature = "ctrlc", unix))] ++static UNIX_HOST_SIGNAL_PENDING: AtomicU32 = AtomicU32::new(0); ++#[cfg(all(feature = "ctrlc", unix))] ++static UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT: AtomicUsize = AtomicUsize::new(0); ++ ++/// Failure to install or bind the explicit Unix host lifecycle supervisor. ++#[cfg(all(feature = "ctrlc", unix))] ++#[derive(thiserror::Error, Debug)] ++pub enum UnixHostLifecycleError { ++ #[error( ++ "Unix host lifecycle supervision is unsupported on this target; use TaskSignalController" ++ )] ++ UnsupportedPlatform, ++ #[error("another Unix host lifecycle supervisor already owns this process")] ++ AlreadyOwned, ++ #[error("this Unix host lifecycle supervisor is already bound to a task")] ++ AlreadyBound, ++ #[error("{operation} failed: {source}")] ++ Io { ++ operation: &'static str, ++ #[source] ++ source: io::Error, ++ }, ++ #[error("{operation} failed for signal {signal}: {source}")] ++ Sigaction { ++ operation: &'static str, ++ signal: libc::c_int, ++ #[source] ++ source: io::Error, ++ }, ++ #[error("buffered {signal:?} could not be delivered when the task was bound: {source}")] ++ BufferedSignalDelivery { ++ signal: Signal, ++ #[source] ++ source: TaskSignalError, ++ }, ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++#[derive(Default)] ++struct UnixHostSignalRoute { ++ target: Option, ++ buffered: u32, ++ bound: bool, ++ stopping: bool, ++} ++ ++/// Exclusive process-scoped Unix lifecycle signal supervision. ++/// ++/// This is an explicit CLI/supervisor adapter, not an embedded-library ++/// default. It captures all of SIGINT, SIGTERM, SIGQUIT, and SIGHUP before a ++/// guest is scheduled, buffers signals until exactly one root task is bound, ++/// and restores the exact previous `sigaction` values when the final owner is ++/// dropped. Use [`TaskSignalController`] directly in embedded applications. ++/// ++/// The low-level errno adapters cover Linux, Android, Apple, the supported BSD ++/// families, and Solaris/illumos. Installation fails without changing ++/// dispositions on any other Unix target; its process owner can still route ++/// signals through [`TaskSignalController`] directly. ++#[cfg(all(feature = "ctrlc", unix))] ++pub struct UnixHostLifecycleSupervisor { ++ resources: UnixHostSignalResources, ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++pub type HostLifecycleSupervisor = UnixHostLifecycleSupervisor; ++ ++#[cfg(all(feature = "ctrlc", unix))] ++impl std::fmt::Debug for UnixHostLifecycleSupervisor { ++ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { ++ f.debug_struct("UnixHostLifecycleSupervisor") ++ .field("read_fd", &self.resources.read_fd) ++ .field("write_fd", &self.resources.write_fd) ++ .finish_non_exhaustive() ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++struct UnixHostSignalResources { ++ read_fd: RawFd, ++ write_fd: RawFd, ++ previous_actions: Vec<(libc::c_int, libc::sigaction)>, ++ route: Arc>, ++ forwarder: Option>, ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++impl UnixHostLifecycleSupervisor { ++ /// Whether the process-global adapter is qualified for this target. ++ pub const fn is_supported() -> bool { ++ cfg!(any( ++ target_os = "linux", ++ target_os = "android", ++ target_os = "macos", ++ target_os = "ios", ++ target_os = "tvos", ++ target_os = "watchos", ++ target_os = "visionos", ++ target_os = "freebsd", ++ target_os = "dragonfly", ++ target_os = "netbsd", ++ target_os = "openbsd", ++ target_os = "solaris", ++ target_os = "illumos", ++ )) ++ } ++ ++ /// Install the exclusive process signal owner and its dedicated forwarder. ++ /// All four dispositions are installed transactionally or the operation ++ /// restores every disposition already changed and fails. ++ pub fn install() -> Result { ++ Self::install_signals(&UNIX_HOST_SIGNALS.map(|entry| entry.0)) ++ } ++ ++ fn install_signals(signals: &[libc::c_int]) -> Result { ++ if !Self::is_supported() { ++ return Err(UnixHostLifecycleError::UnsupportedPlatform); ++ } ++ UNIX_HOST_SIGNAL_OWNER_ACTIVE ++ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) ++ .map_err(|_| UnixHostLifecycleError::AlreadyOwned)?; ++ ++ let (read_fd, write_fd) = match create_unix_signal_pipe() { ++ Ok(fds) => fds, ++ Err(source) => { ++ UNIX_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); ++ return Err(UnixHostLifecycleError::Io { ++ operation: "create host signal self-pipe", ++ source, ++ }); ++ } ++ }; ++ let route = Arc::new(Mutex::new(UnixHostSignalRoute::default())); ++ let thread_route = route.clone(); ++ let forwarder = match std::thread::Builder::new() ++ .name("wasmer-host-signals".to_string()) ++ .spawn(move || unix_host_signal_forwarder(read_fd, thread_route)) ++ { ++ Ok(forwarder) => forwarder, ++ Err(source) => { ++ unsafe { ++ libc::close(read_fd); ++ libc::close(write_fd); ++ } ++ UNIX_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); ++ return Err(UnixHostLifecycleError::Io { ++ operation: "spawn host signal forwarder", ++ source, ++ }); ++ } ++ }; ++ ++ let mut resources = UnixHostSignalResources { ++ read_fd, ++ write_fd, ++ previous_actions: Vec::with_capacity(signals.len()), ++ route, ++ forwarder: Some(forwarder), ++ }; ++ UNIX_HOST_SIGNAL_PENDING.store(0, Ordering::Release); ++ UNIX_HOST_SIGNAL_WRITE_FD.store(write_fd, Ordering::SeqCst); ++ ++ let action = unix_host_signal_action()?; ++ for &signal in signals { ++ let mut previous = MaybeUninit::::uninit(); ++ let rc = unsafe { libc::sigaction(signal, &action, previous.as_mut_ptr()) }; ++ if rc != 0 { ++ return Err(UnixHostLifecycleError::Sigaction { ++ operation: "install host lifecycle handler", ++ signal, ++ source: io::Error::last_os_error(), ++ }); ++ } ++ resources ++ .previous_actions ++ .push((signal, unsafe { previous.assume_init() })); ++ } ++ ++ Ok(Self { resources }) ++ } ++ ++ /// Bind the sole root guest task. Signals captured before this call are ++ /// retained and delivered to this task, even if the dedicated reader has ++ /// not yet observed its self-pipe wakeup when this call returns. ++ pub fn bind_task(&self, task: &TaskJoinHandle) -> Result<(), UnixHostLifecycleError> { ++ let controller = task.signal_controller(); ++ let buffered = { ++ let mut route = lock_unix_signal_route(&self.resources.route); ++ if route.bound { ++ return Err(UnixHostLifecycleError::AlreadyBound); ++ } ++ route.bound = true; ++ route.target = Some(controller.clone()); ++ std::mem::take(&mut route.buffered) ++ }; ++ deliver_unix_signal_mask(buffered, &controller).map_err(|(signal, source)| { ++ UnixHostLifecycleError::BufferedSignalDelivery { signal, source } ++ }) ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++impl Drop for UnixHostSignalResources { ++ fn drop(&mut self) { ++ let mut restoration_failed = false; ++ for (signal, previous) in self.previous_actions.iter().rev() { ++ if unsafe { libc::sigaction(*signal, previous, std::ptr::null_mut()) } != 0 { ++ restoration_failed = true; ++ tracing::error!( ++ host_signal = *signal, ++ error = %io::Error::last_os_error(), ++ "failed to restore previous host signal disposition" ++ ); ++ } ++ } ++ self.previous_actions.clear(); ++ ++ if restoration_failed { ++ // Never close or recycle a descriptor that an unrestored signal ++ // disposition can still reach. Keeping the reader, descriptors, ++ // route, and exclusivity token alive is the only fail-closed state ++ // available from Drop; the detached reader's Arc retains `route`. ++ // This path means the kernel rejected restoration of a disposition ++ // that it previously accepted, so leaking these process-scoped ++ // resources is intentional and safer than a stale-handler UAF. ++ tracing::error!( ++ "host signal restoration was incomplete; retaining the supervisor for process lifetime" ++ ); ++ return; ++ } ++ ++ // The SeqCst protocol closes the stale-handler/reused-fd window: ++ // ++ // * every handler increments before loading the descriptor; ++ // * a handler that loaded the old descriptor is therefore included in ++ // the count observed below and must finish before close; ++ // * a handler whose increment is ordered after the zero observation is ++ // also ordered after this -1 store and cannot load the old value. ++ UNIX_HOST_SIGNAL_WRITE_FD.store(-1, Ordering::SeqCst); ++ while UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT.load(Ordering::SeqCst) != 0 { ++ std::thread::yield_now(); ++ } ++ ++ // No handler can add work after this point. Ask the sole normal bitset ++ // consumer to drain and exit, keeping the target bound until it has ++ // joined. A full pipe needs no extra byte: it already contains a wakeup. ++ { ++ let mut route = lock_unix_signal_route(&self.route); ++ route.stopping = true; ++ } ++ wake_unix_signal_forwarder(self.write_fd); ++ if let Some(forwarder) = self.forwarder.take() ++ && forwarder.join().is_err() ++ { ++ tracing::error!("host signal forwarder panicked during shutdown"); ++ } ++ // Recover pending work if the reader exited because of an unexpected ++ // pipe failure. Under normal operation it has already drained to zero. ++ route_unix_signal_mask( ++ UNIX_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), ++ &self.route, ++ ); ++ { ++ let mut route = lock_unix_signal_route(&self.route); ++ route.target = None; ++ route.buffered = 0; ++ } ++ ++ unsafe { ++ libc::close(self.read_fd); ++ libc::close(self.write_fd); ++ } ++ UNIX_HOST_SIGNAL_PENDING.store(0, Ordering::Release); ++ UNIX_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn unix_host_signal_action() -> Result { ++ let mut action = unsafe { std::mem::zeroed::() }; ++ action.sa_sigaction = unix_host_lifecycle_signal_handler as *const () as usize; ++ action.sa_flags = libc::SA_RESTART; ++ if unsafe { libc::sigemptyset(&mut action.sa_mask) } != 0 { ++ return Err(UnixHostLifecycleError::Io { ++ operation: "initialize host signal mask", ++ source: io::Error::last_os_error(), ++ }); ++ } ++ for (signal, _, _, _) in UNIX_HOST_SIGNALS { ++ if unsafe { libc::sigaddset(&mut action.sa_mask, signal) } != 0 { ++ return Err(UnixHostLifecycleError::Sigaction { ++ operation: "block nested host lifecycle signal", ++ signal, ++ source: io::Error::last_os_error(), ++ }); ++ } ++ } ++ Ok(action) ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++extern "C" fn unix_host_lifecycle_signal_handler(signal: libc::c_int) { ++ UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_add(1, Ordering::SeqCst); ++ let errno = unsafe { unix_errno_location() }; ++ let saved_errno = if errno.is_null() { ++ 0 ++ } else { ++ unsafe { errno.read() } ++ }; ++ ++ if let Some(bit) = unix_host_signal_bit(signal) { ++ let write_fd = UNIX_HOST_SIGNAL_WRITE_FD.load(Ordering::SeqCst); ++ if write_fd >= 0 { ++ UNIX_HOST_SIGNAL_PENDING.fetch_or(bit, Ordering::Release); ++ let wake = [1u8]; ++ loop { ++ let written = unsafe { ++ libc::write(write_fd, wake.as_ptr().cast::(), wake.len()) ++ }; ++ if written >= 0 { ++ break; ++ } ++ let current_errno = if errno.is_null() { ++ 0 ++ } else { ++ unsafe { errno.read() } ++ }; ++ if current_errno != libc::EINTR { ++ break; ++ } ++ } ++ } ++ } ++ ++ if !errno.is_null() { ++ unsafe { errno.write(saved_errno) }; ++ } ++ UNIX_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_sub(1, Ordering::SeqCst); ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn unix_host_signal_bit(signal: libc::c_int) -> Option { ++ UNIX_HOST_SIGNALS ++ .iter() ++ .find_map(|(candidate, bit, _, _)| (*candidate == signal).then_some(*bit)) ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn deliver_unix_signal_mask( ++ mask: u32, ++ controller: &TaskSignalController, ++) -> Result<(), (Signal, TaskSignalError)> { ++ for (_, bit, signal, _) in UNIX_HOST_SIGNALS { ++ if mask & bit != 0 { ++ controller ++ .send_signal(signal) ++ .map_err(|source| (signal, source))?; ++ } ++ } ++ Ok(()) ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn route_unix_signal_mask(mask: u32, route: &Arc>) { ++ if mask == 0 { ++ return; ++ } ++ let target = { ++ let mut route = lock_unix_signal_route(route); ++ match &route.target { ++ Some(target) => Some(target.clone()), ++ None => { ++ route.buffered |= mask; ++ None ++ } ++ } ++ }; ++ if let Some(target) = target ++ && let Err((signal, error)) = deliver_unix_signal_mask(mask, &target) ++ { ++ tracing::warn!(?signal, %error, "failed to forward host lifecycle signal"); ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn unix_host_signal_forwarder(read_fd: RawFd, route: Arc>) { ++ let mut wakeups = [0u8; 64]; ++ loop { ++ let read = unsafe { ++ libc::read( ++ read_fd, ++ wakeups.as_mut_ptr().cast::(), ++ wakeups.len(), ++ ) ++ }; ++ if read < 0 { ++ let error = io::Error::last_os_error(); ++ if error.raw_os_error() == Some(libc::EINTR) { ++ continue; ++ } ++ tracing::error!(%error, "host signal self-pipe read failed"); ++ return; ++ } ++ if read == 0 { ++ return; ++ } ++ route_unix_signal_mask(UNIX_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), &route); ++ if lock_unix_signal_route(&route).stopping { ++ // The writer is quiescent before `stopping` is set. One final swap ++ // closes the small interval between the previous swap and the ++ // stopping observation without competing with another consumer. ++ route_unix_signal_mask(UNIX_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), &route); ++ return; ++ } ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn lock_unix_signal_route( ++ route: &Mutex, ++) -> std::sync::MutexGuard<'_, UnixHostSignalRoute> { ++ route ++ .lock() ++ .unwrap_or_else(|poisoned| poisoned.into_inner()) ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn wake_unix_signal_forwarder(write_fd: RawFd) { ++ let wake = [1u8]; ++ unsafe { ++ libc::write(write_fd, wake.as_ptr().cast::(), wake.len()); ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn create_unix_signal_pipe() -> io::Result<(RawFd, RawFd)> { ++ let mut fds = [-1; 2]; ++ #[cfg(any(target_os = "linux", target_os = "android"))] ++ let result = unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }; ++ #[cfg(not(any(target_os = "linux", target_os = "android")))] ++ let result = unsafe { libc::pipe(fds.as_mut_ptr()) }; ++ if result != 0 { ++ return Err(io::Error::last_os_error()); ++ } ++ ++ let configured = configure_unix_signal_pipe(fds); ++ if let Err(error) = configured { ++ unsafe { ++ libc::close(fds[0]); ++ libc::close(fds[1]); ++ } ++ return Err(error); ++ } ++ Ok((fds[0], fds[1])) ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn configure_unix_signal_pipe(fds: [RawFd; 2]) -> io::Result<()> { ++ #[cfg(not(any(target_os = "linux", target_os = "android")))] ++ { ++ set_unix_fd_flag(fds[0], libc::F_GETFD, libc::F_SETFD, libc::FD_CLOEXEC)?; ++ set_unix_fd_flag(fds[1], libc::F_GETFD, libc::F_SETFD, libc::FD_CLOEXEC)?; ++ } ++ set_unix_fd_flag(fds[1], libc::F_GETFL, libc::F_SETFL, libc::O_NONBLOCK) ++} ++ ++#[cfg(all(feature = "ctrlc", unix))] ++fn set_unix_fd_flag( ++ fd: RawFd, ++ get: libc::c_int, ++ set: libc::c_int, ++ flag: libc::c_int, ++) -> io::Result<()> { ++ let current = unsafe { libc::fcntl(fd, get) }; ++ if current < 0 { ++ return Err(io::Error::last_os_error()); ++ } ++ if unsafe { libc::fcntl(fd, set, current | flag) } < 0 { ++ return Err(io::Error::last_os_error()); ++ } ++ Ok(()) ++} ++ ++#[cfg(all( ++ feature = "ctrlc", ++ unix, ++ any(target_os = "linux", target_os = "dragonfly") ++))] ++unsafe fn unix_errno_location() -> *mut libc::c_int { ++ unsafe { libc::__errno_location() } ++} ++ ++#[cfg(all(feature = "ctrlc", unix, target_os = "android"))] ++unsafe fn unix_errno_location() -> *mut libc::c_int { ++ unsafe { libc::__errno() } ++} ++ ++#[cfg(all( ++ feature = "ctrlc", ++ unix, ++ any( ++ target_os = "macos", ++ target_os = "ios", ++ target_os = "tvos", ++ target_os = "watchos", ++ target_os = "visionos", ++ target_os = "freebsd", ++ ) ++))] ++unsafe fn unix_errno_location() -> *mut libc::c_int { ++ unsafe { libc::__error() } ++} ++ ++#[cfg(all( ++ feature = "ctrlc", ++ unix, ++ any(target_os = "netbsd", target_os = "openbsd") ++))] ++unsafe fn unix_errno_location() -> *mut libc::c_int { ++ unsafe { libc::__errno() } ++} ++ ++#[cfg(all( ++ feature = "ctrlc", ++ unix, ++ any(target_os = "solaris", target_os = "illumos") ++))] ++unsafe fn unix_errno_location() -> *mut libc::c_int { ++ unsafe { libc::___errno() } ++} ++ ++#[cfg(all( ++ feature = "ctrlc", ++ unix, ++ not(any( ++ target_os = "linux", ++ target_os = "android", ++ target_os = "macos", ++ target_os = "ios", ++ target_os = "tvos", ++ target_os = "watchos", ++ target_os = "visionos", ++ target_os = "freebsd", ++ target_os = "dragonfly", ++ target_os = "netbsd", ++ target_os = "openbsd", ++ target_os = "solaris", ++ target_os = "illumos", ++ )) ++))] ++unsafe fn unix_errno_location() -> *mut libc::c_int { ++ std::ptr::null_mut() ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++const WINDOWS_GUEST_SIGNALS: [(u32, Signal); 4] = [ ++ (1 << 0, Signal::Sigint), ++ (1 << 1, Signal::Sigterm), ++ (1 << 2, Signal::Sigquit), ++ (1 << 3, Signal::Sighup), ++]; ++ ++#[cfg(all(feature = "ctrlc", windows))] ++static WINDOWS_HOST_SIGNAL_OWNER_ACTIVE: AtomicBool = AtomicBool::new(false); ++#[cfg(all(feature = "ctrlc", windows))] ++static WINDOWS_HOST_SIGNAL_EVENT: AtomicPtr = AtomicPtr::new(std::ptr::null_mut()); ++#[cfg(all(feature = "ctrlc", windows))] ++static WINDOWS_HOST_SIGNAL_PENDING: AtomicU32 = AtomicU32::new(0); ++#[cfg(all(feature = "ctrlc", windows))] ++static WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT: AtomicUsize = AtomicUsize::new(0); ++ ++/// Failure to install or bind the explicit Windows console lifecycle owner. ++#[cfg(all(feature = "ctrlc", windows))] ++#[derive(thiserror::Error, Debug)] ++pub enum WindowsHostLifecycleError { ++ #[error("another Windows host lifecycle supervisor already owns this process")] ++ AlreadyOwned, ++ #[error("this Windows host lifecycle supervisor is already bound to a task")] ++ AlreadyBound, ++ #[error("{operation} failed: {source}")] ++ Io { ++ operation: &'static str, ++ #[source] ++ source: io::Error, ++ }, ++ #[error("buffered {signal:?} could not be delivered when the task was bound: {source}")] ++ BufferedSignalDelivery { ++ signal: Signal, ++ #[source] ++ source: TaskSignalError, ++ }, ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++#[derive(Default)] ++struct WindowsHostSignalRoute { ++ target: Option, ++ buffered: u32, ++ bound: bool, ++ stopping: bool, ++} ++ ++/// Exclusive process-scoped Windows console lifecycle supervision. ++/// ++/// This adapter is installed only by an explicit process owner. It preserves ++/// the previous Windows handler stack by adding and later removing exactly its ++/// own callback. Console Ctrl+C maps to WASI `SIGINT`, Ctrl+Break to `SIGQUIT`, ++/// close/shutdown to `SIGTERM`, and logoff to `SIGHUP`. ++#[cfg(all(feature = "ctrlc", windows))] ++pub struct WindowsHostLifecycleSupervisor { ++ resources: WindowsHostSignalResources, ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++pub type HostLifecycleSupervisor = WindowsHostLifecycleSupervisor; ++ ++#[cfg(all(feature = "ctrlc", windows))] ++impl std::fmt::Debug for WindowsHostLifecycleSupervisor { ++ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { ++ f.debug_struct("WindowsHostLifecycleSupervisor") ++ .field("event", &self.resources.event) ++ .finish_non_exhaustive() ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++struct WindowsHostSignalResources { ++ event: usize, ++ handler_installed: bool, ++ route: Arc>, ++ forwarder: Option>, ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++impl WindowsHostLifecycleSupervisor { ++ /// Install one exclusive console-control owner and its dedicated forwarder. ++ pub fn install() -> Result { ++ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE ++ .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) ++ .map_err(|_| WindowsHostLifecycleError::AlreadyOwned)?; ++ ++ let event = unsafe { CreateEventW(std::ptr::null(), 0, 0, std::ptr::null()) }; ++ if event.is_null() { ++ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); ++ return Err(WindowsHostLifecycleError::Io { ++ operation: "create host lifecycle event", ++ source: io::Error::last_os_error(), ++ }); ++ } ++ ++ let route = Arc::new(Mutex::new(WindowsHostSignalRoute::default())); ++ let thread_route = route.clone(); ++ let event_address = event as usize; ++ let forwarder = match std::thread::Builder::new() ++ .name("wasmer-host-signals".to_string()) ++ .spawn(move || windows_host_signal_forwarder(event_address as HANDLE, thread_route)) ++ { ++ Ok(forwarder) => forwarder, ++ Err(source) => { ++ unsafe { CloseHandle(event) }; ++ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); ++ return Err(WindowsHostLifecycleError::Io { ++ operation: "spawn host signal forwarder", ++ source, ++ }); ++ } ++ }; ++ ++ let mut resources = WindowsHostSignalResources { ++ event: event_address, ++ handler_installed: false, ++ route, ++ forwarder: Some(forwarder), ++ }; ++ WINDOWS_HOST_SIGNAL_PENDING.store(0, Ordering::Release); ++ WINDOWS_HOST_SIGNAL_EVENT.store(event, Ordering::SeqCst); ++ ++ if unsafe { SetConsoleCtrlHandler(Some(windows_host_lifecycle_handler), 1) } == 0 { ++ return Err(WindowsHostLifecycleError::Io { ++ operation: "install Windows console-control handler", ++ source: io::Error::last_os_error(), ++ }); ++ } ++ resources.handler_installed = true; ++ Ok(Self { resources }) ++ } ++ ++ /// Bind the sole root guest task. Console events received earlier remain ++ /// buffered until the root is available. ++ pub fn bind_task(&self, task: &TaskJoinHandle) -> Result<(), WindowsHostLifecycleError> { ++ let controller = task.signal_controller(); ++ let buffered = { ++ let mut route = lock_windows_signal_route(&self.resources.route); ++ if route.bound { ++ return Err(WindowsHostLifecycleError::AlreadyBound); ++ } ++ route.bound = true; ++ route.target = Some(controller.clone()); ++ std::mem::take(&mut route.buffered) ++ }; ++ deliver_windows_signal_mask(buffered, &controller).map_err(|(signal, source)| { ++ WindowsHostLifecycleError::BufferedSignalDelivery { signal, source } ++ }) ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++impl Drop for WindowsHostSignalResources { ++ fn drop(&mut self) { ++ if self.handler_installed ++ && unsafe { SetConsoleCtrlHandler(Some(windows_host_lifecycle_handler), 0) } == 0 ++ { ++ // As on Unix, retaining the callback target and exclusive owner is ++ // safer than allowing a stale callback to reach a recycled handle. ++ tracing::error!( ++ error = %io::Error::last_os_error(), ++ "failed to remove Windows console-control handler; retaining supervisor for process lifetime" ++ ); ++ return; ++ } ++ ++ WINDOWS_HOST_SIGNAL_EVENT.store(std::ptr::null_mut(), Ordering::SeqCst); ++ while WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT.load(Ordering::SeqCst) != 0 { ++ std::thread::yield_now(); ++ } ++ ++ { ++ let mut route = lock_windows_signal_route(&self.route); ++ route.stopping = true; ++ } ++ wake_windows_signal_forwarder(self.event as HANDLE); ++ if let Some(forwarder) = self.forwarder.take() ++ && forwarder.join().is_err() ++ { ++ tracing::error!("Windows host signal forwarder panicked during shutdown"); ++ } ++ route_windows_signal_mask( ++ WINDOWS_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), ++ &self.route, ++ ); ++ { ++ let mut route = lock_windows_signal_route(&self.route); ++ route.target = None; ++ route.buffered = 0; ++ } ++ ++ unsafe { CloseHandle(self.event as HANDLE) }; ++ WINDOWS_HOST_SIGNAL_PENDING.store(0, Ordering::Release); ++ WINDOWS_HOST_SIGNAL_OWNER_ACTIVE.store(false, Ordering::Release); ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++unsafe extern "system" fn windows_host_lifecycle_handler(control: u32) -> BOOL { ++ WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_add(1, Ordering::SeqCst); ++ let handled = if let Some(bit) = windows_host_signal_bit(control) { ++ let event = WINDOWS_HOST_SIGNAL_EVENT.load(Ordering::SeqCst); ++ if event.is_null() { ++ false ++ } else { ++ WINDOWS_HOST_SIGNAL_PENDING.fetch_or(bit, Ordering::Release); ++ // Only claim the console event when its forwarder was actually ++ // woken. The pending bit intentionally remains set on failure so ++ // a later successful wake can still drain it. ++ (unsafe { SetEvent(event) }) != 0 ++ } ++ } else { ++ false ++ }; ++ WINDOWS_HOST_SIGNAL_HANDLERS_IN_FLIGHT.fetch_sub(1, Ordering::SeqCst); ++ if handled { 1 } else { 0 } ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++fn windows_host_signal_bit(control: u32) -> Option { ++ match control { ++ CTRL_C_EVENT => Some(1 << 0), ++ CTRL_CLOSE_EVENT | CTRL_SHUTDOWN_EVENT => Some(1 << 1), ++ CTRL_BREAK_EVENT => Some(1 << 2), ++ CTRL_LOGOFF_EVENT => Some(1 << 3), ++ _ => None, ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++fn deliver_windows_signal_mask( ++ mask: u32, ++ controller: &TaskSignalController, ++) -> Result<(), (Signal, TaskSignalError)> { ++ for (bit, signal) in WINDOWS_GUEST_SIGNALS { ++ if mask & bit != 0 { ++ controller ++ .send_signal(signal) ++ .map_err(|source| (signal, source))?; ++ } ++ } ++ Ok(()) ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++fn route_windows_signal_mask(mask: u32, route: &Arc>) { ++ if mask == 0 { ++ return; ++ } ++ let target = { ++ let mut route = lock_windows_signal_route(route); ++ match &route.target { ++ Some(target) => Some(target.clone()), ++ None => { ++ route.buffered |= mask; ++ None ++ } ++ } ++ }; ++ if let Some(target) = target ++ && let Err((signal, error)) = deliver_windows_signal_mask(mask, &target) ++ { ++ tracing::warn!(?signal, %error, "failed to forward Windows host lifecycle event"); ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++fn windows_host_signal_forwarder(event: HANDLE, route: Arc>) { ++ loop { ++ let wait = unsafe { WaitForSingleObject(event, INFINITE) }; ++ if wait != WAIT_OBJECT_0 { ++ tracing::error!( ++ error = %io::Error::last_os_error(), ++ wait, ++ "Windows host lifecycle event wait failed" ++ ); ++ return; ++ } ++ route_windows_signal_mask( ++ WINDOWS_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), ++ &route, ++ ); ++ if lock_windows_signal_route(&route).stopping { ++ route_windows_signal_mask( ++ WINDOWS_HOST_SIGNAL_PENDING.swap(0, Ordering::AcqRel), ++ &route, ++ ); ++ return; ++ } ++ } ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++fn wake_windows_signal_forwarder(event: HANDLE) { ++ unsafe { SetEvent(event) }; ++} ++ ++#[cfg(all(feature = "ctrlc", windows))] ++fn lock_windows_signal_route( ++ route: &Mutex, ++) -> std::sync::MutexGuard<'_, WindowsHostSignalRoute> { ++ route ++ .lock() ++ .unwrap_or_else(|poisoned| poisoned.into_inner()) ++} ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use crate::os::task::signal::SignalHandlerAbi; ++ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] ++ use std::time::Instant; ++ use std::{sync::mpsc, time::Duration}; ++ ++ #[derive(Debug)] ++ struct RecordingSignalHandler { ++ sender: mpsc::Sender, ++ } ++ ++ impl SignalHandlerAbi for RecordingSignalHandler { ++ fn signal(&self, signal: u8) -> Result<(), SignalDeliveryError> { ++ self.sender.send(signal).map_err(|_| SignalDeliveryError) ++ } ++ } ++ ++ fn status_and_signals() -> (OwnedTaskStatus, mpsc::Receiver, TaskJoinHandle) { ++ let (sender, receiver) = mpsc::channel(); ++ let mut status = OwnedTaskStatus::default(); ++ status.set_signal_handler(Arc::new(RecordingSignalHandler { sender })); ++ let handle = status.handle(); ++ (status, receiver, handle) ++ } ++ ++ #[test] ++ fn direct_signal_controller_is_platform_neutral_and_rejects_finished_tasks() { ++ let (status, recorded, handle) = status_and_signals(); ++ let controller = handle.signal_controller(); ++ for signal in [ ++ Signal::Sigint, ++ Signal::Sigterm, ++ Signal::Sigquit, ++ Signal::Sighup, ++ ] { ++ controller.send_signal(signal).unwrap(); ++ assert_eq!( ++ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), ++ signal.to_native() as u8 ++ ); ++ } ++ status.set_finished(Ok(Errno::Success.into())); ++ assert!(matches!( ++ controller.send_signal(Signal::Sigterm), ++ Err(TaskSignalError::Terminated) ++ )); ++ } ++ ++ #[cfg(all(feature = "ctrlc", windows))] ++ #[test] ++ fn windows_console_events_have_explicit_wasi_mappings() { ++ assert_eq!(windows_host_signal_bit(CTRL_C_EVENT), Some(1 << 0)); ++ assert_eq!(windows_host_signal_bit(CTRL_CLOSE_EVENT), Some(1 << 1)); ++ assert_eq!(windows_host_signal_bit(CTRL_SHUTDOWN_EVENT), Some(1 << 1)); ++ assert_eq!(windows_host_signal_bit(CTRL_BREAK_EVENT), Some(1 << 2)); ++ assert_eq!(windows_host_signal_bit(CTRL_LOGOFF_EVENT), Some(1 << 3)); ++ assert_eq!(windows_host_signal_bit(u32::MAX), None); ++ } ++ ++ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] ++ static PREVIOUS_HUP_COUNT: AtomicUsize = AtomicUsize::new(0); ++ ++ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] ++ extern "C" fn previous_hup_handler(_: libc::c_int) { ++ PREVIOUS_HUP_COUNT.fetch_add(1, Ordering::SeqCst); ++ } ++ ++ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] ++ fn wait_until(description: &str, predicate: impl Fn() -> bool) { ++ let deadline = Instant::now() + Duration::from_secs(2); ++ while !predicate() { ++ assert!( ++ Instant::now() < deadline, ++ "timed out waiting for {description}" ++ ); ++ std::thread::sleep(Duration::from_millis(1)); ++ } ++ } ++ ++ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] ++ fn run_unix_supervisor_child() { ++ PREVIOUS_HUP_COUNT.store(0, Ordering::SeqCst); ++ let mut previous_hup = MaybeUninit::::uninit(); ++ let mut custom_hup = unsafe { std::mem::zeroed::() }; ++ custom_hup.sa_sigaction = previous_hup_handler as *const () as usize; ++ custom_hup.sa_flags = libc::SA_RESTART; ++ assert_eq!(unsafe { libc::sigemptyset(&mut custom_hup.sa_mask) }, 0); ++ assert_eq!( ++ unsafe { libc::sigaction(libc::SIGHUP, &custom_hup, previous_hup.as_mut_ptr()) }, ++ 0 ++ ); ++ let previous_hup = unsafe { previous_hup.assume_init() }; ++ let mut expected_custom_hup = MaybeUninit::::uninit(); ++ assert_eq!( ++ unsafe { ++ libc::sigaction( ++ libc::SIGHUP, ++ std::ptr::null(), ++ expected_custom_hup.as_mut_ptr(), ++ ) ++ }, ++ 0 ++ ); ++ let expected_custom_hup = unsafe { expected_custom_hup.assume_init() }; ++ ++ // Force a failure after one successful sigaction and prove the setup ++ // guard restores the first disposition and releases exclusivity. ++ assert!(matches!( ++ UnixHostLifecycleSupervisor::install_signals(&[libc::SIGHUP, -1]), ++ Err(UnixHostLifecycleError::Sigaction { signal: -1, .. }) ++ )); ++ assert!(!UNIX_HOST_SIGNAL_OWNER_ACTIVE.load(Ordering::Acquire)); ++ assert_eq!(unsafe { libc::kill(libc::getpid(), libc::SIGHUP) }, 0); ++ wait_until("partial-install rollback handler", || { ++ PREVIOUS_HUP_COUNT.load(Ordering::SeqCst) == 1 ++ }); ++ ++ let supervisor = UnixHostLifecycleSupervisor::install().unwrap(); ++ assert!(matches!( ++ UnixHostLifecycleSupervisor::install(), ++ Err(UnixHostLifecycleError::AlreadyOwned) ++ )); ++ let (status, recorded, handle) = status_and_signals(); ++ ++ // Capture SIGTERM before the root task exists, and require it to be in ++ // the supervisor's pre-bind buffer before binding. ++ assert_eq!(unsafe { libc::kill(libc::getpid(), libc::SIGTERM) }, 0); ++ wait_until("pre-bind SIGTERM buffering", || { ++ lock_unix_signal_route(&supervisor.resources.route).buffered & (1 << 1) != 0 ++ }); ++ supervisor.bind_task(&handle).unwrap(); ++ assert!(matches!( ++ supervisor.bind_task(&handle), ++ Err(UnixHostLifecycleError::AlreadyBound) ++ )); ++ assert_eq!( ++ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), ++ Signal::Sigterm.to_native() as u8 ++ ); ++ ++ // The host callback must not perturb errno in interrupted code. ++ unsafe { unix_errno_location().write(libc::EDOM) }; ++ unix_host_lifecycle_signal_handler(libc::SIGINT); ++ assert_eq!(unsafe { unix_errno_location().read() }, libc::EDOM); ++ assert_eq!( ++ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), ++ Signal::Sigint.to_native() as u8 ++ ); ++ ++ for (host, guest) in [ ++ (libc::SIGQUIT, Signal::Sigquit), ++ (libc::SIGHUP, Signal::Sighup), ++ ] { ++ assert_eq!(unsafe { libc::kill(libc::getpid(), host) }, 0); ++ assert_eq!( ++ recorded.recv_timeout(Duration::from_secs(1)).unwrap(), ++ guest.to_native() as u8 ++ ); ++ } ++ assert_eq!(PREVIOUS_HUP_COUNT.load(Ordering::SeqCst), 1); ++ ++ drop(supervisor); ++ assert!(!UNIX_HOST_SIGNAL_OWNER_ACTIVE.load(Ordering::Acquire)); ++ assert_eq!(unsafe { libc::kill(libc::getpid(), libc::SIGHUP) }, 0); ++ wait_until("restored previous SIGHUP disposition", || { ++ PREVIOUS_HUP_COUNT.load(Ordering::SeqCst) == 2 ++ }); ++ ++ // Repeated ownership cycles must not leak the self-pipe descriptors or ++ // leave a forwarder behind. `/proc/self/fd` counts its own directory ++ // handle in both measurements, so equality remains deterministic. ++ let descriptors_before = std::fs::read_dir("/proc/self/fd").unwrap().count(); ++ for _ in 0..16 { ++ drop(UnixHostLifecycleSupervisor::install().unwrap()); ++ } ++ let descriptors_after = std::fs::read_dir("/proc/self/fd").unwrap().count(); ++ assert_eq!(descriptors_after, descriptors_before); ++ ++ let mut restored_custom_hup = MaybeUninit::::uninit(); ++ assert_eq!( ++ unsafe { ++ libc::sigaction( ++ libc::SIGHUP, ++ std::ptr::null(), ++ restored_custom_hup.as_mut_ptr(), ++ ) ++ }, ++ 0 ++ ); ++ let restored_custom_hup = unsafe { restored_custom_hup.assume_init() }; ++ assert_eq!( ++ restored_custom_hup.sa_sigaction, ++ expected_custom_hup.sa_sigaction ++ ); ++ assert_eq!(restored_custom_hup.sa_flags, expected_custom_hup.sa_flags); ++ for signal in [libc::SIGINT, libc::SIGTERM, libc::SIGQUIT, libc::SIGHUP] { ++ assert_eq!( ++ unsafe { libc::sigismember(&restored_custom_hup.sa_mask, signal) }, ++ unsafe { libc::sigismember(&expected_custom_hup.sa_mask, signal) } ++ ); ++ } ++ ++ assert_eq!( ++ unsafe { libc::sigaction(libc::SIGHUP, &previous_hup, std::ptr::null_mut()) }, ++ 0 ++ ); ++ drop(status); ++ } ++ ++ #[cfg(all(feature = "ctrlc", target_os = "linux", not(target_arch = "wasm32")))] ++ #[test] ++ fn unix_supervisor_real_signals_restore_and_route_exclusively() { ++ const CHILD_ENV: &str = "WASMER_SIGNAL_SUPERVISOR_TEST_CHILD"; ++ if std::env::var_os(CHILD_ENV).is_some() { ++ run_unix_supervisor_child(); ++ return; ++ } ++ ++ let output = std::process::Command::new(std::env::current_exe().unwrap()) ++ .arg("os::task::task_join_handle::tests::unix_supervisor_real_signals_restore_and_route_exclusively") ++ .arg("--exact") ++ .arg("--nocapture") ++ .arg("--test-threads=1") ++ .env(CHILD_ENV, "1") ++ .output() ++ .unwrap(); ++ assert!( ++ output.status.success(), ++ "signal supervisor child failed with {}\nstdout:\n{}\nstderr:\n{}", ++ output.status, ++ String::from_utf8_lossy(&output.stdout), ++ String::from_utf8_lossy(&output.stderr), ++ ); ++ } ++} +diff --git a/lib/wasix/src/os/task/thread.rs b/lib/wasix/src/os/task/thread.rs +index cb7df5f..d079b08 100644 +--- a/lib/wasix/src/os/task/thread.rs ++++ b/lib/wasix/src/os/task/thread.rs +@@ -1,5 +1,5 @@ + use super::{ +- control_plane::TaskCountGuard, ++ control_plane::{ControlPlaneError, TaskCountGuard}, + task_join_handle::{OwnedTaskStatus, TaskJoinHandle}, + }; + use crate::{ +@@ -247,7 +247,7 @@ struct WasiThreadState { + + // Registers the task termination with the ControlPlane on drop. + // Never accessed, since it's a drop guard. +- _task_count_guard: TaskCountGuard, ++ task_count_guard: Mutex>, + } + + static NO_MORE_BYTES: [u8; 0] = [0u8; 0]; +@@ -273,7 +273,7 @@ impl WasiThread { + #[cfg(feature = "journal")] + check_pointing: AtomicBool::new(false), + deep_sleeping: AtomicBool::new(false), +- _task_count_guard: guard, ++ task_count_guard: Mutex::new(Some(guard)), + }), + layout, + start, +@@ -291,6 +291,17 @@ impl WasiThread { + self.state.id + } + ++ pub(crate) fn same_identity(&self, other: &Self) -> bool { ++ Arc::ptr_eq(&self.state, &other.state) ++ } ++ ++ /// Releases this thread's control-plane task reservation exactly once. ++ /// Reusable environments use this when retiring an old execution epoch ++ /// whose thread value may still be referenced by an instance handle. ++ pub(crate) fn retire_task_registration(&self) -> bool { ++ self.state.task_count_guard.lock().unwrap().take().is_some() ++ } ++ + /// Returns true if this thread is the main thread + pub fn is_main(&self) -> bool { + self.state.is_main +@@ -594,13 +605,28 @@ impl WasiThreadHandle { + + impl Drop for WasiThreadHandleProtected { + fn drop(&mut self) { ++ // The handle, not observational WasiThread clones, owns execution ++ // admission. Releasing the last handle must return the task slot even ++ // when an environment snapshot still references the thread state. ++ self.thread.retire_task_registration(); + let id = self.thread.tid(); + if let Some(inner) = Weak::upgrade(&self.inner) { + let mut inner = inner.0.lock().unwrap(); +- if let Some(ctrl) = inner.threads.remove(&id) { ++ let owns_slot = inner ++ .threads ++ .get(&id) ++ .is_some_and(|current| current.same_identity(&self.thread)); ++ if owns_slot { ++ let ctrl = inner ++ .threads ++ .remove(&id) ++ .expect("thread identity was checked under the same lock"); + ctrl.set_status_finished(Ok(Errno::Success.into())); ++ inner.thread_count = inner ++ .thread_count ++ .checked_sub(1) ++ .expect("process thread count underflow"); + } +- inner.thread_count -= 1; + } + } + } +@@ -635,6 +661,10 @@ pub enum WasiThreadError { + /// This will happen if WASM is running in a thread has not been created by the spawn_wasm call + #[error("WASM context is invalid")] + InvalidWasmContext, ++ #[error("Process lifecycle rejected the task - {0}")] ++ ProcessLifecycle(ControlPlaneError), ++ #[error("Failed to capture WASIX store snapshot - {0}")] ++ StoreSnapshotCaptureFailed(crate::StoreSnapshotCaptureError), + } + + impl From for Errno { +@@ -649,6 +679,8 @@ impl From for Errno { + WasiThreadError::InstanceCreateFailed(_) => Errno::Noexec, + WasiThreadError::InitFailed(_) => Errno::Noexec, + WasiThreadError::InvalidWasmContext => Errno::Noexec, ++ WasiThreadError::ProcessLifecycle(_) => Errno::Perm, ++ WasiThreadError::StoreSnapshotCaptureFailed(_) => Errno::Noexec, + } + } + } +diff --git a/lib/wasix/src/syscalls/wasix/futex_wait.rs b/lib/wasix/src/syscalls/wasix/futex_wait.rs +index c9e78a9..7154d3d 100644 +--- a/lib/wasix/src/syscalls/wasix/futex_wait.rs ++++ b/lib/wasix/src/syscalls/wasix/futex_wait.rs +@@ -1,37 +1,36 @@ +-use std::task::Waker; ++use std::{task::Waker, time::Instant}; + + use super::*; + use crate::syscalls::*; + + /// Poller returns true if its triggered and false if it times out + struct FutexPoller { +- state: Arc, ++ registry: WasiFutexRegistry, + poller_idx: u64, + futex_idx: u64, +- expected: u32, + timeout: Option + Send + Sync + 'static>>>, + } + impl Future for FutexPoller { + type Output = bool; + fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll { +- let mut guard = self.state.futexs.lock().unwrap(); +- +- // If the futex itself is no longer registered then it was likely +- // woken by a wake call +- let futex = match guard.futexes.get_mut(&self.futex_idx) { +- Some(f) => f, +- None => return Poll::Ready(true), +- }; +- let waker = match futex.wakers.get_mut(&self.poller_idx) { +- Some(w) => w, +- None => return Poll::Ready(true), +- }; +- +- // Register the waker +- waker.replace(cx.waker().clone()); +- +- // Check for timeout +- drop(guard); ++ let poll_state = self.registry.with(|guard| { ++ // If the futex itself is no longer registered then it was likely ++ // woken by a wake call. ++ let Some(futex) = guard.futexes.get_mut(&self.futex_idx) else { ++ return 0; ++ }; ++ let Some(waker) = futex.wakers.get_mut(&self.poller_idx) else { ++ return 1; ++ }; ++ ++ // Register the waker. ++ waker.replace(cx.waker().clone()); ++ 2 ++ }); ++ if poll_state != 2 { ++ return Poll::Ready(true); ++ } ++ + if let Some(timeout) = self.timeout.as_mut() { + let timeout = timeout.as_mut(); + if timeout.poll(cx).is_ready() { +@@ -46,17 +45,22 @@ impl Future for FutexPoller { + } + impl Drop for FutexPoller { + fn drop(&mut self) { +- let mut guard = self.state.futexs.lock().unwrap(); +- +- let mut should_remove = false; +- if let Some(futex) = guard.futexes.get_mut(&self.futex_idx) { +- if let Some(Some(waker)) = futex.wakers.remove(&self.poller_idx) { +- waker.wake(); ++ let waker_to_wake = self.registry.with(|guard| { ++ let mut should_remove = false; ++ let mut waker_to_wake = None; ++ if let Some(futex) = guard.futexes.get_mut(&self.futex_idx) { ++ if let Some(Some(waker)) = futex.wakers.remove(&self.poller_idx) { ++ waker_to_wake = Some(waker); ++ } ++ should_remove = futex.wakers.is_empty(); + } +- should_remove = futex.wakers.is_empty(); +- } +- if should_remove { +- guard.futexes.remove(&self.futex_idx); ++ if should_remove { ++ guard.futexes.remove(&self.futex_idx); ++ } ++ waker_to_wake ++ }); ++ if let Some(waker) = waker_to_wake { ++ waker.wake(); + } + } + } +@@ -115,8 +119,8 @@ pub(super) fn futex_wait_internal( + }; + Span::current().record("timeout", format!("{timeout:?}")); + +- let state = env.state.clone(); +- let futex_idx: u64 = futex_ptr.offset().into(); ++ let futex_addr: u64 = futex_ptr.offset().into(); ++ let (registry, futex_idx) = wasi_try_ok!(WasiFutexRegistry::resolve(&env.state, futex_addr)); + Span::current().record("futex_idx", futex_idx); + + // We generate a new poller which also registers in the +@@ -125,24 +129,23 @@ pub(super) fn futex_wait_internal( + // removed whenever the wake call is invoked (which could + // be before the poller is polled). + let poller = { +- let mut guard = env.state.futexs.lock().unwrap(); +- guard.poller_seed += 1; +- let poller_idx = guard.poller_seed; +- +- // Create the timeout if one exists + let timeout = timeout.map(|timeout| env.tasks().sleep_now(timeout)); ++ let poller_idx = registry.with(|guard| { ++ guard.poller_seed += 1; ++ let poller_idx = guard.poller_seed; + +- // We insert the futex before we check the condition variable to avoid +- // certain race conditions +- let futex = guard.futexes.entry(futex_idx).or_default(); +- futex.wakers.insert(poller_idx, Default::default()); ++ // We insert the futex before we check the condition variable to avoid ++ // certain race conditions. ++ let futex = guard.futexes.entry(futex_idx).or_default(); ++ futex.wakers.insert(poller_idx, Default::default()); ++ poller_idx ++ }); + + Span::current().record("poller_idx", poller_idx); + FutexPoller { +- state: env.state.clone(), ++ registry, + poller_idx, + futex_idx, +- expected, + timeout, + } + }; +@@ -160,17 +163,16 @@ pub(super) fn futex_wait_internal( + // then the value is not set) - the poller will set it to true + wasi_try_mem_ok!(ret_woken.write(&memory, Bool::False)); + +- // We use asyncify on the poller and potentially go into deep sleep ++ // Wait on the futex while still processing signals. + tracing::trace!("wait on {futex_idx}"); +- let res = __asyncify_with_deep_sleep::(ctx, Box::pin(poller))?; +- if let AsyncifyAction::Finish(ctx, res) = res { +- let mut env = ctx.data(); +- let memory = unsafe { env.memory_view(&ctx) }; +- if res { +- wasi_try_mem_ok!(ret_woken.write(&memory, Bool::True)); +- } else { +- wasi_try_mem_ok!(ret_woken.write(&memory, Bool::False)); +- } ++ let res = __asyncify(&mut ctx, None, async move { Ok(poller.await) })?; ++ let res = wasi_try_ok!(res); ++ let env = ctx.data(); ++ let memory = unsafe { env.memory_view(&ctx) }; ++ if res { ++ wasi_try_mem_ok!(ret_woken.write(&memory, Bool::True)); ++ } else { ++ wasi_try_mem_ok!(ret_woken.write(&memory, Bool::False)); + } + Ok(Errno::Success) + } +diff --git a/lib/wasix/src/syscalls/wasix/futex_wake.rs b/lib/wasix/src/syscalls/wasix/futex_wake.rs +index ae2020d..b63f1ba 100644 +--- a/lib/wasix/src/syscalls/wasix/futex_wake.rs ++++ b/lib/wasix/src/syscalls/wasix/futex_wake.rs +@@ -1,6 +1,19 @@ + use super::*; + use crate::syscalls::*; + ++fn remove_waiter_to_wake(futex: &mut WasiFutex) -> Option<(u64, Option)> { ++ let registered = futex ++ .wakers ++ .iter() ++ .find_map(|(id, waker)| waker.as_ref().map(|_| *id)); ++ if let Some(id) = registered { ++ return futex.wakers.remove(&id).map(|waker| (id, waker)); ++ } ++ ++ let first = futex.wakers.keys().copied().next()?; ++ futex.wakers.remove(&first).map(|waker| (first, waker)) ++} ++ + /// Wake up one thread that's blocked on futex_wait on this futex. + /// Returns true if this actually woke up such a thread, + /// or false if no thread was waiting on this futex. +@@ -18,31 +31,34 @@ pub fn futex_wake( + + let env = ctx.data(); + let memory = unsafe { env.memory_view(&ctx) }; +- let state = env.state.deref(); +- +- let pointer: u64 = futex_ptr.offset().into(); +- Span::current().record("futex_idx", pointer); +- +- let mut woken = false; +- let woken = { +- let mut guard = state.futexs.lock().unwrap(); +- if let Some(futex) = guard.futexes.get_mut(&pointer) { +- let first = futex.wakers.keys().copied().next(); +- if let Some(id) = first +- && let Some(Some(w)) = futex.wakers.remove(&id) +- { +- w.wake(); ++ ++ let futex_addr: u64 = futex_ptr.offset().into(); ++ let (registry, futex_idx) = wasi_try_ok!(WasiFutexRegistry::resolve(&env.state, futex_addr)); ++ Span::current().record("futex_idx", futex_idx); ++ let (woken, waker_to_wake) = registry.with(|guard| { ++ if let Some(futex) = guard.futexes.get_mut(&futex_idx) { ++ let mut waker_to_wake = None; ++ if let Some((_poller_idx, waker)) = remove_waiter_to_wake(futex) { ++ match waker { ++ Some(w) => { ++ waker_to_wake = Some(w); ++ } ++ None => {} ++ } + } + if futex.wakers.is_empty() { +- guard.futexes.remove(&pointer); ++ guard.futexes.remove(&futex_idx); + } +- tracing::trace!("wake(hit) on {pointer}"); +- true ++ tracing::trace!("wake(hit) on {futex_idx}"); ++ (true, waker_to_wake) + } else { +- tracing::trace!("wake(miss) on {pointer}"); +- true ++ tracing::trace!("wake(miss) on {futex_idx}"); ++ (false, None) + } +- }; ++ }); ++ if let Some(waker) = waker_to_wake { ++ waker.wake(); ++ } + Span::current().record("woken", woken); + + let woken = match woken { +@@ -53,3 +69,32 @@ pub fn futex_wake( + + Ok(Errno::Success) + } ++ ++#[cfg(test)] ++mod tests { ++ use futures::task::noop_waker; ++ ++ use super::*; ++ ++ #[test] ++ fn futex_wake_prefers_registered_waker() { ++ let mut futex = WasiFutex::default(); ++ futex.wakers.insert(1, None); ++ futex.wakers.insert(2, Some(noop_waker())); ++ ++ assert!(remove_waiter_to_wake(&mut futex).unwrap().1.is_some()); ++ assert!(futex.wakers.contains_key(&1)); ++ assert!(!futex.wakers.contains_key(&2)); ++ } ++ ++ #[test] ++ fn futex_wake_consumes_unregistered_waiter_when_no_waker_exists() { ++ let mut futex = WasiFutex::default(); ++ futex.wakers.insert(1, None); ++ futex.wakers.insert(2, None); ++ ++ assert!(remove_waiter_to_wake(&mut futex).unwrap().1.is_none()); ++ assert!(!futex.wakers.contains_key(&1)); ++ assert!(futex.wakers.contains_key(&2)); ++ } ++} +diff --git a/lib/wasix/src/syscalls/wasix/futex_wake_all.rs b/lib/wasix/src/syscalls/wasix/futex_wake_all.rs +index 8a74714..88621cc 100644 +--- a/lib/wasix/src/syscalls/wasix/futex_wake_all.rs ++++ b/lib/wasix/src/syscalls/wasix/futex_wake_all.rs +@@ -16,28 +16,29 @@ pub fn futex_wake_all( + + let env = ctx.data(); + let memory = unsafe { env.memory_view(&ctx) }; +- let state = env.state.deref(); + +- let pointer: u64 = futex_ptr.offset().into(); +- //Span::current().record("futex_idx", pointer); +- +- let mut woken = false; +- let woken = { +- let mut guard = state.futexs.lock().unwrap(); +- if let Some(futex) = guard.futexes.remove(&pointer) { +- for waker in futex.wakers { +- if let Some(waker) = waker.1 { +- waker.wake(); ++ let futex_addr: u64 = futex_ptr.offset().into(); ++ let (registry, futex_idx) = wasi_try_ok!(WasiFutexRegistry::resolve(&env.state, futex_addr)); ++ Span::current().record("futex_idx", futex_idx); ++ let (woken, wakers_to_wake) = registry.with(|guard| { ++ if let Some(futex) = guard.futexes.remove(&futex_idx) { ++ let mut wakers_to_wake = Vec::new(); ++ for (poller_idx, waker) in futex.wakers { ++ if let Some(waker) = waker { ++ wakers_to_wake.push(waker); + } + } +- tracing::trace!("wake_all (hit) on {pointer}"); +- true ++ tracing::trace!("wake_all (hit) on {futex_idx}"); ++ (true, wakers_to_wake) + } else { +- tracing::trace!("wake_all (miss) on {pointer}"); +- true ++ tracing::trace!("wake_all (miss) on {futex_idx}"); ++ (false, Vec::new()) + } +- }; +- //Span::current().record("woken", woken); ++ }); ++ for waker in wakers_to_wake { ++ waker.wake(); ++ } ++ Span::current().record("woken", woken); + + let woken = match woken { + false => Bool::False, +diff --git a/lib/wasix/src/syscalls/wasix/proc_exec3.rs b/lib/wasix/src/syscalls/wasix/proc_exec3.rs +index ad76a85..e01ab2e 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_exec3.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_exec3.rs +@@ -152,18 +152,17 @@ pub fn proc_exec3( + // Record the stack offsets before we give up ownership of the wasi_env + let stack_lower = wasi_env.layout.stack_lower; + let stack_upper = wasi_env.layout.stack_upper; ++ let child_tasks = wasi_env.tasks().clone(); + + // Spawn a new process with this current execution environment + let mut err_exit_code: ExitCode = Errno::Success.into(); + + let spawn_result = { + let bin_factory = Box::new(ctx.data().bin_factory.clone()); +- let tasks = wasi_env.tasks().clone(); +- + let mut config = Some(wasi_env); + + match bin_factory.try_built_in(name.clone(), Some(&ctx), &mut config) { +- Ok(a) => Ok(()), ++ Ok(handle) => Ok(handle), + Err(err) => { + if !err.is_not_found() { + error!("builtin failed - {}", err); +@@ -175,9 +174,9 @@ pub fn proc_exec3( + __asyncify_light(ctx.data(), None, async { + let ret = bin_factory.spawn(name_inner, env).await; + match ret { +- Ok(ret) => { ++ Ok(handle) => { + trace!(%child_pid, "spawned sub-process"); +- Ok(()) ++ Ok(handle) + } + Err(err) => { + err_exit_code = conv_spawn_err_to_exit_code(&err); +@@ -203,10 +202,30 @@ pub fn proc_exec3( + ctx.data_mut().vfork = Some(vfork); + return Ok(e); + } +- Ok(()) => { ++ Ok(task_handle) => { ++ // The accepted exec successor now owns the child's command ++ // lifetime. Retain the in-place vfork lease until its returned ++ // task handle is terminal; monitor-admission failure forces and ++ // waits for terminal state before releasing ownership. ++ if let Err(err) = vfork ++ .child_execution ++ .clone() ++ .monitor(task_handle, &child_tasks) ++ { ++ error!(%child_pid, "failed to monitor vfork exec successor: {err}"); ++ } ++ + // We spawned a new process - put the parent env back + ctx.data_mut().swap_inner(&mut vfork.env); + std::mem::swap(ctx.data_mut(), &mut vfork.env); ++ // The ordinary vfork/exec path returns inside the original ++ // parent TaskWasm, so its accepted lease supersedes the ++ // supplemental switch guard. If a deep-sleep continuation ++ // changed that ownership, retain the one guard that remains ++ // authoritative; a later cycle will hand off its duplicate ++ // instead of growing this vector per backend. ++ ctx.data_mut() ++ .restore_parent_execution_guard(vfork.parent_execution); + + let Some(asyncify_info) = vfork.asyncify else { + // vfork without asyncify only forks the WasiEnv, which we have restored +@@ -248,7 +267,7 @@ pub fn proc_exec3( + // on the new module + else { + // Prepare the environment +- let mut wasi_env = ctx.data().clone(); ++ let mut wasi_env = ctx.data_mut().take_for_same_thread_continuation(); + _prepare_wasi(&mut wasi_env, Some(args), envs, None); + + // Get a reference to the runtime +@@ -276,29 +295,13 @@ pub fn proc_exec3( + + match process { + Ok(mut process) => { +- // If we support deep sleeping then we switch to deep sleep mode +- let env = ctx.data(); +- +- let thread = env.thread.clone(); +- +- // The poller will wait for the process to actually finish +- let res = __asyncify_with_deep_sleep::(ctx, async move { +- process +- .wait_finished() +- .await +- .unwrap_or_else(|_| Errno::Child.into()) +- .to_native() +- })?; +- match res { +- AsyncifyAction::Finish(mut ctx, result) => { +- // When we arrive here the process should already be terminated +- let exit_code = ExitCode::from_native(result); +- ctx.data().process.terminate(exit_code); +- WasiEnv::process_signals_and_exit(&mut ctx)?; +- Err(WasiError::Exit(Errno::Unknown.into())) +- } +- AsyncifyAction::Unwind => Ok(Errno::Success), +- } ++ let result = block_on(process.wait_finished()) ++ .unwrap_or_else(|_| Errno::Child.into()) ++ .to_native(); ++ let exit_code = ExitCode::from_native(result); ++ ctx.data().process.terminate(exit_code); ++ WasiEnv::process_signals_and_exit(&mut ctx)?; ++ Err(WasiError::Exit(Errno::Unknown.into())) + } + Err(err) => { + warn!( +diff --git a/lib/wasix/src/syscalls/wasix/proc_exit2.rs b/lib/wasix/src/syscalls/wasix/proc_exit2.rs +index 5ded2f0..d1c3369 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_exit2.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_exit2.rs +@@ -40,9 +40,14 @@ pub fn proc_exit2( + ctx.data_mut().swap_inner(parent_env.as_mut()); + let mut child_env = std::mem::replace(ctx.data_mut(), *parent_env); + +- // Terminate the child process ++ // Parent execution ownership follows the restored environment. The normal ++ // return path is still owned by the original parent TaskWasm; retain the ++ // supplemental switch guard only when a deep-sleep continuation made it ++ // the last authoritative parent owner. ++ ctx.data_mut() ++ .restore_parent_execution_guard(vfork.parent_execution); + child_env.owned_handles.push(vfork.handle); +- child_env.process.terminate(code); ++ vfork.child_execution.finish(Ok(code)); + + let Some(asyncify_info) = vfork.asyncify else { + // vfork without asyncify only forks the WasiEnv, which we have restored +diff --git a/lib/wasix/src/syscalls/wasix/proc_fork.rs b/lib/wasix/src/syscalls/wasix/proc_fork.rs +index 0b282be..a1f7de4 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_fork.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_fork.rs +@@ -1,13 +1,7 @@ + use super::*; +-use crate::{ +- WasiThreadHandle, WasiVForkAsyncify, capture_store_snapshot, +- os::task::OwnedTaskStatus, +- runtime::task_manager::{TaskWasm, TaskWasmRunProperties}, +- state::context_switching::ContextSwitchingEnvironment, +- syscalls::*, +-}; ++use crate::{WasiVForkAsyncify, syscalls::*}; + use serde::{Deserialize, Serialize}; +-use wasmer::Memory; ++use wasmer::AsStoreMut; + + #[derive(Serialize, Deserialize)] + pub(crate) struct ForkResult { +@@ -22,16 +16,11 @@ pub(crate) struct ForkResult { + #[instrument(level = "trace", skip_all, fields(pid = ctx.data().process.pid().raw()), ret)] + pub fn proc_fork( + mut ctx: FunctionEnvMut<'_, WasiEnv>, +- mut copy_memory: Bool, ++ copy_memory: Bool, + pid_ptr: WasmPtr, + ) -> Result { + WasiEnv::do_pending_operations(&mut ctx)?; + +- wasi_try_ok!(ctx.data().ensure_static_module().map_err(|_| { +- warn!("process forking not supported for dynamically linked modules"); +- Errno::Notsup +- })); +- + if let Some(context_switching_environment) = ctx.data().context_switching_environment.as_ref() + && context_switching_environment.active_context_id() + != context_switching_environment.main_context_id() +@@ -56,6 +45,11 @@ pub fn proc_fork( + wasi_try_mem_ok!(pid_ptr.write(&memory, result.pid)); + return Ok(result.ret); + } ++ ++ if copy_memory == Bool::True { ++ warn!("copied-memory fork is unsupported; use vfork followed by exec"); ++ return Ok(Errno::Notsup); ++ } + trace!(%copy_memory, "capturing"); + + if let Some(vfork) = ctx.data().vfork.as_ref() { +@@ -63,11 +57,27 @@ pub fn proc_fork( + return Ok(Errno::Notsup); + } + +- // Fork the environment which will copy all the open file handlers +- // and associate a new context but otherwise shares things like the +- // file system interface. The handle to the forked process is stored +- // in the parent process context +- let (mut child_env, mut child_handle) = match ctx.data().fork() { ++ let supports_asyncify = ctx ++ .data() ++ .inner() ++ .main_module_instance_handles() ++ .supports_asyncify_stack_rewind(); ++ if !supports_asyncify { ++ warn!("process forking requires complete Asyncify stack-rewind exports"); ++ return Ok(Errno::Notsup); ++ } ++ trace!("using Asyncify vfork continuation backend"); ++ ++ let env = ctx.data(); ++ let memory = unsafe { env.memory_view(&ctx) }; ++ ++ // Seed the child return slot before stack capture. The parent continuation ++ // overwrites it with the child pid when it rewinds. ++ wasi_try_mem_ok!(pid_ptr.write(&memory, 0)); ++ ++ // Fork the environment for a vfork continuation. The child shares the ++ // current memory until proc_exec installs a fresh process image. ++ let (mut child_env, child_handle, mut child_registration) = match ctx.data().fork_guarded() { + Ok(p) => p, + Err(err) => { + debug!("could not fork process: {err}"); +@@ -75,279 +85,92 @@ pub fn proc_fork( + return Ok(Errno::Perm); + } + }; +- let child_pid = child_env.process.pid(); +- let child_finished = child_env.process.finished.clone(); +- +- // We write a zero to the PID before we capture the stack +- // so that this is what will be returned to the child +- { +- let mut inner = ctx.data().process.lock(); +- inner.children.push(child_env.process.clone()); +- } +- let env = ctx.data(); +- let memory = unsafe { env.memory_view(&ctx) }; +- +- // Setup some properties in the child environment +- wasi_try_mem_ok!(pid_ptr.write(&memory, 0)); +- let pid = child_env.pid(); +- let tid = child_env.tid(); +- +- // Pass some offsets to the unwind function +- let pid_offset = pid_ptr.offset(); +- +- // If we are not copying the memory then we act like a `vfork` +- // instead which will pretend to be the new process for a period +- // of time until `proc_exec` is called at which point the fork +- // actually occurs +- if copy_memory == Bool::False { +- // Perform the unwind action +- return unwind::(ctx, move |mut ctx, mut memory_stack, rewind_stack| { +- // Grab all the globals and serialize them +- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) +- .serialize() +- .unwrap(); +- let store_data = Bytes::from(store_data); +- +- // We first fork the environment and replace the current environment +- // so that the process can continue to prepare for the real fork as +- // if it had actually forked +- child_env.swap_inner(ctx.data_mut()); +- std::mem::swap(ctx.data_mut(), &mut child_env); +- let previous_vfork = ctx.data_mut().vfork.replace(WasiVFork { +- asyncify: Some(WasiVForkAsyncify { +- rewind_stack: rewind_stack.clone(), +- store_data: store_data.clone(), +- is_64bit: M::is_64bit(), +- }), +- env: Box::new(child_env), +- handle: child_handle, +- }); +- assert!(previous_vfork.is_none()); // Already checked above +- +- // Carry on as if the fork had taken place (which basically means +- // it prevents to be the new process with the old one suspended) +- // Rewind the stack and carry on +- match rewind::( +- ctx, +- Some(memory_stack.freeze()), +- rewind_stack.freeze(), +- store_data, +- ForkResult { +- pid: 0, +- ret: Errno::Success, +- }, +- ) { +- Errno::Success => OnCalledAction::InvokeAgain, +- err => { +- warn!("failed - could not rewind the stack - errno={}", err); +- OnCalledAction::Trap(Box::new(WasiError::Exit(err.into()))) +- } +- } +- }); +- } +- +- // Create the thread that will back this forked process +- let state = env.state.clone(); +- let bin_factory = env.bin_factory.clone(); +- +- // Perform the unwind action +- let snapshot = capture_store_snapshot(&mut ctx.as_store_mut()); +- unwind::(ctx, move |mut ctx, mut memory_stack, rewind_stack| { +- let tasks = ctx.data().tasks().clone(); +- let span = debug_span!( +- "unwind", +- memory_stack_len = memory_stack.len(), +- rewind_stack_len = rewind_stack.len() +- ); +- let _span_guard = span.enter(); +- let memory_stack = memory_stack.freeze(); +- let rewind_stack = rewind_stack.freeze(); +- ++ unwind::(ctx, move |mut ctx, memory_stack, rewind_stack| { + // Grab all the globals and serialize them +- let store_data = snapshot.serialize().unwrap(); ++ let snapshot = match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { ++ Ok(snapshot) => snapshot, ++ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), ++ }; ++ let store_data = { snapshot.serialize().unwrap() }; + let store_data = Bytes::from(store_data); +- +- // Now we use the environment and memory references +- let runtime = child_env.runtime.clone(); +- let tasks = child_env.tasks().clone(); +- let child_memory_stack = memory_stack.clone(); +- let child_rewind_stack = rewind_stack.clone(); +- +- let env_inner = ctx.data().inner(); +- let instance_handles = env_inner.static_module_instance_handles().unwrap(); +- let module = instance_handles.module_clone(); +- let memory = instance_handles.memory_clone(); +- let spawn_type = SpawnType::CopyMemory(memory, ctx.as_store_ref()); +- +- // Spawn a new process with this current execution environment +- let signaler = Box::new(child_env.process.clone()); +- { +- let runtime = runtime.clone(); +- let tasks = tasks.clone(); +- let tasks_outer = tasks.clone(); +- let store_data = store_data.clone(); +- +- let run = move |mut props: TaskWasmRunProperties| { +- let ctx = props.ctx; +- let mut store = props.store; +- +- // Rewind the stack and carry on +- { +- trace!("rewinding child"); +- let mut ctx = ctx.env.clone().into_mut(&mut store); +- let (data, mut store) = ctx.data_and_store_mut(); +- match rewind::( +- ctx, +- Some(child_memory_stack), +- child_rewind_stack, +- store_data.clone(), +- ForkResult { +- pid: 0, +- ret: Errno::Success, +- }, +- ) { +- Errno::Success => OnCalledAction::InvokeAgain, +- err => { +- warn!( +- "wasm rewind failed - could not rewind the stack - errno={}", +- err +- ); +- return; +- } +- }; +- } +- +- // Invoke the start function +- run::(ctx, store, child_handle, None); +- }; +- +- tasks_outer +- .task_wasm( +- TaskWasm::new(Box::new(run), child_env, module, false, false) +- .with_globals(snapshot) +- .with_memory(spawn_type), +- ) +- .map_err(|err| { +- warn!( +- "failed to fork as the process could not be spawned - {}", +- err +- ); +- err +- }) +- .ok(); ++ let parent_tasks = ctx.data().tasks().clone(); ++ ++ // Publish and lease both sides before the in-place process switch. ++ // The current TaskWasm continues to own its original parent lease; ++ // these guards make both identities explicit across deep sleep and ++ // the later parent-resume transition. ++ if let Err(err) = child_registration.commit_child() { ++ warn!("failed to publish forked process: {err}"); ++ return OnCalledAction::Trap(Box::new(WasiError::Exit(Errno::Perm.into()))); ++ } ++ let child_execution = match child_env.process.acquire_execution_guard() { ++ Ok(execution) => execution, ++ Err(err) => { ++ warn!("failed to lease forked child execution: {err}"); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ return OnCalledAction::Trap(Box::new(WasiError::Exit(Errno::Perm.into()))); ++ } ++ }; ++ let parent_execution = match ctx.data().acquire_parent_execution_guard() { ++ Ok(execution) => execution, ++ Err(err) => { ++ warn!("failed to retain vfork parent execution: {err}"); ++ child_execution.finish(Ok(Errno::Canceled.into())); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ return OnCalledAction::Trap(Box::new(WasiError::Exit(Errno::Perm.into()))); ++ } + }; + ++ // Replace the current environment only after both process leases ++ // coexist. The parent environment remains local until rewind has ++ // succeeded, making failure rollback non-blocking and exact. ++ child_env.swap_inner(ctx.data_mut()); ++ std::mem::swap(ctx.data_mut(), &mut child_env); ++ ++ // Carry on as if the fork had taken place (which basically means ++ // it prevents to be the new process with the old one suspended) + // Rewind the stack and carry on + match rewind::( +- ctx, +- Some(memory_stack), +- rewind_stack, +- store_data, ++ ctx.as_mut(), ++ Some(memory_stack.freeze()), ++ rewind_stack.clone().freeze(), ++ store_data.clone(), + ForkResult { +- pid: child_pid.raw() as Pid, ++ pid: 0, + ret: Errno::Success, + }, + ) { +- Errno::Success => OnCalledAction::InvokeAgain, ++ Errno::Success => { ++ parent_execution.arm_fail_closed(); ++ let previous_vfork = ctx.data_mut().vfork.replace(WasiVFork { ++ asyncify: Some(WasiVForkAsyncify { ++ rewind_stack: rewind_stack.clone(), ++ store_data: store_data.clone(), ++ is_64bit: M::is_64bit(), ++ }), ++ env: Box::new(child_env), ++ handle: child_handle, ++ parent_execution, ++ child_execution, ++ }); ++ assert!(previous_vfork.is_none()); // Already checked above ++ child_registration.complete_child_launch(&parent_tasks); ++ OnCalledAction::InvokeAgain ++ } + err => { + warn!("failed - could not rewind the stack - errno={}", err); ++ // Restore the parent before releasing either side of the ++ // failed switch. The original parent TaskWasm still owns ++ // execution, so its supplemental guard is a safe handoff. ++ ctx.data_mut().swap_inner(&mut child_env); ++ let mut failed_child = std::mem::replace(ctx.data_mut(), child_env); ++ failed_child.owned_handles.push(child_handle); ++ child_execution.finish(Ok(err.into())); ++ ctx.data_mut() ++ .restore_parent_execution_guard(parent_execution); ++ child_registration.rollback_child(err.into()); + OnCalledAction::Trap(Box::new(WasiError::Exit(err.into()))) + } + } + }) + } +- +-fn run( +- ctx: WasiFunctionEnv, +- mut store: Store, +- child_handle: WasiThreadHandle, +- rewind_state: Option<(RewindState, RewindResultType)>, +-) -> ExitCode { +- let env = ctx.data(&store); +- let tasks = env.tasks().clone(); +- let pid = env.pid(); +- let tid = env.tid(); +- +- // If we need to rewind then do so +- if let Some((rewind_state, rewind_result)) = rewind_state { +- let mut ctx = ctx.env.clone().into_mut(&mut store); +- let res = rewind_ext::( +- &mut ctx, +- Some(rewind_state.memory_stack), +- rewind_state.rewind_stack, +- rewind_state.store_data, +- rewind_result, +- ); +- if res != Errno::Success { +- return res.into(); +- } +- } +- +- let mut ret: ExitCode = Errno::Success.into(); +- let (mut store, err) = if ctx.data(&store).thread.is_main() { +- trace!(%pid, %tid, "re-invoking main"); +- let start = ctx +- .data(&store) +- .inner() +- .static_module_instance_handles() +- .unwrap() +- .start +- .clone() +- .unwrap(); +- ContextSwitchingEnvironment::run_main_context(&ctx, store, start.into(), vec![]) +- } else { +- trace!(%pid, %tid, "re-invoking thread_spawn"); +- let start = ctx +- .data(&store) +- .inner() +- .static_module_instance_handles() +- .unwrap() +- .thread_spawn +- .clone() +- .unwrap(); +- let params = vec![0i32.into(), 0i32.into()]; +- ContextSwitchingEnvironment::run_main_context(&ctx, store, start.into(), params) +- }; +- if let Err(err) = err { +- match err.downcast::() { +- Ok(WasiError::Exit(exit_code)) => { +- ret = exit_code; +- } +- Ok(WasiError::DeepSleep(deep)) => { +- trace!(%pid, %tid, "entered a deep sleep"); +- +- // Create the respawn function +- let respawn = { +- let tasks = tasks.clone(); +- let rewind_state = deep.rewind; +- move |ctx, store, rewind_result| { +- run::( +- ctx, +- store, +- child_handle, +- Some(( +- rewind_state, +- RewindResultType::RewindWithResult(rewind_result), +- )), +- ); +- } +- }; +- +- /// Spawns the WASM process after a trigger +- unsafe { +- tasks.resume_wasm_after_poller(Box::new(respawn), ctx, store, deep.trigger) +- }; +- return Errno::Success.into(); +- } +- _ => {} +- } +- } +- trace!(%pid, %tid, "child exited (code = {})", ret); +- +- // Clean up the environment and return the result +- ctx.on_exit((&mut store), Some(ret)); +- +- // We drop the handle at the last moment which will close the thread +- drop(child_handle); +- ret +-} +diff --git a/lib/wasix/src/syscalls/wasix/proc_fork_env.rs b/lib/wasix/src/syscalls/wasix/proc_fork_env.rs +index 0c948ee..62ad17d 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_fork_env.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_fork_env.rs +@@ -43,7 +43,7 @@ pub fn proc_fork_env( + // and associate a new context but otherwise shares things like the + // file system interface. The handle to the forked process is stored + // in the parent process context +- let (mut child_env, child_handle) = match env.fork() { ++ let (mut child_env, child_handle, mut child_registration) = match env.fork_guarded() { + Ok(p) => p, + Err(err) => { + tracing::error!("Could not fork process: {err}"); +@@ -55,25 +55,54 @@ pub fn proc_fork_env( + // Write the child's PID to the provided pointer + let memory = unsafe { env.memory_view(&ctx) }; + wasi_try_mem_ok!(child_pid_ptr.write(&memory, child_env.pid().raw())); ++ drop(memory); + ++ // Add the child to the parent's list of children and notify the parent ++ // when the child exits. ++ let commit_result = { child_registration.commit_child() }; ++ if let Err(err) = commit_result { ++ tracing::error!("Could not publish forked process: {err}"); ++ // The ABI promises that failure does not leave a usable child PID. ++ let memory = unsafe { ctx.data().memory_view(&ctx) }; ++ wasi_try_mem_ok!(child_pid_ptr.write(&memory, 0)); ++ return Ok(Errno::Perm); ++ } ++ let child_execution = match child_env.process.acquire_execution_guard() { ++ Ok(execution) => execution, ++ Err(err) => { ++ tracing::error!("Could not lease forked child execution: {err}"); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ let memory = unsafe { ctx.data().memory_view(&ctx) }; ++ wasi_try_mem_ok!(child_pid_ptr.write(&memory, 0)); ++ return Ok(Errno::Perm); ++ } ++ }; ++ let parent_execution = match ctx.data().acquire_parent_execution_guard() { ++ Ok(execution) => execution, ++ Err(err) => { ++ tracing::error!("Could not retain vfork parent execution: {err}"); ++ child_execution.finish(Ok(Errno::Canceled.into())); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ let memory = unsafe { ctx.data().memory_view(&ctx) }; ++ wasi_try_mem_ok!(child_pid_ptr.write(&memory, 0)); ++ return Ok(Errno::Perm); ++ } ++ }; + let parent_env = ctx.data_mut(); +- +- // Add the child to the parent's list of children +- parent_env +- .process +- .lock() +- .children +- .push(child_env.process.clone()); + // Swap the current environment with the child environment + child_env.swap_inner(parent_env); + std::mem::swap(parent_env, &mut child_env); + ++ parent_execution.arm_fail_closed(); + let previous_vfork = parent_env.vfork.replace(WasiVFork { + asyncify: None, + env: Box::new(child_env), + handle: child_handle, ++ parent_execution, ++ child_execution, + }); + assert!(previous_vfork.is_none()); // Already checked at the start of the function ++ child_registration.complete_child_launch(ctx.data().tasks()); + + Ok(Errno::Success) + } +diff --git a/lib/wasix/src/syscalls/wasix/proc_join.rs b/lib/wasix/src/syscalls/wasix/proc_join.rs +index 87c12ea..688b574 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_join.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_join.rs +@@ -7,13 +7,48 @@ use wasmer_wasix_types::wasi::{JoinFlags, JoinStatus, JoinStatusType, JoinStatus + use super::*; + use crate::{WasiProcess, syscalls::*}; + +-#[derive(Serialize, Deserialize)] ++#[derive(Debug, PartialEq, Eq, Serialize, Deserialize)] + enum JoinStatusResult { + Nothing, + ExitNormal(WasiProcessId, ExitCode), + Err(Errno), + } + ++async fn wait_for_any_child(mut process: WasiProcess) -> JoinStatusResult { ++ match process.join_any_child().await { ++ Ok(Some((pid, exit_code))) => { ++ tracing::trace!(%pid, %exit_code, "triggered child join"); ++ trace!(ret_id = pid.raw(), exit_code = exit_code.raw()); ++ JoinStatusResult::ExitNormal(pid, exit_code) ++ } ++ Ok(None) => { ++ tracing::trace!("triggered child join (no child)"); ++ JoinStatusResult::Err(Errno::Child) ++ } ++ Err(err) => { ++ tracing::trace!(%err, "error triggered on child join"); ++ JoinStatusResult::Err(err) ++ } ++ } ++} ++ ++async fn wait_for_explicit_process( ++ parent: Option, ++ process: WasiProcess, ++ pid: WasiProcessId, ++) -> JoinStatusResult { ++ let Some(status) = process.claim_join().await else { ++ return JoinStatusResult::Nothing; ++ }; ++ if let Some(parent) = parent { ++ parent.remove_child_if_same(&process); ++ } ++ process.reap(); ++ let exit_code = status.unwrap_or_else(|_| Errno::Child.into()); ++ tracing::trace!(%exit_code, "triggered child join"); ++ JoinStatusResult::ExitNormal(pid, exit_code) ++} ++ + /// ### `proc_join()` + /// Joins the child process, blocking this one until the other finishes + /// +@@ -118,27 +153,31 @@ pub(super) fn proc_join_internal( + None => { + let mut process = ctx.data_mut().process.clone(); + +- // We wait for any process to exit (if it takes too long +- // then we go into a deep sleep) +- let res = __asyncify_with_deep_sleep::(ctx, async move { +- let child_exit = process.join_any_child().await; +- match child_exit { ++ if flags.contains(JoinFlags::NON_BLOCKING) { ++ let result = match process.try_join_any_child() { + Ok(Some((pid, exit_code))) => { + tracing::trace!(%pid, %exit_code, "triggered child join"); + trace!(ret_id = pid.raw(), exit_code = exit_code.raw()); + JoinStatusResult::ExitNormal(pid, exit_code) + } + Ok(None) => { +- tracing::trace!("triggered child join (no child)"); +- JoinStatusResult::Err(Errno::Child) ++ tracing::trace!("nonblocking child join found no exited child"); ++ JoinStatusResult::Nothing + } + Err(err) => { + tracing::trace!(%err, "error triggered on child join"); + JoinStatusResult::Err(err) + } +- } +- })?; +- return match res { ++ }; ++ return ret_result(ctx, result); ++ } ++ ++ // A postmaster wait must release the instance and linear memory ++ // after the deep-sleep threshold rather than pinning both for the ++ // lifetime of a child. The future performs claim/removal/reap ++ // before its serializable result is captured for rewind. ++ let action = __asyncify_with_deep_sleep::(ctx, wait_for_any_child(process))?; ++ return match action { + AsyncifyAction::Finish(ctx, result) => ret_result(ctx, result), + AsyncifyAction::Unwind => Ok(Errno::Success), + }; +@@ -151,16 +190,16 @@ pub(super) fn proc_join_internal( + + // Waiting for a process that is an explicit child will join it + // meaning it will no longer be a sub-process of the main process +- let mut process = { +- let mut inner = ctx.data().process.lock(); ++ let (mut process, is_child_process) = { ++ let inner = ctx.data().process.lock(); + let process = inner + .children + .iter() + .filter(|c| c.pid == pid) + .map(Clone::clone) + .next(); +- inner.children.retain(|c| c.pid != pid); +- process ++ let is_child_process = process.is_some(); ++ (process, is_child_process) + }; + + // Otherwise it could be the case that we are waiting for a process +@@ -180,21 +219,24 @@ pub(super) fn proc_join_internal( + )); + + if flags.contains(JoinFlags::NON_BLOCKING) { +- if let Some(status) = process.try_join() { ++ if let Some(status) = process.try_claim_join() { ++ if is_child_process { ++ ctx.data().process.remove_child_if_same(&process); ++ } + let exit_code = status.unwrap_or_else(|_| Errno::Child.into()); ++ process.reap(); + ret_result(ctx, JoinStatusResult::ExitNormal(pid, exit_code)) + } else { + ret_result(ctx, JoinStatusResult::Nothing) + } + } else { + // Wait for the process to finish +- let process2 = process.clone(); +- let res = __asyncify_with_deep_sleep::(ctx, async move { +- let exit_code = process.join().await.unwrap_or_else(|_| Errno::Child.into()); +- tracing::trace!(%exit_code, "triggered child join"); +- JoinStatusResult::ExitNormal(pid, exit_code) +- })?; +- match res { ++ let parent = is_child_process.then(|| ctx.data().process.clone()); ++ let action = __asyncify_with_deep_sleep::( ++ ctx, ++ wait_for_explicit_process(parent, process, pid), ++ )?; ++ match action { + AsyncifyAction::Finish(ctx, result) => ret_result(ctx, result), + AsyncifyAction::Unwind => Ok(Errno::Success), + } +@@ -204,3 +246,112 @@ pub(super) fn proc_join_internal( + ret_result(ctx, JoinStatusResult::Nothing) + } + } ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use crate::{WasiControlPlane, os::task::thread::WasiMemoryLayout}; ++ use wasmer_types::ModuleHash; ++ ++ #[test] ++ fn explicit_blocking_wait_claims_and_reaps_before_serialization() { ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, _child_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let child_pid = child.pid(); ++ parent.lock().children.push(child.clone()); ++ child.terminate(Errno::Success.into()); ++ ++ let result = virtual_mio::block_on(wait_for_explicit_process( ++ Some(parent.clone()), ++ child.clone(), ++ child_pid, ++ )); ++ ++ assert_eq!( ++ result, ++ JoinStatusResult::ExitNormal(child_pid, Errno::Success.into()) ++ ); ++ assert!(parent.lock().children.is_empty()); ++ assert!(plane.get_process(child_pid).is_none()); ++ ++ let encoded = bincode::serde::encode_to_vec(&result, bincode::config::legacy()).unwrap(); ++ let (decoded, _): (JoinStatusResult, usize) = ++ bincode::serde::decode_from_slice(&encoded, bincode::config::legacy()).unwrap(); ++ assert_eq!(decoded, result); ++ ++ assert_eq!( ++ virtual_mio::block_on(wait_for_explicit_process(Some(parent), child, child_pid,)), ++ JoinStatusResult::Nothing ++ ); ++ } ++ ++ #[test] ++ fn any_child_blocking_wait_returns_one_serializable_claim() { ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, _child_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let child_pid = child.pid(); ++ parent.lock().children.push(child.clone()); ++ child.terminate(Errno::Success.into()); ++ ++ let result = virtual_mio::block_on(wait_for_any_child(parent.clone())); ++ assert_eq!( ++ result, ++ JoinStatusResult::ExitNormal(child_pid, Errno::Success.into()) ++ ); ++ assert!(parent.lock().children.is_empty()); ++ assert!(plane.get_process(child_pid).is_none()); ++ } ++ ++ #[test] ++ fn blocking_join_payload_survives_beyond_the_deep_sleep_threshold() { ++ use std::{ ++ thread, ++ time::{Duration, Instant}, ++ }; ++ ++ let plane = WasiControlPlane::default(); ++ let (parent, _parent_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let (child, _child_main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let child_pid = child.pid(); ++ parent.lock().children.push(child.clone()); ++ ++ let terminating_child = child.clone(); ++ let terminator = thread::spawn(move || { ++ thread::sleep(Duration::from_millis(75)); ++ terminating_child.terminate(Errno::Success.into()); ++ }); ++ let started = Instant::now(); ++ let result = virtual_mio::block_on(wait_for_any_child(parent.clone())); ++ let elapsed = started.elapsed(); ++ terminator.join().unwrap(); ++ ++ assert!( ++ elapsed >= Duration::from_millis(50), ++ "join completed before the deep-sleep threshold: {elapsed:?}" ++ ); ++ assert_eq!( ++ result, ++ JoinStatusResult::ExitNormal(child_pid, Errno::Success.into()) ++ ); ++ let encoded = bincode::serde::encode_to_vec(&result, bincode::config::legacy()).unwrap(); ++ let (decoded, _): (JoinStatusResult, usize) = ++ bincode::serde::decode_from_slice(&encoded, bincode::config::legacy()).unwrap(); ++ assert_eq!(decoded, result); ++ assert!(parent.lock().children.is_empty()); ++ assert!(plane.get_process(child_pid).is_none()); ++ } ++} +diff --git a/lib/wasix/src/syscalls/wasix/proc_rlimit_get.rs b/lib/wasix/src/syscalls/wasix/proc_rlimit_get.rs +new file mode 100644 +index 0000000..ecb4021 +--- /dev/null ++++ b/lib/wasix/src/syscalls/wasix/proc_rlimit_get.rs +@@ -0,0 +1,38 @@ ++use super::*; ++use crate::syscalls::*; ++ ++const RLIMIT_STACK: u32 = 3; ++const RLIMIT_CORE: u32 = 4; ++const RLIMIT_NOFILE: u32 = 7; ++const RLIM_INFINITY: u64 = u64::MAX; ++ ++/// ### `proc_rlimit_get()` ++/// Returns the current and maximum resource limit for the current process. ++#[instrument(level = "trace", skip_all, fields(%resource, cur = field::Empty, max = field::Empty), ret)] ++pub fn proc_rlimit_get( ++ ctx: FunctionEnvMut<'_, WasiEnv>, ++ resource: u32, ++ ret_cur: WasmPtr, ++ ret_max: WasmPtr, ++) -> Errno { ++ let limit = match resource { ++ RLIMIT_STACK => ctx ++ .data() ++ .runtime ++ .resource_limits() ++ .stack ++ .unwrap_or(RLIM_INFINITY), ++ RLIMIT_CORE => 0, ++ RLIMIT_NOFILE => RLIM_INFINITY, ++ _ => return Errno::Inval, ++ }; ++ ++ Span::current().record("cur", limit); ++ Span::current().record("max", limit); ++ ++ let env = ctx.data(); ++ let memory = unsafe { env.memory_view(&ctx) }; ++ wasi_try_mem!(ret_cur.write(&memory, limit)); ++ wasi_try_mem!(ret_max.write(&memory, limit)); ++ Errno::Success ++} +diff --git a/lib/wasix/src/syscalls/wasix/proc_signal.rs b/lib/wasix/src/syscalls/wasix/proc_signal.rs +index 9c80640..acaa5c2 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_signal.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_signal.rs +@@ -1,4 +1,5 @@ + use super::*; ++use crate::WasiControlPlane; + use crate::syscalls::*; + + /// ### `proc_signal()` +@@ -14,15 +15,69 @@ pub fn proc_signal( + pid: Pid, + sig: Signal, + ) -> Result { +- let process = { +- let pid: WasiProcessId = pid.into(); +- ctx.data().control_plane.get_process(pid) ++ let result = signal_registered_process(&ctx.data().control_plane, pid, sig); ++ ++ WasiEnv::do_pending_operations(&mut ctx)?; ++ ++ Ok(result) ++} ++ ++fn signal_registered_process(control_plane: &WasiControlPlane, pid: Pid, sig: Signal) -> Errno { ++ let pid: WasiProcessId = pid.into(); ++ let Some(process) = control_plane.get_process(pid) else { ++ return Errno::Srch; + }; +- if let Some(process) = process { ++ ++ // POSIX signal zero is a liveness/permission probe. WASIX has no process ++ // permission model here, so registry presence is the complete answer and ++ // no signal may enter the guest's pending queue. ++ if sig != Signal::Signone { + process.signal_process(sig); + } ++ Errno::Success ++} + +- WasiEnv::do_pending_operations(&mut ctx)?; ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use crate::os::task::thread::WasiMemoryLayout; ++ use wasmer_types::ModuleHash; + +- Ok(Errno::Success) ++ #[test] ++ fn absent_pid_returns_srch_for_liveness_probe() { ++ let plane = WasiControlPlane::default(); ++ ++ assert_eq!( ++ signal_registered_process(&plane, 42, Signal::Signone), ++ Errno::Srch ++ ); ++ } ++ ++ #[test] ++ fn signal_zero_observes_existing_pid_without_delivery() { ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ ++ assert_eq!( ++ signal_registered_process(&plane, process.pid().raw(), Signal::Signone), ++ Errno::Success ++ ); ++ assert!(main.pop_signals().is_empty()); ++ } ++ ++ #[test] ++ fn real_signal_delivery_is_unchanged() { ++ let plane = WasiControlPlane::default(); ++ let (process, main) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ ++ assert_eq!( ++ signal_registered_process(&plane, process.pid().raw(), Signal::Sigusr1), ++ Errno::Success ++ ); ++ assert_eq!(main.pop_signals(), vec![Signal::Sigusr1]); ++ } + } +diff --git a/lib/wasix/src/syscalls/wasix/proc_spawn.rs b/lib/wasix/src/syscalls/wasix/proc_spawn.rs +index 8847178..39348b1 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_spawn.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_spawn.rs +@@ -106,22 +106,25 @@ pub fn proc_spawn_internal( + let env = ctx.data(); + + // Fork the current environment and set the new arguments +- let (mut child_env, handle) = match ctx.data().fork() { ++ let (mut child_env, child_handle, mut child_registration) = match ctx.data().fork_guarded() { + Ok(x) => x, + Err(err) => { + // TODO: evaluate the appropriate error code, document it in the spec. + return Ok(Err(Errno::Access)); + } + }; +- let child_process = child_env.process.clone(); + if let Some(args) = args { +- let mut child_state = env.state.fork(); +- child_state.args = std::sync::Mutex::new(args); +- child_env.state = Arc::new(child_state); ++ child_env.state = match env.state.fork_with(move |state| { ++ state.args = std::sync::Mutex::new(args); ++ }) { ++ Ok(state) => state, ++ Err(errno) => { ++ warn!(%errno, "could not fork spawned-process state"); ++ return Ok(Err(errno)); ++ } ++ }; + } + +- // Take ownership of this child +- ctx.data_mut().owned_handles.push(handle); + let env = ctx.data(); + + // Preopen +@@ -230,9 +233,26 @@ pub fn proc_spawn_internal( + // Create the new process + let bin_factory = Box::new(ctx.data().bin_factory.clone()); + let child_pid = child_env.pid(); ++ let child_process = child_env.process.clone(); + + let mut builder = Some(child_env); + ++ // Publish the embryonic child before entering arbitrary built-in code or ++ // Wasm instantiation (which may execute a module start function). ++ let tasks = ctx.data().tasks().clone(); ++ if let Err(err) = child_registration.commit_child() { ++ error!(child_pid = %child_pid, "failed to publish child process: {err}"); ++ return Ok(Err(Errno::Perm)); ++ } ++ let launch_execution = match child_process.acquire_execution_guard() { ++ Ok(execution) => execution, ++ Err(err) => { ++ error!(child_pid = %child_pid, "failed to lease child launch: {err}"); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ return Ok(Err(Errno::Perm)); ++ } ++ }; ++ + // First we try the built in commands + let mut process = match bin_factory.try_built_in(name.clone(), Some(&ctx), &mut builder) { + Ok(a) => a, +@@ -241,23 +261,29 @@ pub fn proc_spawn_internal( + error!("builtin failed - {}", err); + } + // Now we actually spawn the process +- let child_work = bin_factory.spawn(name, builder.take().unwrap()); ++ let env = builder.take().unwrap(); + +- match __asyncify(&mut ctx, None, async move { Ok(child_work.await) })? +- .map_err(|err| Errno::Unknown) +- { +- Ok(Ok(a)) => a, +- Ok(Err(err)) => return Ok(Err(conv_spawn_err_to_errno(&err))), +- Err(err) => return Ok(Err(err)), ++ match block_on(bin_factory.spawn(name, env)) { ++ Ok(a) => a, ++ Err(err) => { ++ let errno = conv_spawn_err_to_errno(&err); ++ launch_execution.finish(Ok(errno.into())); ++ child_registration.rollback_child(errno.into()); ++ return Ok(Err(errno)); ++ } + } + } + }; + +- // Add the process to the environment state +- { +- let mut inner = ctx.data().process.lock(); +- inner.children.push(child_process); ++ if let Err(err) = launch_execution.monitor(process.clone(), &tasks) { ++ error!(child_pid = %child_pid, "failed to monitor child launch: {err}"); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ return Ok(Err(Errno::Canceled)); + } ++ child_registration.complete_child_launch(&tasks); ++ // `child_env` owns its main handle for the full task lifetime. The ++ // syscall's construction handle must not leak into the parent epoch. ++ drop(child_handle); + let env = ctx.data(); + let memory = unsafe { env.memory_view(&ctx) }; + +diff --git a/lib/wasix/src/syscalls/wasix/proc_spawn2.rs b/lib/wasix/src/syscalls/wasix/proc_spawn2.rs +index e447e64..6a813a5 100644 +--- a/lib/wasix/src/syscalls/wasix/proc_spawn2.rs ++++ b/lib/wasix/src/syscalls/wasix/proc_spawn2.rs +@@ -123,7 +123,7 @@ pub fn proc_spawn2( + // and associate a new context but otherwise shares things like the + // file system interface. The handle to the forked process is stored + // in the parent process context +- let (mut child_env, mut child_handle) = match ctx.data().fork() { ++ let (mut child_env, child_handle, mut child_registration) = match ctx.data().fork_guarded() { + Ok(p) => p, + Err(err) => { + debug!("could not fork process: {err}"); +@@ -132,14 +132,10 @@ pub fn proc_spawn2( + } + }; + +- { +- let mut inner = ctx.data().process.lock(); +- inner.children.push(child_env.process.clone()); +- } +- + // Setup some properties in the child environment + let pid = child_env.pid(); + let tid = child_env.tid(); ++ let child_process = child_env.process.clone(); + wasi_try_mem_ok!(ret.write(&memory, pid.raw())); + Span::current() + .record("pid", pid.raw()) +@@ -156,6 +152,25 @@ pub fn proc_spawn2( + + let mut builder = Some(child_env); + ++ // Instance creation can run guest start/initializer code. Make the exact ++ // child identity visible in its parent and registry before resolving any ++ // launch implementation, including arbitrary built-ins. ++ let tasks = ctx.data().tasks().clone(); ++ if let Err(err) = child_registration.commit_child() { ++ error!(child_pid = %pid, "failed to publish child process: {err}"); ++ let _ = ret.write(&memory, 0); ++ return Ok(Errno::Perm); ++ } ++ let launch_execution = match child_process.acquire_execution_guard() { ++ Ok(execution) => execution, ++ Err(err) => { ++ error!(child_pid = %pid, "failed to lease child launch: {err}"); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ let _ = ret.write(&memory, 0); ++ return Ok(Errno::Perm); ++ } ++ }; ++ + let process = match bin_factory.try_built_in(name.clone(), Some(&ctx), &mut builder) { + Ok(a) => Ok(a), + Err(err) => { +@@ -171,8 +186,17 @@ pub fn proc_spawn2( + }; + + match process { +- Ok(_) => { +- ctx.data_mut().owned_handles.push(child_handle); ++ Ok(handle) => { ++ if let Err(err) = launch_execution.monitor(handle, &tasks) { ++ error!(child_pid = %pid, "failed to monitor child launch: {err}"); ++ child_registration.rollback_child(Errno::Canceled.into()); ++ let _ = ret.write(&memory, 0); ++ return Ok(Errno::Canceled); ++ } ++ child_registration.complete_child_launch(&tasks); ++ // The launched child environment owns its main handle. Keeping a ++ // clone in the parent makes RSS and task slots grow per spawn. ++ drop(child_handle); + trace!(child_pid = %pid, "spawned sub-process"); + Ok(Errno::Success) + } +@@ -181,6 +205,9 @@ pub fn proc_spawn2( + + debug!(child_pid = %pid, "process failed with (err={})", err_exit_code); + ++ launch_execution.finish(Ok(err_exit_code)); ++ child_registration.rollback_child(err_exit_code); ++ let _ = ret.write(&memory, 0); + Ok(Errno::Noexec) + } + } +@@ -225,6 +252,8 @@ fn apply_fd_op( + // TODO: verify this is correct + inner: FdInner { + offset: fd_entry.inner.offset.clone(), ++ ofd: fd_entry.inner.ofd.clone(), ++ readdir_cache: fd_entry.inner.readdir_cache.clone(), + rights: fd_entry.inner.rights_inheriting, + fd_flags: { + let mut f = fd_entry.inner.fd_flags; +diff --git a/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs b/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs +index db97b42..1e9af3e 100644 +--- a/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs ++++ b/lib/wasix/src/syscalls/wasix/stack_checkpoint.rs +@@ -45,9 +45,11 @@ pub fn stack_checkpoint( + // Perform the unwind action + unwind::(ctx, move |mut ctx, mut memory_stack, rewind_stack| { + // Grab all the globals and serialize them +- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) +- .serialize() +- .unwrap(); ++ let snapshot = match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { ++ Ok(snapshot) => snapshot, ++ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), ++ }; ++ let store_data = snapshot.serialize().unwrap(); + let env = ctx.data(); + let store_data = Bytes::from(store_data); + +diff --git a/lib/wasix/src/syscalls/wasix/thread_join.rs b/lib/wasix/src/syscalls/wasix/thread_join.rs +index 49c3fb9..ae300f4 100644 +--- a/lib/wasix/src/syscalls/wasix/thread_join.rs ++++ b/lib/wasix/src/syscalls/wasix/thread_join.rs +@@ -34,14 +34,16 @@ pub(super) fn thread_join_internal( + let tid: WasiThreadId = join_tid.into(); + let other_thread = env.process.get_thread(&tid); + if let Some(other_thread) = other_thread { +- let res = __asyncify_with_deep_sleep::(ctx, async move { ++ let res = __asyncify(&mut ctx, None, async move { + other_thread + .join() + .await + .map_err(|err| err.as_exit_code().unwrap_or(ExitCode::from(Errno::Unknown))) + .unwrap_or_else(|a| a) +- .raw() ++ .raw(); ++ Ok(()) + })?; ++ wasi_try_ok!(res); + Ok(Errno::Success) + } else { + Ok(Errno::Success) +diff --git a/lib/wasix/src/syscalls/wasix/thread_sleep.rs b/lib/wasix/src/syscalls/wasix/thread_sleep.rs +index 175f079..ec0614f 100644 +--- a/lib/wasix/src/syscalls/wasix/thread_sleep.rs ++++ b/lib/wasix/src/syscalls/wasix/thread_sleep.rs +@@ -40,9 +40,12 @@ pub(crate) fn thread_sleep_internal( + if duration > 0 { + let duration = Duration::from_nanos(duration); + let tasks = env.tasks().clone(); +- let res = __asyncify_with_deep_sleep::(ctx, async move { ++ if let Err(err) = __asyncify(&mut ctx, None, async move { + tasks.sleep_now(duration).await; +- })?; ++ Ok(()) ++ })? { ++ return Ok(err); ++ } + } + Ok(Errno::Success) + } +diff --git a/lib/wasix/src/syscalls/wasix/thread_spawn.rs b/lib/wasix/src/syscalls/wasix/thread_spawn.rs +index 1e0b4bd..d681ecb 100644 +--- a/lib/wasix/src/syscalls/wasix/thread_spawn.rs ++++ b/lib/wasix/src/syscalls/wasix/thread_spawn.rs +@@ -8,7 +8,7 @@ use crate::{ + os::task::thread::WasiMemoryLayout, + runtime::{ + TaintReason, +- task_manager::{TaskWasm, TaskWasmRunProperties}, ++ task_manager::{TaskWasm, TaskWasmAcceptedExecutionGuard, TaskWasmRunProperties}, + }, + state::context_switching::ContextSwitchingEnvironment, + syscalls::*, +@@ -60,6 +60,10 @@ pub fn thread_spawn_internal_from_wasi( + ) -> Result { + // Now we use the environment and memory references + let env = ctx.data(); ++ if env.vfork.is_some() { ++ tracing::warn!("thread_spawn is undefined in a vfork child"); ++ return Err(Errno::Notsup); ++ } + let memory = unsafe { env.memory_view(&ctx) }; + let runtime = env.runtime.clone(); + let tasks = env.tasks().clone(); +@@ -133,10 +137,7 @@ pub fn thread_spawn_internal_using_layout( + let linker = env_inner.linker().cloned(); + + // We capture some local variables +- let state = env.state.clone(); +- let mut thread_env = env.clone(); +- thread_env.thread = thread_handle.as_thread(); +- thread_env.layout = layout; ++ let mut thread_env = env.clone_for_thread_spawn(thread_handle.as_thread(), layout); + + // TODO: Currently asynchronous threading does not work with multi + // threading in JS but it does work for the main thread. This will +@@ -151,9 +152,18 @@ pub fn thread_spawn_internal_using_layout( + // calls into the process + let mut execute_module = { + let thread_handle = thread_handle; +- move |ctx: WasiFunctionEnv, mut store: Store| { ++ move |ctx: WasiFunctionEnv, ++ mut store: Store, ++ execution_guard: Option| { + // Call the thread +- call_module::(ctx, store, start_ptr_offset, thread_handle, rewind_state) ++ call_module::( ++ ctx, ++ store, ++ start_ptr_offset, ++ thread_handle, ++ rewind_state, ++ execution_guard, ++ ) + } + }; + +@@ -172,10 +182,10 @@ pub fn thread_spawn_internal_using_layout( + // Now spawn a thread + trace!("threading: spawning background thread"); + let run = move |props: TaskWasmRunProperties| { +- execute_module(props.ctx, props.store); ++ execute_module(props.ctx, props.store, props.execution_guard); + }; + +- let mut task_wasm = TaskWasm::new(Box::new(run), thread_env, thread_module, false, false) ++ let task_wasm = TaskWasm::new(Box::new(run), thread_env, thread_module, false, false) + .with_memory(spawn_type); + + tasks.task_wasm(task_wasm).map_err(Into::::into)?; +@@ -270,6 +280,10 @@ fn handle_thread_result( + .on_taint(TaintReason::DlSymbolResolutionFailed(symbol.clone())); + Ok(Some(ExitCode::from(129))) + } ++ Ok(WasiError::StoreSnapshot(err)) => { ++ eprintln!("Thread {tid} of process {pid} failed to snapshot store state: {err}"); ++ Ok(Some(ExitCode::from(129))) ++ } + Err(err) => { + eprintln!("Thread {tid} of process {pid} failed with runtime error: {err}"); + env.data(&store) +@@ -287,6 +301,7 @@ fn call_module( + start_ptr_offset: M::Offset, + thread_handle: Arc, + rewind_state: Option<(RewindState, RewindResultType)>, ++ execution_guard: Option, + ) { + let env = ctx.data(&store); + let tasks = env.tasks().clone(); +@@ -302,6 +317,9 @@ fn call_module( + rewind_result, + ); + if res != Errno::Success { ++ let exit_code = res.into(); ++ ctx.data().blocking_on_exit(Some(exit_code)); ++ thread_handle.set_status_finished(Ok(exit_code)); + return; + } + } +@@ -311,11 +329,12 @@ fn call_module( + + // If it went to deep sleep then we need to handle that + if let Err(deep) = ret { ++ let terminal_thread = thread_handle.as_thread(); + // Create the callback that will be invoked when the thread respawns after a deep sleep + let rewind = deep.rewind; + let respawn = { + let tasks = tasks.clone(); +- move |ctx, store, trigger_res| { ++ move |ctx, store, trigger_res, execution_guard| { + // Call the thread + call_module::( + ctx, +@@ -323,14 +342,29 @@ fn call_module( + start_ptr_offset, + thread_handle, + Some((rewind, RewindResultType::RewindWithResult(trigger_res))), ++ execution_guard, + ); + } + }; + +- /// Spawns the WASM process after a trigger +- unsafe { +- tasks.resume_wasm_after_poller(Box::new(respawn), ctx, store, deep.trigger) +- }; ++ // Spawns the WASM process after a trigger. Successor rejection must ++ // terminalize this exact thread before the current callback releases ++ // its execution lease. ++ if let Err(err) = unsafe { ++ tasks.resume_wasm_after_poller( ++ Box::new(respawn), ++ ctx, ++ store, ++ deep.trigger, ++ execution_guard, ++ ) ++ } { ++ tracing::warn!("failed to resume thread after deep sleep: {err}"); ++ terminal_thread.set_status_finished(Err(crate::RuntimeError::new(format!( ++ "failed to resume thread after deep sleep: {err}" ++ )) ++ .into())); ++ } + return; + }; + +-- diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0006-wasix-epoll-and-socket-readiness.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0006-wasix-epoll-and-socket-readiness.patch new file mode 100644 index 000000000..0e60cdd0e --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0006-wasix-epoll-and-socket-readiness.patch @@ -0,0 +1,2974 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 6/9] wasix: connect epoll and socket readiness lifetimes + +Bridge the virtual network readiness model into WASIX sockets, poll and epoll. +Listener/accepted descriptor reuse and close must not leave stale registrations +or discard another open-file description's readiness. Depends on descriptor, +process and virtual-io/net changes in this series; retained probes exercise +external connections and descriptor lifecycle. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/lib/wasix/src/net/socket.rs b/lib/wasix/src/net/socket.rs +index 1ca5276..d2460d4 100644 +--- a/lib/wasix/src/net/socket.rs ++++ b/lib/wasix/src/net/socket.rs +@@ -396,13 +396,10 @@ impl InodeSocket { + &self, + tasks: &dyn VirtualTaskManager, + net: &dyn VirtualNetworking, +- _backlog: usize, ++ backlog: usize, + ) -> Result, Errno> { +- let timeout = self +- .opt_time(TimeType::AcceptTimeout) +- .ok() +- .flatten() +- .unwrap_or(Duration::from_secs(30)); ++ let accept_timeout = self.opt_time(TimeType::AcceptTimeout).ok().flatten(); ++ let listen_timeout = accept_timeout.unwrap_or(Duration::from_secs(30)); + + let socket = { + let inner = self.inner.protected.read().unwrap(); +@@ -419,7 +416,7 @@ impl InodeSocket { + let reuse_addr = props.reuse_addr; + drop(inner); + +- net.listen_tcp(addr, only_v6, reuse_port, reuse_addr) ++ net.listen_tcp_with_backlog(addr, only_v6, reuse_port, reuse_addr, backlog) + } + ty => { + tracing::warn!( +@@ -441,7 +438,7 @@ impl InodeSocket { + let reuse_addr = props.reuse_addr; + drop(inner); + +- net.listen_tcp(addr, only_v6, reuse_port, reuse_addr) ++ net.listen_tcp_with_backlog(addr, only_v6, reuse_port, reuse_addr, backlog) + } + ty => { + tracing::warn!( +@@ -481,10 +478,10 @@ impl InodeSocket { + let socket = socket.map_err(net_error_into_wasi_err)?; + Ok(Some(InodeSocket::new(InodeSocketKind::TcpListener { + socket, +- accept_timeout: Some(timeout), ++ accept_timeout, + }))) + }, +- _ = tasks.sleep_now(timeout) => Err(Errno::Timedout) ++ _ = tasks.sleep_now(listen_timeout) => Err(Errno::Timedout) + } + } + +@@ -497,20 +494,11 @@ impl InodeSocket { + struct SocketAccepter<'a> { + sock: &'a InodeSocket, + nonblocking: bool, +- handler_registered: bool, +- } +- impl Drop for SocketAccepter<'_> { +- fn drop(&mut self) { +- if self.handler_registered { +- let mut inner = self.sock.inner.protected.write().unwrap(); +- inner.remove_handler(); +- } +- } + } + impl Future for SocketAccepter<'_> { + type Output = Result<(Box, SocketAddr), Errno>; + fn poll( +- mut self: Pin<&mut Self>, ++ self: Pin<&mut Self>, + cx: &mut std::task::Context<'_>, + ) -> std::task::Poll { + loop { +@@ -521,16 +509,17 @@ impl InodeSocket { + Err(NetworkError::WouldBlock) if self.nonblocking => { + Poll::Ready(Err(Errno::Again)) + } +- Err(NetworkError::WouldBlock) if !self.handler_registered => { +- let res = socket.set_handler(cx.waker().into()); +- if let Err(err) = res { +- return Poll::Ready(Err(net_error_into_wasi_err(err))); ++ Err(NetworkError::WouldBlock) => { ++ match socket.poll_read_ready(cx) { ++ Poll::Ready(Ok(_)) => {} ++ Poll::Ready(Err(err)) => { ++ return Poll::Ready(Err(net_error_into_wasi_err(err))); ++ } ++ Poll::Pending => return Poll::Pending, + } + drop(inner); +- self.handler_registered = true; + continue; + } +- Err(NetworkError::WouldBlock) => Poll::Pending, + Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), + }, + InodeSocketKind::PreSocket { .. } => Poll::Ready(Err(Errno::Notconn)), +@@ -543,7 +532,6 @@ impl InodeSocket { + let acceptor = SocketAccepter { + sock: self, + nonblocking, +- handler_registered: false, + }; + if let Some(timeout) = timeout { + tokio::select! { +@@ -1114,6 +1102,69 @@ impl InodeSocket { + } + } + ++ pub fn try_send_now(&self, buf: &[u8]) -> Result { ++ let mut inner = self.inner.protected.write().unwrap(); ++ let res = { ++ match &mut inner.kind { ++ InodeSocketKind::Raw(socket) => socket.try_send(buf), ++ InodeSocketKind::TcpStream { socket, .. } => socket.try_send(buf), ++ InodeSocketKind::UdpSocket { socket, peer } => { ++ if let Some(peer) = peer { ++ socket.try_send_to(buf, *peer) ++ } else { ++ Err(NetworkError::NotConnected) ++ } ++ } ++ InodeSocketKind::PreSocket { .. } => return Err(Errno::Notconn), ++ InodeSocketKind::RemoteSocket { is_dead, .. } => match is_dead { ++ true => return Err(Errno::Connreset), ++ false => return Ok(buf.len()), ++ }, ++ _ => return Err(Errno::Notsup), ++ } ++ }; ++ match res { ++ Ok(amt) => Ok(amt), ++ Err(NetworkError::WouldBlock) => Err(Errno::Again), ++ Err(err) => Err(net_error_into_wasi_err(err)), ++ } ++ } ++ ++ pub fn try_recv_now(&self, buf: &mut [MaybeUninit], peek: bool) -> Result { ++ let mut inner = self.inner.protected.write().unwrap(); ++ let res = { ++ match &mut inner.kind { ++ InodeSocketKind::Raw(socket) => socket.try_recv(buf, peek), ++ InodeSocketKind::TcpStream { socket, .. } => socket.try_recv(buf, peek), ++ InodeSocketKind::UdpSocket { socket, peer } => { ++ if let Some(peer) = peer { ++ match socket.try_recv_from(buf, peek) { ++ Ok((amt, addr)) if addr == *peer => Ok(amt), ++ Ok(_) => Err(NetworkError::WouldBlock), ++ Err(err) => Err(err), ++ } ++ } else { ++ match socket.try_recv_from(buf, peek) { ++ Ok((amt, _)) => Ok(amt), ++ Err(err) => Err(err), ++ } ++ } ++ } ++ InodeSocketKind::RemoteSocket { is_dead, .. } => match is_dead { ++ true => return Ok(0), ++ false => return Err(Errno::Again), ++ }, ++ InodeSocketKind::PreSocket { .. } => return Err(Errno::Notconn), ++ _ => return Err(Errno::Notsup), ++ } ++ }; ++ match res { ++ Ok(amt) => Ok(amt), ++ Err(NetworkError::WouldBlock) => Err(Errno::Again), ++ Err(err) => Err(net_error_into_wasi_err(err)), ++ } ++ } ++ + pub async fn send( + &self, + tasks: &dyn VirtualTaskManager, +@@ -1125,59 +1176,51 @@ impl InodeSocket { + inner: &'a InodeSocketInner, + data: &'b [u8], + nonblocking: bool, +- handler_registered: bool, +- } +- impl Drop for SocketSender<'_, '_> { +- fn drop(&mut self) { +- if self.handler_registered { +- let mut inner = self.inner.protected.write().unwrap(); +- inner.remove_handler(); +- } +- } + } + impl Future for SocketSender<'_, '_> { + type Output = Result; +- fn poll( +- mut self: Pin<&mut Self>, +- cx: &mut std::task::Context<'_>, +- ) -> Poll { ++ fn poll(self: Pin<&mut Self>, cx: &mut std::task::Context<'_>) -> Poll { + loop { + let mut inner = self.inner.protected.write().unwrap(); +- let res = match &mut inner.kind { +- InodeSocketKind::Raw(socket) => socket.try_send(self.data), +- InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), +- InodeSocketKind::UdpSocket { socket, peer } => { +- if let Some(peer) = peer { +- socket.try_send_to(self.data, *peer) +- } else { +- Err(NetworkError::NotConnected) ++ let res = { ++ match &mut inner.kind { ++ InodeSocketKind::Raw(socket) => socket.try_send(self.data), ++ InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), ++ InodeSocketKind::UdpSocket { socket, peer } => { ++ if let Some(peer) = peer { ++ socket.try_send_to(self.data, *peer) ++ } else { ++ Err(NetworkError::NotConnected) ++ } + } ++ InodeSocketKind::PreSocket { .. } => { ++ return Poll::Ready(Err(Errno::Notconn)); ++ } ++ InodeSocketKind::RemoteSocket { is_dead, .. } => { ++ return match is_dead { ++ true => Poll::Ready(Err(Errno::Connreset)), ++ false => Poll::Ready(Ok(self.data.len())), ++ }; ++ } ++ _ => return Poll::Ready(Err(Errno::Notsup)), + } +- InodeSocketKind::PreSocket { .. } => { +- return Poll::Ready(Err(Errno::Notconn)); +- } +- InodeSocketKind::RemoteSocket { is_dead, .. } => { +- return match is_dead { +- true => Poll::Ready(Err(Errno::Connreset)), +- false => Poll::Ready(Ok(self.data.len())), +- }; +- } +- _ => return Poll::Ready(Err(Errno::Notsup)), + }; + return match res { + Ok(amt) => Poll::Ready(Ok(amt)), + Err(NetworkError::WouldBlock) if self.nonblocking => { + Poll::Ready(Err(Errno::Again)) + } +- Err(NetworkError::WouldBlock) if !self.handler_registered => { +- inner +- .set_handler(cx.waker().into()) +- .map_err(net_error_into_wasi_err)?; ++ Err(NetworkError::WouldBlock) => { ++ match inner.poll_write_ready(cx) { ++ Poll::Ready(Ok(_)) => {} ++ Poll::Ready(Err(err)) => { ++ return Poll::Ready(Err(crate::utils::map_io_err(err))); ++ } ++ Poll::Pending => return Poll::Pending, ++ } + drop(inner); +- self.handler_registered = true; + continue; + } +- Err(NetworkError::WouldBlock) => Poll::Pending, + Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), + }; + } +@@ -1188,7 +1231,6 @@ impl InodeSocket { + inner: &self.inner, + data: buf, + nonblocking, +- handler_registered: false, + }; + if let Some(timeout) = timeout { + tokio::select! { +@@ -1213,55 +1255,49 @@ impl InodeSocket { + data: &'b [u8], + addr: SocketAddr, + nonblocking: bool, +- handler_registered: bool, +- } +- impl Drop for SocketSender<'_, '_> { +- fn drop(&mut self) { +- if self.handler_registered { +- let mut inner = self.inner.protected.write().unwrap(); +- inner.remove_handler(); +- } +- } + } + impl Future for SocketSender<'_, '_> { + type Output = Result; +- fn poll( +- mut self: Pin<&mut Self>, +- cx: &mut std::task::Context<'_>, +- ) -> Poll { ++ fn poll(self: Pin<&mut Self>, cx: &mut std::task::Context<'_>) -> Poll { + loop { + let mut inner = self.inner.protected.write().unwrap(); +- let res = match &mut inner.kind { +- InodeSocketKind::Icmp(socket) => socket.try_send_to(self.data, self.addr), +- InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), +- InodeSocketKind::UdpSocket { socket, .. } => { +- socket.try_send_to(self.data, self.addr) +- } +- InodeSocketKind::PreSocket { .. } => { +- return Poll::Ready(Err(Errno::Notconn)); +- } +- InodeSocketKind::RemoteSocket { is_dead, .. } => { +- return match is_dead { +- true => Poll::Ready(Err(Errno::Connreset)), +- false => Poll::Ready(Ok(self.data.len())), +- }; ++ let res = { ++ match &mut inner.kind { ++ InodeSocketKind::Icmp(socket) => { ++ socket.try_send_to(self.data, self.addr) ++ } ++ InodeSocketKind::TcpStream { socket, .. } => socket.try_send(self.data), ++ InodeSocketKind::UdpSocket { socket, .. } => { ++ socket.try_send_to(self.data, self.addr) ++ } ++ InodeSocketKind::PreSocket { .. } => { ++ return Poll::Ready(Err(Errno::Notconn)); ++ } ++ InodeSocketKind::RemoteSocket { is_dead, .. } => { ++ return match is_dead { ++ true => Poll::Ready(Err(Errno::Connreset)), ++ false => Poll::Ready(Ok(self.data.len())), ++ }; ++ } ++ _ => return Poll::Ready(Err(Errno::Notsup)), + } +- _ => return Poll::Ready(Err(Errno::Notsup)), + }; + return match res { + Ok(amt) => Poll::Ready(Ok(amt)), + Err(NetworkError::WouldBlock) if self.nonblocking => { + Poll::Ready(Err(Errno::Again)) + } +- Err(NetworkError::WouldBlock) if !self.handler_registered => { +- inner +- .set_handler(cx.waker().into()) +- .map_err(net_error_into_wasi_err)?; +- self.handler_registered = true; ++ Err(NetworkError::WouldBlock) => { ++ match inner.poll_write_ready(cx) { ++ Poll::Ready(Ok(_)) => {} ++ Poll::Ready(Err(err)) => { ++ return Poll::Ready(Err(crate::utils::map_io_err(err))); ++ } ++ Poll::Pending => return Poll::Pending, ++ } + drop(inner); + continue; + } +- Err(NetworkError::WouldBlock) => Poll::Pending, + Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), + }; + } +@@ -1273,7 +1309,6 @@ impl InodeSocket { + data: buf, + addr, + nonblocking, +- handler_registered: false, + }; + if let Some(timeout) = timeout { + tokio::select! { +@@ -1298,15 +1333,6 @@ impl InodeSocket { + data: &'b mut [MaybeUninit], + nonblocking: bool, + peek: bool, +- handler_registered: bool, +- } +- impl Drop for SocketReceiver<'_, '_> { +- fn drop(&mut self) { +- if self.handler_registered { +- let mut inner = self.inner.protected.write().unwrap(); +- inner.remove_handler(); +- } +- } + } + impl Future for SocketReceiver<'_, '_> { + type Output = Result; +@@ -1317,51 +1343,54 @@ impl InodeSocket { + loop { + let peek = self.peek; + let mut inner = self.inner.protected.write().unwrap(); +- let res = match &mut inner.kind { +- InodeSocketKind::Raw(socket) => socket.try_recv(self.data, peek), +- InodeSocketKind::TcpStream { socket, .. } => { +- socket.try_recv(self.data, peek) +- } +- InodeSocketKind::UdpSocket { socket, peer } => { +- if let Some(peer) = peer { +- match socket.try_recv_from(self.data, peek) { +- Ok((amt, addr)) if addr == *peer => Ok(amt), +- Ok(_) => Err(NetworkError::WouldBlock), +- Err(err) => Err(err), +- } +- } else { +- match socket.try_recv_from(self.data, peek) { +- Ok((amt, _)) => Ok(amt), +- Err(err) => Err(err), ++ let res = { ++ match &mut inner.kind { ++ InodeSocketKind::Raw(socket) => socket.try_recv(self.data, peek), ++ InodeSocketKind::TcpStream { socket, .. } => { ++ socket.try_recv(self.data, peek) ++ } ++ InodeSocketKind::UdpSocket { socket, peer } => { ++ if let Some(peer) = peer { ++ match socket.try_recv_from(self.data, peek) { ++ Ok((amt, addr)) if addr == *peer => Ok(amt), ++ Ok(_) => Err(NetworkError::WouldBlock), ++ Err(err) => Err(err), ++ } ++ } else { ++ match socket.try_recv_from(self.data, peek) { ++ Ok((amt, _)) => Ok(amt), ++ Err(err) => Err(err), ++ } + } + } ++ InodeSocketKind::RemoteSocket { is_dead, .. } => { ++ return match is_dead { ++ true => Poll::Ready(Ok(0)), ++ false => Poll::Pending, ++ }; ++ } ++ InodeSocketKind::PreSocket { .. } => { ++ return Poll::Ready(Err(Errno::Notconn)); ++ } ++ _ => return Poll::Ready(Err(Errno::Notsup)), + } +- InodeSocketKind::RemoteSocket { is_dead, .. } => { +- return match is_dead { +- true => Poll::Ready(Ok(0)), +- false => Poll::Pending, +- }; +- } +- InodeSocketKind::PreSocket { .. } => { +- return Poll::Ready(Err(Errno::Notconn)); +- } +- _ => return Poll::Ready(Err(Errno::Notsup)), + }; + return match res { + Ok(amt) => Poll::Ready(Ok(amt)), + Err(NetworkError::WouldBlock) if self.nonblocking => { + Poll::Ready(Err(Errno::Again)) + } +- Err(NetworkError::WouldBlock) if !self.handler_registered => { +- inner +- .set_handler(cx.waker().into()) +- .map_err(net_error_into_wasi_err)?; +- self.handler_registered = true; ++ Err(NetworkError::WouldBlock) => { ++ match inner.poll_read_ready_direct(cx) { ++ Poll::Ready(Ok(_)) => {} ++ Poll::Ready(Err(err)) => { ++ return Poll::Ready(Err(crate::utils::map_io_err(err))); ++ } ++ Poll::Pending => return Poll::Pending, ++ } + drop(inner); + continue; + } +- +- Err(NetworkError::WouldBlock) => Poll::Pending, + Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), + }; + } +@@ -1373,7 +1402,6 @@ impl InodeSocket { + data: buf, + nonblocking, + peek, +- handler_registered: false, + }; + if let Some(timeout) = timeout { + tokio::select! { +@@ -1398,15 +1426,6 @@ impl InodeSocket { + data: &'b mut [MaybeUninit], + nonblocking: bool, + peek: bool, +- handler_registered: bool, +- } +- impl Drop for SocketReceiver<'_, '_> { +- fn drop(&mut self) { +- if self.handler_registered { +- let mut inner = self.inner.protected.write().unwrap(); +- inner.remove_handler(); +- } +- } + } + impl Future for SocketReceiver<'_, '_> { + type Output = Result<(usize, SocketAddr), Errno>; +@@ -1415,8 +1434,8 @@ impl InodeSocket { + cx: &mut std::task::Context<'_>, + ) -> Poll { + let peek = self.peek; +- let mut inner = self.inner.protected.write().unwrap(); + loop { ++ let mut inner = self.inner.protected.write().unwrap(); + let res = match &mut inner.kind { + InodeSocketKind::Icmp(socket) => socket.try_recv_from(self.data, peek), + InodeSocketKind::UdpSocket { socket, .. } => { +@@ -1440,14 +1459,17 @@ impl InodeSocket { + Err(NetworkError::WouldBlock) if self.nonblocking => { + Poll::Ready(Err(Errno::Again)) + } +- Err(NetworkError::WouldBlock) if !self.handler_registered => { +- inner +- .set_handler(cx.waker().into()) +- .map_err(net_error_into_wasi_err)?; +- self.handler_registered = true; ++ Err(NetworkError::WouldBlock) => { ++ match inner.poll_read_ready(cx) { ++ Poll::Ready(Ok(_)) => {} ++ Poll::Ready(Err(err)) => { ++ return Poll::Ready(Err(crate::utils::map_io_err(err))); ++ } ++ Poll::Pending => return Poll::Pending, ++ } ++ drop(inner); + continue; + } +- Err(NetworkError::WouldBlock) => Poll::Pending, + Err(err) => Poll::Ready(Err(net_error_into_wasi_err(err))), + }; + } +@@ -1459,7 +1481,6 @@ impl InodeSocket { + data: buf, + nonblocking, + peek, +- handler_registered: false, + }; + if let Some(timeout) = timeout { + tokio::select! { +@@ -1501,22 +1522,6 @@ impl InodeSocket { + } + + impl InodeSocketProtected { +- pub fn remove_handler(&mut self) { +- match &mut self.kind { +- InodeSocketKind::TcpListener { socket, .. } => socket.remove_handler(), +- InodeSocketKind::TcpStream { socket, .. } => socket.remove_handler(), +- InodeSocketKind::UdpSocket { socket, .. } => socket.remove_handler(), +- InodeSocketKind::Raw(socket) => socket.remove_handler(), +- InodeSocketKind::Icmp(socket) => socket.remove_handler(), +- InodeSocketKind::PreSocket { props, .. } => { +- props.handler.take(); +- } +- InodeSocketKind::RemoteSocket { props, .. } => { +- props.handler.take(); +- } +- } +- } +- + pub fn poll_read_ready(&mut self, cx: &mut Context<'_>) -> Poll> { + match &mut self.kind { + InodeSocketKind::TcpListener { socket, .. } => socket.poll_read_ready(cx), +@@ -1533,6 +1538,14 @@ impl InodeSocketProtected { + .map_err(net_error_into_io_err) + } + ++ pub fn poll_read_ready_direct(&mut self, cx: &mut Context<'_>) -> Poll> { ++ match &mut self.kind { ++ InodeSocketKind::TcpStream { socket, .. } => socket.poll_read_ready_direct(cx), ++ _ => return self.poll_read_ready(cx), ++ } ++ .map_err(net_error_into_io_err) ++ } ++ + pub fn poll_write_ready(&mut self, cx: &mut Context<'_>) -> Poll> { + match &mut self.kind { + InodeSocketKind::TcpListener { socket, .. } => socket.poll_write_ready(cx), +@@ -1591,7 +1604,7 @@ pub(crate) fn all_socket_rights() -> Rights { + + #[cfg(test)] + mod tests { +- use super::{InodeSocket, InodeSocketKind, WasiSocketStatus}; ++ use super::{InodeSocket, InodeSocketKind, TimeType, WasiSocketStatus}; + use std::{ + mem::MaybeUninit, + net::{Ipv4Addr, Shutdown, SocketAddr}, +@@ -1606,12 +1619,13 @@ mod tests { + use virtual_mio::InterestHandler; + use virtual_net::{ + NetworkError, Result as NetResult, SocketStatus, VirtualConnectedSocket, VirtualIoSource, +- VirtualSocket, VirtualTcpSocket, ++ VirtualSocket, VirtualTcpListener, VirtualTcpSocket, + }; + + #[derive(Debug)] + struct MockTcpSocket { + read_calls: Arc, ++ direct_read_calls: Arc, + write_calls: Arc, + status: Arc, + } +@@ -1634,6 +1648,11 @@ mod tests { + Poll::Ready(Ok(3)) + } + ++ fn poll_read_ready_direct(&mut self, _cx: &mut Context<'_>) -> Poll> { ++ self.direct_read_calls.fetch_add(1, Ordering::Relaxed); ++ Poll::Ready(Ok(11)) ++ } ++ + fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { + self.write_calls.fetch_add(1, Ordering::Relaxed); + self.status.store(MOCK_STATUS_OPENED, Ordering::Relaxed); +@@ -1746,6 +1765,46 @@ mod tests { + } + } + ++ #[derive(Debug)] ++ struct MockTcpListener; ++ ++ impl VirtualIoSource for MockTcpListener { ++ fn remove_handler(&mut self) {} ++ ++ fn poll_read_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { ++ Poll::Pending ++ } ++ ++ fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { ++ Poll::Ready(Ok(0)) ++ } ++ } ++ ++ impl VirtualTcpListener for MockTcpListener { ++ fn try_accept(&mut self) -> NetResult<(Box, SocketAddr)> { ++ Err(NetworkError::WouldBlock) ++ } ++ ++ fn set_handler( ++ &mut self, ++ _handler: Box, ++ ) -> NetResult<()> { ++ Ok(()) ++ } ++ ++ fn addr_local(&self) -> NetResult { ++ Ok(SocketAddr::from((Ipv4Addr::LOCALHOST, 0))) ++ } ++ ++ fn set_ttl(&mut self, _ttl: u8) -> NetResult<()> { ++ Ok(()) ++ } ++ ++ fn ttl(&self) -> NetResult { ++ Ok(64) ++ } ++ } ++ + #[test] + fn inode_socket_poll_write_ready_uses_write_path() { + let read_calls = Arc::new(AtomicUsize::new(0)); +@@ -1754,6 +1813,7 @@ mod tests { + let mut inode = InodeSocket::new(InodeSocketKind::TcpStream { + socket: Box::new(MockTcpSocket { + read_calls: read_calls.clone(), ++ direct_read_calls: Arc::new(AtomicUsize::new(0)), + write_calls: write_calls.clone(), + status, + }), +@@ -1770,12 +1830,39 @@ mod tests { + assert_eq!(write_calls.load(Ordering::Relaxed), 1); + } + ++ #[test] ++ fn inode_socket_poll_read_ready_direct_uses_tcp_direct_path() { ++ let read_calls = Arc::new(AtomicUsize::new(0)); ++ let direct_read_calls = Arc::new(AtomicUsize::new(0)); ++ let mut inner = super::InodeSocketProtected { ++ kind: InodeSocketKind::TcpStream { ++ socket: Box::new(MockTcpSocket { ++ read_calls: read_calls.clone(), ++ direct_read_calls: direct_read_calls.clone(), ++ write_calls: Arc::new(AtomicUsize::new(0)), ++ status: Arc::new(AtomicUsize::new(MOCK_STATUS_OPENED)), ++ }), ++ write_timeout: None, ++ read_timeout: None, ++ }, ++ }; ++ ++ let waker = futures::task::noop_waker(); ++ let mut cx = Context::from_waker(&waker); ++ let ready = inner.poll_read_ready_direct(&mut cx); ++ ++ assert!(matches!(ready, Poll::Ready(Ok(11)))); ++ assert_eq!(read_calls.load(Ordering::Relaxed), 0); ++ assert_eq!(direct_read_calls.load(Ordering::Relaxed), 1); ++ } ++ + #[test] + fn inode_socket_status_tracks_tcp_socket_status() { + let status = Arc::new(AtomicUsize::new(MOCK_STATUS_OPENING)); + let inode = InodeSocket::new(InodeSocketKind::TcpStream { + socket: Box::new(MockTcpSocket { + read_calls: Arc::new(AtomicUsize::new(0)), ++ direct_read_calls: Arc::new(AtomicUsize::new(0)), + write_calls: Arc::new(AtomicUsize::new(0)), + status: status.clone(), + }), +@@ -1787,4 +1874,14 @@ mod tests { + status.store(MOCK_STATUS_OPENED, Ordering::Relaxed); + assert!(matches!(inode.status().unwrap(), WasiSocketStatus::Opened)); + } ++ ++ #[test] ++ fn tcp_listener_accept_timeout_defaults_to_unset() { ++ let inode = InodeSocket::new(InodeSocketKind::TcpListener { ++ socket: Box::new(MockTcpListener), ++ accept_timeout: None, ++ }); ++ ++ assert_eq!(inode.opt_time(TimeType::AcceptTimeout).unwrap(), None); ++ } + } +diff --git a/lib/wasix/src/os/epoll/mod.rs b/lib/wasix/src/os/epoll/mod.rs +index 9b5ace4..7e97ee0 100644 +--- a/lib/wasix/src/os/epoll/mod.rs ++++ b/lib/wasix/src/os/epoll/mod.rs +@@ -24,7 +24,7 @@ + //! 3. Enqueue exactly one `ReadyItem` per subscription while `enqueued == true`. + //! 4. Wake one waiter via `Notify`. + //! +-//! ### Consumer path (`drain_ready_events` used by `epoll_wait`) ++//! ### Consumer path (`wait_for_ready_events` used by `epoll_wait`) + //! 1. Pop `ReadyItem` from the ready queue. + //! 2. Resolve the current subscription and drop stale/missing entries. + //! 3. Atomically take readiness bits, clear `enqueued`, and map bits to output events. +@@ -40,7 +40,7 @@ + use std::{ + collections::VecDeque, + sync::{ +- Arc, Mutex as StdMutex, ++ Arc, Mutex as StdMutex, Weak, + atomic::{AtomicBool, AtomicU8, AtomicU64, Ordering}, + }, + }; +@@ -48,17 +48,16 @@ use std::{ + use fnv::FnvHashMap; + use serde::{Deserialize, Serialize}; + use tokio::sync::Notify; +-use virtual_mio::{InterestHandler, InterestType}; ++use virtual_mio::{InterestHandler, InterestHandlerRegistration, InterestType}; + use virtual_net::net_error_into_io_err; + use wasmer_wasix_types::wasi::{ +- EpollEventCtl, EpollType, Errno, Eventtype, Fd as WasiFd, Subscription, ++ EpollEventCtl, EpollType, Errno, Eventtype, Fd as WasiFd, Rights, Subscription, + SubscriptionFsReadwrite, SubscriptionUnion, + }; + + use crate::{ +- fs::{InodeValFilePollGuard, InodeValFilePollGuardMode}, +- state::{PollEvent, PollEventBuilder, WasiState}, +- syscalls::poll_fd_guard, ++ fs::{EpollRegistrationGuard, Fd, InodeValFilePollGuard, InodeValFilePollGuardMode}, ++ state::{PollEvent, PollEventBuilder}, + }; + + const READABLE_BIT: u8 = 1 << 0; +@@ -66,11 +65,6 @@ const WRITABLE_BIT: u8 = 1 << 1; + const HUP_BIT: u8 = 1 << 2; + const ERR_BIT: u8 = 1 << 3; + +-static EPOLL_ENQUEUE_ATTEMPTS: AtomicU64 = AtomicU64::new(0); +-static EPOLL_ENQUEUE_DEDUPE_HITS: AtomicU64 = AtomicU64::new(0); +-static EPOLL_STALE_GENERATION_DROPS: AtomicU64 = AtomicU64::new(0); +-static EPOLL_EMPTY_DEQUEUE_ENTRIES: AtomicU64 = AtomicU64::new(0); +- + #[derive(Debug, Clone, Serialize, Deserialize)] + pub struct EpollFd { + /// Event mask configured by the caller (`epoll_ctl`). +@@ -128,55 +122,65 @@ impl EpollFd { + } + } + ++/// Linux identifies an epoll interest by the numeric fd used for ADD plus the ++/// underlying open file description. The fd can later be closed and reused ++/// while a dup/fork alias keeps the old description alive. ++#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] ++pub(crate) struct EpollSubscriptionKey { ++ fd: WasiFd, ++ ofd_id: u64, ++} ++ ++impl EpollSubscriptionKey { ++ pub(crate) fn new(fd: WasiFd, ofd_id: u64) -> Self { ++ Self { fd, ofd_id } ++ } ++ ++ pub(crate) fn fd(self) -> WasiFd { ++ self.fd ++ } ++ ++ fn ofd_id(self) -> u64 { ++ self.ofd_id ++ } ++} ++ + #[derive(Debug)] + pub struct EpollJoinGuard { + /// Underlying poll registration guard. + fd_guard: InodeValFilePollGuard, ++ /// Removes only this subscription's handler from the OFD fanout. ++ _handler_registration: Option, + } + + impl EpollJoinGuard { +- fn new(fd_guard: InodeValFilePollGuard) -> Self { +- Self { fd_guard } +- } +-} +- +-impl Drop for EpollJoinGuard { +- fn drop(&mut self) { +- // Dropping a subscription must detach its interest handler from the source. +- match &self.fd_guard.mode { +- InodeValFilePollGuardMode::File(_) => { +- // Intentionally ignored, epoll doesn't work with files +- } +- InodeValFilePollGuardMode::Socket { inner } => { +- let mut inner = inner.protected.write().unwrap(); +- inner.remove_handler(); +- } +- InodeValFilePollGuardMode::EventNotifications(inner) => { +- inner.remove_interest_handler(); +- } +- InodeValFilePollGuardMode::DuplexPipe { pipe } => { +- let inner = pipe.write().unwrap(); +- inner.remove_interest_handler(); +- } +- InodeValFilePollGuardMode::PipeRx { rx } => { +- let inner = rx.write().unwrap(); +- inner.remove_interest_handler(); +- } +- InodeValFilePollGuardMode::PipeTx { .. } => { +- // Intentionally ignored, the sending end of a pipe can't have an interest handler +- } ++ fn new( ++ fd_guard: InodeValFilePollGuard, ++ handler_registration: Option, ++ ) -> Self { ++ Self { ++ fd_guard, ++ _handler_registration: handler_registration, + } + } + } + + #[derive(Debug)] + pub struct EpollState { +- /// Active subscriptions keyed by watched fd. +- subscriptions: StdMutex>>, ++ /// Serializes the multi-step state/source transaction performed by ++ /// epoll_ctl. Source registration can block or fail, so the subscription ++ /// map alone cannot make ADD/MOD rollback atomic with another ctl call. ++ ctl: StdMutex<()>, ++ /// Active subscriptions keyed by the ADD fd and its open-file-description. ++ subscriptions: StdMutex>>, + /// Ready queue of subscriptions with potentially pending bits. + ready: StdMutex>, + /// Wake primitive for blocked `epoll_wait`. + notify: Notify, ++ /// Final close is idempotent and prevents new registrations. ++ closed: AtomicBool, ++ /// Never-reused identity for queued readiness and replacement detection. ++ next_registration_id: AtomicU64, + } + + impl Default for EpollState { +@@ -189,34 +193,65 @@ impl EpollState { + /// Creates a fresh epoll runtime state. + pub fn new() -> Self { + Self { ++ ctl: StdMutex::new(()), + subscriptions: StdMutex::new(FnvHashMap::default()), + ready: StdMutex::new(VecDeque::new()), + notify: Notify::new(), ++ closed: AtomicBool::new(false), ++ next_registration_id: AtomicU64::new(1), + } + } + ++ fn allocate_registration_id(&self) -> u64 { ++ self.next_registration_id ++ .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { ++ current.checked_add(1) ++ }) ++ .expect("epoll registration identity space exhausted") ++ } ++ ++ pub(crate) fn with_ctl_transaction(&self, transaction: impl FnOnce() -> T) -> T { ++ let _ctl = self.ctl.lock().unwrap(); ++ transaction() ++ } ++ + #[cfg(test)] +- fn insert_subscription(&self, fd: WasiFd, state: Arc) { +- self.subscriptions.lock().unwrap().insert(fd, state); ++ fn insert_subscription(&self, key: EpollSubscriptionKey, state: Arc) { ++ self.subscriptions.lock().unwrap().insert(key, state); + } + +- fn restore_subscription(&self, fd: WasiFd, previous: Option>) { +- let mut subscriptions = self.subscriptions.lock().unwrap(); +- subscriptions.remove(&fd); +- if let Some(previous) = previous { +- subscriptions.insert(fd, previous); +- } ++ fn subscription(&self, key: EpollSubscriptionKey) -> Option> { ++ self.subscriptions.lock().unwrap().get(&key).cloned() ++ } ++ ++ #[cfg(test)] ++ pub(crate) fn contains_exact_subscription( ++ &self, ++ key: EpollSubscriptionKey, ++ expected: &Arc, ++ ) -> bool { ++ self.subscription(key) ++ .is_some_and(|current| Arc::ptr_eq(¤t, expected)) + } + +- fn subscription(&self, fd: WasiFd) -> Option> { +- self.subscriptions.lock().unwrap().get(&fd).cloned() ++ #[cfg(test)] ++ pub(crate) fn subscription_count(&self) -> usize { ++ self.subscriptions.lock().unwrap().len() + } + +- fn enqueue_ready(&self, fd: WasiFd, generation: u64) { +- self.ready +- .lock() +- .unwrap() +- .push_back(ReadyItem { fd, generation }); ++ fn enqueue_ready(&self, key: EpollSubscriptionKey, registration_id: u64) { ++ let mut ready = self.ready.lock().unwrap(); ++ // Linearize enqueue against final close. If enqueue wins this lock, ++ // close clears the item afterward; if close publishes first, no new ++ // item may enter a permanently closed epoll queue. ++ if self.closed.load(Ordering::Acquire) { ++ return; ++ } ++ ready.push_back(ReadyItem { ++ key, ++ registration_id, ++ }); ++ drop(ready); + self.notify.notify_one(); + } + +@@ -224,90 +259,215 @@ impl EpollState { + self.ready.lock().unwrap().pop_front() + } + +- /// Waits until a producer enqueues readiness and notifies. +- pub async fn wait(&self) { +- self.notify.notified().await; ++ fn ready_queue_snapshot(&self) -> Vec { ++ self.ready.lock().unwrap().iter().copied().collect() ++ } ++ ++ fn refresh_level_readiness(&self) -> bool { ++ if self.is_closed() { ++ return false; ++ } ++ let ready_items = self.ready_queue_snapshot(); ++ let refreshed_ready_queue = !ready_items.is_empty(); ++ let subscriptions: Vec<_> = { ++ let subscriptions = self.subscriptions.lock().unwrap(); ++ if refreshed_ready_queue { ++ ready_items ++ .iter() ++ .filter_map(|item| { ++ let sub_state = subscriptions.get(&item.key)?.clone(); ++ (sub_state.registration_id() == item.registration_id) ++ .then_some((item.key, sub_state)) ++ }) ++ .collect() ++ } else { ++ subscriptions ++ .iter() ++ .map(|(&key, sub_state)| (key, sub_state.clone())) ++ .collect() ++ } ++ }; ++ for (key, sub_state) in subscriptions { ++ let current_bits = if refreshed_ready_queue { ++ let pending_bits = sub_state.pending_bits(); ++ if pending_bits != 0 { ++ pending_bits ++ } else { ++ sub_state.poll_current_level_bits() ++ } ++ } else { ++ sub_state.poll_current_level_bits() ++ }; ++ if current_bits == 0 { ++ continue; ++ } ++ ++ sub_state ++ .pending_bits ++ .fetch_or(current_bits, Ordering::AcqRel); ++ if sub_state.mark_enqueued() { ++ self.enqueue_ready(key, sub_state.registration_id()); ++ } ++ } ++ refreshed_ready_queue + } + + pub(crate) fn prepare_add( + &self, +- fd: WasiFd, ++ key: EpollSubscriptionKey, + event: &EpollEventCtl, + ) -> Result<(EpollFd, Arc), Errno> { + let mut subscriptions = self.subscriptions.lock().unwrap(); +- if subscriptions.contains_key(&fd) { ++ if self.closed.load(Ordering::Acquire) { ++ return Err(Errno::Badf); ++ } ++ if subscriptions.contains_key(&key) { + return Err(Errno::Exist); + } + +- let (epoll_fd, sub_state) = self.build_pending_subscription(fd, event, 1); +- subscriptions.insert(fd, sub_state.clone()); ++ let (epoll_fd, sub_state) = self.build_pending_subscription(key, event); ++ subscriptions.insert(key, sub_state.clone()); + Ok((epoll_fd, sub_state)) + } + + pub(crate) fn prepare_mod( + &self, +- fd: WasiFd, ++ key: EpollSubscriptionKey, + event: &EpollEventCtl, + ) -> Result<(EpollFd, Arc, Arc), Errno> { + let mut subscriptions = self.subscriptions.lock().unwrap(); +- let Some(previous) = subscriptions.remove(&fd) else { ++ if self.closed.load(Ordering::Acquire) { ++ return Err(Errno::Badf); ++ } ++ let Some(previous) = subscriptions.remove(&key) else { + return Err(Errno::Noent); + }; +- tracing::trace!(fd, "unregistering waker"); ++ tracing::trace!(fd = key.fd(), ofd_id = key.ofd_id(), "unregistering waker"); + +- let (epoll_fd, sub_state) = +- self.build_pending_subscription(fd, event, previous.next_generation()); +- subscriptions.insert(fd, sub_state.clone()); ++ let (epoll_fd, sub_state) = self.build_pending_subscription(key, event); ++ subscriptions.insert(key, sub_state.clone()); + Ok((epoll_fd, sub_state, previous)) + } + +- pub(crate) fn apply_del(&self, fd: WasiFd) -> Result<(), Errno> { +- let removed = self +- .subscriptions +- .lock() +- .unwrap() +- .remove(&fd) +- .ok_or(Errno::Noent)?; +- removed.detach_joins(); ++ pub(crate) fn prepare_restore( ++ &self, ++ key: EpollSubscriptionKey, ++ epoll_fd: EpollFd, ++ ) -> Result, Errno> { ++ let mut subscriptions = self.subscriptions.lock().unwrap(); ++ if self.closed.load(Ordering::Acquire) { ++ return Err(Errno::Badf); ++ } ++ if subscriptions.contains_key(&key) { ++ return Err(Errno::Exist); ++ } ++ let registration_id = self.allocate_registration_id(); ++ let sub_state = Arc::new(EpollSubState::new(epoll_fd, registration_id)); ++ subscriptions.insert(key, sub_state.clone()); ++ Ok(sub_state) ++ } ++ ++ pub(crate) fn apply_del(&self, key: EpollSubscriptionKey) -> Result<(), Errno> { ++ let removed = self.subscriptions.lock().unwrap().remove(&key); ++ let Some(removed) = removed else { ++ return Err(Errno::Noent); ++ }; ++ removed.deactivate_and_detach(); + Ok(()) + } + +- pub(crate) fn rollback_registration(&self, fd: WasiFd, previous: Option>) { +- self.restore_subscription(fd, previous); ++ pub(crate) fn rollback_registration( ++ &self, ++ key: EpollSubscriptionKey, ++ expected: &Arc, ++ ) -> bool { ++ self.remove_if_same(key, expected) + } + + fn build_pending_subscription( + &self, +- fd: WasiFd, ++ key: EpollSubscriptionKey, + event: &EpollEventCtl, +- generation: u64, + ) -> (EpollFd, Arc) { +- let epoll_fd = EpollFd::from_event_ctl(fd, event); ++ let registration_id = self.allocate_registration_id(); ++ let epoll_fd = EpollFd::from_event_ctl(key.fd(), event); + tracing::trace!( + peb = ?event.events, + ptr = ?event.ptr, + data1 = event.data1, + data2 = event.data2, +- fd, ++ fd = key.fd(), ++ ofd_id = key.ofd_id(), + "registering waker" + ); +- let sub_state = Arc::new(EpollSubState::new(epoll_fd.clone(), generation)); ++ let sub_state = Arc::new(EpollSubState::new(epoll_fd.clone(), registration_id)); + (epoll_fd, sub_state) + } ++ ++ pub(crate) fn remove_if_same( ++ &self, ++ key: EpollSubscriptionKey, ++ expected: &Arc, ++ ) -> bool { ++ let removed = { ++ let mut subscriptions = self.subscriptions.lock().unwrap(); ++ let is_same = subscriptions ++ .get(&key) ++ .is_some_and(|current| Arc::ptr_eq(current, expected)); ++ is_same.then(|| subscriptions.remove(&key)).flatten() ++ }; ++ let Some(removed) = removed else { ++ return false; ++ }; ++ removed.deactivate_and_detach(); ++ true ++ } ++ ++ /// Closes the epoll open file description. This is idempotent and drops ++ /// subscription resources outside the state locks. ++ pub(crate) fn close(&self) { ++ if self.closed.swap(true, Ordering::AcqRel) { ++ return; ++ } ++ let subscriptions = { ++ let mut subscriptions = self.subscriptions.lock().unwrap(); ++ subscriptions ++ .drain() ++ .map(|(_, sub)| sub) ++ .collect::>() ++ }; ++ for subscription in subscriptions { ++ subscription.deactivate_and_detach(); ++ } ++ self.ready.lock().unwrap().clear(); ++ self.notify.notify_waiters(); ++ } ++ ++ pub(crate) fn is_closed(&self) -> bool { ++ self.closed.load(Ordering::Acquire) ++ } + } + + #[derive(Debug)] + pub struct EpollSubState { + /// Snapshot of user-visible metadata. + fd_meta: StdMutex, +- /// Guard ownership for all attached handlers. +- joins: StdMutex>, ++ /// Lifecycle-owned resources. Active is checked under the same lock used ++ /// to attach resources so final close cannot race in a dead handler. ++ resources: StdMutex, + /// Atomic readiness bitset (EPOLLIN/OUT/HUP/ERR). + pending_bits: AtomicU8, + /// Queue dedupe flag: whether this sub already has a ready-queue entry. + enqueued: AtomicBool, +- /// Generation used to invalidate stale queue entries after DEL/MOD. +- generation: AtomicU64, ++ /// Never-reused identity used to invalidate stale queue entries. ++ registration_id: AtomicU64, ++} ++ ++#[derive(Debug)] ++struct EpollSubResources { ++ active: bool, ++ joins: Vec, ++ close_registration: Option, + } + + impl EpollSubState { +@@ -315,32 +475,68 @@ impl EpollSubState { + pub fn new(fd_meta: EpollFd, generation: u64) -> Self { + Self { + fd_meta: StdMutex::new(fd_meta), +- joins: StdMutex::new(Vec::new()), ++ resources: StdMutex::new(EpollSubResources { ++ active: true, ++ joins: Vec::new(), ++ close_registration: None, ++ }), + pending_bits: AtomicU8::new(0), + enqueued: AtomicBool::new(false), +- generation: AtomicU64::new(generation), ++ registration_id: AtomicU64::new(generation), + } + } + +- /// Returns `generation + 1` without mutating the current subscription. +- /// +- /// Callers use this to seed the generation of a replacement subscription. +- pub fn next_generation(&self) -> u64 { +- self.generation.load(Ordering::Acquire).saturating_add(1) ++ /// Adds a registration guard that will detach handlers when dropped. ++ pub fn add_join(&self, join: EpollJoinGuard) -> Result<(), EpollJoinGuard> { ++ let mut resources = self.resources.lock().unwrap(); ++ if !resources.active { ++ return Err(join); ++ } ++ resources.joins.push(join); ++ Ok(()) + } + +- /// Adds a registration guard that will detach handlers when dropped. +- pub fn add_join(&self, join: EpollJoinGuard) { +- self.joins.lock().unwrap().push(join); ++ pub(crate) fn attach_close_registration( ++ &self, ++ registration: EpollRegistrationGuard, ++ ) -> Result<(), EpollRegistrationGuard> { ++ let mut resources = self.resources.lock().unwrap(); ++ if !resources.active { ++ return Err(registration); ++ } ++ resources.close_registration = Some(registration); ++ Ok(()) ++ } ++ ++ /// Marks this subscription inactive, extracts owned resources under the ++ /// lifecycle lock, then drops them after releasing it. ++ pub(crate) fn deactivate_and_detach(&self) { ++ let (joins, close_registration) = { ++ let mut resources = self.resources.lock().unwrap(); ++ if !resources.active && resources.joins.is_empty() { ++ return; ++ } ++ resources.active = false; ++ ( ++ std::mem::take(&mut resources.joins), ++ resources.close_registration.take(), ++ ) ++ }; ++ drop(joins); ++ drop(close_registration); + } + +- /// Detaches and drops all registered handlers for this subscription. +- pub fn detach_joins(&self) { +- self.joins.lock().unwrap().clear(); ++ pub(crate) fn is_active(&self) -> bool { ++ self.resources.lock().unwrap().active + } + +- fn generation(&self) -> u64 { +- self.generation.load(Ordering::Acquire) ++ #[cfg(test)] ++ fn join_count(&self) -> usize { ++ self.resources.lock().unwrap().joins.len() ++ } ++ ++ fn registration_id(&self) -> u64 { ++ self.registration_id.load(Ordering::Acquire) + } + + pub(crate) fn fd_meta(&self) -> EpollFd { +@@ -369,14 +565,53 @@ impl EpollSubState { + fn clear_enqueued(&self) { + self.enqueued.store(false, Ordering::Release); + } ++ ++ fn poll_current_level_bits(&self) -> u8 { ++ let event = self.fd_meta(); ++ let mask_bits = epoll_mask_to_pending_bits(event.events()); ++ let mut bits = 0; ++ ++ let resources = self.resources.lock().unwrap(); ++ for join in resources.joins.iter() { ++ for readiness in join.fd_guard.poll_immediate_ready() { ++ if let Some(bit) = epoll_type_to_pending_bit(readiness) { ++ bits |= bit; ++ } ++ } ++ } ++ ++ bits & mask_bits ++ } ++ ++ fn has_tcp_listener_join(&self) -> bool { ++ self.resources.lock().unwrap().joins.iter().any(|join| { ++ let InodeValFilePollGuardMode::Socket { inner } = &join.fd_guard.mode else { ++ return false; ++ }; ++ matches!( ++ &inner.protected.read().unwrap().kind, ++ crate::net::socket::InodeSocketKind::TcpListener { .. } ++ ) ++ }) ++ } ++ ++ fn validate_current_level_bits(&self, bits: u8) -> u8 { ++ if !self.has_tcp_listener_join() { ++ return bits; ++ } ++ ++ let current_bits = self.poll_current_level_bits(); ++ let edge_bits = bits & (HUP_BIT | ERR_BIT); ++ (bits & current_bits) | edge_bits ++ } + } + + #[derive(Debug, Clone, Copy)] + struct ReadyItem { +- /// Watched fd key used to resolve current subscription state. +- fd: WasiFd, +- /// Generation snapshot captured when enqueued. +- generation: u64, ++ /// Exact subscription key used to resolve current state. ++ key: EpollSubscriptionKey, ++ /// Never-reused registration identity captured when enqueued. ++ registration_id: u64, + } + + /// Maps epoll readiness flags into internal pending-bit positions. +@@ -447,6 +682,7 @@ fn epoll_mask_to_pending_bits(mask: EpollType) -> u8 { + } + + fn prime_immediate_writable_if_applicable( ++ key: EpollSubscriptionKey, + event: &EpollFd, + fd_guard: &InodeValFilePollGuard, + epoll_state: &Arc, +@@ -472,18 +708,18 @@ fn prime_immediate_writable_if_applicable( + .pending_bits + .fetch_or(WRITABLE_BIT, Ordering::AcqRel); + if sub_state.mark_enqueued() { +- epoll_state.enqueue_ready(event.fd(), sub_state.generation()); ++ epoll_state.enqueue_ready(key, sub_state.registration_id()); + } + } + + /// Re-enqueues a subscription if new pending bits arrived during/after consumer drain. + fn repair_ready_queue_after_drain( + epoll_state: &Arc, +- fd: WasiFd, ++ key: EpollSubscriptionKey, + sub_state: &Arc, + ) { + if sub_state.pending_bits() != 0 && sub_state.mark_enqueued() { +- epoll_state.enqueue_ready(fd, sub_state.generation()); ++ epoll_state.enqueue_ready(key, sub_state.registration_id()); + } + } + +@@ -502,14 +738,11 @@ pub(crate) fn drain_ready_events( + let Some(item) = epoll_state.dequeue_ready() else { + break; + }; +- +- let Some(sub_state) = epoll_state.subscription(item.fd) else { +- epoll_empty_dequeue_entry(); ++ let Some(sub_state) = epoll_state.subscription(item.key) else { + continue; + }; + +- if sub_state.generation() != item.generation { +- epoll_stale_generation_drop(); ++ if sub_state.registration_id() != item.registration_id { + continue; + } + +@@ -517,8 +750,13 @@ pub(crate) fn drain_ready_events( + sub_state.clear_enqueued(); + + if bits == 0 { +- repair_ready_queue_after_drain(epoll_state, item.fd, &sub_state); +- epoll_empty_dequeue_entry(); ++ repair_ready_queue_after_drain(epoll_state, item.key, &sub_state); ++ continue; ++ } ++ ++ let bits = sub_state.validate_current_level_bits(bits); ++ if bits == 0 { ++ repair_ready_queue_after_drain(epoll_state, item.key, &sub_state); + continue; + } + +@@ -539,7 +777,7 @@ pub(crate) fn drain_ready_events( + .pending_bits + .fetch_or(undispatched_bits, Ordering::AcqRel); + } +- repair_ready_queue_after_drain(epoll_state, item.fd, &sub_state); ++ repair_ready_queue_after_drain(epoll_state, item.key, &sub_state); + + if ret.len() >= maxevents { + break; +@@ -548,59 +786,112 @@ pub(crate) fn drain_ready_events( + ret + } + ++/// Waits until at least one ready event can be drained. ++/// ++/// `tokio::Notify` requires the waiter to be enabled before the queue is ++/// checked. Otherwise a producer can enqueue readiness and call `notify_one` ++/// after an empty drain but before the `Notified` future is registered. ++pub(crate) async fn wait_for_ready_events( ++ epoll_state: &Arc, ++ maxevents: usize, ++) -> Vec<(EpollFd, EpollType)> { ++ loop { ++ let notified = epoll_state.notify.notified(); ++ tokio::pin!(notified); ++ notified.as_mut().enable(); ++ ++ let refreshed_ready_queue = epoll_state.refresh_level_readiness(); ++ let ret = drain_ready_events(epoll_state, maxevents); ++ if !ret.is_empty() { ++ return ret; ++ } ++ if refreshed_ready_queue { ++ continue; ++ } ++ if epoll_state.is_closed() { ++ return Vec::new(); ++ } ++ ++ notified.as_mut().await; ++ } ++} ++ + #[derive(Debug)] + struct EpollHandler { +- /// Watched fd associated with the subscription. +- fd: WasiFd, ++ /// Exact numeric-fd/open-file-description subscription identity. ++ key: EpollSubscriptionKey, + /// Parent epoll state for queueing and wakeups. +- epoll_state: Arc, ++ epoll_state: Weak, + /// Per-subscription state updated by interest callbacks. +- sub_state: Arc, ++ sub_state: Weak, + } + + impl EpollHandler { +- fn new(fd: WasiFd, epoll_state: Arc, sub_state: Arc) -> Box { ++ fn new( ++ key: EpollSubscriptionKey, ++ epoll_state: &Arc, ++ sub_state: &Arc, ++ ) -> Box { + Box::new(Self { +- fd, +- epoll_state, +- sub_state, ++ key, ++ epoll_state: Arc::downgrade(epoll_state), ++ sub_state: Arc::downgrade(sub_state), + }) + } + } + ++impl Drop for EpollHandler { ++ fn drop(&mut self) { ++ let Some(epoll_state) = self.epoll_state.upgrade() else { ++ return; ++ }; ++ let Some(sub_state) = self.sub_state.upgrade() else { ++ return; ++ }; ++ epoll_state.remove_if_same(self.key, &sub_state); ++ } ++} ++ + impl InterestHandler for EpollHandler { + /// Producer path: + /// set pending bits, enqueue once, and wake one waiter. + fn push_interest(&mut self, interest: InterestType) { +- EPOLL_ENQUEUE_ATTEMPTS.fetch_add(1, Ordering::Relaxed); ++ let Some(sub_state) = self.sub_state.upgrade() else { ++ return; ++ }; ++ let Some(epoll_state) = self.epoll_state.upgrade() else { ++ return; ++ }; ++ if !sub_state.is_active() { ++ return; ++ } + let bit = interest_to_pending_bit(interest); +- if !self.sub_state.set_pending(bit) { +- EPOLL_ENQUEUE_DEDUPE_HITS.fetch_add(1, Ordering::Relaxed); ++ if !sub_state.set_pending(bit) { + return; + } + +- if self.sub_state.mark_enqueued() { +- self.epoll_state +- .enqueue_ready(self.fd, self.sub_state.generation()); +- } else { +- EPOLL_ENQUEUE_DEDUPE_HITS.fetch_add(1, Ordering::Relaxed); ++ if sub_state.mark_enqueued() { ++ epoll_state.enqueue_ready(self.key, sub_state.registration_id()); + } + } + + /// Clears one readiness bit from this subscription only. + fn pop_interest(&mut self, interest: InterestType) -> bool { ++ let Some(sub_state) = self.sub_state.upgrade() else { ++ return false; ++ }; + let bit = interest_to_pending_bit(interest); +- let old = self +- .sub_state +- .pending_bits +- .fetch_and(!bit, Ordering::AcqRel); ++ let old = sub_state.pending_bits.fetch_and(!bit, Ordering::AcqRel); + (old & bit) != 0 + } + + /// Checks whether this subscription currently has a readiness bit set. + fn has_interest(&self, interest: InterestType) -> bool { ++ let Some(sub_state) = self.sub_state.upgrade() else { ++ return false; ++ }; + let bit = interest_to_pending_bit(interest); +- (self.sub_state.pending_bits() & bit) != 0 ++ (sub_state.pending_bits() & bit) != 0 + } + } + +@@ -608,7 +899,8 @@ impl InterestHandler for EpollHandler { + /// + /// `None` means the fd kind does not support handler attachment for epoll. + pub(crate) fn register_epoll_handler( +- state: &Arc, ++ fd_entry: &Fd, ++ key: EpollSubscriptionKey, + event: &EpollFd, + epoll_state: Arc, + sub_state: Arc, +@@ -637,57 +929,128 @@ pub(crate) fn register_epoll_handler( + }, + }; + +- let fd_guard = poll_fd_guard(state, peb.build(), event.fd(), s)?; +- let handler = EpollHandler::new(event.fd(), epoll_state.clone(), sub_state.clone()); ++ let requires_access = match s.type_ { ++ Eventtype::FdRead => Rights::FD_READ, ++ Eventtype::FdWrite => Rights::FD_WRITE, ++ _ => Rights::empty(), ++ }; ++ if !(fd_entry.inner.rights.contains(Rights::POLL_FD_READWRITE) ++ && fd_entry.inner.rights.contains(requires_access)) ++ { ++ return Err(Errno::Access); ++ } ++ let fd_guard = { ++ let guard = fd_entry.inode.read(); ++ InodeValFilePollGuard::new(event.fd(), peb.build(), s, &guard).ok_or(Errno::Badf)? ++ }; + + match &fd_guard.mode { + InodeValFilePollGuardMode::File(_) => { + // Intentionally ignored, epoll doesn't work with files + return Ok(None); + } ++ InodeValFilePollGuardMode::PipeTx { .. } => { ++ // The sending end of a pipe can't have an interest handler, since we ++ // only support "readable" interest on pipes; they're considered to ++ // always be writable. ++ prime_immediate_writable_if_applicable(key, event, &fd_guard, &epoll_state, &sub_state); ++ return Ok(Some(EpollJoinGuard::new(fd_guard, None))); ++ } ++ _ => {} ++ } ++ ++ // One source may feed multiple epolls and multiple dup-backed numeric fds. ++ // The OFD-owned fanout installs once on the source and gives this ++ // subscription an independently removable handler token. ++ let fanout = fd_entry.inner.ofd.interest_fanout(); ++ let handler_registration = fanout.register(EpollHandler::new(key, &epoll_state, &sub_state)); ++ match &fd_guard.mode { + InodeValFilePollGuardMode::Socket { inner, .. } => { + let mut inner = inner.protected.write().unwrap(); +- inner.set_handler(handler).map_err(net_error_into_io_err)?; +- drop(inner); ++ inner ++ .set_handler(Box::new(fanout.clone())) ++ .map_err(net_error_into_io_err)?; ++ } ++ InodeValFilePollGuardMode::EventNotifications(inner) => { ++ inner.set_interest_handler(Box::new(fanout.clone())); + } +- InodeValFilePollGuardMode::EventNotifications(inner) => inner.set_interest_handler(handler), + InodeValFilePollGuardMode::DuplexPipe { pipe } => { + let inner = pipe.write().unwrap(); +- inner.set_interest_handler(handler); ++ inner.set_interest_handler(Box::new(fanout.clone())); + } + InodeValFilePollGuardMode::PipeRx { rx } => { + let inner = rx.write().unwrap(); +- inner.set_interest_handler(handler); ++ inner.set_interest_handler(Box::new(fanout.clone())); + } +- InodeValFilePollGuardMode::PipeTx { .. } => { +- // The sending end of a pipe can't have an interest handler, since we +- // only support "readable" interest on pipes; they're considered to +- // always be writable. +- prime_immediate_writable_if_applicable(event, &fd_guard, &epoll_state, &sub_state); +- return Ok(None); ++ InodeValFilePollGuardMode::File(_) | InodeValFilePollGuardMode::PipeTx { .. } => { ++ unreachable!("non-handler fd kinds returned above") + } + } + +- prime_immediate_writable_if_applicable(event, &fd_guard, &epoll_state, &sub_state); +- +- Ok(Some(EpollJoinGuard::new(fd_guard))) +-} ++ prime_immediate_writable_if_applicable(key, event, &fd_guard, &epoll_state, &sub_state); + +-/// Increments stale-generation dequeue metric. +-pub(crate) fn epoll_stale_generation_drop() { +- EPOLL_STALE_GENERATION_DROPS.fetch_add(1, Ordering::Relaxed); +-} +- +-/// Increments empty dequeue metric. +-pub(crate) fn epoll_empty_dequeue_entry() { +- EPOLL_EMPTY_DEQUEUE_ENTRIES.fetch_add(1, Ordering::Relaxed); ++ Ok(Some(EpollJoinGuard::new( ++ fd_guard, ++ Some(handler_registration), ++ ))) + } + + #[cfg(test)] + mod tests { + use super::*; +- use std::sync::RwLock; ++ use crate::net::socket::{InodeSocket, InodeSocketKind}; ++ use std::{ ++ io::Write, ++ net::{Ipv4Addr, SocketAddr}, ++ sync::{Barrier, RwLock}, ++ task::{Context, Poll}, ++ thread, ++ }; + use virtual_fs::Pipe; ++ use virtual_mio::InterestHandler; ++ use virtual_net::{ ++ NetworkError, Result as NetResult, VirtualIoSource, VirtualTcpListener, VirtualTcpSocket, ++ }; ++ ++ #[derive(Debug)] ++ struct PendingTcpListener; ++ ++ impl VirtualIoSource for PendingTcpListener { ++ fn remove_handler(&mut self) {} ++ ++ fn poll_read_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { ++ Poll::Pending ++ } ++ ++ fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { ++ Poll::Pending ++ } ++ } ++ ++ impl VirtualTcpListener for PendingTcpListener { ++ fn try_accept(&mut self) -> NetResult<(Box, SocketAddr)> { ++ Err(NetworkError::WouldBlock) ++ } ++ ++ fn set_handler( ++ &mut self, ++ _handler: Box, ++ ) -> NetResult<()> { ++ Ok(()) ++ } ++ ++ fn addr_local(&self) -> NetResult { ++ Ok(SocketAddr::from((Ipv4Addr::LOCALHOST, 0))) ++ } ++ ++ fn set_ttl(&mut self, _ttl: u8) -> NetResult<()> { ++ Ok(()) ++ } ++ ++ fn ttl(&self) -> NetResult { ++ Ok(64) ++ } ++ } + + fn test_epoll_event_ctl(fd: WasiFd) -> EpollEventCtl { + EpollEventCtl { +@@ -699,6 +1062,10 @@ mod tests { + } + } + ++ fn test_key(fd: WasiFd) -> EpollSubscriptionKey { ++ EpollSubscriptionKey::new(fd, 10_000 + u64::from(fd)) ++ } ++ + fn test_epoll_handler(fd: WasiFd) -> (Arc, Arc, Box) { + let epoll_state = Arc::new(EpollState::new()); + let sub_state = Arc::new(EpollSubState::new( +@@ -714,7 +1081,7 @@ mod tests { + ), + 1, + )); +- let handler = EpollHandler::new(fd, epoll_state.clone(), sub_state.clone()); ++ let handler = EpollHandler::new(test_key(fd), &epoll_state, &sub_state); + (epoll_state, sub_state, handler) + } + +@@ -734,6 +1101,29 @@ mod tests { + )) + } + ++ fn test_pipe_tx_join(fd: WasiFd) -> EpollJoinGuard { ++ let (tx, _rx) = Pipe::new().split(); ++ EpollJoinGuard::new( ++ InodeValFilePollGuard { ++ fd, ++ peb: PollEventBuilder::new().build(), ++ subscription: Subscription { ++ userdata: 0, ++ type_: Eventtype::FdRead, ++ data: SubscriptionUnion { ++ fd_readwrite: SubscriptionFsReadwrite { ++ file_descriptor: fd, ++ }, ++ }, ++ }, ++ mode: InodeValFilePollGuardMode::PipeTx { ++ tx: Arc::new(RwLock::new(Box::new(tx))), ++ }, ++ }, ++ None, ++ ) ++ } ++ + #[test] + fn epoll_fd_from_event_ctl_uses_explicit_fd() { + let event = test_epoll_event_ctl(1234); +@@ -756,8 +1146,8 @@ mod tests { + EpollFd::new(EpollType::EPOLLIN, 0, 11, 0, 0), + 1, + )); +- let mut handler1 = EpollHandler::new(10, epoll_state.clone(), sub_state1.clone()); +- let mut handler2 = EpollHandler::new(11, epoll_state.clone(), sub_state2.clone()); ++ let mut handler1 = EpollHandler::new(test_key(10), &epoll_state, &sub_state1); ++ let mut handler2 = EpollHandler::new(test_key(11), &epoll_state, &sub_state2); + + handler1.push_interest(InterestType::Readable); + handler2.push_interest(InterestType::Readable); +@@ -805,6 +1195,22 @@ mod tests { + ); + } + ++ #[test] ++ fn epoll_handler_does_not_retain_closed_subscription_graph() { ++ let (epoll_state, sub_state, mut handler) = test_epoll_handler(8); ++ let epoll_state_weak = Arc::downgrade(&epoll_state); ++ let sub_state_weak = Arc::downgrade(&sub_state); ++ ++ drop(epoll_state); ++ drop(sub_state); ++ ++ assert!(epoll_state_weak.upgrade().is_none()); ++ assert!(sub_state_weak.upgrade().is_none()); ++ handler.push_interest(InterestType::Readable); ++ assert!(!handler.has_interest(InterestType::Readable)); ++ assert!(!handler.pop_interest(InterestType::Readable)); ++ } ++ + #[test] + fn epoll_type_to_pending_bit_has_stable_mapping() { + assert_eq!( +@@ -865,10 +1271,10 @@ mod tests { + sub_b.pending_bits.store(readable_bit, Ordering::Release); + sub_b.enqueued.store(true, Ordering::Release); + +- epoll_state.insert_subscription(10, sub_a); +- epoll_state.insert_subscription(11, sub_b); +- epoll_state.enqueue_ready(10, 1); +- epoll_state.enqueue_ready(11, 1); ++ epoll_state.insert_subscription(test_key(10), sub_a); ++ epoll_state.insert_subscription(test_key(11), sub_b); ++ epoll_state.enqueue_ready(test_key(10), 1); ++ epoll_state.enqueue_ready(test_key(11), 1); + + let events = drain_ready_events(&epoll_state, 8); + assert_eq!(events.len(), 2); +@@ -889,8 +1295,8 @@ mod tests { + sub.pending_bits + .store(READABLE_BIT | WRITABLE_BIT, Ordering::Release); + sub.enqueued.store(true, Ordering::Release); +- epoll_state.insert_subscription(90, sub.clone()); +- epoll_state.enqueue_ready(90, 1); ++ epoll_state.insert_subscription(test_key(90), sub.clone()); ++ epoll_state.enqueue_ready(test_key(90), 1); + + let first = drain_ready_events(&epoll_state, 1); + assert_eq!(first.len(), 1); +@@ -908,7 +1314,7 @@ mod tests { + } + + #[test] +- fn drain_ready_events_drops_stale_generation_items() { ++ fn drain_ready_events_drops_stale_registration_items() { + let epoll_state = Arc::new(EpollState::new()); + + let sub = test_sub_state(22, 2); +@@ -916,19 +1322,138 @@ mod tests { + sub.pending_bits.store(readable_bit, Ordering::Release); + sub.enqueued.store(true, Ordering::Release); + +- epoll_state.insert_subscription(22, sub.clone()); +- epoll_state.enqueue_ready(22, 1); ++ epoll_state.insert_subscription(test_key(22), sub.clone()); ++ epoll_state.enqueue_ready(test_key(22), 1); + + let events = drain_ready_events(&epoll_state, 8); + assert!( + events.is_empty(), +- "stale generation items must not emit events" ++ "stale registration items must not emit events" + ); + assert_eq!( + sub.pending_bits.load(Ordering::Acquire), + readable_bit, +- "stale dequeue must not clear pending bits for current generation" ++ "stale dequeue must not clear pending bits for current registration" ++ ); ++ } ++ ++ #[test] ++ fn refresh_level_readiness_requeues_current_pipe_readiness() { ++ let epoll_state = Arc::new(EpollState::new()); ++ let sub = Arc::new(EpollSubState::new( ++ EpollFd::new(EpollType::EPOLLIN, 0, 77, 0, 0), ++ 1, ++ )); ++ epoll_state.insert_subscription(test_key(77), sub.clone()); ++ ++ let (mut tx, rx) = Pipe::new().split(); ++ tx.write_all(b"ready").unwrap(); ++ sub.add_join(EpollJoinGuard::new( ++ InodeValFilePollGuard { ++ fd: 77, ++ peb: PollEventBuilder::new() ++ .add(PollEvent::PollIn) ++ .add(PollEvent::PollError) ++ .add(PollEvent::PollHangUp) ++ .build(), ++ subscription: Subscription { ++ userdata: 0, ++ type_: Eventtype::FdRead, ++ data: SubscriptionUnion { ++ fd_readwrite: SubscriptionFsReadwrite { ++ file_descriptor: 77, ++ }, ++ }, ++ }, ++ mode: InodeValFilePollGuardMode::PipeRx { ++ rx: Arc::new(RwLock::new(Box::new(rx))), ++ }, ++ }, ++ None, ++ )) ++ .unwrap(); ++ ++ epoll_state.refresh_level_readiness(); ++ ++ let events = drain_ready_events(&epoll_state, 8); ++ assert_eq!(events.len(), 1); ++ assert_eq!(events[0].0.fd(), 77); ++ assert_eq!(events[0].1, EpollType::EPOLLIN); ++ } ++ ++ #[test] ++ fn drain_ready_events_drops_stale_tcp_listener_readiness() { ++ let epoll_state = Arc::new(EpollState::new()); ++ let sub = Arc::new(EpollSubState::new( ++ EpollFd::new(EpollType::EPOLLIN, 0, 78, 0, 0), ++ 1, ++ )); ++ epoll_state.insert_subscription(test_key(78), sub.clone()); ++ ++ let listener = InodeSocket::new(InodeSocketKind::TcpListener { ++ socket: Box::new(PendingTcpListener), ++ accept_timeout: None, ++ }); ++ sub.add_join(EpollJoinGuard::new( ++ InodeValFilePollGuard { ++ fd: 78, ++ peb: PollEventBuilder::new() ++ .add(PollEvent::PollIn) ++ .add(PollEvent::PollError) ++ .add(PollEvent::PollHangUp) ++ .build(), ++ subscription: Subscription { ++ userdata: 0, ++ type_: Eventtype::FdRead, ++ data: SubscriptionUnion { ++ fd_readwrite: SubscriptionFsReadwrite { ++ file_descriptor: 78, ++ }, ++ }, ++ }, ++ mode: InodeValFilePollGuardMode::Socket { ++ inner: listener.inner.clone(), ++ }, ++ }, ++ None, ++ )) ++ .unwrap(); ++ sub.pending_bits.store(READABLE_BIT, Ordering::Release); ++ sub.enqueued.store(true, Ordering::Release); ++ epoll_state.enqueue_ready(test_key(78), 1); ++ ++ let events = drain_ready_events(&epoll_state, 8); ++ assert!( ++ events.is_empty(), ++ "stale producer readiness must not be dispatched after level revalidation" + ); ++ assert_eq!(sub.pending_bits(), 0); ++ assert!(!sub.enqueued.load(Ordering::Acquire)); ++ assert_eq!(epoll_state.ready.lock().unwrap().len(), 0); ++ } ++ ++ #[tokio::test] ++ async fn wait_for_ready_events_observes_async_producer() { ++ let (epoll_state, sub_state, mut handler) = test_epoll_handler(33); ++ epoll_state.insert_subscription(test_key(33), sub_state); ++ ++ let producer = tokio::spawn(async move { ++ tokio::task::yield_now().await; ++ handler.push_interest(InterestType::Readable); ++ handler ++ }); ++ ++ let events = tokio::time::timeout( ++ std::time::Duration::from_secs(1), ++ wait_for_ready_events(&epoll_state, 8), ++ ) ++ .await ++ .expect("waiter should wake when the producer enqueues readiness"); ++ ++ let _handler = producer.await.unwrap(); ++ assert_eq!(events.len(), 1); ++ assert_eq!(events[0].0.fd(), 33); ++ assert_eq!(events[0].1, EpollType::EPOLLIN); + } + + #[test] +@@ -939,11 +1464,11 @@ mod tests { + sub.pending_bits.store(writable_bit, Ordering::Release); + sub.enqueued.store(false, Ordering::Release); + +- repair_ready_queue_after_drain(&epoll_state, 44, &sub); ++ repair_ready_queue_after_drain(&epoll_state, test_key(44), &sub); + + assert!(sub.enqueued.load(Ordering::Acquire)); + let queued = epoll_state.ready.lock().unwrap().pop_front().unwrap(); +- assert_eq!(queued.fd, 44); ++ assert_eq!(queued.key.fd(), 44); + } + + #[test] +@@ -951,31 +1476,241 @@ mod tests { + let epoll_state = Arc::new(EpollState::new()); + let event = test_epoll_event_ctl(55); + let sub = Arc::new(EpollSubState::new(EpollFd::from_event_ctl(55, &event), 1)); +- epoll_state.insert_subscription(55, sub.clone()); ++ epoll_state.insert_subscription(test_key(55), sub.clone()); + + let (tx, _rx) = Pipe::new().split(); +- sub.add_join(EpollJoinGuard::new(InodeValFilePollGuard { +- fd: 55, +- peb: PollEventBuilder::new().build(), +- subscription: Subscription { +- userdata: 0, +- type_: Eventtype::FdRead, +- data: SubscriptionUnion { +- fd_readwrite: SubscriptionFsReadwrite { +- file_descriptor: 55, ++ sub.add_join(EpollJoinGuard::new( ++ InodeValFilePollGuard { ++ fd: 55, ++ peb: PollEventBuilder::new().build(), ++ subscription: Subscription { ++ userdata: 0, ++ type_: Eventtype::FdRead, ++ data: SubscriptionUnion { ++ fd_readwrite: SubscriptionFsReadwrite { ++ file_descriptor: 55, ++ }, + }, + }, ++ mode: InodeValFilePollGuardMode::PipeTx { ++ tx: Arc::new(RwLock::new(Box::new(tx))), ++ }, + }, +- mode: InodeValFilePollGuardMode::PipeTx { +- tx: Arc::new(RwLock::new(Box::new(tx))), +- }, +- })); ++ None, ++ )) ++ .unwrap(); + + let leaked_ref = sub.clone(); +- assert_eq!(leaked_ref.joins.lock().unwrap().len(), 1); ++ assert_eq!(leaked_ref.join_count(), 1); ++ ++ epoll_state.apply_del(test_key(55)).unwrap(); ++ ++ assert_eq!(leaked_ref.join_count(), 0); ++ } ++ ++ #[test] ++ fn add_join_racing_epoll_final_close_cannot_retain_guard() { ++ // Exercise both valid mutex orderings repeatedly: either ADD attaches ++ // first and close drains it, or close wins and ADD rejects the guard. ++ for iteration in 0..64 { ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(56, 50_000 + iteration); ++ let (_, subscription) = state.prepare_add(key, &test_epoll_event_ctl(56)).unwrap(); ++ let barrier = Arc::new(Barrier::new(3)); ++ ++ let add_subscription = subscription.clone(); ++ let add_barrier = barrier.clone(); ++ let add = thread::spawn(move || { ++ add_barrier.wait(); ++ drop(add_subscription.add_join(test_pipe_tx_join(56))); ++ }); ++ ++ let close_state = state.clone(); ++ let close_barrier = barrier.clone(); ++ let close = thread::spawn(move || { ++ close_barrier.wait(); ++ close_state.close(); ++ }); ++ ++ barrier.wait(); ++ add.join().unwrap(); ++ close.join().unwrap(); ++ ++ assert!(state.is_closed()); ++ assert!(!subscription.is_active()); ++ assert_eq!(subscription.join_count(), 0); ++ } ++ } ++ ++ #[test] ++ fn readiness_enqueue_racing_epoll_final_close_leaves_queue_drained() { ++ for iteration in 0..64 { ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(57, 60_000 + iteration); ++ let (_, subscription) = state.prepare_add(key, &test_epoll_event_ctl(57)).unwrap(); ++ let mut handler = EpollHandler::new(key, &state, &subscription); ++ let barrier = Arc::new(Barrier::new(3)); ++ ++ let push_barrier = barrier.clone(); ++ let push = thread::spawn(move || { ++ push_barrier.wait(); ++ handler.push_interest(InterestType::Readable); ++ handler ++ }); ++ ++ let close_state = state.clone(); ++ let close_barrier = barrier.clone(); ++ let close = thread::spawn(move || { ++ close_barrier.wait(); ++ close_state.close(); ++ }); ++ ++ barrier.wait(); ++ let _handler = push.join().unwrap(); ++ close.join().unwrap(); ++ ++ assert!(state.is_closed()); ++ assert!(state.ready.lock().unwrap().is_empty()); ++ assert_eq!(state.subscription_count(), 0); ++ assert!(!subscription.is_active()); ++ } ++ } ++ ++ #[test] ++ fn ofd_aware_keys_allow_dup_backed_old_watch_and_reused_numeric_fd() { ++ let state = Arc::new(EpollState::new()); ++ let old_key = EpollSubscriptionKey::new(5, 1001); ++ let new_key = EpollSubscriptionKey::new(5, 1002); ++ let mut old_event = test_epoll_event_ctl(5); ++ old_event.data1 = 1; ++ let mut new_event = test_epoll_event_ctl(5); ++ new_event.data1 = 2; ++ let (_, old_sub) = state.prepare_add(old_key, &old_event).unwrap(); ++ let (_, new_sub) = state.prepare_add(new_key, &new_event).unwrap(); ++ ++ old_sub.pending_bits.store(READABLE_BIT, Ordering::Release); ++ old_sub.enqueued.store(true, Ordering::Release); ++ new_sub.pending_bits.store(READABLE_BIT, Ordering::Release); ++ new_sub.enqueued.store(true, Ordering::Release); ++ state.enqueue_ready(old_key, old_sub.registration_id()); ++ state.enqueue_ready(new_key, new_sub.registration_id()); ++ ++ let events = drain_ready_events(&state, 8); ++ assert_eq!(events.len(), 2); ++ assert!(events.iter().all(|(event, _)| event.fd() == 5)); ++ let payloads = events ++ .iter() ++ .map(|(event, _)| event.data1()) ++ .collect::>(); ++ assert_eq!(payloads, std::collections::HashSet::from([1, 2])); ++ } + +- epoll_state.apply_del(55).unwrap(); ++ #[test] ++ fn del_add_same_ofd_key_cannot_replay_stale_ready_item() { ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(6, 2001); ++ let mut old_event = test_epoll_event_ctl(6); ++ old_event.data1 = 1; ++ let (_, old_sub) = state.prepare_add(key, &old_event).unwrap(); ++ old_sub.pending_bits.store(READABLE_BIT, Ordering::Release); ++ old_sub.enqueued.store(true, Ordering::Release); ++ state.enqueue_ready(key, old_sub.registration_id()); ++ state.apply_del(key).unwrap(); ++ ++ let mut new_event = test_epoll_event_ctl(6); ++ new_event.data1 = 2; ++ let (_, new_sub) = state.prepare_add(key, &new_event).unwrap(); ++ new_sub.pending_bits.store(READABLE_BIT, Ordering::Release); ++ new_sub.enqueued.store(true, Ordering::Release); ++ state.enqueue_ready(key, new_sub.registration_id()); ++ ++ let events = drain_ready_events(&state, 8); ++ assert_eq!(events.len(), 1); ++ assert_eq!(events[0].0.data1(), 2); ++ } ++ ++ #[test] ++ fn delayed_old_handler_drop_cannot_remove_replacement_subscription() { ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(7, 3001); ++ let (_, old_sub) = state.prepare_add(key, &test_epoll_event_ctl(7)).unwrap(); ++ let old_handler = EpollHandler::new(key, &state, &old_sub); ++ state.apply_del(key).unwrap(); ++ let (_, replacement) = state.prepare_add(key, &test_epoll_event_ctl(7)).unwrap(); + +- assert_eq!(leaked_ref.joins.lock().unwrap().len(), 0); ++ drop(old_handler); ++ ++ assert!(state.contains_exact_subscription(key, &replacement)); ++ assert!(replacement.is_active()); ++ } ++ ++ #[test] ++ fn failed_older_mod_cannot_rollback_over_later_subscription() { ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(71, 7001); ++ let (_, original) = state.prepare_add(key, &test_epoll_event_ctl(71)).unwrap(); ++ let (_, first_provisional, replaced_original) = ++ state.prepare_mod(key, &test_epoll_event_ctl(72)).unwrap(); ++ assert!(Arc::ptr_eq(&original, &replaced_original)); ++ ++ // Force the exact historical interleave: an older MOD is between map ++ // publication and source installation while a later MOD replaces it. ++ let later_published = Arc::new(Barrier::new(2)); ++ let later_state = state.clone(); ++ let later_barrier = later_published.clone(); ++ let first_for_later = first_provisional.clone(); ++ let later = thread::spawn(move || { ++ let (_, later_subscription, replaced_first) = later_state ++ .prepare_mod(key, &test_epoll_event_ctl(73)) ++ .unwrap(); ++ assert!(Arc::ptr_eq(&replaced_first, &first_for_later)); ++ replaced_first.deactivate_and_detach(); ++ later_barrier.wait(); ++ later_subscription ++ }); ++ ++ later_published.wait(); ++ let later_subscription = later.join().unwrap(); ++ ++ // The first operation now reports a source-install failure. Its ++ // rollback must be conditional on exact identity and restoration must ++ // not overwrite the later successful registration. ++ assert!(!state.rollback_registration(key, &first_provisional)); ++ assert!(matches!( ++ state.prepare_restore(key, replaced_original.fd_meta()), ++ Err(Errno::Exist) ++ )); ++ assert!(state.contains_exact_subscription(key, &later_subscription)); ++ assert!(later_subscription.is_active()); ++ } ++ ++ #[test] ++ fn fanout_delivers_to_two_epolls_and_removing_one_keeps_the_other() { ++ let first_state = Arc::new(EpollState::new()); ++ let second_state = Arc::new(EpollState::new()); ++ let first_key = EpollSubscriptionKey::new(8, 4001); ++ let second_key = EpollSubscriptionKey::new(9, 4001); ++ let (_, first_sub) = first_state ++ .prepare_add(first_key, &test_epoll_event_ctl(8)) ++ .unwrap(); ++ let (_, second_sub) = second_state ++ .prepare_add(second_key, &test_epoll_event_ctl(9)) ++ .unwrap(); ++ let mut fanout = virtual_mio::InterestHandlerFanout::default(); ++ let first_registration = ++ fanout.register(EpollHandler::new(first_key, &first_state, &first_sub)); ++ let _second_registration = ++ fanout.register(EpollHandler::new(second_key, &second_state, &second_sub)); ++ ++ fanout.push_interest(InterestType::Readable); ++ assert_eq!(drain_ready_events(&first_state, 8).len(), 1); ++ assert_eq!(drain_ready_events(&second_state, 8).len(), 1); ++ ++ drop(first_registration); ++ assert_eq!(first_state.subscription_count(), 0); ++ assert_eq!(fanout.handler_count(), 1); ++ fanout.push_interest(InterestType::Readable); ++ assert!(drain_ready_events(&first_state, 8).is_empty()); ++ assert_eq!(drain_ready_events(&second_state, 8).len(), 1); + } + } +diff --git a/lib/wasix/src/syscalls/wasi/poll_oneoff.rs b/lib/wasix/src/syscalls/wasi/poll_oneoff.rs +index 5cd9781..1cc026c 100644 +--- a/lib/wasix/src/syscalls/wasi/poll_oneoff.rs ++++ b/lib/wasix/src/syscalls/wasi/poll_oneoff.rs +@@ -474,14 +474,9 @@ where + return Ok(Errno::Success); + } + +- // We use asyncify with a deep sleep to wait on new IO events +- let res = __asyncify_with_deep_sleep::, Errno>, _>( +- ctx, +- Box::pin(trigger), +- )?; +- if let AsyncifyAction::Finish(mut ctx, events) = res { +- let events = events.map(|events| events.into_iter().map(EventResult::into_event).collect()); +- process_events(&ctx, events); +- } ++ // Wait on new IO events while still processing signals. ++ let events = __asyncify(&mut ctx, None, Box::pin(trigger))?; ++ let events = events.map(|events| events.into_iter().map(EventResult::into_event).collect()); ++ process_events(&ctx, events); + Ok(Errno::Success) + } +diff --git a/lib/wasix/src/syscalls/wasix/epoll_ctl.rs b/lib/wasix/src/syscalls/wasix/epoll_ctl.rs +index 14cf332..2302ca1 100644 +--- a/lib/wasix/src/syscalls/wasix/epoll_ctl.rs ++++ b/lib/wasix/src/syscalls/wasix/epoll_ctl.rs +@@ -3,12 +3,84 @@ use wasmer_wasix_types::wasi::{EpollCtl, EpollEvent, EpollEventCtl, Subscription + use super::*; + use crate::{ + WasiInodes, +- fs::{InodeValFilePollGuard, InodeValFilePollGuardJoin}, +- os::epoll::register_epoll_handler, ++ fs::{Fd, InodeValFilePollGuard, InodeValFilePollGuardJoin}, ++ os::epoll::{EpollFd, EpollState, EpollSubState, EpollSubscriptionKey, register_epoll_handler}, + state::PollEventSet, + syscalls::*, + }; + ++fn install_subscription( ++ target: &Fd, ++ key: EpollSubscriptionKey, ++ epoll_fd: &EpollFd, ++ state: &Arc, ++ subscription: &Arc, ++) -> Result<(), Errno> { ++ let close_registration = target ++ .inner ++ .ofd ++ .register_epoll(state, subscription, key) ++ .ok_or(Errno::Badf)?; ++ subscription ++ .attach_close_registration(close_registration) ++ .map_err(|_| Errno::Badf)?; ++ ++ if let Some(join) = ++ register_epoll_handler(target, key, epoll_fd, state.clone(), subscription.clone())? ++ { ++ subscription.add_join(join).map_err(|_| Errno::Badf)?; ++ } ++ Ok(()) ++} ++ ++fn add_subscription( ++ target: &Fd, ++ key: EpollSubscriptionKey, ++ state: &Arc, ++ event: &EpollEventCtl, ++) -> Result<(), Errno> { ++ let (epoll_fd, subscription) = state.prepare_add(key, event)?; ++ match install_subscription(target, key, &epoll_fd, state, &subscription) { ++ Ok(()) => Ok(()), ++ Err(err) => { ++ state.rollback_registration(key, &subscription); ++ Err(err) ++ } ++ } ++} ++ ++fn modify_subscription( ++ target: &Fd, ++ key: EpollSubscriptionKey, ++ state: &Arc, ++ event: &EpollEventCtl, ++) -> Result<(), Errno> { ++ let (epoll_fd, subscription, old_subscription) = state.prepare_mod(key, event)?; ++ let old_epoll_fd = old_subscription.fd_meta(); ++ old_subscription.deactivate_and_detach(); ++ ++ match install_subscription(target, key, &epoll_fd, state, &subscription) { ++ Ok(()) => Ok(()), ++ Err(err) => { ++ state.rollback_registration(key, &subscription); ++ let restored = state.prepare_restore(key, old_epoll_fd.clone())?; ++ if let Err(reinstall_err) = ++ install_subscription(target, key, &old_epoll_fd, state, &restored) ++ { ++ state.rollback_registration(key, &restored); ++ tracing::warn!( ++ fd = key.fd(), ++ ?err, ++ ?reinstall_err, ++ "failed to reinstall previous epoll handler after MOD failure" ++ ); ++ return Err(reinstall_err); ++ } ++ Err(err) ++ } ++ } ++} ++ + /// ### `epoll_ctl()` + /// Modifies an epoll interest list + /// Output: +@@ -69,99 +141,176 @@ pub(crate) fn epoll_ctl_internal( + event_ctl: Option<&EpollEventCtl>, + ) -> Result, WasiError> { + let env = ctx.data(); +- let fd_entry = wasi_try_ok_ok!(env.state.fs.get_fd(epfd)); +- +- let mut inode_guard = fd_entry.inode.read(); +- match inode_guard.deref() { +- Kind::Epoll { state } => { +- let res = match op { +- EpollCtl::Add => { +- let Some(event) = event_ctl else { +- return Ok(Err(Errno::Inval)); +- }; +- let (epoll_fd, sub_state) = match state.prepare_add(fd, event) { +- Ok(v) => v, +- Err(err) => return Ok(Err(err)), +- }; +- +- match register_epoll_handler( +- &env.state, +- &epoll_fd, +- state.clone(), +- sub_state.clone(), +- ) { +- Ok(fd_guard) => { +- if let Some(fd_guard) = fd_guard { +- sub_state.add_join(fd_guard); +- } +- Ok(()) +- } +- Err(err) => { +- state.rollback_registration(fd, None); +- Err(err) +- } +- } +- } +- EpollCtl::Mod => { +- let Some(event) = event_ctl else { +- return Ok(Err(Errno::Inval)); +- }; +- let (epoll_fd, sub_state, old_subscription) = match state.prepare_mod(fd, event) +- { +- Ok(v) => v, +- Err(err) => return Ok(Err(err)), +- }; +- // Detach the previous generation before installing the new +- // handler so dropping old guards cannot remove the new one. +- old_subscription.detach_joins(); +- +- match register_epoll_handler( +- &env.state, +- &epoll_fd, +- state.clone(), +- sub_state.clone(), +- ) { +- Ok(fd_guard) => { +- if let Some(fd_guard) = fd_guard { +- sub_state.add_join(fd_guard); +- } +- Ok(()) +- } +- Err(err) => { +- state.rollback_registration(fd, Some(old_subscription.clone())); +- let old_epoll_fd = old_subscription.fd_meta(); +- match register_epoll_handler( +- &env.state, +- &old_epoll_fd, +- state.clone(), +- old_subscription.clone(), +- ) { +- Ok(fd_guard) => { +- if let Some(fd_guard) = fd_guard { +- old_subscription.add_join(fd_guard); +- } +- } +- Err(reinstall_err) => { +- // Do not leave a restored subscription without handlers. +- state.rollback_registration(fd, None); +- tracing::warn!( +- fd, +- ?err, +- ?reinstall_err, +- "failed to reinstall previous epoll handler after MOD failure" +- ); +- return Ok(Err(reinstall_err)); +- } +- } +- Err(err) +- } +- } +- } +- EpollCtl::Del => state.apply_del(fd), +- EpollCtl::Unknown => Err(Errno::Inval), +- }; +- Ok(res) ++ let epoll_entry = wasi_try_ok_ok!(env.state.fs.get_fd(epfd)); ++ let state = { ++ let inode_guard = epoll_entry.inode.read(); ++ match inode_guard.deref() { ++ Kind::Epoll { state } => state.clone(), ++ _ => return Ok(Err(Errno::Inval)), ++ } ++ }; ++ if epfd == fd { ++ return Ok(Err(Errno::Inval)); ++ } ++ ++ // Capture the target once. Numeric fd reuse after this point must not make ++ // registration attach to a different open file description. ++ let target = wasi_try_ok_ok!(env.state.fs.get_fd(fd)); ++ let key = EpollSubscriptionKey::new(fd, target.inner.ofd.id()); ++ ++ let result = state.with_ctl_transaction(|| match op { ++ EpollCtl::Add => event_ctl ++ .ok_or(Errno::Inval) ++ .and_then(|event| add_subscription(&target, key, &state, event)), ++ EpollCtl::Mod => event_ctl ++ .ok_or(Errno::Inval) ++ .and_then(|event| modify_subscription(&target, key, &state, event)), ++ EpollCtl::Del => state.apply_del(key), ++ EpollCtl::Unknown => Err(Errno::Inval), ++ }); ++ Ok(result) ++} ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use crate::{ ++ fs::{FdInner, InodeVal, OpenFileDescription}, ++ net::socket::{InodeSocket, InodeSocketKind}, ++ os::epoll::drain_ready_events, ++ }; ++ use std::{ ++ borrow::Cow, ++ net::{Ipv4Addr, SocketAddr}, ++ sync::{Arc, Mutex, RwLock, atomic::AtomicU64}, ++ task::{Context, Poll}, ++ }; ++ use virtual_net::{ ++ InterestHandler, NetworkError, Result as NetResult, VirtualIoSource, VirtualTcpListener, ++ VirtualTcpSocket, ++ }; ++ use wasmer_wasix_types::wasi::EpollType; ++ ++ #[derive(Debug)] ++ struct FailSecondHandlerInstall { ++ calls: Arc, ++ handler: Arc>>>, ++ } ++ ++ impl VirtualIoSource for FailSecondHandlerInstall { ++ fn remove_handler(&mut self) { ++ self.handler.lock().unwrap().take(); ++ } ++ ++ fn poll_read_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { ++ Poll::Pending ++ } ++ ++ fn poll_write_ready(&mut self, _cx: &mut Context<'_>) -> Poll> { ++ Poll::Pending ++ } ++ } ++ ++ impl VirtualTcpListener for FailSecondHandlerInstall { ++ fn try_accept(&mut self) -> NetResult<(Box, SocketAddr)> { ++ Err(NetworkError::WouldBlock) ++ } ++ ++ fn set_handler( ++ &mut self, ++ handler: Box, ++ ) -> NetResult<()> { ++ let call = self.calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst) + 1; ++ if call == 2 { ++ return Err(NetworkError::IOError); ++ } ++ *self.handler.lock().unwrap() = Some(handler); ++ Ok(()) + } +- _ => Ok(Err(Errno::Inval)), ++ ++ fn addr_local(&self) -> NetResult { ++ Ok(SocketAddr::from((Ipv4Addr::LOCALHOST, 0))) ++ } ++ ++ fn set_ttl(&mut self, _ttl: u8) -> NetResult<()> { ++ Ok(()) ++ } ++ ++ fn ttl(&self) -> NetResult { ++ Ok(64) ++ } ++ } ++ ++ #[test] ++ fn failed_mod_rebuilds_active_old_subscription_with_fresh_identity() { ++ let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); ++ let installed_handler = Arc::new(Mutex::new(None)); ++ let socket = InodeSocket::new(InodeSocketKind::TcpListener { ++ socket: Box::new(FailSecondHandlerInstall { ++ calls: calls.clone(), ++ handler: installed_handler.clone(), ++ }), ++ accept_timeout: None, ++ }); ++ let inodes = WasiInodes::new(); ++ let inode = inodes.add_inode_val(InodeVal { ++ stat: RwLock::new(Default::default()), ++ is_preopened: false, ++ name: RwLock::new(Cow::Borrowed("mod-rollback-listener")), ++ kind: RwLock::new(Kind::Socket { socket }), ++ }); ++ let ofd = OpenFileDescription::new(); ++ let rights = Rights::POLL_FD_READWRITE | Rights::FD_READ; ++ let target = Fd { ++ inner: FdInner { ++ rights, ++ rights_inheriting: rights, ++ flags: Fdflags::empty(), ++ offset: Arc::new(AtomicU64::new(0)), ++ ofd: ofd.clone(), ++ readdir_cache: Default::default(), ++ fd_flags: Fdflagsext::empty(), ++ }, ++ open_flags: 0, ++ inode, ++ is_stdio: false, ++ }; ++ target.acquire_descriptor(); ++ ++ let state = Arc::new(EpollState::new()); ++ let key = EpollSubscriptionKey::new(42, ofd.id()); ++ let old_event = EpollEventCtl { ++ events: EpollType::EPOLLIN, ++ ptr: 0, ++ fd: 42, ++ data1: 11, ++ data2: 0, ++ }; ++ let new_event = EpollEventCtl { ++ data1: 22, ++ ..old_event ++ }; ++ add_subscription(&target, key, &state, &old_event).unwrap(); ++ ++ assert_eq!( ++ modify_subscription(&target, key, &state, &new_event), ++ Err(Errno::Pipe) ++ ); ++ assert_eq!(calls.load(std::sync::atomic::Ordering::SeqCst), 3); ++ assert_eq!(state.subscription_count(), 1); ++ assert_eq!(ofd.epoll_registration_count(), 1); ++ ++ installed_handler ++ .lock() ++ .unwrap() ++ .as_mut() ++ .unwrap() ++ .push_interest(virtual_mio::InterestType::Error); ++ let events = drain_ready_events(&state, 8); ++ assert_eq!(events.len(), 1); ++ assert_eq!(events[0].0.data1(), 11); ++ ++ target.release_descriptor(); ++ assert_eq!(state.subscription_count(), 0); + } + } +diff --git a/lib/wasix/src/syscalls/wasix/epoll_wait.rs b/lib/wasix/src/syscalls/wasix/epoll_wait.rs +index d1b56a3..d99411f 100644 +--- a/lib/wasix/src/syscalls/wasix/epoll_wait.rs ++++ b/lib/wasix/src/syscalls/wasix/epoll_wait.rs +@@ -4,7 +4,7 @@ use super::*; + use crate::{ + WasiInodes, + fs::{InodeValFilePollGuard, InodeValFilePollGuardJoin}, +- os::epoll::{EpollFd, drain_ready_events}, ++ os::epoll::{EpollFd, wait_for_ready_events}, + state::PollEventSet, + syscalls::*, + }; +@@ -49,19 +49,7 @@ pub fn epoll_wait( + // We enter a controlled loop that will continuously poll and react to + // epoll events until something of interest needs to be returned to the + // caller or a timeout happens +- let work = { +- async move { +- // Loop until some events of interest are returned +- loop { +- let ret = drain_ready_events(&epoll_state, maxevents); +- if !ret.is_empty() { +- return Ok(ret); +- } +- +- epoll_state.wait().await; +- } +- } +- }; ++ let work = async move { Ok(wait_for_ready_events(&epoll_state, maxevents).await) }; + + // Build the trigger using the timeout + let trigger = { +@@ -128,6 +116,10 @@ pub fn epoll_wait( + wasi_try_mem!(ret_nevents.write(&memory, M::ZERO)); + Errno::Success + } ++ Err(Errno::Intr) => { ++ tracing::trace!("epoll interrupted by signal"); ++ Errno::Intr ++ } + Err(err) => { + tracing::warn!("failed to epoll during deep sleep - {}", err); + err +@@ -143,14 +135,7 @@ pub fn epoll_wait( + return Ok(process_events(&ctx, events)); + } + +- // We use asyncify with a deep sleep to wait on new IO events +- let res = __asyncify_with_deep_sleep::, Errno>, _>( +- ctx, +- Box::pin(trigger), +- )?; +- if let AsyncifyAction::Finish(mut ctx, events) = res { +- Ok(process_events(&ctx, events)) +- } else { +- Ok(Errno::Success) +- } ++ // Wait on new IO events while still processing signals. ++ let events = __asyncify(&mut ctx, None, Box::pin(trigger))?; ++ Ok(process_events(&ctx, events)) + } +diff --git a/lib/wasix/src/syscalls/wasix/sock_accept.rs b/lib/wasix/src/syscalls/wasix/sock_accept.rs +index b851666..76aedb0 100644 +--- a/lib/wasix/src/syscalls/wasix/sock_accept.rs ++++ b/lib/wasix/src/syscalls/wasix/sock_accept.rs +@@ -26,19 +26,18 @@ pub fn sock_accept( + + ctx = wasi_try_ok!(maybe_snapshot::(ctx)?); + +- let env = ctx.data(); +- let (memory, state, _) = unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; +- + let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); + + let (fd, _, _) = wasi_try_ok!(sock_accept_internal( +- env, ++ &mut ctx, + sock, + fd_flags, + nonblocking, + None + )?); + ++ let env = ctx.data(); ++ let memory = unsafe { env.memory_view(&ctx) }; + wasi_try_mem_ok!(ro_fd.write(&memory, fd)); + + Ok(Errno::Success) +@@ -67,13 +66,10 @@ pub fn sock_accept_v2( + ) -> Result { + WasiEnv::do_pending_operations(&mut ctx)?; + +- let env = ctx.data(); +- let (memory, state, _) = unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; +- + let nonblocking = fd_flags.contains(Fdflags::NONBLOCK); + + let (fd, local_addr, peer_addr) = wasi_try_ok!(sock_accept_internal( +- env, ++ &mut ctx, + sock, + fd_flags, + nonblocking, +@@ -98,7 +94,7 @@ pub fn sock_accept_v2( + } + + let env = ctx.data(); +- let (memory, state, _) = unsafe { env.get_memory_and_wasi_state_and_inodes(&ctx, 0) }; ++ let memory = unsafe { env.memory_view(&ctx) }; + wasi_try_mem_ok!(ro_fd.write(&memory, fd)); + wasi_try_ok!(crate::net::write_ip_port( + &memory, +@@ -111,37 +107,46 @@ pub fn sock_accept_v2( + } + + pub(crate) fn sock_accept_internal( +- env: &WasiEnv, ++ ctx: &mut FunctionEnvMut<'_, WasiEnv>, + sock: WasiFd, + mut fd_flags: Fdflags, + mut nonblocking: bool, + with_fd: Option, + ) -> Result, WasiError> { +- let state = env.state(); +- let inodes = &state.inodes; +- +- let tasks = env.tasks().clone(); +- let (child, local_addr, peer_addr, fd_flags) = wasi_try_ok_ok!(__sock_asyncify( +- env, +- sock, +- Rights::SOCK_ACCEPT, +- move |socket, fd| async move { +- if fd.inner.flags.contains(Fdflags::NONBLOCK) { +- fd_flags.set(Fdflags::NONBLOCK, true); +- nonblocking = true; ++ let tasks = ctx.data().tasks().clone(); ++ let accept = { ++ let fd_entry = wasi_try_ok_ok!(ctx.data().state.fs.get_fd(sock)); ++ if !fd_entry.inner.rights.contains(Rights::SOCK_ACCEPT) { ++ return Ok(Err(Errno::Access)); ++ } ++ ++ let inode = fd_entry.inode.clone(); ++ let mut guard = inode.write(); ++ match guard.deref_mut() { ++ Kind::Socket { socket } => { ++ let socket = socket.clone(); ++ drop(guard); ++ ++ async move { ++ if fd_entry.inner.flags.contains(Fdflags::NONBLOCK) { ++ fd_flags.set(Fdflags::NONBLOCK, true); ++ nonblocking = true; ++ } ++ let timeout = socket.opt_time(TimeType::AcceptTimeout).ok().flatten(); ++ let local_addr = socket.addr_local()?; ++ socket ++ .accept(tasks.deref(), nonblocking, timeout) ++ .await ++ .map(|a| (a.0, local_addr, a.1, fd_flags)) ++ } + } +- let timeout = socket +- .opt_time(TimeType::AcceptTimeout) +- .ok() +- .flatten() +- .unwrap_or(Duration::from_secs(30)); +- let local_addr = socket.addr_local()?; +- socket +- .accept(tasks.deref(), nonblocking, Some(timeout)) +- .await +- .map(|a| (a.0, local_addr, a.1, fd_flags)) +- }, +- )); ++ _ => return Ok(Err(Errno::Notsock)), ++ } ++ }; ++ let (child, local_addr, peer_addr, fd_flags) = wasi_try_ok_ok!(__asyncify(ctx, None, accept)?); ++ ++ let state = ctx.data().state(); ++ let inodes = &state.inodes; + + let kind = Kind::Socket { + socket: InodeSocket::new(InodeSocketKind::TcpStream { +@@ -159,11 +164,6 @@ pub(crate) fn sock_accept_internal( + new_flags.set(Fdflags::NONBLOCK, true); + } + +- let mut new_flags = Fdflags::empty(); +- if fd_flags.contains(Fdflags::NONBLOCK) { +- new_flags.set(Fdflags::NONBLOCK, true); +- } +- + let rights = Rights::all_socket(); + let fd = wasi_try_ok_ok!(if let Some(fd) = with_fd { + state +diff --git a/lib/wasix/src/syscalls/wasix/sock_recv.rs b/lib/wasix/src/syscalls/wasix/sock_recv.rs +index f507a5b..cf5b823 100644 +--- a/lib/wasix/src/syscalls/wasix/sock_recv.rs ++++ b/lib/wasix/src/syscalls/wasix/sock_recv.rs +@@ -29,11 +29,12 @@ pub fn sock_recv( + WasiEnv::do_pending_operations(&mut ctx)?; + + let env = ctx.data(); +- let fd_entry = wasi_try_ok!(env.state.fs.get_fd(sock)); +- let guard = fd_entry.inode.read(); +- // Some guests route socket-like wakeups through pipe-backed fds. +- let use_read = matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeRx { .. }); +- drop(guard); ++ let fd_entry = wasi_try_ok!({ env.state.fs.get_fd(sock) }); ++ let use_read = { ++ let guard = fd_entry.inode.read(); ++ // Some guests route socket-like wakeups through pipe-backed fds. ++ matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeRx { .. }) ++ }; + if use_read { + fd_read(ctx, sock, ri_data, ri_data_len, ro_data_len) + } else { +@@ -116,23 +117,28 @@ pub(super) fn sock_recv_internal( + + let peek = (ri_flags & __WASI_SOCK_RECV_INPUT_PEEK) != 0; + let nonblocking_flag = (ri_flags & __WASI_SOCK_RECV_INPUT_DONT_WAIT) != 0; ++ + let data = wasi_try_ok_ok!(__sock_asyncify( + env, + sock, + Rights::SOCK_RECV, + |socket, fd| async move { +- let iovs_arr = ri_data +- .slice(&memory, ri_data_len) +- .map_err(mem_error_to_wasi)?; +- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ let iovs_arr = { ++ let iovs_arr = ri_data ++ .slice(&memory, ri_data_len) ++ .map_err(mem_error_to_wasi)?; ++ iovs_arr.access().map_err(mem_error_to_wasi) ++ }?; + + let mut total_read = 0; + for iovs in iovs_arr.iter() { +- let mut buf = WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi)? +- .access() +- .map_err(mem_error_to_wasi)?; ++ let mut buf = { ++ WasmPtr::::new(iovs.buf) ++ .slice(&memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi) ++ }?; + + let nonblocking = nonblocking_flag || fd.inner.flags.contains(Fdflags::NONBLOCK); + let timeout = socket +diff --git a/lib/wasix/src/syscalls/wasix/sock_send.rs b/lib/wasix/src/syscalls/wasix/sock_send.rs +index bfc196b..939e8a4 100644 +--- a/lib/wasix/src/syscalls/wasix/sock_send.rs ++++ b/lib/wasix/src/syscalls/wasix/sock_send.rs +@@ -28,12 +28,13 @@ pub fn sock_send( + WasiEnv::do_pending_operations(&mut ctx)?; + + let env = ctx.data(); +- let fd_entry = wasi_try_ok!(env.state.fs.get_fd(fd)); ++ let fd_entry = wasi_try_ok!({ env.state.fs.get_fd(fd) }); + let enable_journal = env.enable_journal; +- let guard = fd_entry.inode.read(); +- // Some guests route socket-like wakeups through pipe-backed fds. +- let use_write = matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeTx { .. }); +- drop(guard); ++ let use_write = { ++ let guard = fd_entry.inode.read(); ++ // Some guests route socket-like wakeups through pipe-backed fds. ++ matches!(guard.deref(), Kind::DuplexPipe { .. } | Kind::PipeTx { .. }) ++ }; + + let bytes_written = if use_write { + let offset = { fd_entry.inner.offset.load(Ordering::Acquire) as usize }; +@@ -108,16 +109,20 @@ pub(crate) fn sock_send_internal( + + match si_data { + FdWriteSource::Iovs { iovs, iovs_len } => { +- let iovs_arr = iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; +- let iovs_arr = iovs_arr.access().map_err(mem_error_to_wasi)?; ++ let iovs_arr = { ++ let iovs_arr = iovs.slice(&memory, iovs_len).map_err(mem_error_to_wasi)?; ++ iovs_arr.access().map_err(mem_error_to_wasi) ++ }?; + + let mut sent = 0usize; + for iovs in iovs_arr.iter() { +- let buf = WasmPtr::::new(iovs.buf) +- .slice(&memory, iovs.buf_len) +- .map_err(mem_error_to_wasi)? +- .access() +- .map_err(mem_error_to_wasi)?; ++ let buf = { ++ WasmPtr::::new(iovs.buf) ++ .slice(&memory, iovs.buf_len) ++ .map_err(mem_error_to_wasi)? ++ .access() ++ .map_err(mem_error_to_wasi) ++ }?; + let local_sent = match socket + .send( + env.tasks().deref(), +diff --git a/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs b/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs +index f7ddbeb..bf020e3 100644 +--- a/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs ++++ b/lib/wasix/src/syscalls/wasix/sock_set_opt_flag.rs +@@ -48,6 +48,13 @@ pub(crate) fn sock_set_opt_flag_internal( + Ok(o) => o, + Err(_) => return Ok(Err(Errno::Inval)), + }; ++ match (&option, flag) { ++ (crate::net::socket::WasiSocketOption::NoDelay, true) => {} ++ (crate::net::socket::WasiSocketOption::NoDelay, false) => {} ++ (crate::net::socket::WasiSocketOption::KeepAlive, true) => {} ++ (crate::net::socket::WasiSocketOption::KeepAlive, false) => {} ++ _ => {} ++ } + wasi_try_ok_ok!(__sock_actor_mut( + ctx, + sock, diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0007-wasix-instance-linker-and-sealed-runtime.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0007-wasix-instance-linker-and-sealed-runtime.patch new file mode 100644 index 000000000..e4e6b8fde --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0007-wasix-instance-linker-and-sealed-runtime.patch @@ -0,0 +1,11082 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 7/9] wasix: integrate instance lifecycle and sealed loading + +Integrate the memory-image, linker, environment, runtime and task-manager +changes required by concurrent exec and the sealed Postmaster carrier. This +is the remaining coupled integration boundary, including preinitialized memory +and loader audit policy. It is deliberately not presented as one generic +upstream optimization; preserving its aggregate behavior is the prerequisite +for further hunk-level extraction and platform qualification. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/lib/wasix/Cargo.toml b/lib/wasix/Cargo.toml +index 9071afa..8b53f98 100644 +--- a/lib/wasix/Cargo.toml ++++ b/lib/wasix/Cargo.toml +@@ -126,7 +126,6 @@ toml.workspace = true + pin-utils.workspace = true + wasmparser.workspace = true + crossbeam-channel.workspace = true +-bus.workspace = true + + [target.'cfg(not(any(target_arch = "riscv64", target_arch = "loongarch64")))'.dependencies.reqwest] + workspace = true +@@ -148,7 +147,11 @@ termios.workspace = true + + [target.'cfg(windows)'.dependencies] + windows-sys = { workspace = true, features = [ ++ "Win32_Foundation", ++ "Win32_Security", ++ "Win32_System_Console", + "Win32_System_SystemInformation", ++ "Win32_System_Threading", + ] } + + [target.'cfg(not(target_arch = "wasm32"))'.dependencies] +@@ -196,7 +199,9 @@ features = ["wasm_js"] + default = ["sys-default"] + + time = ["tokio/time"] +-ctrlc = ["tokio/signal"] ++# Enables the explicit Unix host-lifecycle adapter. Merely enabling this ++# feature never installs process-global signal dispositions. ++ctrlc = [] + + webc_runner_rt_wcgi = [ + "hyper", +diff --git a/lib/wasix/src/lib.rs b/lib/wasix/src/lib.rs +index 9986bae..42a6ee3 100644 +--- a/lib/wasix/src/lib.rs ++++ b/lib/wasix/src/lib.rs +@@ -15,6 +15,9 @@ + //! [WASI plugin example](https://github.com/wasmerio/wasmer/blob/main/examples/plugin.rs) + //! for an example of how to extend WASI using the WASI FS API. + ++/// The exact `wasmer-wasix` crate version used by this runtime. ++pub const VERSION: &str = env!("CARGO_PKG_VERSION"); ++ + #[cfg(all( + not(feature = "sys"), + not(feature = "js"), +@@ -65,7 +68,7 @@ mod state; + mod syscalls; + mod utils; + +-use std::sync::Arc; ++use std::{collections::HashSet, sync::Arc}; + + #[allow(unused_imports)] + use bytes::{Bytes, BytesMut}; +@@ -77,7 +80,7 @@ pub use wasmer_wasix_types; + + use wasmer::{ + AsStoreMut, Exports, FunctionEnv, Imports, Memory32, MemoryAccessError, MemorySize, +- RuntimeError, imports, namespace, ++ RuntimeError, imports, + }; + + pub use virtual_fs; +@@ -92,6 +95,28 @@ pub use virtual_net::{ + }; + use wasmer_wasix_types::wasi::{Errno, ExitCode}; + ++/// Product-specific host ABI for PostgreSQL postmaster capabilities that are ++/// not part of portable WASIX. ++pub const OLIPHAUNT_POSTMASTER_V1_NAMESPACE: &str = "oliphaunt_postmaster_v1"; ++ ++type RequiredImports<'a> = HashSet<(&'a str, &'a str)>; ++ ++macro_rules! namespace_for_required_imports { ++ ($required:expr, $namespace:expr; $( $import_name:expr => $import_item:expr ),* $(,)?) => {{ ++ let mut namespace = Exports::new(); ++ ++ $( ++ if $required.map_or(true, |required| { ++ required.contains(&($namespace, $import_name)) ++ }) { ++ namespace.insert($import_name, $import_item); ++ } ++ )* ++ ++ namespace ++ }}; ++} ++ + pub use crate::{ + fs::{Fd, VIRTUAL_ROOT_FD, WasiFs, WasiInodes, default_fs_backing}, + os::{ +@@ -99,21 +124,33 @@ pub use crate::{ + command::{BuiltinCommand, VirtualCommand}, + task::{ + control_plane::WasiControlPlane, +- process::{WasiProcess, WasiProcessId}, ++ process::{WasiProcess, WasiProcessExecutionGuard, WasiProcessId}, + thread::{WasiThread, WasiThreadError, WasiThreadHandle, WasiThreadId}, + }, + }, + rewind::*, +- runtime::{PluggableRuntime, Runtime, task_manager::VirtualTaskManager}, ++ runtime::{PluggableRuntime, ResourceLimits, Runtime, task_manager::VirtualTaskManager}, + state::{ +- ALL_RIGHTS, WasiEnv, WasiEnvBuilder, WasiEnvInit, WasiFunctionEnv, +- WasiModuleInstanceHandles, WasiModuleTreeHandles, WasiStateCreationError, ++ ALL_RIGHTS, DETERMINISTIC_START_ANALYZER_POLICY, DETERMINISTIC_START_GLOBAL_EFFECTS, ++ DETERMINISTIC_START_MEMORY_EFFECTS, DETERMINISTIC_START_MEMORY_READS, ++ DETERMINISTIC_START_PROOF_SCHEMA, DETERMINISTIC_START_TABLE_EFFECTS, ++ DeterministicStartProof, IntrinsicFileImmutability, PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, ++ PREINITIALIZED_MEMORY_IMAGE_PHASE, PreinitializedMemoryImage, ++ PreinitializedMemoryImageBacking, PreinitializedMemoryImageCapture, ++ PreinitializedMemoryImageHandle, PreinitializedMemoryImageLoadAudit, ++ PreinitializedMemoryImageLoader, PreinitializedMemoryImageMetadata, ++ PreinitializedMemoryImageMode, PreinitializedMemoryImageRuntimeAudit, WasiEnv, ++ WasiEnvBuilder, WasiEnvInit, WasiFunctionEnv, WasiModuleInstanceHandles, ++ WasiModuleTreeHandles, WasiStateCreationError, intrinsic_file_immutability, + }, + syscalls::{journal::wait_for_snapshot, rewind, rewind_ext, types, unwind}, + utils::is_wasix_module, + utils::{ + WasiVersion, get_wasi_version, get_wasi_versions, is_wasi_module, +- store::{StoreSnapshot, capture_store_snapshot, restore_store_snapshot}, ++ store::{ ++ StoreSnapshot, StoreSnapshotCaptureError, capture_store_snapshot, ++ restore_store_snapshot, ++ }, + }, + }; + +@@ -131,6 +168,8 @@ pub enum WasiError { + UnknownWasiVersion, + #[error("Dynamically-linked symbol not found or has bad type: {0}")] + DlSymbolResolutionFailed(String), ++ #[error("failed to capture WASIX store snapshot: {0}")] ++ StoreSnapshot(#[from] StoreSnapshotCaptureError), + } + + pub type WasiResult = Result, WasiError>; +@@ -332,6 +371,14 @@ pub struct WasiVFork { + /// Handle of the thread we have forked (dropping this handle + /// will signal that the thread is dead) + pub handle: WasiThreadHandle, ++ ++ /// Keeps the suspended parent non-reapable even if the child crosses a ++ /// deep-sleep TaskWasm boundary before returning control. ++ pub(crate) parent_execution: WasiProcessExecutionGuard, ++ ++ /// Owns execution admission for the child while it runs in-place inside ++ /// the parent's Wasmer instance, before exec creates a successor task. ++ pub(crate) child_execution: WasiProcessExecutionGuard, + } + + #[derive(Debug, Clone)] +@@ -350,6 +397,8 @@ impl Clone for WasiVFork { + asyncify: self.asyncify.clone(), + env: Box::new(self.env.as_ref().clone()), + handle: self.handle.clone(), ++ parent_execution: self.parent_execution.clone(), ++ child_execution: self.child_execution.clone(), + } + } + } +@@ -370,7 +419,7 @@ pub fn generate_import_object_from_env( + WasiVersion::Wasix64v1 => generate_import_object_wasix64_v1(store, ctx), + }; + +- let exports_wasi_generic = wasi_exports_generic(store, ctx); ++ let exports_wasi_generic = wasi_exports_generic(store, ctx, None); + + let imports_wasi_generic = imports! { + "wasi" => exports_wasi_generic, +@@ -378,20 +427,49 @@ pub fn generate_import_object_from_env( + + imports.extend(&imports_wasi_generic); + ++ // Explicit-version callers do not provide the module's import set, so ++ // register the small product ABI unfiltered. The main/dylink paths below ++ // still materialize only imports actually requested by each module. ++ let exports_oliphaunt_postmaster = oliphaunt_postmaster_exports(store, ctx, None); ++ let imports_oliphaunt_postmaster = imports! { ++ OLIPHAUNT_POSTMASTER_V1_NAMESPACE => exports_oliphaunt_postmaster, ++ }; ++ imports.extend(&imports_oliphaunt_postmaster); ++ + imports + } + +-fn wasi_exports_generic(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { ++fn oliphaunt_postmaster_exports( ++ mut store: &mut impl AsStoreMut, ++ env: &FunctionEnv, ++ required: Option<&RequiredImports<'_>>, ++) -> Exports { ++ use syscalls::*; ++ ++ namespace_for_required_imports! { required, OLIPHAUNT_POSTMASTER_V1_NAMESPACE; ++ "fd_sync_range" => Function::new_typed_with_env(&mut store, env, fd_sync_range), ++ } ++} ++ ++fn wasi_exports_generic( ++ mut store: &mut impl AsStoreMut, ++ env: &FunctionEnv, ++ required: Option<&RequiredImports<'_>>, ++) -> Exports { + use syscalls::*; +- let namespace = namespace! { ++ let namespace = namespace_for_required_imports! { required, "wasi"; + "thread-spawn" => Function::new_typed_with_env(&mut store, env, thread_spawn::), + }; + namespace + } + +-fn wasi_unstable_exports(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { ++fn wasi_unstable_exports( ++ mut store: &mut impl AsStoreMut, ++ env: &FunctionEnv, ++ required: Option<&RequiredImports<'_>>, ++) -> Exports { + use syscalls::*; +- let namespace = namespace! { ++ let namespace = namespace_for_required_imports! { required, "wasi_unstable"; + "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), + "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), + "clock_res_get" => Function::new_typed_with_env(&mut store, env, clock_res_get::), +@@ -445,9 +523,10 @@ fn wasi_unstable_exports(mut store: &mut impl AsStoreMut, env: &FunctionEnv, ++ required: Option<&RequiredImports<'_>>, + ) -> Exports { + use syscalls::*; +- let namespace = namespace! { ++ let namespace = namespace_for_required_imports! { required, "wasi_snapshot_preview1"; + "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), + "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), + "clock_res_get" => Function::new_typed_with_env(&mut store, env, clock_res_get::), +@@ -499,11 +578,15 @@ fn wasi_snapshot_preview1_exports( + namespace + } + +-fn wasix_exports_32(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { ++fn wasix_exports_32( ++ mut store: &mut impl AsStoreMut, ++ env: &FunctionEnv, ++ required: Option<&RequiredImports<'_>>, ++) -> Exports { + let engine_supports_async = store.as_store_ref().engine().supports_async(); + + use syscalls::*; +- let namespace = namespace! { ++ let namespace = namespace_for_required_imports! { required, "wasix_32v1"; + "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), + "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), + "call_dynamic" => Function::new_typed_with_env(&mut store, env, call_dynamic::), +@@ -576,10 +659,14 @@ fn wasix_exports_32(mut store: &mut impl AsStoreMut, env: &FunctionEnv) + "proc_spawn2" => Function::new_typed_with_env(&mut store, env, proc_spawn2::), + "proc_id" => Function::new_typed_with_env(&mut store, env, proc_id::), + "proc_parent" => Function::new_typed_with_env(&mut store, env, proc_parent::), ++ "proc_rlimit_get" => Function::new_typed_with_env(&mut store, env, proc_rlimit_get::), + "random_get" => Function::new_typed_with_env(&mut store, env, random_get::), + "tty_get" => Function::new_typed_with_env(&mut store, env, tty_get::), + "tty_set" => Function::new_typed_with_env(&mut store, env, tty_set::), + "getcwd" => Function::new_typed_with_env(&mut store, env, getcwd::), ++ "mem_mmap" => Function::new_typed_with_env(&mut store, env, mem_mmap::), ++ "mem_munmap" => Function::new_typed_with_env(&mut store, env, mem_munmap::), ++ "mem_msync" => Function::new_typed_with_env(&mut store, env, mem_msync::), + "chdir" => Function::new_typed_with_env(&mut store, env, chdir::), + "dl_invalid_handle" => Function::new_typed_with_env(&mut store, env, dl_invalid_handle), + "dlopen" => Function::new_typed_with_env(&mut store, env, dlopen::), +@@ -646,11 +733,15 @@ fn wasix_exports_32(mut store: &mut impl AsStoreMut, env: &FunctionEnv) + namespace + } + +-fn wasix_exports_64(mut store: &mut impl AsStoreMut, env: &FunctionEnv) -> Exports { ++fn wasix_exports_64( ++ mut store: &mut impl AsStoreMut, ++ env: &FunctionEnv, ++ required: Option<&RequiredImports<'_>>, ++) -> Exports { + let engine_supports_async = store.as_store_ref().engine().supports_async(); + + use syscalls::*; +- let namespace = namespace! { ++ let namespace = namespace_for_required_imports! { required, "wasix_64v1"; + "args_get" => Function::new_typed_with_env(&mut store, env, args_get::), + "args_sizes_get" => Function::new_typed_with_env(&mut store, env, args_sizes_get::), + "call_dynamic" => Function::new_typed_with_env(&mut store, env, call_dynamic::), +@@ -723,10 +814,14 @@ fn wasix_exports_64(mut store: &mut impl AsStoreMut, env: &FunctionEnv) + "proc_spawn2" => Function::new_typed_with_env(&mut store, env, proc_spawn2::), + "proc_id" => Function::new_typed_with_env(&mut store, env, proc_id::), + "proc_parent" => Function::new_typed_with_env(&mut store, env, proc_parent::), ++ "proc_rlimit_get" => Function::new_typed_with_env(&mut store, env, proc_rlimit_get::), + "random_get" => Function::new_typed_with_env(&mut store, env, random_get::), + "tty_get" => Function::new_typed_with_env(&mut store, env, tty_get::), + "tty_set" => Function::new_typed_with_env(&mut store, env, tty_set::), + "getcwd" => Function::new_typed_with_env(&mut store, env, getcwd::), ++ "mem_mmap" => Function::new_typed_with_env(&mut store, env, mem_mmap::), ++ "mem_munmap" => Function::new_typed_with_env(&mut store, env, mem_munmap::), ++ "mem_msync" => Function::new_typed_with_env(&mut store, env, mem_msync::), + "chdir" => Function::new_typed_with_env(&mut store, env, chdir::), + "dl_invalid_handle" => Function::new_typed_with_env(&mut store, env, dl_invalid_handle), + "dlopen" => Function::new_typed_with_env(&mut store, env, dlopen::), +@@ -796,15 +891,23 @@ fn wasix_exports_64(mut store: &mut impl AsStoreMut, env: &FunctionEnv) + // TODO: split function into two variants, one for JS and one for sys. + // (this will make code less messy) + fn import_object_for_all_wasi_versions( +- _module: &wasmer::Module, ++ module: &wasmer::Module, + store: &mut impl AsStoreMut, + env: &FunctionEnv, + ) -> Imports { +- let exports_wasi_generic = wasi_exports_generic(store, env); +- let exports_wasi_unstable = wasi_unstable_exports(store, env); +- let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env); +- let exports_wasix_32v1 = wasix_exports_32(store, env); +- let exports_wasix_64v1 = wasix_exports_64(store, env); ++ let required: RequiredImports<'_> = module ++ .info() ++ .imports ++ .keys() ++ .map(|import| (import.module.as_str(), import.field.as_str())) ++ .collect(); ++ let required = Some(&required); ++ let exports_wasi_generic = wasi_exports_generic(store, env, required); ++ let exports_wasi_unstable = wasi_unstable_exports(store, env, required); ++ let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env, required); ++ let exports_wasix_32v1 = wasix_exports_32(store, env, required); ++ let exports_wasix_64v1 = wasix_exports_64(store, env, required); ++ let exports_oliphaunt_postmaster = oliphaunt_postmaster_exports(store, env, required); + + // Allowed due to JS feature flag complications. + #[allow(unused_mut)] +@@ -814,17 +917,174 @@ fn import_object_for_all_wasi_versions( + "wasi_snapshot_preview1" => exports_wasi_snapshot_preview1, + "wasix_32v1" => exports_wasix_32v1, + "wasix_64v1" => exports_wasix_64v1, ++ OLIPHAUNT_POSTMASTER_V1_NAMESPACE => exports_oliphaunt_postmaster, + }; + + imports + } + ++#[cfg(test)] ++mod required_import_tests { ++ use super::*; ++ ++ #[test] ++ fn filtered_namespace_only_materializes_exact_module_imports() { ++ fn retained() {} ++ fn must_not_materialize() -> wasmer::Function { ++ panic!("filtered import expression was evaluated") ++ } ++ ++ let required = RequiredImports::from([("wasix_32v1", "retained")]); ++ let mut store = wasmer::Store::default(); ++ let exports = namespace_for_required_imports! { ++ Some(&required), "wasix_32v1"; ++ "retained" => wasmer::Function::new_typed(&mut store, retained), ++ "wrong-name" => must_not_materialize(), ++ }; ++ ++ assert_eq!(exports.len(), 1); ++ assert!(exports.get_function("retained").is_ok()); ++ assert!(exports.get_function("wrong-name").is_err()); ++ } ++ ++ #[test] ++ fn filtered_namespace_distinguishes_equal_names_in_other_abis() { ++ fn must_not_materialize() -> wasmer::Function { ++ panic!("an import from another ABI namespace was materialized") ++ } ++ ++ let required = RequiredImports::from([("wasi_snapshot_preview1", "fd_read")]); ++ let exports = namespace_for_required_imports! { ++ Some(&required), "wasix_32v1"; ++ "fd_read" => must_not_materialize(), ++ }; ++ ++ assert!(exports.is_empty()); ++ } ++ ++ #[tokio::test] ++ async fn oliphaunt_postmaster_fd_sync_range_has_exact_versioned_abi() { ++ use wasmer::{FunctionType, Type}; ++ ++ let engine = wasmer::Engine::default(); ++ let mut store = wasmer::Store::new(engine.clone()); ++ let func_env = WasiEnv::builder("fd-sync-range-abi-test") ++ .engine(engine) ++ .finalize(&mut store) ++ .unwrap(); ++ let required = ++ RequiredImports::from([(OLIPHAUNT_POSTMASTER_V1_NAMESPACE, "fd_sync_range")]); ++ let exports = oliphaunt_postmaster_exports(&mut store, &func_env.env, Some(&required)); ++ let function = exports.get_function("fd_sync_range").unwrap(); ++ ++ assert_eq!( ++ function.ty(&store), ++ FunctionType::new([Type::I32, Type::I64, Type::I64, Type::I32], [Type::I32]) ++ ); ++ } ++ ++ #[tokio::test] ++ async fn explicit_wasi_import_object_includes_versioned_postmaster_namespace() { ++ let engine = wasmer::Engine::default(); ++ let mut store = wasmer::Store::new(engine.clone()); ++ let func_env = WasiEnv::builder("fd-sync-range-import-test") ++ .engine(engine) ++ .finalize(&mut store) ++ .unwrap(); ++ let imports = ++ generate_import_object_from_env(&mut store, &func_env.env, WasiVersion::Wasix32v1); ++ let exports = imports ++ .get_namespace_exports(OLIPHAUNT_POSTMASTER_V1_NAMESPACE) ++ .unwrap(); ++ ++ assert_eq!(exports.len(), 1); ++ assert!(exports.get_function("fd_sync_range").is_ok()); ++ } ++ ++ #[cfg(feature = "sys")] ++ fn dynamic_mem_mmap_test_instance() -> (wasmer::Store, wasmer::Instance, WasiFunctionEnv) { ++ use wasmer::sys::{BaseTunables, NativeEngineExt}; ++ use wasmer::{Module, Pages}; ++ ++ let mut engine = wasmer::Engine::default(); ++ engine.set_tunables(BaseTunables { ++ static_memory_bound: Pages(0), ++ static_memory_offset_guard_size: 0, ++ dynamic_memory_offset_guard_size: 0, ++ }); ++ let mut store = wasmer::Store::new(engine.clone()); ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (import "wasix_32v1" "mem_mmap" ++ (func $mem_mmap ++ (param i32 i32 i32 i32 i32 i64 i32) ++ (result i32))) ++ (memory (export "memory") 1 2) ++ (func (export "map") (param $prot i32) (result i32) ++ i32.const 0 ++ i32.const 4096 ++ local.get $prot ++ i32.const 1 ++ i32.const -1 ++ i64.const 0 ++ i32.const 16 ++ call $mem_mmap))"#, ++ ) ++ .unwrap(); ++ let (instance, env) = WasiEnv::builder("direct-mem-mmap-test") ++ .engine(engine) ++ .instantiate(module, &mut store) ++ .unwrap(); ++ (store, instance, env) ++ } ++ ++ #[cfg(feature = "sys")] ++ #[tokio::test] ++ async fn direct_mem_mmap_import_rejects_weaker_protection_contracts() { ++ let (mut store, instance, _env) = dynamic_mem_mmap_test_instance(); ++ let map = instance ++ .exports ++ .get_typed_function::(&store, "map") ++ .unwrap(); ++ ++ assert_eq!( ++ map.call(&mut store, 0x01).unwrap(), ++ wasmer_wasix_types::wasi::Errno::Inval as i32 ++ ); ++ assert_eq!( ++ map.call(&mut store, 0x02).unwrap(), ++ wasmer_wasix_types::wasi::Errno::Inval as i32 ++ ); ++ } ++ ++ #[cfg(feature = "sys")] ++ #[tokio::test] ++ async fn unsupported_persistent_remap_rejects_before_grow_or_state_publication() { ++ let (mut store, instance, env) = dynamic_mem_mmap_test_instance(); ++ let memory = instance.exports.get_memory("memory").unwrap(); ++ let map = instance ++ .exports ++ .get_typed_function::(&store, "map") ++ .unwrap(); ++ let size_before = memory.size(&store); ++ ++ assert!(!memory.supports_persistent_shared_fixed_remap(&store)); ++ assert_eq!( ++ map.call(&mut store, 0x01 | 0x02).unwrap(), ++ wasmer_wasix_types::wasi::Errno::Notsup as i32 ++ ); ++ assert_eq!(memory.size(&store), size_before); ++ assert!(env.data(&store).state.shared_memory_mappings().is_empty()); ++ } ++} ++ + /// Combines a state generating function with the import list for legacy WASI + fn generate_import_object_snapshot0( + store: &mut impl AsStoreMut, + env: &FunctionEnv, + ) -> Imports { +- let exports_unstable = wasi_unstable_exports(store, env); ++ let exports_unstable = wasi_unstable_exports(store, env, None); + imports! { + "wasi_unstable" => exports_unstable + } +@@ -834,7 +1094,7 @@ fn generate_import_object_snapshot1( + store: &mut impl AsStoreMut, + env: &FunctionEnv, + ) -> Imports { +- let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env); ++ let exports_wasi_snapshot_preview1 = wasi_snapshot_preview1_exports(store, env, None); + imports! { + "wasi_snapshot_preview1" => exports_wasi_snapshot_preview1 + } +@@ -845,7 +1105,7 @@ fn generate_import_object_wasix32_v1( + store: &mut impl AsStoreMut, + env: &FunctionEnv, + ) -> Imports { +- let exports_wasix_32v1 = wasix_exports_32(store, env); ++ let exports_wasix_32v1 = wasix_exports_32(store, env, None); + imports! { + "wasix_32v1" => exports_wasix_32v1 + } +@@ -855,7 +1115,7 @@ fn generate_import_object_wasix64_v1( + store: &mut impl AsStoreMut, + env: &FunctionEnv, + ) -> Imports { +- let exports_wasix_64v1 = wasix_exports_64(store, env); ++ let exports_wasix_64v1 = wasix_exports_64(store, env, None); + imports! { + "wasix_64v1" => exports_wasix_64v1 + } +diff --git a/lib/wasix/src/os/command/builtins/cmd_wasmer.rs b/lib/wasix/src/os/command/builtins/cmd_wasmer.rs +index 4f60993..a383abd 100644 +--- a/lib/wasix/src/os/command/builtins/cmd_wasmer.rs ++++ b/lib/wasix/src/os/command/builtins/cmd_wasmer.rs +@@ -76,10 +76,17 @@ impl CmdWasmer { + let mut env = config.take().ok_or(SpawnError::UnknownError)?; + + // Set the arguments of the environment by replacing the state +- let mut state = env.state.fork(); + args.insert(0, what.clone()); +- state.args = std::sync::Mutex::new(args); +- env.state = Arc::new(state); ++ env.state = env ++ .state ++ .fork_with(move |state| { ++ state.args = std::sync::Mutex::new(args); ++ }) ++ .map_err(|errno| { ++ SpawnError::Other(Box::new(std::io::Error::other(format!( ++ "could not fork command state during shared-memory transition: {errno}" ++ )))) ++ })?; + + let file_path = if what.starts_with('/') { + PathBuf::from(&what) +diff --git a/lib/wasix/src/os/console/mod.rs b/lib/wasix/src/os/console/mod.rs +index 1281f50..2755cdd 100644 +--- a/lib/wasix/src/os/console/mod.rs ++++ b/lib/wasix/src/os/console/mod.rs +@@ -324,7 +324,6 @@ mod tests { + /// + /// See [#4284](https://github.com/wasmerio/wasmer/issues/4284) for more. + #[test] +- #[cfg_attr(not(feature = "host-reqwest"), ignore = "Requires a HTTP client")] + #[ignore = "Unconditionally aborts (CC #4284)"] + fn test_console_dash_tty_with_args_and_env() { + let tokio_rt = tokio::runtime::Runtime::new().unwrap(); +diff --git a/lib/wasix/src/runners/wasi.rs b/lib/wasix/src/runners/wasi.rs +index c416457..f17df68 100644 +--- a/lib/wasix/src/runners/wasi.rs ++++ b/lib/wasix/src/runners/wasi.rs +@@ -1,6 +1,6 @@ + //! WebC container support for running WASI modules + +-use std::{path::PathBuf, sync::Arc}; ++use std::{collections::HashMap, path::PathBuf, sync::Arc}; + + use anyhow::{Context, Error}; + use tracing::Instrument; +@@ -9,25 +9,134 @@ use wasmer::{Engine, Module}; + use wasmer_types::ModuleHash; + use webc::metadata::{Command, annotations::Wasi}; + ++#[cfg(feature = "journal")] ++use crate::journal::{DynJournal, DynReadableJournal, SnapshotTrigger}; ++#[cfg(all(feature = "ctrlc", any(unix, windows)))] ++use crate::os::task::HostLifecycleSupervisor; + use crate::{ + Runtime, WasiEnvBuilder, WasiError, WasiRuntimeError, + bin_factory::BinaryPackage, + capabilities::Capabilities, +- journal::{DynJournal, DynReadableJournal, SnapshotTrigger}, ++ os::task::{TaskJoinHandle, terminate_and_reap_abandoned_task}, + runners::{MappedDirectory, MountedDirectory, wasi_common::CommonWasiOptions}, + runtime::task_manager::VirtualTaskManagerExt, ++ state::{PreinitializedMemoryImageHandle, PreinitializedMemoryImageMode}, + }; ++use wasmer_wasix_types::wasi::{ExitCode, Signal}; + + use super::wasi_common::{ + ExistingMountConflictBehavior, MAPPED_CURRENT_DIR_DEFAULT_PATH, MappedCommand, + }; + ++/// Owns a spawned root until its authoritative completion has been observed. ++/// ++/// The admitted watcher future can still be canceled or panic after guest ++/// spawn. A plain `TaskJoinHandle` drop does not terminate anything, so this ++/// guard transfers every abnormal path to an independent, non-Tokio reaper. ++struct RootTaskLifecycleGuard { ++ task: Option, ++} ++ ++impl RootTaskLifecycleGuard { ++ fn new(task: TaskJoinHandle) -> Self { ++ Self { task: Some(task) } ++ } ++ ++ fn task(&self) -> &TaskJoinHandle { ++ self.task ++ .as_ref() ++ .expect("armed root lifecycle guard has no task") ++ } ++ ++ async fn wait_finished(mut self) -> Result> { ++ let result = self ++ .task ++ .as_mut() ++ .expect("armed root lifecycle guard has no task") ++ .wait_finished() ++ .await; ++ // Authoritative terminal status was observed; disarm the hot path so ++ // normal completion never creates a reaper thread. ++ self.task.take(); ++ result ++ } ++ ++ async fn terminate_and_reap(mut self, bind_error: Error) -> Error { ++ let result = terminate_root_after_lifecycle_bind_failure( ++ self.task ++ .as_mut() ++ .expect("armed root lifecycle guard has no task"), ++ bind_error, ++ ) ++ .await; ++ self.task.take(); ++ result ++ } ++} ++ ++impl Drop for RootTaskLifecycleGuard { ++ fn drop(&mut self) { ++ if let Some(task) = self.task.take() { ++ terminate_and_reap_abandoned_task(task, "wasmer-root-reaper", "root task", None, None); ++ } ++ } ++} ++ ++async fn terminate_root_after_lifecycle_bind_failure( ++ task: &mut TaskJoinHandle, ++ bind_error: Error, ++) -> Error { ++ let signal_error = task.send_signal(Signal::Sigkill).err(); ++ let completion = task.wait_finished().await; ++ let cleanup = match (signal_error, completion) { ++ (None, Ok(code)) => format!( ++ "root task terminated and reaped with exit code {}", ++ code.raw() ++ ), ++ (None, Err(error)) => format!("root task terminated and reaped with error: {error}"), ++ (Some(signal_error), Ok(code)) => format!( ++ "root task finished and was reaped with exit code {} after SIGKILL delivery failed: {signal_error}", ++ code.raw() ++ ), ++ (Some(signal_error), Err(error)) => format!( ++ "root task finished and was reaped with error {error} after SIGKILL delivery failed: {signal_error}" ++ ), ++ }; ++ bind_error.context(cleanup) ++} ++ ++/// Spawn, optionally bind, and join one root inside a single admitted future. ++/// ++/// Keeping `spawn_root` inside this async function is deliberate: constructing ++/// the future has no guest-side effects. A task manager that rejects (or ++/// panics during) `spawn_and_block_on` admission drops this future without ++/// creating a root that has no waiter. ++async fn spawn_and_wait_for_root( ++ spawn_root: Spawn, ++ bind_root: Bind, ++) -> Result>, Error> ++where ++ Spawn: FnOnce() -> Result, ++ Bind: FnOnce(&TaskJoinHandle) -> Result<(), Error>, ++{ ++ let task_handle = spawn_root()?; ++ let root = RootTaskLifecycleGuard::new(task_handle); ++ if let Err(bind_error) = bind_root(root.task()) { ++ return Err(root.terminate_and_reap(bind_error).await); ++ } ++ Ok(root.wait_finished().await) ++} ++ + #[derive(Debug, Default, Clone)] + pub struct WasiRunner { + wasi: CommonWasiOptions, + stdin: Option, + stdout: Option, + stderr: Option, ++ sealed_modules: HashMap)>, ++ preinitialized_memory_image: Option, ++ #[cfg(all(feature = "ctrlc", any(unix, windows)))] ++ host_lifecycle_supervisor: Option>, + } + + pub enum PackageOrHash<'a> { +@@ -184,6 +293,48 @@ impl WasiRunner { + self + } + ++ /// Registers an immutable guest executable for every process spawned by ++ /// this runner. A non-empty registry closes executable lookup: paths not ++ /// present in it are denied before consulting the guest filesystem. ++ pub fn with_sealed_module( ++ &mut self, ++ guest_path: impl Into, ++ module_hash: ModuleHash, ++ preinitialized_memory_image: Option, ++ ) -> &mut Self { ++ self.sealed_modules.insert( ++ guest_path.into(), ++ (module_hash, preinitialized_memory_image), ++ ); ++ self ++ } ++ ++ /// Applies or captures a preinitialized memory image for only the initial ++ /// module executed by this runner. Exec aliases carry their own sealed ++ /// image through the immutable executable registry. ++ pub fn with_preinitialized_memory_image( ++ &mut self, ++ image: PreinitializedMemoryImageMode, ++ ) -> &mut Self { ++ self.preinitialized_memory_image = Some(image); ++ self ++ } ++ ++ /// Routes this runner's single root task through an already-installed, ++ /// process-scoped host lifecycle supervisor. ++ /// ++ /// `WasiRunner` never installs host signal dispositions itself. Embedders ++ /// therefore retain host policy by default, while a process owner such as ++ /// the CLI can opt in before any guest is scheduled. ++ #[cfg(all(feature = "ctrlc", any(unix, windows)))] ++ pub fn with_host_lifecycle_supervisor( ++ &mut self, ++ supervisor: Arc, ++ ) -> &mut Self { ++ self.host_lifecycle_supervisor = Some(supervisor); ++ self ++ } ++ + #[cfg(feature = "journal")] + pub fn with_snapshot_trigger(&mut self, on: SnapshotTrigger) -> &mut Self { + self.wasi.snapshot_on.push(on); +@@ -372,9 +523,8 @@ impl WasiRunner { + None, + )?; + +- #[cfg(feature = "ctrlc")] +- { +- builder = builder.attach_ctrl_c(); ++ for (guest_path, (module_hash, image)) in &runner.sealed_modules { ++ builder.add_sealed_module(guest_path.clone(), *module_hash, image.clone()); + } + + #[cfg(feature = "journal")] +@@ -412,15 +562,50 @@ impl WasiRunner { + let env = builder.build()?; + let runtime = env.runtime.clone(); + let tasks = runtime.task_manager().clone(); ++ let root_process = env.process.clone(); ++ let control_plane = env.control_plane.clone(); ++ ++ #[cfg(all(feature = "ctrlc", any(unix, windows)))] ++ let host_lifecycle_supervisor = runner.host_lifecycle_supervisor.clone(); ++ let root_task = async move { ++ let result = spawn_and_wait_for_root( ++ move || { ++ crate::bin_factory::spawn_exec_module_with_preinitialized_memory_image( ++ module, ++ env, ++ &runtime, ++ runner.preinitialized_memory_image, ++ ) ++ .context("Spawn failed") ++ }, ++ move |task_handle| { ++ #[cfg(not(all(feature = "ctrlc", any(unix, windows))))] ++ let _ = task_handle; ++ ++ #[cfg(all(feature = "ctrlc", any(unix, windows)))] ++ if let Some(supervisor) = &host_lifecycle_supervisor ++ && let Err(error) = supervisor.bind_task(task_handle) ++ { ++ return Err(Error::new(error).context( ++ "Unable to bind the root task to host lifecycle supervision", ++ )); ++ } ++ ++ Ok(()) ++ }, ++ ) ++ .await?; ++ control_plane ++ .wait_for_process_tree_quiescence(&root_process) ++ .await; ++ Ok::<_, Error>(result) ++ } ++ .in_current_span(); + +- let mut task_handle = +- crate::bin_factory::spawn_exec_module(module, env, &runtime).context("Spawn failed")?; +- +- #[cfg(feature = "ctrlc")] +- task_handle.install_ctrlc_handler(); +- let task_handle = async move { task_handle.wait_finished().await }.in_current_span(); +- +- let result = tasks.spawn_and_block_on(task_handle)?; ++ // Spawn, binding, fail-closed cleanup, and the normal join all run ++ // inside the same admitted future. No guest effect precedes watcher ++ // admission, and bind failure never needs a second admission. ++ let result = tasks.spawn_and_block_on(root_task)??; + let exit_code = result + .map_err(|err| { + // We do our best to recover the error +@@ -515,6 +700,8 @@ impl WasiRunner { + let command_name = command_name.to_string(); + let tasks = runtime.task_manager().clone(); + let pkg = pkg.clone(); ++ #[cfg(all(feature = "ctrlc", any(unix, windows)))] ++ let host_lifecycle_supervisor = self.host_lifecycle_supervisor.clone(); + + // Wrapping the call to `spawn_and_block_on` in a call to `spawn_await` could help to prevent deadlocks + // because then blocking in here won't block the tokio runtime +@@ -522,16 +709,21 @@ impl WasiRunner { + // See run_wasm above for a possible fix + let exit_code = tasks.spawn_and_block_on( + async move { +- let mut task_handle = +- crate::bin_factory::spawn_exec(pkg, &command_name, env, &runtime) +- .await +- .context("Spawn failed")?; +- +- #[cfg(feature = "ctrlc")] +- task_handle.install_ctrlc_handler(); ++ let task_handle = crate::bin_factory::spawn_exec(pkg, &command_name, env, &runtime) ++ .await ++ .context("Spawn failed")?; ++ let root = RootTaskLifecycleGuard::new(task_handle); ++ ++ #[cfg(all(feature = "ctrlc", any(unix, windows)))] ++ if let Some(supervisor) = &host_lifecycle_supervisor ++ && let Err(error) = supervisor.bind_task(root.task()) ++ { ++ let bind_error = Error::new(error) ++ .context("Unable to bind the root task to host lifecycle supervision"); ++ return Err(root.terminate_and_reap(bind_error).await); ++ } + +- task_handle +- .wait_finished() ++ root.wait_finished() + .await + .map_err(|err| { + // We do our best to recover the error +@@ -594,6 +786,9 @@ fn wasi_runtime_error_to_owned(err: &WasiRuntimeError) -> WasiRuntimeError { + WasiRuntimeError::Wasi(WasiError::DlSymbolResolutionFailed(symbol)) => { + WasiRuntimeError::Wasi(WasiError::DlSymbolResolutionFailed(symbol.clone())) + } ++ WasiRuntimeError::Wasi(WasiError::StoreSnapshot(err)) => { ++ WasiRuntimeError::Wasi(WasiError::StoreSnapshot(err.clone())) ++ } + WasiRuntimeError::ControlPlane(a) => WasiRuntimeError::ControlPlane(a.clone()), + WasiRuntimeError::Runtime(a) => WasiRuntimeError::Runtime(a.clone()), + WasiRuntimeError::Thread(a) => WasiRuntimeError::Thread(a.clone()), +@@ -601,10 +796,180 @@ fn wasi_runtime_error_to_owned(err: &WasiRuntimeError) -> WasiRuntimeError { + } + } + ++#[cfg(all(test, feature = "ctrlc", any(unix, windows)))] ++mod host_lifecycle_tests { ++ use std::{ ++ future::Future, ++ sync::{Arc, mpsc}, ++ task::{Context as TaskContext, Poll}, ++ time::Duration, ++ }; ++ ++ use wasmer::FromToNativeWasmType; ++ ++ use super::*; ++ use crate::os::task::{ ++ OwnedTaskStatus, TaskStatus, ++ signal::{SignalDeliveryError, SignalHandlerAbi}, ++ }; ++ ++ #[derive(Debug)] ++ struct FinishingSignalHandler { ++ sender: mpsc::Sender, ++ } ++ ++ impl SignalHandlerAbi for FinishingSignalHandler { ++ fn signal(&self, signal: u8) -> Result<(), SignalDeliveryError> { ++ self.sender.send(signal).map_err(|_| SignalDeliveryError) ++ } ++ } ++ ++ fn finishing_root() -> (Arc, mpsc::Receiver, TaskJoinHandle) { ++ let (sender, receiver) = mpsc::channel(); ++ let mut owned = OwnedTaskStatus::default(); ++ owned.set_signal_handler(Arc::new(FinishingSignalHandler { sender })); ++ let owned = Arc::new(owned); ++ let task = owned.handle(); ++ (owned, receiver, task) ++ } ++ ++ fn finish_root_after_kill( ++ owned: Arc, ++ receiver: mpsc::Receiver, ++ ) -> std::thread::JoinHandle<()> { ++ std::thread::spawn(move || { ++ assert_eq!( ++ receiver.recv_timeout(Duration::from_secs(1)).unwrap(), ++ Signal::Sigkill.to_native() as u8 ++ ); ++ owned.set_finished(Ok(0u16.into())); ++ }) ++ } ++ ++ #[test] ++ fn wasi_runner_never_owns_host_lifecycle_by_default() { ++ assert!(WasiRunner::new().host_lifecycle_supervisor.is_none()); ++ } ++ ++ #[test] ++ fn lifecycle_bind_failure_kills_and_reaps_spawned_root() { ++ let (owned, receiver, mut task) = finishing_root(); ++ let finisher = finish_root_after_kill(owned.clone(), receiver); ++ ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let error = runtime.block_on(terminate_root_after_lifecycle_bind_failure( ++ &mut task, ++ anyhow::anyhow!("synthetic bind failure"), ++ )); ++ ++ finisher.join().unwrap(); ++ assert!(matches!(owned.status(), TaskStatus::Finished(_))); ++ assert!(error.to_string().contains("terminated and reaped")); ++ assert!(format!("{error:#}").contains("synthetic bind failure")); ++ } ++ ++ #[test] ++ fn bind_panic_terminates_and_reaps_spawned_root() { ++ let (owned, receiver, task) = finishing_root(); ++ let finisher = finish_root_after_kill(owned.clone(), receiver); ++ let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ virtual_mio::block_on(spawn_and_wait_for_root( ++ move || Ok(task), ++ |_| -> Result<(), Error> { panic!("synthetic bind panic") }, ++ )) ++ })); ++ ++ assert!(result.is_err()); ++ finisher.join().unwrap(); ++ assert!(matches!(owned.status(), TaskStatus::Finished(_))); ++ } ++ ++ #[test] ++ fn dropping_admitted_watcher_terminates_and_reaps_spawned_root() { ++ let (owned, receiver, task) = finishing_root(); ++ let finisher = finish_root_after_kill(owned.clone(), receiver); ++ let mut watcher = Box::pin(spawn_and_wait_for_root(move || Ok(task), |_| Ok(()))); ++ let waker = futures::task::noop_waker(); ++ let mut context = TaskContext::from_waker(&waker); ++ assert!(matches!(watcher.as_mut().poll(&mut context), Poll::Pending)); ++ ++ drop(watcher); ++ finisher.join().unwrap(); ++ assert!(matches!(owned.status(), TaskStatus::Finished(_))); ++ } ++} ++ + #[cfg(test)] + mod tests { + use super::*; + ++ #[derive(Debug)] ++ struct RejectingSharedTaskManager; ++ ++ impl crate::runtime::task_manager::VirtualTaskManager for RejectingSharedTaskManager { ++ fn sleep_now( ++ &self, ++ _time: std::time::Duration, ++ ) -> std::pin::Pin + Send + Sync + 'static>> ++ { ++ Box::pin(async {}) ++ } ++ ++ fn task_shared( ++ &self, ++ _task: Box futures::future::BoxFuture<'static, ()> + Send + 'static>, ++ ) -> Result<(), crate::WasiThreadError> { ++ Err(crate::WasiThreadError::Unsupported) ++ } ++ ++ fn task_wasm( ++ &self, ++ _task: crate::runtime::task_manager::TaskWasm, ++ ) -> Result<(), crate::WasiThreadError> { ++ unreachable!("root spawn must not reach task_wasm before shared-task admission") ++ } ++ ++ fn task_dedicated( ++ &self, ++ _task: Box, ++ ) -> Result<(), crate::WasiThreadError> { ++ unreachable!("root admission does not use a dedicated task") ++ } ++ ++ fn thread_parallelism(&self) -> Result { ++ Ok(1) ++ } ++ } ++ ++ #[test] ++ fn direct_root_spawn_has_no_effect_before_watcher_admission() { ++ use std::sync::atomic::{AtomicBool, Ordering}; ++ ++ let spawned = Arc::new(AtomicBool::new(false)); ++ let spawn_observer = spawned.clone(); ++ let root_task = spawn_and_wait_for_root( ++ move || { ++ spawn_observer.store(true, Ordering::SeqCst); ++ Err(anyhow::anyhow!("root spawn should not be polled")) ++ }, ++ |_| Ok(()), ++ ); ++ assert!(!spawned.load(Ordering::SeqCst)); ++ ++ // VirtualTaskManagerExt currently panics on a rejected task_shared ++ // admission. The admitted future must be dropped unpolled in that ++ // path, so its root-spawn closure must remain untouched. ++ let manager = Arc::new(RejectingSharedTaskManager); ++ let admission = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ let _ = manager.spawn_and_block_on(root_task); ++ })); ++ assert!(admission.is_err()); ++ assert!(!spawned.load(Ordering::SeqCst)); ++ } ++ + #[test] + fn send_and_sync() { + fn assert_send() {} +diff --git a/lib/wasix/src/runners/wasi_common.rs b/lib/wasix/src/runners/wasi_common.rs +index afb8669..a52c008 100644 +--- a/lib/wasix/src/runners/wasi_common.rs ++++ b/lib/wasix/src/runners/wasi_common.rs +@@ -5,6 +5,7 @@ use std::{ + }; + + use anyhow::{Context, Error}; ++#[cfg(feature = "host-fs")] + use tokio::runtime::Handle; + use virtual_fs::{ + ArcFileSystem, ExactMountConflictMode, FileSystem, MountFileSystem, OverlayFileSystem, +@@ -12,12 +13,13 @@ use virtual_fs::{ + }; + use webc::metadata::annotations::Wasi as WasiAnnotation; + ++#[cfg(feature = "journal")] ++use crate::journal::{DynJournal, DynReadableJournal, SnapshotTrigger}; + use crate::{ + WasiEnvBuilder, + bin_factory::{BinaryPackage, BinaryPackageMounts}, + capabilities::Capabilities, + fs::WasiFsRoot, +- journal::{DynJournal, DynReadableJournal, SnapshotTrigger}, + }; + + pub const MAPPED_CURRENT_DIR_DEFAULT_PATH: &str = "/home"; +@@ -48,10 +50,15 @@ pub(crate) struct CommonWasiOptions { + pub(crate) is_home_mapped: bool, + pub(crate) injected_packages: Vec, + pub(crate) capabilities: Capabilities, ++ #[cfg(feature = "journal")] + pub(crate) read_only_journals: Vec>, ++ #[cfg(feature = "journal")] + pub(crate) writable_journals: Vec>, ++ #[cfg(feature = "journal")] + pub(crate) snapshot_on: Vec, ++ #[cfg(feature = "journal")] + pub(crate) snapshot_interval: Option, ++ #[cfg(feature = "journal")] + pub(crate) stop_running_after_snapshot: bool, + pub(crate) skip_stdio_during_bootstrap: bool, + pub(crate) current_dir: Option, +diff --git a/lib/wasix/src/runtime/mod.rs b/lib/wasix/src/runtime/mod.rs +index 2023d22..e810d39 100644 +--- a/lib/wasix/src/runtime/mod.rs ++++ b/lib/wasix/src/runtime/mod.rs +@@ -1,6 +1,7 @@ + pub mod module_cache; + pub mod package_loader; + pub mod resolver; ++pub mod sealed_loader_audit; + pub mod task_manager; + + use self::module_cache::CacheError; +@@ -147,6 +148,16 @@ impl<'a> ModuleInput<'a> { + } + } + ++/// Process resource limits exposed by a WASIX runtime. ++/// ++/// Limits are runtime policy: embedders may choose conservative values that ++/// reflect host-side constraints that are not directly visible in guest memory. ++#[derive(Debug, Clone, Copy, Default)] ++pub struct ResourceLimits { ++ /// Effective process stack limit in bytes. ++ pub stack: Option, ++} ++ + /// Runtime components used when running WebAssembly programs. + /// + /// Think of this as the "System" in "WebAssembly Systems Interface". +@@ -161,6 +172,11 @@ where + /// Retrieve the active [`VirtualTaskManager`]. + fn task_manager(&self) -> &Arc; + ++ /// Process resource limits reported to WASIX programs. ++ fn resource_limits(&self) -> ResourceLimits { ++ ResourceLimits::default() ++ } ++ + /// A package loader. + fn package_loader(&self) -> Arc { + Arc::new(UnsupportedPackageLoader) +@@ -377,6 +393,9 @@ pub async fn load_module( + + match result { + Ok(module) => return Ok(module), ++ Err(error) if module_cache.is_authoritative() => { ++ return Err(crate::SpawnError::CacheError(error)); ++ } + Err(CacheError::NotFound) => {} + Err(other) => { + tracing::warn!( +@@ -459,6 +478,7 @@ pub struct PluggableRuntime { + pub source: Arc, + pub engine: Engine, + pub module_cache: Arc, ++ pub resource_limits: ResourceLimits, + pub tty: Option>, + #[cfg(feature = "journal")] + pub read_only_journals: Vec>, +@@ -500,6 +520,7 @@ impl PluggableRuntime { + source: Arc::new(source), + package_loader: Arc::new(loader), + module_cache: Arc::new(module_cache::in_memory()), ++ resource_limits: ResourceLimits::default(), + #[cfg(feature = "journal")] + read_only_journals: Vec::new(), + #[cfg(feature = "journal")] +@@ -522,6 +543,11 @@ impl PluggableRuntime { + self + } + ++ pub fn set_resource_limits(&mut self, resource_limits: ResourceLimits) -> &mut Self { ++ self.resource_limits = resource_limits; ++ self ++ } ++ + pub fn set_tty(&mut self, tty: Arc) -> &mut Self { + self.tty = Some(tty); + self +@@ -627,6 +653,10 @@ impl Runtime for PluggableRuntime { + &self.rt + } + ++ fn resource_limits(&self) -> ResourceLimits { ++ self.resource_limits ++ } ++ + fn tty(&self) -> Option<&(dyn TtyBridge + Send + Sync)> { + self.tty.as_deref() + } +@@ -688,6 +718,7 @@ pub struct OverriddenRuntime { + source: Option>, + engine: Option, + module_cache: Option>, ++ resource_limits: Option, + tty: Option>, + additional_imports: Vec, + instance_callbacks: Vec, +@@ -708,6 +739,7 @@ impl OverriddenRuntime { + source: None, + engine: None, + module_cache: None, ++ resource_limits: None, + tty: None, + additional_imports: Vec::new(), + instance_callbacks: Vec::new(), +@@ -756,6 +788,11 @@ impl OverriddenRuntime { + self + } + ++ pub fn with_resource_limits(mut self, resource_limits: ResourceLimits) -> Self { ++ self.resource_limits.replace(resource_limits); ++ self ++ } ++ + pub fn with_tty(mut self, tty: Arc) -> Self { + self.tty.replace(tty); + self +@@ -820,6 +857,11 @@ impl Runtime for OverriddenRuntime { + } + } + ++ fn resource_limits(&self) -> ResourceLimits { ++ self.resource_limits ++ .unwrap_or_else(|| self.inner.resource_limits()) ++ } ++ + fn source(&self) -> Arc { + if let Some(source) = self.source.clone() { + source +@@ -930,3 +972,58 @@ impl Runtime for OverriddenRuntime { + } + } + } ++ ++#[cfg(all(test, not(target_arch = "wasm32")))] ++mod authoritative_cache_tests { ++ use super::*; ++ ++ #[derive(Debug)] ++ struct AuthoritativeMissCache; ++ ++ #[async_trait::async_trait] ++ impl ModuleCache for AuthoritativeMissCache { ++ fn is_authoritative(&self) -> bool { ++ true ++ } ++ ++ async fn load(&self, _key: ModuleHash, _engine: &Engine) -> Result { ++ Err(CacheError::NotFound) ++ } ++ ++ async fn contains(&self, _key: ModuleHash, _engine: &Engine) -> Result { ++ Ok(false) ++ } ++ ++ async fn save( ++ &self, ++ _key: ModuleHash, ++ _engine: &Engine, ++ _module: &Module, ++ ) -> Result<(), CacheError> { ++ Err(CacheError::other(std::io::Error::new( ++ std::io::ErrorKind::PermissionDenied, ++ "authoritative cache is immutable", ++ ))) ++ } ++ } ++ ++ #[tokio::test] ++ async fn authoritative_miss_is_terminal_without_compilation_fallback() { ++ let engine = Engine::default(); ++ let cache: Arc = Arc::new(AuthoritativeMissCache); ++ assert!(cache.is_authoritative()); ++ ++ let error = load_module( ++ &engine, ++ cache.as_ref(), ++ ModuleInput::Bytes(Cow::Borrowed(b"\0asm\x01\0\0\0")), ++ None, ++ ) ++ .await ++ .unwrap_err(); ++ assert!(matches!( ++ error, ++ crate::SpawnError::CacheError(CacheError::NotFound) ++ )); ++ } ++} +diff --git a/lib/wasix/src/runtime/module_cache/filesystem.rs b/lib/wasix/src/runtime/module_cache/filesystem.rs +index 675c1bf..fd657ca 100644 +--- a/lib/wasix/src/runtime/module_cache/filesystem.rs ++++ b/lib/wasix/src/runtime/module_cache/filesystem.rs +@@ -42,10 +42,11 @@ impl FileSystemCache { + /// A tokio reactor must be available + #[tracing::instrument(level = "debug", skip_all, fields(? path))] + async fn tokio_load(path: PathBuf, engine: Engine) -> Result { +- let bytes = read_file(&path).await?; +- let deserialized = tokio::task::spawn_blocking(move || deserialize(&bytes, &engine)) +- .await +- .unwrap(); ++ let deserialize_path = path.clone(); ++ let deserialized = ++ tokio::task::spawn_blocking(move || deserialize_file(&deserialize_path, &engine)) ++ .await ++ .unwrap(); + match deserialized { + Ok(m) => { + tracing::debug!("Cache hit!"); +@@ -174,8 +175,8 @@ impl ModuleCache for FileSystemCache { + } + } + +-async fn read_file(path: &Path) -> Result, CacheError> { +- match tokio::fs::read(path).await { ++fn read_file(path: &Path) -> Result, CacheError> { ++ match std::fs::read(path) { + Ok(bytes) => Ok(bytes), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => Err(CacheError::NotFound), + Err(error) => Err(CacheError::FileRead { +@@ -185,7 +186,7 @@ async fn read_file(path: &Path) -> Result, CacheError> { + } + } + +-fn deserialize(bytes: &[u8], engine: &Engine) -> Result { ++fn deserialize_file(path: &Path, engine: &Engine) -> Result { + // We used to compress our compiled modules using LZW encoding in the past. + // This was removed because it has a negative impact on startup times for + // "wasmer run", so all new compiled modules should be saved directly to +@@ -201,18 +202,22 @@ fn deserialize(bytes: &[u8], engine: &Engine) -> Result { + // - ModuleCache::save(): 2.4s, 72MB binary + // - ModuleCache::load(): 822ms + +- match unsafe { Module::deserialize(engine, bytes) } { ++ match unsafe { Module::deserialize_from_file(engine, path) } { + // The happy case + Ok(m) => Ok(m), + Err(wasmer::DeserializeError::Incompatible(_)) => { ++ let bytes = read_file(path)?; + let bytes = weezl::decode::Decoder::new(weezl::BitOrder::Msb, 8) +- .decode(bytes) ++ .decode(&bytes) + .map_err(CacheError::other)?; + + let m = unsafe { Module::deserialize(engine, bytes)? }; + + Ok(m) + } ++ Err(wasmer::DeserializeError::Io(e)) if e.kind() == std::io::ErrorKind::NotFound => { ++ Err(CacheError::NotFound) ++ } + Err(e) => Err(CacheError::Deserialize(e)), + } + } +diff --git a/lib/wasix/src/runtime/module_cache/types.rs b/lib/wasix/src/runtime/module_cache/types.rs +index 3c4510c..0c1abd0 100644 +--- a/lib/wasix/src/runtime/module_cache/types.rs ++++ b/lib/wasix/src/runtime/module_cache/types.rs +@@ -23,6 +23,16 @@ use crate::runtime::module_cache::{FallbackCache, progress::ModuleLoadProgressRe + /// + #[async_trait::async_trait] + pub trait ModuleCache: Debug { ++ /// Whether this cache is the authoritative module closure. ++ /// ++ /// An authoritative cache is a policy boundary rather than an optimization: ++ /// a miss or load error must be returned to the caller without attempting to ++ /// compile the supplied WebAssembly bytes. This is used by sealed, ++ /// precompiled-only runtimes. ++ fn is_authoritative(&self) -> bool { ++ false ++ } ++ + /// Load a module based on its hash. + async fn load(&self, key: ModuleHash, engine: &Engine) -> Result; + +@@ -95,6 +105,10 @@ where + D: Deref + Debug + Send + Sync, + C: ModuleCache + Send + Sync + ?Sized, + { ++ fn is_authoritative(&self) -> bool { ++ (**self).is_authoritative() ++ } ++ + async fn load(&self, key: ModuleHash, engine: &Engine) -> Result { + (**self).load(key, engine).await + } +diff --git a/lib/wasix/src/runtime/sealed_loader_audit.rs b/lib/wasix/src/runtime/sealed_loader_audit.rs +new file mode 100644 +index 0000000..d974629 +--- /dev/null ++++ b/lib/wasix/src/runtime/sealed_loader_audit.rs +@@ -0,0 +1,310 @@ ++//! Non-faulting host-page-cache evidence for sealed loader inputs. ++//! ++//! These helpers describe advisory calls and Linux `mincore(2)` observations; ++//! they never participate in artifact admission or integrity decisions. ++ ++use std::fs::File; ++ ++/// Outcome of one or more best-effort file-cache advisory calls. ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub struct FileAdviceAudit { ++ pub supported: bool, ++ pub calls: u64, ++ pub successes: u64, ++ /// The first nonzero error returned by an advisory call. ++ pub first_errno: Option, ++} ++ ++impl FileAdviceAudit { ++ pub const fn unsupported() -> Self { ++ Self { ++ supported: false, ++ calls: 0, ++ successes: 0, ++ first_errno: None, ++ } ++ } ++ ++ pub const fn not_applicable() -> Self { ++ Self { ++ supported: true, ++ calls: 0, ++ successes: 0, ++ first_errno: None, ++ } ++ } ++} ++ ++/// A point-in-time observation of file-backed page-cache residency. ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub struct FileResidencyAudit { ++ pub state: FileResidencyState, ++ pub page_size: Option, ++ pub total_pages: Option, ++ pub resident_pages: Option, ++ pub resident_bytes: Option, ++ pub errno: Option, ++} ++ ++impl FileResidencyAudit { ++ pub const fn unsupported() -> Self { ++ Self { ++ state: FileResidencyState::Unsupported, ++ page_size: None, ++ total_pages: None, ++ resident_pages: None, ++ resident_bytes: None, ++ errno: None, ++ } ++ } ++ ++ pub const fn not_applicable() -> Self { ++ Self { ++ state: FileResidencyState::NotApplicable, ++ page_size: None, ++ total_pages: None, ++ resident_pages: None, ++ resident_bytes: None, ++ errno: None, ++ } ++ } ++ ++ #[cfg(target_os = "linux")] ++ const fn error(errno: i32) -> Self { ++ Self { ++ state: FileResidencyState::Error, ++ page_size: None, ++ total_pages: None, ++ resident_pages: None, ++ resident_bytes: None, ++ errno: Some(errno), ++ } ++ } ++} ++ ++/// Portable state attached to a residency checkpoint. ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub enum FileResidencyState { ++ Measured, ++ Unsupported, ++ NotApplicable, ++ Error, ++} ++ ++impl FileResidencyState { ++ pub const fn as_str(self) -> &'static str { ++ match self { ++ Self::Measured => "measured", ++ Self::Unsupported => "unsupported-platform", ++ Self::NotApplicable => "not-applicable", ++ Self::Error => "error", ++ } ++ } ++} ++ ++/// Apply sequential and no-reuse hints before a one-shot verified read. ++#[cfg(target_os = "linux")] ++pub fn advise_file_for_one_shot_read(file: &File) -> FileAdviceAudit { ++ use std::os::fd::AsRawFd; ++ ++ let results = [ ++ // SAFETY: `file` owns this live descriptor and no memory is accessed. ++ unsafe { libc::posix_fadvise(file.as_raw_fd(), 0, 0, libc::POSIX_FADV_SEQUENTIAL) }, ++ // SAFETY: as above; this is a best-effort page-replacement hint. ++ unsafe { libc::posix_fadvise(file.as_raw_fd(), 0, 0, libc::POSIX_FADV_NOREUSE) }, ++ ]; ++ FileAdviceAudit { ++ supported: true, ++ calls: results.len() as u64, ++ successes: results.iter().filter(|&&result| result == 0).count() as u64, ++ first_errno: results.into_iter().find(|&result| result != 0), ++ } ++} ++ ++#[cfg(not(target_os = "linux"))] ++pub fn advise_file_for_one_shot_read(_file: &File) -> FileAdviceAudit { ++ FileAdviceAudit::unsupported() ++} ++ ++/// Ask the kernel to release clean source pages after verified activation. ++#[cfg(target_os = "linux")] ++pub fn advise_file_away(file: &File) -> FileAdviceAudit { ++ use std::os::fd::AsRawFd; ++ ++ // SAFETY: `file` owns this live descriptor. DONTNEED neither changes file ++ // contents nor forms part of the loader's correctness contract. ++ let result = unsafe { libc::posix_fadvise(file.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) }; ++ FileAdviceAudit { ++ supported: true, ++ calls: 1, ++ successes: u64::from(result == 0), ++ first_errno: (result != 0).then_some(result), ++ } ++} ++ ++#[cfg(not(target_os = "linux"))] ++pub fn advise_file_away(_file: &File) -> FileAdviceAudit { ++ FileAdviceAudit::unsupported() ++} ++ ++/// Observe file-backed residency without reading or faulting payload bytes. ++/// ++/// Linux maps the descriptor `PROT_NONE`, asks `mincore(2)` for the existing ++/// page-cache vector, and immediately unmaps it. The mapping itself cannot be ++/// dereferenced and therefore cannot make a cold artifact resident. ++#[cfg(target_os = "linux")] ++pub fn file_residency(file: &File, logical_bytes: u64) -> FileResidencyAudit { ++ use std::os::fd::AsRawFd; ++ ++ fn errno_or(fallback: i32) -> i32 { ++ std::io::Error::last_os_error() ++ .raw_os_error() ++ .filter(|errno| *errno != 0) ++ .unwrap_or(fallback) ++ } ++ ++ let metadata = match file.metadata() { ++ Ok(metadata) => metadata, ++ Err(error) => { ++ return FileResidencyAudit::error(error.raw_os_error().unwrap_or(libc::EIO)); ++ } ++ }; ++ if metadata.len() != logical_bytes { ++ return FileResidencyAudit::error(libc::ESTALE); ++ } ++ ++ let page_size = unsafe { libc::sysconf(libc::_SC_PAGESIZE) }; ++ if page_size <= 0 { ++ return FileResidencyAudit::error(errno_or(libc::EINVAL)); ++ } ++ let page_size = page_size as u64; ++ if logical_bytes == 0 { ++ return FileResidencyAudit { ++ state: FileResidencyState::Measured, ++ page_size: Some(page_size), ++ total_pages: Some(0), ++ resident_pages: Some(0), ++ resident_bytes: Some(0), ++ errno: None, ++ }; ++ } ++ ++ let Ok(mapping_len) = usize::try_from(logical_bytes) else { ++ return FileResidencyAudit::error(libc::EOVERFLOW); ++ }; ++ let total_pages = logical_bytes.div_ceil(page_size); ++ let Ok(vector_len) = usize::try_from(total_pages) else { ++ return FileResidencyAudit::error(libc::EOVERFLOW); ++ }; ++ let mut vector = Vec::new(); ++ if vector.try_reserve_exact(vector_len).is_err() { ++ return FileResidencyAudit::error(libc::ENOMEM); ++ } ++ vector.resize(vector_len, 0_u8); ++ ++ // SAFETY: this creates a new non-dereferenceable mapping over the live ++ // descriptor. No loader mapping or executable allocation is modified. ++ let address = unsafe { ++ libc::mmap( ++ std::ptr::null_mut(), ++ mapping_len, ++ libc::PROT_NONE, ++ libc::MAP_SHARED, ++ file.as_raw_fd(), ++ 0, ++ ) ++ }; ++ if address == libc::MAP_FAILED { ++ return FileResidencyAudit::error(errno_or(libc::EIO)); ++ } ++ ++ // SAFETY: `address` names the full live mapping and `vector` has exactly ++ // one byte for every native page in that range. ++ let mincore_result = unsafe { libc::mincore(address, mapping_len, vector.as_mut_ptr()) }; ++ let mincore_errno = (mincore_result != 0).then(|| errno_or(libc::EIO)); ++ // SAFETY: this releases exactly the mapping created above. ++ let unmap_result = unsafe { libc::munmap(address, mapping_len) }; ++ let unmap_errno = (unmap_result != 0).then(|| errno_or(libc::EIO)); ++ if let Some(errno) = mincore_errno.or(unmap_errno) { ++ return FileResidencyAudit::error(errno); ++ } ++ ++ let resident_pages = vector.iter().filter(|entry| **entry & 1 != 0).count() as u64; ++ let resident_bytes = vector ++ .iter() ++ .enumerate() ++ .filter(|(_, entry)| **entry & 1 != 0) ++ .map(|(index, _)| { ++ let offset = index as u64 * page_size; ++ (logical_bytes - offset).min(page_size) ++ }) ++ .sum(); ++ FileResidencyAudit { ++ state: FileResidencyState::Measured, ++ page_size: Some(page_size), ++ total_pages: Some(total_pages), ++ resident_pages: Some(resident_pages), ++ resident_bytes: Some(resident_bytes), ++ errno: None, ++ } ++} ++ ++#[cfg(not(target_os = "linux"))] ++pub fn file_residency(_file: &File, _logical_bytes: u64) -> FileResidencyAudit { ++ FileResidencyAudit::unsupported() ++} ++ ++#[cfg(all(test, target_os = "linux"))] ++mod tests { ++ use super::*; ++ use std::io::{Read, Seek, SeekFrom, Write}; ++ ++ #[test] ++ fn mincore_checkpoint_does_not_move_or_read_the_descriptor() { ++ let mut file = tempfile::tempfile().unwrap(); ++ let bytes = vec![0x5a; 8193]; ++ file.write_all(&bytes).unwrap(); ++ file.seek(SeekFrom::Start(7)).unwrap(); ++ ++ let audit = file_residency(&file, bytes.len() as u64); ++ assert_eq!(audit.state, FileResidencyState::Measured); ++ assert_eq!(audit.total_pages, Some(3)); ++ assert_eq!(audit.resident_pages, Some(3)); ++ assert_eq!(audit.resident_bytes, Some(bytes.len() as u64)); ++ ++ let mut byte = [0_u8; 1]; ++ file.read_exact(&mut byte).unwrap(); ++ assert_eq!(byte[0], 0x5a); ++ assert_eq!(file.stream_position().unwrap(), 8); ++ } ++ ++ #[test] ++ fn mincore_checkpoint_rejects_a_changed_file_size() { ++ let mut file = tempfile::tempfile().unwrap(); ++ file.write_all(b"stable-size").unwrap(); ++ ++ let audit = file_residency(&file, 12); ++ assert_eq!(audit.state, FileResidencyState::Error); ++ assert_eq!(audit.errno, Some(libc::ESTALE)); ++ assert_eq!(audit.resident_pages, None); ++ assert_eq!(audit.resident_bytes, None); ++ } ++ ++ #[test] ++ fn advisory_audits_preserve_call_and_errno_shape() { ++ let mut file = tempfile::tempfile().unwrap(); ++ file.write_all(b"advice").unwrap(); ++ let read = advise_file_for_one_shot_read(&file); ++ assert!(read.supported); ++ assert_eq!(read.calls, 2); ++ assert_eq!(read.successes + u64::from(read.first_errno.is_some()), 2); ++ ++ let eviction = advise_file_away(&file); ++ assert!(eviction.supported); ++ assert_eq!(eviction.calls, 1); ++ assert_eq!( ++ eviction.successes, ++ u64::from(eviction.first_errno.is_none()) ++ ); ++ } ++} +diff --git a/lib/wasix/src/runtime/task_manager/mod.rs b/lib/wasix/src/runtime/task_manager/mod.rs +index 2a6457e..2418f25 100644 +--- a/lib/wasix/src/runtime/task_manager/mod.rs ++++ b/lib/wasix/src/runtime/task_manager/mod.rs +@@ -4,7 +4,7 @@ pub mod tokio; + + use std::ops::Deref; + use std::task::{Context, Poll}; +-use std::{pin::Pin, time::Duration}; ++use std::{pin::Pin, sync::Arc, time::Duration}; + + use bytes::Bytes; + use derive_more::Debug; +@@ -17,7 +17,13 @@ use wasmer_wasix_types::wasi::{Errno, ExitCode}; + + use crate::syscalls::AsyncifyFuture; + use crate::{StoreSnapshot, WasiEnv, WasiFunctionEnv, WasiThread, capture_store_snapshot}; +-use crate::{os::task::thread::WasiThreadError, state::Linker}; ++use crate::{ ++ os::task::{ ++ process::{WasiProcessExecutionLease, WasiProcessStartGate}, ++ thread::WasiThreadError, ++ }, ++ state::{Linker, PreinitializedMemoryImageMode}, ++}; + + pub use virtual_mio::waker::*; + +@@ -49,15 +55,157 @@ pub enum SpawnMemoryTypeOrStore { + StoreAndMemory(wasmer::Store, Option), + } + +-pub type WasmResumeTask = dyn FnOnce(WasiFunctionEnv, Store, Bytes) + Send + 'static; ++pub type WasmResumeTask = dyn FnOnce(WasiFunctionEnv, Store, Bytes, Option) ++ + Send ++ + 'static; + + pub type WasmResumeTrigger = dyn FnOnce() -> Pin> + Send + 'static>> + + Send + + Sync; + ++/// Fail-closed ownership for a `TaskWasm` that has not entered its accepted ++/// run callback yet. ++/// ++/// Memory construction, Wasmer instantiation, async trigger polling, and the ++/// blocking worker queue all happen before the callback owns execution. If ++/// any of those stages drops the task, this guard terminalizes the exact WASI ++/// thread before releasing the process-quiescence lease. The only successful ++/// disarm boundary is [`Self::accept_callback`], invoked from inside the ++/// worker callback immediately before calling `TaskWasmRun`. ++#[derive(Debug)] ++#[must_use = "dropping pending TaskWasm execution terminalizes its thread"] ++pub struct TaskWasmExecutionGuard { ++ thread: WasiThread, ++ lease: Option, ++ owner_identity: Arc<()>, ++} ++ ++impl TaskWasmExecutionGuard { ++ fn new(thread: WasiThread, lease: WasiProcessExecutionLease) -> Self { ++ Self { ++ thread, ++ lease: Some(lease), ++ owner_identity: Arc::new(()), ++ } ++ } ++ ++ /// Transfers responsibility to an accepted `TaskWasmRun` callback. ++ /// ++ /// Task-manager implementations must call this only from within the ++ /// worker closure, after every fallible pre-run stage has completed and ++ /// immediately before invoking the callback. The callback must publish a ++ /// terminal thread status or acquire an accepted successor lease before ++ /// the returned guard is dropped. ++ fn into_accepted_callback(mut self) -> TaskWasmAcceptedExecutionGuard { ++ TaskWasmAcceptedExecutionGuard { ++ thread: self.thread.clone(), ++ lease: self.lease.take(), ++ owner_identity: Arc::clone(&self.owner_identity), ++ } ++ } ++ ++ fn bind_pending_owner(&self, env: &mut WasiEnv) { ++ env.bind_pending_task_wasm_execution(&self.thread, &self.owner_identity); ++ } ++ ++ /// Accepts this callback only if [`TaskWasm::new`] bound this exact guard's ++ /// unforgeable identity to the physical instance in `env`. ++ /// ++ /// Every [`VirtualTaskManager`] implementation must invoke this from the ++ /// accepted worker closure after all fallible setup and immediately before ++ /// constructing [`TaskWasmRunProperties`]. The exact-token assertion and ++ /// private raw conversion make alternate task managers follow the same ++ /// vfork ownership contract as the built-in Tokio implementation. ++ pub fn accept_callback(self, env: &mut WasiEnv) -> TaskWasmAcceptedExecutionGuard { ++ let accepted = self.into_accepted_callback(); ++ env.accept_task_wasm_execution(&accepted); ++ accepted ++ } ++} ++ ++impl Drop for TaskWasmExecutionGuard { ++ fn drop(&mut self) { ++ let Some(lease) = self.lease.take() else { ++ return; ++ }; ++ if self.thread.try_join().is_none() { ++ self.thread ++ .set_status_finished(Err(crate::RuntimeError::new( ++ "TaskWasm execution canceled before callback acceptance", ++ ) ++ .into())); ++ } ++ // Exact-thread terminal status must happen-before process quiescence. ++ drop(lease); ++ } ++} ++ ++/// Fail-closed ownership after a `TaskWasm` callback has been admitted. ++/// ++/// Arbitrary callback code may panic or abandon its properties. Keeping the ++/// exact thread beside the process lease makes that path safe by construction: ++/// dropping an armed guard publishes terminal thread status before releasing ++/// process quiescence. A deep-sleep continuation can disarm it only through ++/// [`Self::admit_successor`], which performs successor admission itself and ++/// releases the predecessor lease only after admission returns successfully. ++#[derive(Debug)] ++#[must_use = "dropping accepted TaskWasm execution terminalizes its thread"] ++pub struct TaskWasmAcceptedExecutionGuard { ++ thread: WasiThread, ++ lease: Option, ++ owner_identity: Arc<()>, ++} ++ ++impl TaskWasmAcceptedExecutionGuard { ++ pub(crate) fn thread(&self) -> &WasiThread { ++ &self.thread ++ } ++ ++ pub(crate) fn owner_identity(&self) -> &Arc<()> { ++ &self.owner_identity ++ } ++ ++ fn admit_successor(mut self, admission: impl FnOnce() -> Result) -> Result { ++ let accepted = admission()?; ++ // `admission` constructed and transferred the successor TaskWasm, ++ // whose pending guard acquired its lease in TaskWasm::new. Taking the ++ // predecessor lease only after Ok makes successor-before-predecessor ++ // ordering structural, including unwinding inside `admission`. ++ let lease = self ++ .lease ++ .take() ++ .expect("accepted TaskWasm execution handed off twice"); ++ drop(lease); ++ Ok(accepted) ++ } ++} ++ ++impl Drop for TaskWasmAcceptedExecutionGuard { ++ fn drop(&mut self) { ++ let Some(lease) = self.lease.take() else { ++ return; ++ }; ++ if self.thread.try_join().is_none() { ++ self.thread ++ .set_status_finished(Err(crate::RuntimeError::new( ++ "TaskWasm callback ended without terminal status or an accepted successor", ++ ) ++ .into())); ++ } ++ // Exact-thread terminal status must happen-before process quiescence. ++ drop(lease); ++ } ++} ++ + /// The properties passed to the task + #[derive(derive_more::Debug)] + pub struct TaskWasmRunProperties { ++ /// Accepted execution ownership. It is deliberately the first field so ++ /// abandoning the complete properties value terminalizes before dropping ++ /// the environment's owning handles. Callbacks that move it out retain ++ /// the same fail-closed guarantee through the guard's own destructor. ++ #[debug(ignore)] ++ pub execution_guard: Option, + pub ctx: WasiFunctionEnv, + pub store: Store, + /// The result of the asynchronous trigger serialized into bytes using the bincode serializer +@@ -94,6 +242,15 @@ pub type TaskWasmRecycle = dyn FnOnce(TaskWasmRecycleProperties) + Send + 'stati + + /// Represents a WASM task that will be executed on a dedicated thread + pub struct TaskWasm<'a> { ++ /// Held fail-closed from construction through every pre-callback stage. ++ /// This is deliberately the first field so abandoning the whole task ++ /// terminalizes before its environment can drop owning thread handles. ++ /// Task managers disarm it only from inside the accepted run callback. ++ pub execution_lease: ++ Result, ++ /// Immutable proof that child publication committed before construction; ++ /// Wasmer instantiation itself may execute guest start functions. ++ pub(crate) start_gate: WasiProcessStartGate, + pub run: Box, + pub recycle: Option>, + pub env: WasiEnv, +@@ -103,17 +260,36 @@ pub struct TaskWasm<'a> { + pub trigger: Option>, + pub update_layout: bool, + pub call_initialize: bool, ++ pub preinitialized_memory_image: Option, + pub pre_run: Option>, + } + + impl<'a> TaskWasm<'a> { + pub fn new( + run: Box, +- env: WasiEnv, ++ mut env: WasiEnv, + module: Module, + update_layout: bool, + call_initialize: bool, + ) -> Self { ++ let execution_lease = if env.process.guest_start_is_committed() { ++ env.process ++ .acquire_execution_lease() ++ .map(|lease| TaskWasmExecutionGuard::new(env.thread.clone(), lease)) ++ } else { ++ Err( ++ crate::os::task::control_plane::ControlPlaneError::ProcessNotPublished { ++ pid: env.pid().raw(), ++ }, ++ ) ++ }; ++ if let Ok(execution) = &execution_lease { ++ // Bind structurally in the constructor: every successfully ++ // constructed TaskWasm is ready for guest imports that may run ++ // during instantiation, independent of task-manager behavior. ++ execution.bind_pending_owner(&mut env); ++ } ++ let start_gate = env.process.start_gate(); + let shared_memory = module.imports().memories().next().map(|a| *a.ty()); + Self { + run, +@@ -127,11 +303,25 @@ impl<'a> TaskWasm<'a> { + trigger: None, + update_layout, + call_initialize, ++ preinitialized_memory_image: None, + recycle: None, + pre_run: None, ++ execution_lease, ++ start_gate, + } + } + ++ /// Must be called by task-manager implementations before allocating or ++ /// instantiating guest state. Keeping lifecycle failure inside TaskWasm ++ /// preserves the long-standing infallible constructor while making task ++ /// acceptance fail closed. ++ pub fn validate_lifecycle(&self) -> Result<(), WasiThreadError> { ++ self.execution_lease ++ .as_ref() ++ .map(|_| ()) ++ .map_err(|err| WasiThreadError::ProcessLifecycle(err.clone())) ++ } ++ + pub fn with_memory(mut self, spawn_type: SpawnType<'a>) -> Self { + self.spawn_type = spawn_type; + self +@@ -163,6 +353,14 @@ impl<'a> TaskWasm<'a> { + self.pre_run.replace(pre_run); + self + } ++ ++ pub fn with_preinitialized_memory_image( ++ mut self, ++ image: PreinitializedMemoryImageMode, ++ ) -> Self { ++ self.preinitialized_memory_image = Some(image); ++ self ++ } + } + + /// A task executor backed by a thread pool. +@@ -199,7 +397,7 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { + // browser otherwise creation will fail. + let _ = ty.maximum.get_or_insert(wasmer_types::Pages::max_value()); + +- let mem = Memory::new(&mut store, ty).map_err(|err| { ++ let mem = { Memory::new(&mut store, ty) }.map_err(|err| { + tracing::error!( + error = &err as &dyn std::error::Error, + memory_type=?ty, +@@ -210,7 +408,7 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { + Ok(Some(mem)) + } + SpawnType::ShareMemory(mem, old_store) => { +- let mem = mem.share_in_store(&old_store, store).map_err(|err| { ++ let mem = { mem.share_in_store(&old_store, store) }.map_err(|err| { + tracing::warn!( + error = &err as &dyn std::error::Error, + "could not clone memory", +@@ -220,7 +418,7 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { + Ok(Some(mem)) + } + SpawnType::CopyMemory(mem, old_store) => { +- let mem = mem.copy_to_store(&old_store, store).map_err(|err| { ++ let mem = { mem.copy_to_store(&old_store, store) }.map_err(|err| { + tracing::warn!( + error = &err as &dyn std::error::Error, + "could not copy memory", +@@ -259,6 +457,12 @@ pub trait VirtualTaskManager: std::fmt::Debug + Send + Sync + 'static { + /// + /// This is primarily used inside the context of a syscall and allows + /// the transfer of things like [`wasmer::Module`] across threads. ++ /// Implementations must call [`TaskWasm::validate_lifecycle`] before any ++ /// guest memory allocation, instantiation, initialization, or execution. ++ /// From inside the accepted worker closure, after all other fallible setup, ++ /// they must convert the pending guard with ++ /// [`TaskWasmExecutionGuard::accept_callback`] against the callback's ++ /// concrete [`WasiEnv`] immediately before invoking [`TaskWasmRun`]. + fn task_wasm(&self, task: TaskWasm) -> Result<(), WasiThreadError>; + + /// Run a blocking operation on the thread pool. +@@ -361,6 +565,7 @@ impl dyn VirtualTaskManager { + ctx: WasiFunctionEnv, + mut store: Store, + trigger: Pin>, ++ current_execution: Option, + ) -> Result<(), WasiThreadError> { + // This poller will process any signals when the main working function is idle + struct AsyncifyPollerOwned { +@@ -384,60 +589,70 @@ impl dyn VirtualTaskManager { + } + } + +- let snapshot = capture_store_snapshot(&mut store.as_store_mut()); +- let env = ctx.data(&store); +- let env_inner = env.inner(); +- let handles = env_inner +- .static_module_instance_handles() +- .ok_or(WasiThreadError::Unsupported)?; +- let module = handles.module_clone(); +- let memory = handles.memory_clone(); ++ let snapshot = capture_store_snapshot(&mut store.as_store_mut()) ++ .map_err(WasiThreadError::StoreSnapshotCaptureFailed)?; ++ let env = ctx.data_mut(&mut store); ++ let (module, memory) = { ++ let env_inner = env.inner(); ++ let handles = env_inner ++ .static_module_instance_handles() ++ .ok_or(WasiThreadError::Unsupported)?; ++ (handles.module_clone(), handles.memory_clone()) ++ }; + let thread = env.thread.clone(); +- let env = env.clone(); ++ let observer_env = env.clone(); ++ let continuation_env = env.take_for_same_thread_continuation(); + + let thread_inner = thread.clone(); +- self.task_wasm( +- TaskWasm::new( +- Box::new(move |props| { +- let result = props +- .trigger_result +- .expect("If there is no result then its likely the trigger did not run"); +- let result = match result { +- Ok(r) => r, +- Err(exit_code) => { +- thread.set_status_finished(Ok(exit_code)); +- return; +- } +- }; +- task(props.ctx, props.store, result) +- }), +- env.clone(), +- module, +- false, +- false, ++ let admit_successor = || { ++ self.task_wasm( ++ TaskWasm::new( ++ Box::new(move |mut props| { ++ let result = props.trigger_result.expect( ++ "If there is no result then its likely the trigger did not run", ++ ); ++ let result = match result { ++ Ok(r) => r, ++ Err(exit_code) => { ++ thread.set_status_finished(Ok(exit_code)); ++ return; ++ } ++ }; ++ task(props.ctx, props.store, result, props.execution_guard.take()) ++ }), ++ continuation_env, ++ module, ++ false, ++ false, ++ ) ++ .with_memory(SpawnType::ShareMemory(memory, store.as_store_ref())) ++ .with_globals(snapshot) ++ .with_trigger(Box::new(move || { ++ Box::pin(async move { ++ let mut poller = AsyncifyPollerOwned { ++ thread: thread_inner, ++ trigger, ++ }; ++ let res = Pin::new(&mut poller).await; ++ let res = match res { ++ Ok(res) => res, ++ Err(exit_code) => { ++ observer_env.thread.set_status_finished(Ok(exit_code)); ++ return Err(exit_code); ++ } ++ }; ++ ++ tracing::trace!("deep sleep woken - res.len={}", res.len()); ++ Ok(res) ++ }) ++ })), + ) +- .with_memory(SpawnType::ShareMemory(memory, store.as_store_ref())) +- .with_globals(snapshot) +- .with_trigger(Box::new(move || { +- Box::pin(async move { +- let mut poller = AsyncifyPollerOwned { +- thread: thread_inner, +- trigger, +- }; +- let res = Pin::new(&mut poller).await; +- let res = match res { +- Ok(res) => res, +- Err(exit_code) => { +- env.thread.set_status_finished(Ok(exit_code)); +- return Err(exit_code); +- } +- }; +- +- tracing::trace!("deep sleep woken - res.len={}", res.len()); +- Ok(res) +- }) +- })), +- ) ++ }; ++ ++ match current_execution { ++ Some(execution) => execution.admit_successor(admit_successor), ++ None => admit_successor(), ++ } + } + } + +@@ -504,3 +719,156 @@ where + Box::new(receiver.map_err(|e| Box::new(e).into())) + } + } ++ ++#[cfg(test)] ++mod lifecycle_tests { ++ use std::{sync::mpsc, thread, time::Duration}; ++ ++ use wasmer_types::ModuleHash; ++ use wasmer_wasix_types::wasi::Errno; ++ ++ use crate::{WasiControlPlane, os::task::thread::WasiMemoryLayout}; ++ ++ use super::{TaskWasmAcceptedExecutionGuard, TaskWasmExecutionGuard}; ++ ++ fn pending_execution() -> ( ++ crate::WasiProcess, ++ crate::WasiThreadHandle, ++ TaskWasmExecutionGuard, ++ ) { ++ let plane = WasiControlPlane::default(); ++ let (process, handle) = plane ++ .new_process_with_main_thread(ModuleHash::random(), WasiMemoryLayout::default()) ++ .unwrap(); ++ let lease = process.acquire_execution_lease().unwrap(); ++ let guard = TaskWasmExecutionGuard::new(handle.as_thread(), lease); ++ (process, handle, guard) ++ } ++ ++ fn accepted_execution() -> ( ++ crate::WasiProcess, ++ crate::WasiThreadHandle, ++ TaskWasmAcceptedExecutionGuard, ++ ) { ++ let (process, handle, pending) = pending_execution(); ++ (process, handle, pending.into_accepted_callback()) ++ } ++ ++ #[test] ++ fn canceled_task_wasm_terminalizes_exact_thread_before_quiescence() { ++ let (process, handle, guard) = pending_execution(); ++ let observer_process = process.clone(); ++ let observer_thread = handle.as_thread(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = thread::spawn(move || { ++ observer_process.wait_for_execution_quiescence_blocking(); ++ observed_tx.send(observer_thread.try_join()).unwrap(); ++ }); ++ ++ drop(guard); ++ ++ let status = observed_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .unwrap() ++ .expect("thread was not terminal when process became quiescent"); ++ assert!(status.is_err()); ++ assert_eq!(process.lock().execution_leases, 0); ++ observer.join().unwrap(); ++ drop(handle); ++ } ++ ++ #[test] ++ fn closed_worker_queue_drops_pending_execution_fail_closed() { ++ let (process, handle, guard) = pending_execution(); ++ let queued_callback: Box = Box::new(move || { ++ let _accepted = guard.into_accepted_callback(); ++ }); ++ ++ // A closed queue owns and drops its callback without invoking it. ++ drop(queued_callback); ++ ++ assert!(handle.try_join().is_some_and(|status| status.is_err())); ++ assert_eq!(process.lock().execution_leases, 0); ++ drop(handle); ++ } ++ ++ #[test] ++ fn accepted_callback_conversion_is_the_only_successful_disarm() { ++ let (process, handle, guard) = pending_execution(); ++ let lease = guard.into_accepted_callback(); ++ ++ assert!(handle.try_join().is_none()); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ handle.set_status_finished(Ok(Errno::Success.into())); ++ drop(lease); ++ assert_eq!(process.lock().execution_leases, 0); ++ assert_eq!(handle.try_join().unwrap().unwrap(), Errno::Success.into()); ++ } ++ ++ #[test] ++ fn accepted_callback_panic_terminalizes_before_quiescence() { ++ let (process, handle, accepted) = accepted_execution(); ++ let observer_process = process.clone(); ++ let observer_thread = handle.as_thread(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = thread::spawn(move || { ++ observer_process.wait_for_execution_quiescence_blocking(); ++ let _ = observed_tx.send(observer_thread.try_join()); ++ }); ++ ++ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ let _ = accepted.admit_successor(|| -> Result<(), ()> { ++ panic!("synthetic accepted-callback panic") ++ }); ++ })); ++ ++ assert!(panic.is_err()); ++ let status = observed_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .unwrap() ++ .expect("accepted callback released quiescence before terminal status"); ++ assert!(status.is_err()); ++ assert_eq!(process.lock().execution_leases, 0); ++ observer.join().unwrap(); ++ drop(handle); ++ } ++ ++ #[test] ++ fn rejected_successor_keeps_predecessor_until_fail_closed_drop() { ++ let (process, handle, accepted) = accepted_execution(); ++ ++ let result = accepted.admit_successor(|| { ++ assert_eq!(process.lock().execution_leases, 1); ++ Err::<(), _>("synthetic admission rejection") ++ }); ++ ++ assert_eq!(result.unwrap_err(), "synthetic admission rejection"); ++ assert!(handle.try_join().is_some_and(|status| status.is_err())); ++ assert_eq!(process.lock().execution_leases, 0); ++ drop(handle); ++ } ++ ++ #[test] ++ fn deep_sleep_handoff_keeps_thread_live_until_successor_guard_finishes() { ++ let (process, handle, accepted) = accepted_execution(); ++ let mut successor = None; ++ ++ accepted ++ .admit_successor(|| { ++ assert_eq!(process.lock().execution_leases, 1); ++ let lease = process.acquire_execution_lease().unwrap(); ++ successor = Some(TaskWasmExecutionGuard::new(handle.as_thread(), lease)); ++ assert_eq!(process.lock().execution_leases, 2); ++ Ok::<_, ()>(()) ++ }) ++ .unwrap(); ++ ++ assert!(handle.try_join().is_none()); ++ assert_eq!(process.lock().execution_leases, 1); ++ drop(successor); ++ assert!(handle.try_join().is_some_and(|status| status.is_err())); ++ assert_eq!(process.lock().execution_leases, 0); ++ drop(handle); ++ } ++} +diff --git a/lib/wasix/src/runtime/task_manager/tokio.rs b/lib/wasix/src/runtime/task_manager/tokio.rs +index e390436..36c1785 100644 +--- a/lib/wasix/src/runtime/task_manager/tokio.rs ++++ b/lib/wasix/src/runtime/task_manager/tokio.rs +@@ -1,5 +1,5 @@ + use std::sync::Mutex; +-use std::{num::NonZeroUsize, pin::Pin, sync::Arc, time::Duration}; ++use std::{pin::Pin, sync::Arc, time::Duration}; + + use futures::{Future, future::BoxFuture}; + use tokio::runtime::{Handle, Runtime}; +@@ -74,6 +74,33 @@ impl std::fmt::Debug for ThreadPool { + pub struct TokioTaskManager { + rt: RuntimeOrHandle, + pool: Arc, ++ config: TokioTaskManagerConfig, ++} ++ ++/// Host worker policy for blocking WASIX tasks. ++/// ++/// A small persistent core avoids retaining one host thread and allocator arena ++/// for every guest task ever observed. The pool can still grow to `max_threads` ++/// while guests are active, and excess workers retire after `idle_timeout`. ++#[derive(Clone, Copy, Debug, PartialEq, Eq)] ++pub struct TokioTaskManagerConfig { ++ pub core_threads: usize, ++ pub max_threads: usize, ++ pub idle_timeout: Duration, ++} ++ ++impl Default for TokioTaskManagerConfig { ++ fn default() -> Self { ++ let concurrency = std::thread::available_parallelism() ++ .map(usize::from) ++ .unwrap_or(1); ++ ++ Self { ++ core_threads: 1, ++ max_threads: 200usize.max(concurrency.saturating_mul(100)), ++ idle_timeout: Duration::from_secs(1), ++ } ++ } + } + + impl TokioTaskManager { +@@ -81,20 +108,33 @@ impl TokioTaskManager { + where + I: Into, + { +- let concurrency = std::thread::available_parallelism() +- .unwrap_or(NonZeroUsize::new(1).unwrap()) +- .get(); +- let max_threads = 200usize.max(concurrency * 100); ++ Self::new_with_config(rt, TokioTaskManagerConfig::default()) ++ } ++ ++ pub fn new_with_config(rt: I, config: TokioTaskManagerConfig) -> Self ++ where ++ I: Into, ++ { ++ assert!( ++ config.core_threads > 0, ++ "core_threads must be greater than 0" ++ ); ++ assert!( ++ config.max_threads >= config.core_threads, ++ "max_threads must be greater than or equal to core_threads" ++ ); + + Self { + rt: rt.into(), + pool: Arc::new(ThreadPool { + inner: rusty_pool::Builder::new() + .name("TokioTaskManager Thread Pool".to_string()) +- .core_size(max_threads) +- .max_size(max_threads) ++ .core_size(config.core_threads) ++ .max_size(config.max_threads) ++ .keep_alive(config.idle_timeout) + .build(), + }), ++ config, + } + } + +@@ -105,6 +145,10 @@ impl TokioTaskManager { + pub fn pool_handle(&self) -> Arc { + self.pool.clone() + } ++ ++ pub fn config(&self) -> TokioTaskManagerConfig { ++ self.config ++ } + } + + impl Default for TokioTaskManager { +@@ -140,10 +184,19 @@ impl VirtualTaskManager for TokioTaskManager { + + /// See [`VirtualTaskManager::task_wasm`]. + fn task_wasm(&self, task: TaskWasm) -> Result<(), WasiThreadError> { ++ // This must precede memory creation and instance construction: Wasmer ++ // may execute a module start function while instantiating. ++ task.validate_lifecycle()?; + let run = task.run; + let recycle = task.recycle; + let env = task.env; ++ let lifecycle_thread = env.thread.clone(); + let pre_run = task.pre_run; ++ let preinitialized_memory_image = task.preinitialized_memory_image; ++ let execution_guard = task ++ .execution_lease ++ .expect("TaskWasm lifecycle was validated above"); ++ let start_gate = task.start_gate; + + let make_memory: SpawnMemoryTypeOrStore = match &task.spawn_type { + SpawnType::CreateMemory | SpawnType::NewLinkerInstanceGroup(..) => { +@@ -152,7 +205,15 @@ impl VirtualTaskManager for TokioTaskManager { + SpawnType::CreateMemoryOfType(t) => SpawnMemoryTypeOrStore::Type(*t), + SpawnType::ShareMemory(_, _) | SpawnType::CopyMemory(_, _) => { + let mut store = env.runtime().new_store(); +- let memory = self.build_memory(&mut store.as_store_mut(), &task.spawn_type)?; ++ let memory = self ++ .build_memory(&mut store.as_store_mut(), &task.spawn_type) ++ .map_err(|err| { ++ lifecycle_thread.set_status_finished(Err(crate::RuntimeError::new( ++ format!("Wasm task memory construction failed: {err}"), ++ ) ++ .into())); ++ err ++ })?; + SpawnMemoryTypeOrStore::StoreAndMemory(store, memory) + } + }; +@@ -161,9 +222,8 @@ impl VirtualTaskManager for TokioTaskManager { + // See the comment below for why we can't do it there yet. + // + // For now block_in_place at least ensures that we don't block the async runtime +- let ret = tokio::task::block_in_place(move || { +- if let SpawnType::NewLinkerInstanceGroup(linker, func_env, mut store) = task.spawn_type +- { ++ let ret = tokio::task::block_in_place(move || match task.spawn_type { ++ SpawnType::NewLinkerInstanceGroup(linker, func_env, mut store) => { + WasiFunctionEnv::new_with_store( + task.module, + env, +@@ -171,21 +231,37 @@ impl VirtualTaskManager for TokioTaskManager { + make_memory, + task.update_layout, + task.call_initialize, ++ preinitialized_memory_image, + Some((linker, &mut func_env.into_mut(&mut store))), + ) +- } else { +- WasiFunctionEnv::new_with_store( +- task.module, +- env, +- task.globals, +- make_memory, +- task.update_layout, +- task.call_initialize, +- None, +- ) + } ++ _ => WasiFunctionEnv::new_with_store( ++ task.module, ++ env, ++ task.globals, ++ make_memory, ++ task.update_layout, ++ task.call_initialize, ++ preinitialized_memory_image, ++ None, ++ ), + }); + ++ // `new_with_store` can fail after a task has been accepted but before ++ // its run callback (and its normal run guard) exists. Publish terminal ++ // status while the execution lease is still held instead of relying ++ // on environment/handle field drop order. ++ let (mut ctx, mut store) = match ret { ++ Ok(context) => context, ++ Err(err) => { ++ lifecycle_thread.set_status_finished(Err(crate::RuntimeError::new(format!( ++ "Wasm task instantiation failed: {err}" ++ )) ++ .into())); ++ return Err(err); ++ } ++ }; ++ + if let Some(trigger) = task.trigger { + tracing::trace!("spawning task_wasm trigger in async pool"); + // In principle, we'd need to create this in the `pool.execute` function below, that is +@@ -217,11 +293,11 @@ impl VirtualTaskManager for TokioTaskManager { + // pool's one), and let it fail for runtimes that don't support entities created in a + // thread that's not the one in which execution happens in; this until we can clone + // stores. +- let (mut ctx, mut store) = ret?; +- + let mut trigger = trigger(); + let pool = self.pool.clone(); + self.rt.handle().spawn(async move { ++ debug_assert!(start_gate.is_committed()); ++ + // We wait for either the trigger or for a snapshot to take place + let result = loop { + let env = ctx.data(&store); +@@ -264,12 +340,17 @@ impl VirtualTaskManager for TokioTaskManager { + + // Build the task that will go on the callback + pool.execute(move || { +- // Invoke the callback ++ // This is the sole successful disarm boundary: all ++ // fallible setup and queueing stages are behind us, and ++ // the accepted callback now owns terminal/successor state. ++ let execution_guard = ++ execution_guard.accept_callback(ctx.data_mut(&mut store)); + run(TaskWasmRunProperties { + ctx, + store, + trigger_result: Some(result), + recycle, ++ execution_guard: Some(execution_guard), + }); + }); + }); +@@ -281,32 +362,26 @@ impl VirtualTaskManager for TokioTaskManager { + // Run the callback on a dedicated thread + self.pool.execute(move || { + tracing::trace!("task_wasm started in blocking thread"); +- let (mut ctx, mut store) = match ret { +- Ok(x) => { +- sx.send(Ok(())).unwrap(); +- x +- } +- Err(c) => { +- sx.send(Err(c)).unwrap(); +- return; +- } +- }; ++ let _ = sx.send(()); ++ ++ debug_assert!(start_gate.is_committed()); + + if let Some(pre_run) = pre_run { + block_on(pre_run(&mut ctx, &mut store)); + } + +- // Invoke the callback ++ // Disarm only after the queue and pre-run stages completed. ++ let execution_guard = execution_guard.accept_callback(ctx.data_mut(&mut store)); + run(TaskWasmRunProperties { + ctx, + store, + trigger_result: None, + recycle, ++ execution_guard: Some(execution_guard), + }); + }); + +- rx.recv() +- .map_err(|_| WasiThreadError::InvalidWasmContext)??; ++ rx.recv().map_err(|_| WasiThreadError::InvalidWasmContext)?; + } + Ok(()) + } +@@ -361,3 +436,490 @@ impl Drop for SleepNow { + } + } + } ++ ++#[cfg(test)] ++mod tests { ++ use std::sync::atomic::{AtomicBool, Ordering}; ++ use std::sync::{Arc, Barrier, Mutex, mpsc}; ++ use std::time::{Duration, Instant}; ++ ++ use wasmer::{AsStoreRef, Engine, Function, Imports, Memory, MemoryType, Module, Store}; ++ use wasmer_wasix_types::wasi::Errno; ++ ++ use crate::{ ++ PluggableRuntime, WasiControlPlane, WasiEnv, WasiProcess, WasiThreadError, ++ runtime::task_manager::{SpawnType, TaskWasm}, ++ }; ++ ++ use super::{TokioTaskManager, TokioTaskManagerConfig}; ++ ++ fn task_wasm_test_env( ++ runtime: &tokio::runtime::Runtime, ++ engine: &Engine, ++ name: &str, ++ ) -> WasiEnv { ++ let mut wasix_runtime = ++ PluggableRuntime::new(Arc::new(TokioTaskManager::new(runtime.handle().clone()))); ++ wasix_runtime.set_engine(engine.clone()); ++ WasiEnv::builder(name) ++ .runtime(Arc::new(wasix_runtime)) ++ .build() ++ .unwrap() ++ } ++ ++ #[test] ++ fn non_core_workers_retire_after_the_idle_timeout() { ++ const TASKS: usize = 8; ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let config = TokioTaskManagerConfig { ++ core_threads: 1, ++ max_threads: TASKS, ++ idle_timeout: Duration::from_millis(20), ++ }; ++ let manager = TokioTaskManager::new_with_config(runtime.handle().clone(), config); ++ let pool = manager.pool_handle(); ++ let barrier = Arc::new(Barrier::new(TASKS + 1)); ++ ++ for _ in 0..TASKS { ++ let barrier = Arc::clone(&barrier); ++ pool.execute(move || { ++ barrier.wait(); ++ }); ++ } ++ ++ barrier.wait(); ++ pool.join(); ++ assert_eq!(manager.config(), config); ++ assert_eq!(pool.get_current_worker_count(), TASKS); ++ ++ let deadline = Instant::now() + Duration::from_secs(2); ++ while pool.get_current_worker_count() != config.core_threads && Instant::now() < deadline { ++ std::thread::sleep(Duration::from_millis(10)); ++ } ++ ++ assert_eq!(pool.get_current_worker_count(), config.core_threads); ++ } ++ ++ #[test] ++ fn memory_construction_failure_terminalizes_before_releasing_lease() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ let env = task_wasm_test_env(&runtime, &engine, "memory-failure-test"); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let tasks = env.tasks().clone(); ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (memory (export "memory") 1) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let mut old_store = Store::new(engine.clone()); ++ let non_shared = Memory::new(&mut old_store, MemoryType::new(1, Some(1), false)).unwrap(); ++ let run_called = Arc::new(AtomicBool::new(false)); ++ let run_called_inner = Arc::clone(&run_called); ++ ++ let result = tasks.task_wasm( ++ TaskWasm::new( ++ Box::new(move |_| run_called_inner.store(true, Ordering::Release)), ++ env, ++ module, ++ false, ++ false, ++ ) ++ .with_memory(SpawnType::ShareMemory(non_shared, old_store.as_store_ref())), ++ ); ++ ++ assert!(matches!( ++ result, ++ Err(WasiThreadError::MemoryCreateFailed(_)) ++ )); ++ assert!(!run_called.load(Ordering::Acquire)); ++ assert!(thread.try_join().is_some_and(|status| status.is_err())); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn instantiation_failure_terminalizes_before_releasing_lease() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ let env = task_wasm_test_env(&runtime, &engine, "instantiation-failure-test"); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let tasks = env.tasks().clone(); ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (import "wasi_snapshot_preview1" "proc_exit" (func $proc_exit (param i32))) ++ (import "missing" "function" (func $missing)) ++ (memory (export "memory") 1) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let run_called = Arc::new(AtomicBool::new(false)); ++ let run_called_inner = Arc::clone(&run_called); ++ ++ let result = tasks.task_wasm(TaskWasm::new( ++ Box::new(move |_| run_called_inner.store(true, Ordering::Release)), ++ env, ++ module, ++ false, ++ false, ++ )); ++ ++ assert!(result.is_err(), "missing import unexpectedly instantiated"); ++ assert!(!run_called.load(Ordering::Acquire)); ++ assert!(thread.try_join().is_some(), "thread remained nonterminal"); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn panicking_custom_task_wasm_callback_terminalizes_before_quiescence() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ let env = task_wasm_test_env(&runtime, &engine, "callback-panic-test"); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let tasks = env.tasks().clone(); ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (memory (export "memory") 1) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let (entered_tx, entered_rx) = mpsc::channel(); ++ ++ tasks ++ .task_wasm(TaskWasm::new( ++ Box::new(move |_props| { ++ let _ = entered_tx.send(()); ++ panic!("synthetic custom TaskWasmRun panic"); ++ }), ++ env, ++ module, ++ false, ++ false, ++ )) ++ .unwrap(); ++ entered_rx.recv_timeout(Duration::from_secs(2)).unwrap(); ++ ++ let observer_process = process.clone(); ++ let observer_thread = thread.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = std::thread::spawn(move || { ++ observer_process.wait_for_execution_quiescence_blocking(); ++ let _ = observed_tx.send(observer_thread.try_join()); ++ }); ++ let status = observed_rx ++ .recv_timeout(Duration::from_secs(2)) ++ .unwrap() ++ .expect("custom callback released quiescence before terminal status"); ++ ++ assert!(status.is_err()); ++ assert_eq!(process.lock().execution_leases, 0); ++ observer.join().unwrap(); ++ } ++ ++ #[test] ++ fn accepted_restored_process_adopts_its_single_deferred_parent_guard() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ let mut env = task_wasm_test_env(&runtime, &engine, "restored-parent-handoff-test"); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let tasks = env.tasks().clone(); ++ let prior_owner = Arc::new(()); ++ let deferred = process ++ .acquire_supplemental_execution_guard(thread.clone(), Arc::downgrade(&prior_owner)) ++ .unwrap(); ++ deferred.arm_fail_closed(); ++ env.deferred_parent_execution = Some(deferred); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (memory (export "memory") 1) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let callback_process = process.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ tasks ++ .task_wasm(TaskWasm::new( ++ Box::new(move |props| { ++ let env = props.ctx.data(&props.store); ++ observed_tx ++ .send(( ++ env.deferred_parent_execution.is_none(), ++ callback_process.lock().execution_leases, ++ )) ++ .unwrap(); ++ env.thread.set_status_finished(Ok(Errno::Success.into())); ++ }), ++ env, ++ module, ++ false, ++ false, ++ )) ++ .unwrap(); ++ ++ assert_eq!( ++ observed_rx.recv_timeout(Duration::from_secs(2)).unwrap(), ++ (true, 1) ++ ); ++ process.wait_for_execution_quiescence_blocking(); ++ assert_eq!(process.lock().execution_leases, 0); ++ assert!(thread.try_join().is_some()); ++ } ++ ++ #[test] ++ fn task_wasm_constructor_binds_owner_before_manager_instantiation() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ let env = task_wasm_test_env(&runtime, &engine, "constructor-owner-bind-test"); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (memory (export "memory") 1) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let mut task = TaskWasm::new(Box::new(|_| unreachable!()), env, module, false, false); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ let supplemental = task.env.acquire_parent_execution_guard().unwrap(); ++ assert_eq!(process.lock().execution_leases, 2); ++ task.env.restore_parent_execution_guard(supplemental); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ drop(task); ++ process.wait_for_execution_quiescence_blocking(); ++ assert!(thread.try_join().is_some_and(|status| status.is_err())); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn pending_task_drop_with_deferred_owner_is_fail_closed_and_bounded() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ let mut env = task_wasm_test_env(&runtime, &engine, "deferred-queue-drop-test"); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let predecessor = Arc::new(()); ++ env.bind_pending_task_wasm_execution(&thread, &predecessor); ++ let deferred = env.acquire_parent_execution_guard().unwrap(); ++ deferred.arm_fail_closed(); ++ env.deferred_parent_execution = Some(deferred); ++ drop(predecessor); ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (memory (export "memory") 1) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let task = TaskWasm::new(Box::new(|_| unreachable!()), env, module, false, false); ++ assert_eq!(process.lock().execution_leases, 2); ++ ++ drop(task); ++ ++ process.wait_for_execution_quiescence_blocking(); ++ assert!(thread.try_join().is_some()); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn foreign_pending_token_cannot_accept_same_guest_thread() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ let env = task_wasm_test_env(&runtime, &engine, "foreign-owner-bind-test"); ++ let process = env.process.clone(); ++ let thread = env.thread.clone(); ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (memory (export "memory") 1) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ let task_a = TaskWasm::new( ++ Box::new(|_| unreachable!()), ++ env.clone(), ++ module.clone(), ++ false, ++ false, ++ ); ++ let mut task_b = TaskWasm::new(Box::new(|_| unreachable!()), env, module, false, false); ++ assert_eq!(process.lock().execution_leases, 2); ++ let pending_a = task_a.execution_lease.unwrap(); ++ ++ let panic = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { ++ let _ = pending_a.accept_callback(&mut task_b.env); ++ })); ++ assert!(panic.is_err()); ++ assert!(thread.try_join().is_some_and(|status| status.is_err())); ++ assert_eq!(process.lock().execution_leases, 1); ++ ++ let observer_process = process.clone(); ++ let observer_thread = thread.clone(); ++ let (observed_tx, observed_rx) = mpsc::channel(); ++ let observer = std::thread::spawn(move || { ++ observer_process.wait_for_execution_quiescence_blocking(); ++ observed_tx.send(observer_thread.try_join()).unwrap(); ++ }); ++ assert!(observed_rx.recv_timeout(Duration::from_millis(20)).is_err()); ++ drop(task_b); ++ assert!( ++ observed_rx ++ .recv_timeout(Duration::from_secs(1)) ++ .unwrap() ++ .is_some_and(|status| status.is_err()) ++ ); ++ observer.join().unwrap(); ++ assert_eq!(process.lock().execution_leases, 0); ++ } ++ ++ #[test] ++ fn module_start_observes_child_parent_after_exact_publication() { ++ let runtime = tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ let _runtime_guard = runtime.enter(); ++ let engine = Engine::default(); ++ ++ type ExpectedPublication = (WasiProcess, WasiProcess, WasiControlPlane); ++ let expected_publication = Arc::new(Mutex::new(None::)); ++ let (start_observed_tx, start_observed_rx) = mpsc::channel(); ++ let mut wasix_runtime = ++ PluggableRuntime::new(Arc::new(TokioTaskManager::new(runtime.handle().clone()))); ++ wasix_runtime.set_engine(engine.clone()); ++ let expected_for_import = Arc::clone(&expected_publication); ++ wasix_runtime.with_additional_imports(move |_module, store| { ++ let expected_for_start = Arc::clone(&expected_for_import); ++ let start_observed_tx = start_observed_tx.clone(); ++ let observe_publication = Function::new_typed(store, move || { ++ let expected = expected_for_start.lock().unwrap(); ++ let (parent, child, plane) = expected ++ .as_ref() ++ .expect("module start ran before the test installed its expectation"); ++ let registered = plane ++ .get_process(child.pid()) ++ .is_some_and(|process| process.same_identity(child)); ++ let adopted = parent ++ .lock() ++ .children ++ .iter() ++ .any(|process| process.same_identity(child)); ++ start_observed_tx ++ .send((child.ppid().raw(), registered, adopted)) ++ .unwrap(); ++ }); ++ let mut imports = Imports::new(); ++ imports.define("test", "observe_publication", observe_publication); ++ Ok(imports) ++ }); ++ let parent_env = WasiEnv::builder("publication-before-start-test") ++ .runtime(Arc::new(wasix_runtime)) ++ .build() ++ .unwrap(); ++ let parent = parent_env.process.clone(); ++ let plane = parent_env.control_plane.clone(); ++ let tasks = parent_env.tasks().clone(); ++ let (child_env, child_handle, mut registration) = parent_env.fork_guarded().unwrap(); ++ let child = child_env.process.clone(); ++ ++ let module = Module::new( ++ &engine, ++ br#"(module ++ (import "test" "observe_publication" (func $observe_publication)) ++ (memory (export "memory") 1) ++ (func $constructor call $observe_publication) ++ (start $constructor) ++ (func (export "_start")))"#, ++ ) ++ .unwrap(); ++ ++ assert!(matches!( ++ TaskWasm::new( ++ Box::new(|_| unreachable!()), ++ child_env.clone(), ++ module.clone(), ++ false, ++ false ++ ) ++ .validate_lifecycle(), ++ Err(crate::WasiThreadError::ProcessLifecycle( ++ crate::os::task::control_plane::ControlPlaneError::ProcessNotPublished { .. } ++ )) ++ )); ++ ++ registration.commit_child().unwrap(); ++ *expected_publication.lock().unwrap() = Some((parent.clone(), child, plane)); ++ let (run_tx, run_rx) = mpsc::channel(); ++ let expected_parent_pid = parent.pid().raw(); ++ let run = move |props: crate::runtime::task_manager::TaskWasmRunProperties| { ++ props ++ .ctx ++ .data(&props.store) ++ .thread ++ .set_status_finished(Ok(Errno::Success.into())); ++ run_tx.send(()).unwrap(); ++ }; ++ ++ tasks ++ .task_wasm(TaskWasm::new( ++ Box::new(run), ++ child_env, ++ module, ++ false, ++ false, ++ )) ++ .unwrap(); ++ registration.complete_child_launch(&tasks); ++ let (observed_parent, registered, adopted) = start_observed_rx ++ .recv_timeout(Duration::from_secs(2)) ++ .unwrap(); ++ assert_eq!(observed_parent, expected_parent_pid); ++ assert!(registered); ++ assert!(adopted); ++ run_rx.recv_timeout(Duration::from_secs(2)).unwrap(); ++ drop(child_handle); ++ } ++} +diff --git a/lib/wasix/src/state/builder.rs b/lib/wasix/src/state/builder.rs +index b66c0e1..9dc9baf 100644 +--- a/lib/wasix/src/state/builder.rs ++++ b/lib/wasix/src/state/builder.rs +@@ -23,7 +23,7 @@ use crate::{ + fs::{WasiFs, WasiFsRoot, WasiInodes}, + os::command::VirtualCommand, + os::task::control_plane::{ControlPlaneConfig, ControlPlaneError, WasiControlPlane}, +- state::WasiState, ++ state::{PreinitializedMemoryImageHandle, WasiState}, + syscalls::types::{__WASI_STDERR_FILENO, __WASI_STDIN_FILENO, __WASI_STDOUT_FILENO}, + }; + use wasmer_types::ModuleHash; +@@ -79,6 +79,11 @@ pub struct WasiEnvBuilder { + + pub(super) module_hash: Option, + ++ /// Exact immutable guest executable aliases. Modules are resolved lazily ++ /// from the runtime's authoritative cache by hash. ++ pub(super) sealed_modules: ++ HashMap)>, ++ + /// List of host commands to map into the WASI instance. + pub(super) map_commands: HashMap, + /// Indicates if internal builtin commands should be disabled. +@@ -127,6 +132,7 @@ impl std::fmt::Debug for WasiEnvBuilder { + .field("builtin_commands_count", &self.builtin_commands.len()) + .field("engine_override_exists", &self.engine.is_some()) + .field("runtime_override_exists", &self.runtime.is_some()) ++ .field("sealed_module_count", &self.sealed_modules.len()) + .finish() + } + } +@@ -394,6 +400,23 @@ impl WasiEnvBuilder { + self + } + ++ /// Registers an immutable guest executable for this process tree. ++ /// ++ /// Once any path is registered, exact matches are resolved by hash and all ++ /// other executable paths are denied before guest-filesystem lookup. ++ pub fn add_sealed_module( ++ &mut self, ++ guest_path: impl Into, ++ module_hash: ModuleHash, ++ preinitialized_memory_image: Option, ++ ) -> &mut Self { ++ self.sealed_modules.insert( ++ guest_path.into(), ++ (module_hash, preinitialized_memory_image), ++ ); ++ self ++ } ++ + /// Adds a container this module inherits from. + /// + /// This will make all of the container's files and commands available to the +@@ -1004,6 +1027,10 @@ impl WasiEnvBuilder { + clock_offset: Default::default(), + envs: std::sync::Mutex::new(conv_env_vars(self.envs)), + signals: std::sync::Mutex::new(self.signals.iter().map(|s| (s.sig, s.disp)).collect()), ++ shared_futex_registries: Default::default(), ++ shared_memory_mappings: Default::default(), ++ shared_memory_exec_phase: Default::default(), ++ shared_memory_exec_reservations: Default::default(), + }; + + let runtime = self.runtime.unwrap_or_else(|| { +@@ -1046,7 +1073,8 @@ impl WasiEnvBuilder { + let disable_default_builtins = self.disable_default_builtins; + let builtin_commands = self.builtin_commands; + +- let mut bin_factory = BinFactory::new(runtime.clone()); ++ let mut bin_factory = ++ BinFactory::new_with_sealed_modules(runtime.clone(), self.sealed_modules); + if disable_default_builtins { + bin_factory.clear_builtin_commands(); + } +@@ -1148,7 +1176,7 @@ impl WasiEnvBuilder { + .map(|ty| wasmer::Memory::new(store, ty)) + .transpose() + .map_err(WasiThreadError::MemoryCreateFailed)?; +- Ok(env.instantiate(module, store, memory, true, call_init, None)?) ++ Ok(env.instantiate(module, store, memory, true, call_init, None, false, None)?) + } + } + +diff --git a/lib/wasix/src/state/env.rs b/lib/wasix/src/state/env.rs +index 783291f..915aa13 100644 +--- a/lib/wasix/src/state/env.rs ++++ b/lib/wasix/src/state/env.rs +@@ -1,5 +1,6 @@ + #[cfg(feature = "journal")] + use crate::journal::{DynJournal, JournalEffector, SnapshotTrigger}; ++use crate::runtime::task_manager::TaskWasmAcceptedExecutionGuard; + use crate::{ + Runtime, VirtualTaskManager, WasiControlPlane, WasiEnvBuilder, WasiError, WasiFunctionEnv, + WasiResult, WasiRuntimeError, WasiStateCreationError, WasiThreadError, WasiVFork, +@@ -8,8 +9,8 @@ use crate::{ + fs::{WasiFsRoot, WasiInodes}, + import_object_for_all_wasi_versions, + os::task::{ +- control_plane::ControlPlaneError, +- process::{WasiProcess, WasiProcessId}, ++ control_plane::{ControlPlaneError, WasiProcessRegistrationGuard}, ++ process::{WasiProcess, WasiProcessExecutionGuard, WasiProcessId}, + thread::{WasiMemoryLayout, WasiThread, WasiThreadHandle, WasiThreadId}, + }, + syscalls::platform_clock_time_get, +@@ -21,7 +22,7 @@ use std::{ + ops::Deref, + path::{Path, PathBuf}, + str, +- sync::Arc, ++ sync::{Arc, Weak}, + time::Duration, + }; + use virtual_fs::{FileSystem, FsError, VirtualFile}; +@@ -41,7 +42,10 @@ use wasmer_wasix_types::{ + use webc::metadata::annotations::Wasi; + + pub use super::handles::*; +-use super::{Linker, WasiState, context_switching::ContextSwitchingEnvironment, conv_env_vars}; ++use super::{ ++ Linker, PreinitializedMemoryImageMode, WasiState, ++ context_switching::ContextSwitchingEnvironment, conv_env_vars, ++}; + + async fn write_readonly_buffer_to_fs( + fs: &WasiFsRoot, +@@ -127,6 +131,10 @@ impl WasiEnvInit { + args: std::sync::Mutex::new(self.state.args.lock().unwrap().clone()), + envs: std::sync::Mutex::new(self.state.envs.lock().unwrap().deref().clone()), + signals: std::sync::Mutex::new(self.state.signals.lock().unwrap().deref().clone()), ++ shared_futex_registries: Default::default(), ++ shared_memory_mappings: Default::default(), ++ shared_memory_exec_phase: Default::default(), ++ shared_memory_exec_reservations: Default::default(), + preopen: self.state.preopen.clone(), + }, + runtime: self.runtime.clone(), +@@ -171,6 +179,14 @@ pub struct WasiEnv { + /// List of the handles that are owned by this context + /// (this can be used to ensure that threads own themselves or others) + pub owned_handles: Vec, ++ /// The sole deferred execution owner retained across an in-place vfork ++ /// process switch. Nested vfork is unsupported, so an `Option` encodes the ++ /// real cardinality and makes per-switch accumulation impossible. ++ pub(crate) deferred_parent_execution: Option, ++ /// Weak identity of the accepted TaskWasm that owns the physical module ++ /// instance currently stored in `inner`. It is deliberately not cloned; ++ /// `swap_inner` moves it with that physical instance across vfork. ++ current_task_wasm_owner: Weak<()>, + /// Implementation of the WASI runtime. + pub runtime: Arc, + +@@ -231,6 +247,10 @@ impl Clone for WasiEnv { + bin_factory: self.bin_factory.clone(), + inner: Default::default(), + owned_handles: self.owned_handles.clone(), ++ // Generic clones are non-owning. Only an explicitly declared ++ // same-thread continuation may alias a deferred vfork owner. ++ deferred_parent_execution: None, ++ current_task_wasm_owner: Weak::new(), + runtime: self.runtime.clone(), + capabilities: self.capabilities.clone(), + enable_deep_sleep: self.enable_deep_sleep, +@@ -244,7 +264,136 @@ impl Clone for WasiEnv { + } + } + ++fn retire_finished_process_tree( ++ control_plane: &WasiControlPlane, ++ process: &WasiProcess, ++) -> Result<(), ControlPlaneError> { ++ let descendants = process.begin_epoch_retirement()?; ++ for descendant in &descendants { ++ control_plane.retire_process_epoch(descendant)?; ++ descendant.retire_epoch_task_registrations(); ++ } ++ control_plane.retire_process_epoch(process)?; ++ process.retire_epoch_task_registrations(); ++ for descendant in &descendants { ++ descendant.clear_retired_children(); ++ } ++ process.clear_retired_children(); ++ Ok(()) ++} ++ + impl WasiEnv { ++ /// Binds a pending TaskWasm's unforgeable identity before Wasmer can run ++ /// guest start functions during instantiation. The pending guard remains ++ /// fail-closed; this only establishes which physical execution may request ++ /// an in-place vfork ownership handoff. ++ pub(crate) fn bind_pending_task_wasm_execution( ++ &mut self, ++ thread: &WasiThread, ++ owner_identity: &Arc<()>, ++ ) { ++ assert!( ++ thread.same_identity(&self.thread), ++ "pending TaskWasm thread does not match its environment" ++ ); ++ let owner = Arc::downgrade(owner_identity); ++ assert!( ++ self.current_task_wasm_owner.upgrade().is_none() ++ || Weak::ptr_eq(&self.current_task_wasm_owner, &owner), ++ "environment is already bound to a different TaskWasm" ++ ); ++ self.current_task_wasm_owner = owner; ++ } ++ ++ /// Transfers bounded process-switch ownership to an accepted TaskWasm for ++ /// the exact restored guest thread. The accepted successor already holds ++ /// its own lease, so predecessor release remains successor-before- ++ /// predecessor without consulting process-wide lease counts. ++ pub(crate) fn accept_task_wasm_execution(&mut self, accepted: &TaskWasmAcceptedExecutionGuard) { ++ assert!( ++ accepted.thread().same_identity(&self.thread), ++ "accepted TaskWasm thread does not match its environment" ++ ); ++ let accepted_owner = Arc::downgrade(accepted.owner_identity()); ++ assert!( ++ Weak::ptr_eq(&self.current_task_wasm_owner, &accepted_owner) ++ && self.current_task_wasm_owner.upgrade().is_some(), ++ "accepted TaskWasm does not match the pending owner bound before instantiation" ++ ); ++ let Some(guard) = self.deferred_parent_execution.take() else { ++ return; ++ }; ++ assert!( ++ guard.try_handoff_to_accepted_task_wasm(&self.process, accepted.thread()), ++ "deferred parent execution does not match its accepted successor" ++ ); ++ } ++ ++ pub(crate) fn clear_task_wasm_execution(&mut self) { ++ self.current_task_wasm_owner = Weak::new(); ++ self.deferred_parent_execution.take(); ++ } ++ ++ /// Restores a supplemental owner after switching back to its process. The ++ /// common path hands it directly to the still-accepted TaskWasm. If no ++ /// such successor exists, one fail-closed owner is retained until a later ++ /// accepted callback can adopt it. ++ pub(crate) fn restore_parent_execution_guard(&mut self, guard: WasiProcessExecutionGuard) { ++ assert!( ++ guard.matches_process_thread(&self.process, &self.thread), ++ "restored parent execution does not match its environment" ++ ); ++ if guard.try_handoff_to_current_task_wasm( ++ &self.process, ++ &self.thread, ++ &self.current_task_wasm_owner, ++ ) { ++ return; ++ } ++ assert!( ++ self.deferred_parent_execution.is_none(), ++ "nested deferred parent execution ownership is unsupported" ++ ); ++ self.deferred_parent_execution = Some(guard); ++ } ++ ++ pub(crate) fn acquire_parent_execution_guard( ++ &self, ++ ) -> Result { ++ self.process.acquire_supplemental_execution_guard( ++ self.thread.clone(), ++ self.current_task_wasm_owner.clone(), ++ ) ++ } ++ ++ /// Clones process-visible WASI state for a newly spawned guest thread ++ /// without copying execution ownership that belongs to the calling thread. ++ pub(crate) fn clone_for_thread_spawn( ++ &self, ++ thread: WasiThread, ++ layout: WasiMemoryLayout, ++ ) -> Self { ++ assert!( ++ self.vfork.is_none(), ++ "a vfork child cannot spawn a guest thread" ++ ); ++ let mut env = self.clone(); ++ env.deferred_parent_execution.take(); ++ env.current_task_wasm_owner = Weak::new(); ++ env.thread = thread; ++ env.layout = layout; ++ env ++ } ++ ++ /// Clones an environment for a successor TaskWasm of the exact same guest ++ /// thread. This is the only clone path allowed to carry the one deferred ++ /// parent execution owner across deep sleep or exec continuation. ++ pub(crate) fn take_for_same_thread_continuation(&mut self) -> Self { ++ let mut env = self.clone(); ++ env.deferred_parent_execution = self.deferred_parent_execution.take(); ++ env ++ } ++ + /// Construct a new [`WasiEnvBuilder`] that allows customizing an environment. + pub fn builder(program_name: impl Into) -> WasiEnvBuilder { + WasiEnvBuilder::new(program_name) +@@ -252,13 +401,32 @@ impl WasiEnv { + + /// Forking the WasiState is used when either fork or vfork is called + pub fn fork(&self) -> Result<(Self, WasiThreadHandle), ControlPlaneError> { +- let process = self.control_plane.new_process(self.process.module_hash)?; +- let handle = process.new_thread(self.layout.clone(), ThreadStartType::MainThread)?; ++ let (env, handle, mut registration) = self.fork_guarded()?; ++ registration.commit_child()?; ++ registration.complete_child_launch(self.tasks()); ++ Ok((env, handle)) ++ } ++ ++ /// Internal fork transaction whose process registration is committed only ++ /// when the syscall crosses its externally observable success boundary. ++ pub(crate) fn fork_guarded( ++ &self, ++ ) -> Result<(Self, WasiThreadHandle, WasiProcessRegistrationGuard), ControlPlaneError> { ++ let (process, handle, registration) = self ++ .control_plane ++ .new_child_process_with_main_thread_guarded( ++ &self.process, ++ self.process.module_hash, ++ self.layout.clone(), ++ )?; + + let thread = handle.as_thread(); + thread.copy_stack_from(&self.thread); + +- let state = Arc::new(self.state.fork()); ++ let state = self.state.fork_with(|_| {}).map_err(|errno| { ++ tracing::warn!(%errno, "could not fork process state"); ++ ControlPlaneError::SharedMemoryForkUnavailable ++ })?; + + let bin_factory = self.bin_factory.clone(); + +@@ -272,7 +440,11 @@ impl WasiEnv { + bin_factory, + state, + inner: Default::default(), +- owned_handles: Vec::new(), ++ // The child owns its main-thread lifetime. Parent environments ++ // must never accumulate handles for already-joined children. ++ owned_handles: vec![handle.clone()], ++ deferred_parent_execution: None, ++ current_task_wasm_owner: Weak::new(), + runtime: self.runtime.clone(), + capabilities: self.capabilities.clone(), + enable_deep_sleep: self.enable_deep_sleep, +@@ -283,7 +455,7 @@ impl WasiEnv { + disable_fs_cleanup: self.disable_fs_cleanup, + context_switching_environment: None, + }; +- Ok((new_env, handle)) ++ Ok((new_env, handle, registration)) + } + + pub fn pid(&self) -> WasiProcessId { +@@ -297,22 +469,35 @@ impl WasiEnv { + /// Returns true if this WASM process will need and try to use + /// asyncify while its running which normally means. + pub fn will_use_asyncify(&self) -> bool { +- self.inner() +- .static_module_instance_handles() +- .map(|handles| self.enable_deep_sleep || handles.has_stack_checkpoint) +- .unwrap_or(false) ++ let handles = self.inner().main_module_instance_handles(); ++ self.enable_deep_sleep || handles.has_stack_checkpoint + } + + /// Re-initializes this environment so that it can be executed again + pub fn reinit(&mut self) -> Result<(), WasiStateCreationError> { ++ // Verify the whole old process tree before mutating registry, handle, ++ // descriptor, or task-count state. Rejection therefore leaves the ++ // reusable environment exactly as it was. Success seals the epoch; ++ // any later filesystem/setup error is terminal for that old epoch and ++ // a second reinit attempt is rejected as ProcessRetiring. ++ retire_finished_process_tree(&self.control_plane, &self.process)?; ++ self.owned_handles.clear(); ++ self.deferred_parent_execution.take(); ++ self.current_task_wasm_owner = Weak::new(); ++ + // If the cleanup logic is enabled then we need to rebuild the + // file descriptors which would have been destroyed when the + // main thread exited + if !self.disable_fs_cleanup { + // First we clear any open files as the descriptors would + // otherwise clash +- if let Ok(mut map) = self.state.fs.fd_map.write() { +- map.clear(); ++ let removed = if let Ok(mut map) = self.state.fs.fd_map.write() { ++ map.drain_deferred() ++ } else { ++ Vec::new() ++ }; ++ for fd in removed { ++ fd.release_descriptor(); + } + self.state.fs.preopen_fds.write().unwrap().clear(); + *self.state.fs.current_dir.lock().unwrap() = "/".to_string(); +@@ -331,21 +516,17 @@ impl WasiEnv { + .map_err(WasiStateCreationError::WasiFsSetupError)?; + } + +- // The process and thread state need to be reset +- self.process = WasiProcess::new( +- self.process.pid, +- self.process.module_hash, +- self.process.compute.clone(), +- ); +- self.thread = WasiThread::new( +- self.thread.pid(), +- self.thread.tid(), +- self.thread.is_main(), +- self.process.finished.clone(), +- self.process.compute.must_upgrade().register_task()?, +- self.thread.memory_layout().clone(), +- self.thread.thread_start_type(), +- ); ++ // Allocate a fresh PID and register the replacement main thread through ++ // the same guarded process/thread transaction used by spawn and fork. ++ let module_hash = self.process.module_hash; ++ let (process, handle, mut registration) = self ++ .control_plane ++ .new_process_with_main_thread_guarded(module_hash, self.layout.clone())?; ++ let thread = handle.as_thread(); ++ registration.commit(); ++ self.process = process; ++ self.thread = thread; ++ self.owned_handles.push(handle); + + Ok(()) + } +@@ -377,13 +558,12 @@ impl WasiEnv { + self.try_inner() + .map(|handles| { + handles +- .static_module_instance_handles() +- .map(|handles| { +- handles.asyncify_get_state.is_some() +- && handles.asyncify_start_rewind.is_some() +- && handles.asyncify_start_unwind.is_some() +- }) +- .unwrap_or(false) ++ .main_module_instance_handles() ++ .supports_asyncify_stack_rewind() ++ && handles ++ .main_module_instance_handles() ++ .asyncify_get_state ++ .is_some() + }) + .unwrap_or(false) + } +@@ -398,10 +578,11 @@ impl WasiEnv { + init: WasiEnvInit, + module_hash: ModuleHash, + ) -> Result { +- let process = if let Some(p) = init.process { +- p ++ let (process, process_registration) = if let Some(p) = init.process { ++ (p, None) + } else { +- init.control_plane.new_process(module_hash)? ++ let (process, registration) = init.control_plane.new_process_guarded(module_hash)?; ++ (process, Some(registration)) + }; + + #[cfg(feature = "journal")] +@@ -418,6 +599,7 @@ impl WasiEnv { + process.new_thread(layout.clone(), ThreadStartType::MainThread)? + }; + ++ let state = Arc::new(init.state); + let mut env = Self { + control_plane: init.control_plane, + process, +@@ -425,9 +607,11 @@ impl WasiEnv { + layout, + vfork: None, + poll_seed: 0, +- state: Arc::new(init.state), ++ state, + inner: Default::default(), + owned_handles: Vec::new(), ++ deferred_parent_execution: None, ++ current_task_wasm_owner: Weak::new(), + #[cfg(feature = "journal")] + enable_journal: init.runtime.active_journal().is_some(), + #[cfg(not(feature = "journal"))] +@@ -455,6 +639,10 @@ impl WasiEnv { + #[cfg(feature = "sys")] + env.map_commands(init.mapped_commands.clone())?; + ++ if let Some(mut registration) = process_registration { ++ registration.commit(); ++ } ++ + Ok(env) + } + +@@ -467,15 +655,40 @@ impl WasiEnv { + memory: Option, + update_layout: bool, + call_initialize: bool, +- parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut)>, ++ preinitialized_memory_image: Option, ++ fresh_zeroed_memory: bool, ++ parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut<'_, WasiEnv>)>, + ) -> Result<(Instance, WasiFunctionEnv), WasiThreadError> { + let pid = self.process.pid(); + + let mut store = store.as_store_mut(); + let engine = self.runtime().engine(); +- let mut func_env = WasiFunctionEnv::new(&mut store, self); ++ let mut func_env = { WasiFunctionEnv::new(&mut store, self) }; + + let is_dl = super::linker::is_dynamically_linked(&module); ++ if !is_dl ++ && func_env ++ .data(&store) ++ .state ++ .has_shared_memory_exec_reservations() ++ { ++ return Err(WasiThreadError::LinkError(Arc::new( ++ super::linker::LinkError::SharedMemoryExecReservation( ++ "fresh exec images with inherited mappings require an imported linear memory that can be reserved before module start" ++ .to_string(), ++ ), ++ ))); ++ } ++ if preinitialized_memory_image.is_some() && (!is_dl || parent_linker_and_ctx.is_some()) { ++ let reason = if !is_dl { ++ "preinitialized memory images require a fresh dynamically linked main module" ++ } else { ++ "preinitialized memory images cannot replace shared instance-group memory" ++ }; ++ return Err(WasiThreadError::LinkError(Arc::new( ++ super::linker::LinkError::PreinitializedMemoryImage(reason.to_string()), ++ ))); ++ } + if is_dl { + let linker = match parent_linker_and_ctx { + Some((linker, ctx)) => linker.create_instance_group(ctx, &mut store, &mut func_env), +@@ -502,15 +715,19 @@ impl WasiEnv { + }; + + // TODO: make stack size configurable +- Linker::new( +- engine, +- &module, +- &mut store, +- memory, +- &mut func_env, +- 8 * 1024 * 1024, +- &ld_library_path, +- ) ++ { ++ Linker::new( ++ engine, ++ &module, ++ &mut store, ++ memory, ++ &mut func_env, ++ 8 * 1024 * 1024, ++ &ld_library_path, ++ preinitialized_memory_image, ++ fresh_zeroed_memory, ++ ) ++ } + } + }; + +@@ -539,8 +756,7 @@ impl WasiEnv { + import_object.define("env", "memory", memory); + } + let runtime = func_env.data(&store).runtime.clone(); +- let additional_imports = runtime +- .additional_imports(&module, &mut store) ++ let additional_imports = { runtime.additional_imports(&module, &mut store) } + .map_err(|err| WasiThreadError::AdditionalImportCreationFailed(Arc::new(err)))?; + + for ((namespace, name), value) in &additional_imports { +@@ -564,7 +780,7 @@ impl WasiEnv { + }); + + // Construct the instance. +- let instance = match Instance::new(&mut store, &module, &import_object) { ++ let instance = match { Instance::new(&mut store, &module, &import_object) } { + Ok(a) => a, + Err(err) => { + tracing::error!( +@@ -579,16 +795,14 @@ impl WasiEnv { + } + }; + +- runtime +- .configure_new_instance(&module, &mut store, &instance, imported_memory.as_ref()) +- .map_err(|err| WasiThreadError::AdditionalImportCreationFailed(Arc::new(err)))?; ++ { ++ runtime.configure_new_instance(&module, &mut store, &instance, imported_memory.as_ref()) ++ } ++ .map_err(|err| WasiThreadError::AdditionalImportCreationFailed(Arc::new(err)))?; + + let handles = match imported_memory { +- Some(memory) => WasiModuleTreeHandles::Static(WasiModuleInstanceHandles::new( +- memory, +- &store, +- instance.clone(), +- None, ++ Some(memory) => Ok(WasiModuleTreeHandles::Static( ++ WasiModuleInstanceHandles::new(memory, &store, instance.clone(), None), + )), + None => { + let exported_memory = instance +@@ -605,23 +819,22 @@ impl WasiEnv { + .ok_or(WasiThreadError::ExportError(ExportError::Missing( + "No imported or exported memory found".to_owned(), + )))?; +- WasiModuleTreeHandles::Static(WasiModuleInstanceHandles::new( +- exported_memory, +- &store, +- instance.clone(), +- None, ++ Ok(WasiModuleTreeHandles::Static( ++ WasiModuleInstanceHandles::new(exported_memory, &store, instance.clone(), None), + )) + } +- }; ++ }?; + + // Initialize the WASI environment +- if let Err(err) = func_env.initialize_handles_and_layout( +- &mut store, +- instance.clone(), +- handles, +- None, +- update_layout, +- ) { ++ if let Err(err) = { ++ func_env.initialize_handles_and_layout( ++ &mut store, ++ instance.clone(), ++ handles, ++ None, ++ update_layout, ++ ) ++ } { + tracing::error!( + %pid, + error = &err as &dyn std::error::Error, +@@ -635,7 +848,7 @@ impl WasiEnv { + + // If this module exports an _initialize function, run that first. + if call_initialize && let Ok(initialize) = instance.exports.get_function("_initialize") { +- let initialize_result = initialize.call(&mut store, &[]); ++ let initialize_result = { initialize.call(&mut store, &[]) }; + if let Err(err) = initialize_result { + func_env + .data(&store) +@@ -796,11 +1009,13 @@ impl WasiEnv { + } + } + ++ let mut processed_guest_signal = false; + for signal in signals { + // Skip over Sigwakeup, which is host-side-only + if matches!(signal, Signal::Sigwakeup) { + continue; + } ++ processed_guest_signal = true; + + tracing::trace!( + pid=%ctx.data().pid(), +@@ -836,7 +1051,7 @@ impl WasiEnv { + "signal processed", + ); + } +- Ok(true) ++ Ok(processed_guest_signal) + } else { + tracing::trace!("no signal handler"); + Ok(false) +@@ -915,13 +1130,10 @@ impl WasiEnv { + #[doc(hidden)] + pub(crate) fn swap_inner(&mut self, other: &mut Self) { + std::mem::swap(&mut self.inner, &mut other.inner); +- } +- +- /// Helper function to ensure the module isn't dynamically linked, needed since +- /// we only support a subset of WASIX functionality for dynamically linked modules. +- /// Specifically, anything that requires asyncify is not supported right now. +- pub(crate) fn ensure_static_module(&self) -> Result<(), ()> { +- self.inner.get().unwrap().ensure_static_module() ++ std::mem::swap( ++ &mut self.current_task_wasm_owner, ++ &mut other.current_task_wasm_owner, ++ ); + } + + /// Tries to clone the instance from this environment, but only if it's a static +@@ -1259,6 +1471,8 @@ impl WasiEnv { + pub fn on_exit(&self, process_exit_code: Option) -> BoxFuture<'static, ()> { + const CLEANUP_TIMEOUT: Duration = Duration::from_secs(10); + ++ if process_exit_code.is_some() {} ++ + // If snap-shooting is enabled then we should record an event that the thread has exited. + #[cfg(feature = "journal")] + if self.should_journal() && self.has_active_journal() { +@@ -1348,3 +1562,578 @@ impl WasiEnv { + } + } + } ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use wasmer::Engine; ++ ++ fn enter_tokio_runtime() -> Option { ++ #[cfg(not(target_arch = "wasm32"))] ++ { ++ Some( ++ tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build() ++ .unwrap(), ++ ) ++ } ++ ++ #[cfg(target_arch = "wasm32")] ++ { ++ None ++ } ++ } ++ ++ #[test] ++ fn repeated_reinit_uses_fresh_registered_epoch_and_plateaus_counts() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-reinit-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let mut previous_pid = env.pid(); ++ ++ for _ in 0..8 { ++ env.process.terminate(Errno::Success.into()); ++ env.reinit().unwrap(); ++ ++ assert_ne!(env.pid(), previous_pid); ++ previous_pid = env.pid(); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ assert_eq!(env.owned_handles.len(), 1); ++ assert_eq!(env.process.active_threads(), 1); ++ assert_eq!(env.process.all_threads(), vec![env.tid()]); ++ let registered = plane.get_process(env.pid()).unwrap(); ++ assert!(registered.same_identity(&env.process)); ++ } ++ ++ env.process.terminate(Errno::Success.into()); ++ plane.retire_process_epoch(&env.process).unwrap(); ++ env.thread.retire_task_registration(); ++ env.owned_handles.clear(); ++ assert_eq!(plane.registered_process_count(), 0); ++ assert_eq!(plane.active_task_count(), 0); ++ } ++ ++ #[test] ++ fn repeated_child_construction_keeps_main_handle_owned_by_child() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let env = WasiEnv::builder("child-handle-plateau-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let baseline_handles = env.owned_handles.len(); ++ let baseline_tasks = plane.active_task_count(); ++ let baseline_processes = plane.registered_process_count(); ++ ++ for _ in 0..32 { ++ let (child_env, construction_handle, registration) = env.fork_guarded().unwrap(); ++ assert_eq!(env.owned_handles.len(), baseline_handles); ++ assert_eq!(child_env.owned_handles.len(), 1); ++ assert_eq!(plane.active_task_count(), baseline_tasks + 1); ++ assert_eq!(plane.registered_process_count(), baseline_processes); ++ ++ // Aborting an unpublished launch drops the child-owned clone; the ++ // parent never becomes a lifetime owner for the child main task. ++ drop(registration); ++ drop(construction_handle); ++ drop(child_env); ++ assert_eq!(env.owned_handles.len(), baseline_handles); ++ assert_eq!(plane.active_task_count(), baseline_tasks); ++ assert_eq!(plane.registered_process_count(), baseline_processes); ++ assert_eq!(env.process.lock().pending_child_publications, 0); ++ } ++ } ++ ++ #[test] ++ fn reinit_rejects_a_running_epoch_without_mutation() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-running-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let pid = env.pid(); ++ let error = env.reinit().unwrap_err(); ++ ++ assert!(matches!( ++ error, ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessStillRunning { ++ pid: running_pid ++ }) if running_pid == pid.raw() ++ )); ++ assert_eq!(env.pid(), pid); ++ assert_eq!(env.control_plane.registered_process_count(), 1); ++ assert_eq!(env.control_plane.active_task_count(), 1); ++ assert!(!env.process.lock().retiring); ++ } ++ ++ #[test] ++ fn reinit_requires_terminal_status_and_execution_quiescence() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-execution-lease-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let old_pid = env.pid(); ++ let lease = env.process.acquire_execution_lease().unwrap(); ++ env.process.terminate(Errno::Success.into()); ++ ++ assert_eq!( ++ env.reinit().unwrap_err(), ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessStillRunning { ++ pid: old_pid.raw(), ++ }) ++ ); ++ assert_eq!(env.pid(), old_pid); ++ assert!(!env.process.lock().retiring); ++ assert!(plane.get_process(old_pid).is_some()); ++ ++ drop(lease); ++ env.reinit().unwrap(); ++ assert_ne!(env.pid(), old_pid); ++ assert!(plane.get_process(old_pid).is_none()); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ } ++ ++ #[test] ++ fn reinit_rejects_live_background_thread_without_mutating_epoch() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-live-thread-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let pid = env.pid(); ++ let background = env ++ .process ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) ++ .unwrap(); ++ env.process.terminate(Errno::Success.into()); ++ ++ let error = env.reinit().unwrap_err(); ++ assert_eq!( ++ error, ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveThreads { ++ pid: pid.raw(), ++ count: 1, ++ }) ++ ); ++ assert_eq!(env.pid(), pid); ++ assert_eq!(env.process.active_threads(), 2); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 2); ++ assert!(!env.process.lock().retiring); ++ ++ drop(background); ++ env.reinit().unwrap(); ++ assert_ne!(env.pid(), pid); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ } ++ ++ #[test] ++ fn reinit_rejects_live_child_then_reaps_finished_child_exactly() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-live-child-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let parent_pid = env.pid(); ++ let (child, child_main, mut child_registration) = plane ++ .new_child_process_with_main_thread_guarded( ++ &env.process, ++ ModuleHash::random(), ++ WasiMemoryLayout::default(), ++ ) ++ .unwrap(); ++ child_registration.commit_child().unwrap(); ++ child_registration.complete_child_launch(env.tasks()); ++ let child_pid = child.pid(); ++ env.process.terminate(Errno::Success.into()); ++ ++ let error = env.reinit().unwrap_err(); ++ assert_eq!( ++ error, ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveChildren { ++ pid: parent_pid.raw(), ++ count: 1, ++ }) ++ ); ++ assert_eq!(env.pid(), parent_pid); ++ assert_eq!(env.process.lock().children.len(), 1); ++ assert_eq!(plane.registered_process_count(), 2); ++ assert_eq!(plane.active_task_count(), 2); ++ assert!(!env.process.lock().retiring); ++ ++ child.terminate(Errno::Success.into()); ++ drop(child_main); ++ env.reinit().unwrap(); ++ assert_ne!(env.pid(), parent_pid); ++ assert!(plane.get_process(child_pid).is_none()); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ } ++ ++ #[test] ++ fn reinit_cannot_cross_inflight_child_publication() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-pending-child-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let parent_pid = env.pid(); ++ let (child_env, child_handle, child_registration) = env.fork_guarded().unwrap(); ++ env.process.terminate(Errno::Success.into()); ++ ++ let error = env.reinit().unwrap_err(); ++ assert_eq!( ++ error, ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveChildren { ++ pid: parent_pid.raw(), ++ count: 1, ++ }) ++ ); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 2); ++ assert!(!env.process.lock().retiring); ++ ++ drop(child_registration); ++ drop(child_handle); ++ drop(child_env); ++ env.reinit().unwrap(); ++ assert_ne!(env.pid(), parent_pid); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ } ++ ++ #[test] ++ fn sealed_epoch_is_terminal_after_later_reset_failure() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-terminal-reset-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let pid = env.pid(); ++ env.process.terminate(Errno::Success.into()); ++ assert!(env.process.begin_epoch_retirement().unwrap().is_empty()); ++ ++ let error = env.reinit().unwrap_err(); ++ assert_eq!( ++ error, ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessRetiring { ++ pid: pid.raw(), ++ }) ++ ); ++ assert_eq!(env.pid(), pid); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ ++ plane.retire_process_epoch(&env.process).unwrap(); ++ env.process.retire_epoch_task_registrations(); ++ env.owned_handles.clear(); ++ } ++ ++ #[test] ++ fn public_fork_publishes_and_adopts_before_releasing_parent_permit() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("public-fork-adoption-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let (child_env, child_handle) = env.fork().unwrap(); ++ let child_pid = child_env.pid(); ++ ++ assert!( ++ env.process ++ .lock() ++ .children ++ .iter() ++ .any(|child| child.same_identity(&child_env.process)) ++ ); ++ assert!( ++ plane ++ .get_process(child_pid) ++ .unwrap() ++ .same_identity(&child_env.process) ++ ); ++ assert_eq!(plane.registered_process_count(), 2); ++ assert_eq!(plane.active_task_count(), 2); ++ ++ child_env.process.terminate(Errno::Success.into()); ++ drop(child_handle); ++ env.process.terminate(Errno::Success.into()); ++ env.reinit().unwrap(); ++ assert!(plane.get_process(child_pid).is_none()); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ } ++ ++ #[test] ++ fn reinit_recursively_retires_finished_grandchildren() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-grandchild-retirement-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let (child_env, child_handle) = env.fork().unwrap(); ++ let child_pid = child_env.pid(); ++ let (grandchild_env, grandchild_handle) = child_env.fork().unwrap(); ++ let grandchild_pid = grandchild_env.pid(); ++ ++ grandchild_env.process.terminate(Errno::Success.into()); ++ drop(grandchild_handle); ++ child_env.process.terminate(Errno::Success.into()); ++ drop(child_handle); ++ env.process.terminate(Errno::Success.into()); ++ ++ env.reinit().unwrap(); ++ assert!(plane.get_process(child_pid).is_none()); ++ assert!(plane.get_process(grandchild_pid).is_none()); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ assert!(env.process.lock().children.is_empty()); ++ } ++ ++ #[test] ++ fn descendant_validation_failure_seals_no_process_in_tree() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-tree-validation-transaction-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let (child_env, child_handle) = env.fork().unwrap(); ++ let (grandchild_env, grandchild_handle) = child_env.fork().unwrap(); ++ ++ child_env.process.terminate(Errno::Success.into()); ++ env.process.terminate(Errno::Success.into()); ++ assert!(matches!( ++ env.reinit().unwrap_err(), ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessHasLiveChildren { ++ count: 1, ++ .. ++ }) ++ )); ++ assert!(!env.process.lock().retiring); ++ assert!(!child_env.process.lock().retiring); ++ assert!(!grandchild_env.process.lock().retiring); ++ ++ grandchild_env.process.terminate(Errno::Success.into()); ++ drop(grandchild_handle); ++ drop(child_handle); ++ env.reinit().unwrap(); ++ assert_eq!(env.control_plane.registered_process_count(), 1); ++ assert_eq!(env.control_plane.active_task_count(), 1); ++ } ++ ++ #[test] ++ fn stale_environment_cannot_fork_or_start_threads_after_reinit() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-stale-epoch-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let stale = env.clone(); ++ let stale_pid = stale.pid(); ++ let plane = env.control_plane.clone(); ++ ++ env.process.terminate(Errno::Success.into()); ++ env.reinit().unwrap(); ++ assert_ne!(env.pid(), stale_pid); ++ ++ assert_eq!( ++ stale.fork().unwrap_err(), ++ ControlPlaneError::ProcessRetiring { ++ pid: stale_pid.raw(), ++ } ++ ); ++ assert_eq!( ++ stale ++ .process ++ .new_thread(WasiMemoryLayout::default(), ThreadStartType::MainThread) ++ .unwrap_err(), ++ ControlPlaneError::ProcessRetiring { ++ pid: stale_pid.raw(), ++ } ++ ); ++ assert_eq!(plane.registered_process_count(), 1); ++ assert_eq!(plane.active_task_count(), 1); ++ assert!(plane.get_process(stale_pid).is_none()); ++ assert!(stale.process.lock().children.is_empty()); ++ assert_eq!(stale.process.lock().pending_child_publications, 0); ++ } ++ ++ #[test] ++ fn retargeted_thread_clone_drops_calling_thread_execution_ownership() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("thread-clone-ownership-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let owner = Arc::new(()); ++ let main = env.thread.clone(); ++ env.bind_pending_task_wasm_execution(&main, &owner); ++ let deferred = env.acquire_parent_execution_guard().unwrap(); ++ env.deferred_parent_execution = Some(deferred); ++ assert_eq!(env.process.lock().execution_leases, 1); ++ ++ let spawned = env ++ .process ++ .new_thread( ++ WasiMemoryLayout::default(), ++ ThreadStartType::ThreadSpawn { start_ptr: 0 }, ++ ) ++ .unwrap(); ++ let spawned_thread = spawned.as_thread(); ++ let cloned = ++ env.clone_for_thread_spawn(spawned_thread.clone(), WasiMemoryLayout::default()); ++ assert!(cloned.thread.same_identity(&spawned_thread)); ++ assert!(cloned.deferred_parent_execution.is_none()); ++ assert!(cloned.current_task_wasm_owner.upgrade().is_none()); ++ assert!(cloned.vfork.is_none()); ++ assert!(env.deferred_parent_execution.is_some()); ++ assert_eq!(env.process.lock().execution_leases, 1); ++ ++ let deferred = env.deferred_parent_execution.take().unwrap(); ++ env.restore_parent_execution_guard(deferred); ++ assert_eq!(env.process.lock().execution_leases, 0); ++ drop(spawned); ++ } ++ ++ #[test] ++ fn swap_inner_uses_exact_physical_owner_for_immediate_and_deferred_restore() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ ++ // Ordinary vfork restore returns to the same physical TaskWasm owner, ++ // so the supplemental parent lease is consumed immediately. ++ { ++ let mut current = WasiEnv::builder("ordinary-vfork-owner-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let parent_process = current.process.clone(); ++ let parent_thread = current.thread.clone(); ++ let owner = Arc::new(()); ++ current.bind_pending_task_wasm_execution(&parent_thread, &owner); ++ let supplemental = current.acquire_parent_execution_guard().unwrap(); ++ let (mut stored, _child_handle, mut registration) = current.fork_guarded().unwrap(); ++ registration.commit_child().unwrap(); ++ registration.complete_child_launch(current.tasks()); ++ ++ stored.swap_inner(&mut current); ++ std::mem::swap(&mut current, &mut stored); ++ assert!(!current.process.same_identity(&stored.process)); ++ stored.swap_inner(&mut current); ++ std::mem::swap(&mut current, &mut stored); ++ assert!(current.process.same_identity(&parent_process)); ++ current.restore_parent_execution_guard(supplemental); ++ assert!(current.deferred_parent_execution.is_none()); ++ assert_eq!(parent_process.lock().execution_leases, 0); ++ } ++ ++ // A child successor has a different physical owner. Restoration must ++ // retain exactly one parent lease and transfer it to a later exact ++ // parent continuation before releasing it. ++ { ++ let mut current = WasiEnv::builder("deferred-vfork-owner-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let parent_process = current.process.clone(); ++ let parent_thread = current.thread.clone(); ++ let predecessor = Arc::new(()); ++ current.bind_pending_task_wasm_execution(&parent_thread, &predecessor); ++ let supplemental = current.acquire_parent_execution_guard().unwrap(); ++ let (mut stored, _child_handle, mut registration) = current.fork_guarded().unwrap(); ++ registration.commit_child().unwrap(); ++ registration.complete_child_launch(current.tasks()); ++ ++ stored.swap_inner(&mut current); ++ std::mem::swap(&mut current, &mut stored); ++ drop(predecessor); ++ let child_successor = Arc::new(()); ++ let child_thread = current.thread.clone(); ++ current.bind_pending_task_wasm_execution(&child_thread, &child_successor); ++ ++ stored.swap_inner(&mut current); ++ std::mem::swap(&mut current, &mut stored); ++ current.restore_parent_execution_guard(supplemental); ++ assert!(current.deferred_parent_execution.is_some()); ++ assert_eq!(parent_process.lock().execution_leases, 1); ++ ++ let mut parent_successor = current.take_for_same_thread_continuation(); ++ assert!(current.deferred_parent_execution.is_none()); ++ let successor_identity = Arc::new(()); ++ let successor_lease = parent_process.acquire_execution_lease().unwrap(); ++ parent_successor.bind_pending_task_wasm_execution(&parent_thread, &successor_identity); ++ let deferred = parent_successor.deferred_parent_execution.take().unwrap(); ++ assert!(deferred.try_handoff_to_accepted_task_wasm(&parent_process, &parent_thread)); ++ assert_eq!(parent_process.lock().execution_leases, 1); ++ drop(successor_lease); ++ assert_eq!(parent_process.lock().execution_leases, 0); ++ drop(child_successor); ++ } ++ } ++ ++ #[test] ++ fn post_seal_main_thread_admission_failure_is_terminal_and_leak_free() { ++ let runtime = enter_tokio_runtime(); ++ let _runtime_guard = runtime.as_ref().map(|runtime| runtime.enter()); ++ let mut env = WasiEnv::builder("dcgi-post-seal-admission-test") ++ .engine(Engine::default()) ++ .build() ++ .unwrap(); ++ let plane = env.control_plane.clone(); ++ let old_pid = env.pid(); ++ env.process.terminate(Errno::Success.into()); ++ plane.fail_next_task_admission(); ++ ++ assert_eq!( ++ env.reinit().unwrap_err(), ++ WasiStateCreationError::ControlPlane(ControlPlaneError::TaskLimitReached { ++ max: usize::MAX, ++ }) ++ ); ++ assert_eq!(env.pid(), old_pid); ++ assert!(env.process.lock().retiring); ++ assert!(plane.get_process(old_pid).is_none()); ++ assert_eq!(plane.registered_process_count(), 0); ++ assert_eq!(plane.active_task_count(), 0); ++ ++ assert_eq!( ++ env.reinit().unwrap_err(), ++ WasiStateCreationError::ControlPlane(ControlPlaneError::ProcessRetiring { ++ pid: old_pid.raw(), ++ }) ++ ); ++ assert_eq!(plane.registered_process_count(), 0); ++ assert_eq!(plane.active_task_count(), 0); ++ } ++} +diff --git a/lib/wasix/src/state/func_env.rs b/lib/wasix/src/state/func_env.rs +index 914edc2..e5c514c 100644 +--- a/lib/wasix/src/state/func_env.rs ++++ b/lib/wasix/src/state/func_env.rs +@@ -17,7 +17,7 @@ use crate::{ + utils::{get_wasi_version, get_wasi_versions, store::restore_store_snapshot}, + }; + +-use super::Linker; ++use super::{Linker, PreinitializedMemoryImageMode}; + + /// The default stack size for WASIX - the number itself is the default that compilers + /// have used in the past when compiling WASM apps. +@@ -46,12 +46,17 @@ impl WasiFunctionEnv { + spawn_type: SpawnMemoryTypeOrStore, + update_layout: bool, + call_initialize: bool, +- parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut)>, ++ preinitialized_memory_image: Option, ++ parent_linker_and_ctx: Option<(Linker, &mut FunctionEnvMut<'_, WasiEnv>)>, + ) -> Result<(Self, Store), WasiThreadError> { + // Create a new store and put the memory object in it + // (but only if it has imported memory) +- let (memory, store): (Option, Option) = match spawn_type { +- SpawnMemoryTypeOrStore::New => (None, None), ++ let (memory, store, fresh_zeroed_memory): ( ++ Option, ++ Option, ++ bool, ++ ) = match spawn_type { ++ SpawnMemoryTypeOrStore::New => (None, None, true), + SpawnMemoryTypeOrStore::Type(mut ty) => { + ty.shared = true; + +@@ -61,7 +66,7 @@ impl WasiFunctionEnv { + // browser otherwise creation will fail. + let _ = ty.maximum.get_or_insert(wasmer_types::Pages::max_value()); + +- let mem = Memory::new(&mut store, ty).map_err(|err| { ++ let mem = { Memory::new(&mut store, ty) }.map_err(|err| { + tracing::error!( + error = &err as &dyn std::error::Error, + memory_type=?ty, +@@ -69,27 +74,36 @@ impl WasiFunctionEnv { + ); + WasiThreadError::MemoryCreateFailed(err) + })?; +- (Some(mem), Some(store)) ++ (Some(mem), Some(store), true) + } +- SpawnMemoryTypeOrStore::StoreAndMemory(s, m) => (m, Some(s)), ++ SpawnMemoryTypeOrStore::StoreAndMemory(s, m) => (m, Some(s), false), + }; + +- let mut store = store.unwrap_or_else(|| env.runtime().new_store()); ++ let mut store = match store { ++ Some(store) => store, ++ None => env.runtime().new_store(), ++ }; + +- let (_, ctx) = env.instantiate( +- module, +- &mut store, +- memory, +- update_layout, +- call_initialize, +- parent_linker_and_ctx, +- )?; ++ let (_, ctx) = { ++ env.instantiate( ++ module, ++ &mut store, ++ memory, ++ update_layout, ++ call_initialize, ++ preinitialized_memory_image, ++ fresh_zeroed_memory, ++ parent_linker_and_ctx, ++ ) ++ }?; + + // FIXME: shouldn't this happen _before_ instantiating, so the startup code in the instance + // has access to the globals? + // Set all the globals + if let Some(snapshot) = store_snapshot { +- restore_store_snapshot(&mut store, &snapshot); ++ restore_store_snapshot(&mut store, &snapshot).map_err(|err| { ++ WasiThreadError::InitFailed(std::sync::Arc::new(anyhow::Error::new(err))) ++ })?; + } + + Ok((ctx, store)) +diff --git a/lib/wasix/src/state/handles/mod.rs b/lib/wasix/src/state/handles/mod.rs +index 4fa1030..6de50be 100644 +--- a/lib/wasix/src/state/handles/mod.rs ++++ b/lib/wasix/src/state/handles/mod.rs +@@ -47,9 +47,6 @@ pub struct WasiModuleInstanceHandles { + /// Points to the start of the TLS area + pub(crate) tls_base: Option, + +- /// Main function that will be invoked (name = "_start") +- pub(crate) start: Option>, +- + /// Function thats invoked to initialize the WASM module (name = "_initialize") + // TODO: review allow... + #[allow(dead_code)] +@@ -140,7 +137,6 @@ impl WasiModuleInstanceHandles { + stack_low: instance.exports.get_global("__stack_low").cloned().ok(), + stack_high: instance.exports.get_global("__stack_high").cloned().ok(), + tls_base: instance.exports.get_global("__tls_base").cloned().ok(), +- start: instance.exports.get_typed_function(store, "_start").ok(), + initialize: instance + .exports + .get_typed_function(store, "_initialize") +@@ -187,6 +183,13 @@ impl WasiModuleInstanceHandles { + self.instance.module().clone() + } + ++ pub(crate) fn supports_asyncify_stack_rewind(&self) -> bool { ++ self.asyncify_start_unwind.is_some() ++ && self.asyncify_stop_unwind.is_some() ++ && self.asyncify_start_rewind.is_some() ++ && self.asyncify_stop_rewind.is_some() ++ } ++ + /// Providers safe access to the memory + /// (it must be initialized before it can be used) + pub fn memory_view<'a>(&'a self, store: &'a (impl AsStoreRef + ?Sized)) -> MemoryView<'a> { +@@ -245,6 +248,68 @@ impl From for Errno { + } + } + ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use wasmer::{Instance, Module, Store, imports}; ++ ++ fn supports_asyncify_stack_rewind(wat: &str) -> bool { ++ let mut store = Store::default(); ++ let module = Module::new(&store, wat).unwrap(); ++ let instance = Instance::new(&mut store, &module, &imports! {}).unwrap(); ++ let memory = instance.exports.get_memory("memory").unwrap().clone(); ++ let handles = WasiModuleInstanceHandles::new(memory, &store, instance, None); ++ ++ handles.supports_asyncify_stack_rewind() ++ } ++ ++ #[test] ++ fn stack_rewind_is_unsupported_without_exports() { ++ let supported = supports_asyncify_stack_rewind( ++ r#" ++ (module ++ (memory (export "memory") 1) ++ ) ++ "#, ++ ); ++ ++ assert!(!supported); ++ } ++ ++ #[test] ++ fn stack_rewind_requires_complete_asyncify_exports() { ++ let supported = supports_asyncify_stack_rewind( ++ r#" ++ (module ++ (memory (export "memory") 1) ++ (func (export "asyncify_start_unwind") (param i32)) ++ (func (export "asyncify_stop_unwind")) ++ (func (export "asyncify_start_rewind") (param i32)) ++ ) ++ "#, ++ ); ++ ++ assert!(!supported); ++ } ++ ++ #[test] ++ fn stack_rewind_supports_complete_asyncify_exports() { ++ let supported = supports_asyncify_stack_rewind( ++ r#" ++ (module ++ (memory (export "memory") 1) ++ (func (export "asyncify_start_unwind") (param i32)) ++ (func (export "asyncify_stop_unwind")) ++ (func (export "asyncify_start_rewind") (param i32)) ++ (func (export "asyncify_stop_rewind")) ++ ) ++ "#, ++ ); ++ ++ assert!(supported); ++ } ++} ++ + impl WasiModuleTreeHandles { + /// Can be used to get the `WasiModuleInstanceHandles` of the main module. + /// If access to the side modules' instance handles is required, one must go +@@ -290,16 +355,6 @@ impl WasiModuleTreeHandles { + } + } + +- /// Helper function to ensure the module isn't dynamically linked, needed since +- /// we only support a subset of WASIX functionality for dynamically linked modules. +- /// Specifically, anything that requires asyncify is not supported right now. +- pub(crate) fn ensure_static_module(&self) -> Result<(), ()> { +- match self { +- WasiModuleTreeHandles::Static(_) => Ok(()), +- _ => Err(()), +- } +- } +- + /// Providers safe access to the memory + /// (it must be initialized before it can be used) + pub fn memory_view<'a>(&'a self, store: &'a (impl AsStoreRef + ?Sized)) -> MemoryView<'a> { +diff --git a/lib/wasix/src/state/linker.rs b/lib/wasix/src/state/linker.rs +index 0abec18..4090d25 100644 +--- a/lib/wasix/src/state/linker.rs ++++ b/lib/wasix/src/state/linker.rs +@@ -266,22 +266,21 @@ use std::{ + ops::{Deref, DerefMut}, + path::{Path, PathBuf}, + sync::{ +- Arc, Barrier, Mutex, MutexGuard, RwLock, RwLockWriteGuard, TryLockError, ++ Arc, Barrier, Condvar, Mutex, MutexGuard, RwLock, RwLockWriteGuard, TryLockError, + atomic::{AtomicBool, Ordering}, + }, + }; + +-use bus::Bus; + use derive_more::Debug; + use shared_buffer::OwnedBuffer; + use tracing::trace; + use virtual_fs::{AsyncReadExt, FileSystem, FsError}; + use virtual_mio::block_on; + use wasmer::{ +- AsStoreMut, AsStoreRef, Engine, ExportError, Exportable, Extern, ExternType, Function, +- FunctionEnv, FunctionEnvMut, FunctionType, Global, GlobalType, ImportType, Imports, Instance, +- InstantiationError, Memory, MemoryError, Module, RuntimeError, StoreMut, Table, Tag, Type, +- Value, WASM_PAGE_SIZE, WasmTypeList, ++ AsStoreMut, Engine, ExportError, Extern, ExternType, Function, FunctionEnv, FunctionEnvMut, ++ FunctionType, Global, GlobalType, ImportType, Imports, Instance, InstantiationError, Memory, ++ MemoryError, Module, RuntimeError, StoreMut, Table, Tag, Type, Value, WASM_PAGE_SIZE, ++ WasmTypeList, + }; + use wasmer_wasix_types::wasix::WasiMemoryLayout; + +@@ -291,7 +290,15 @@ use crate::{ + runtime::module_cache::HashedModuleData, + }; + +-use super::{WasiModuleInstanceHandles, WasiState}; ++use super::{PreinitializedMemoryImageMode, WasiModuleInstanceHandles, WasiState}; ++ ++/// Linear witness that ordinary WebAssembly start returned normally for the ++/// newly instantiated main module. The private field can be constructed only ++/// in this linker module, and the non-`Copy` value is consumed by image ++/// application at the immediately following boundary. ++pub(crate) struct OrdinaryModuleStartCompleted { ++ _private: (), ++} + + // Module handle 1 is always the main module. Side modules get handles starting from the next one after the main module. + pub static MAIN_MODULE_HANDLE: ModuleHandle = ModuleHandle(1); +@@ -323,6 +330,224 @@ impl Display for ModuleHandle { + + const DEFAULT_RUNTIME_PATH: [&str; 3] = ["/lib", "/usr/lib", "/usr/local/lib"]; + ++/// Exports retained eagerly by WASIX's dynamic linker. ++/// ++/// These are the exact fixed entry points consumed while constructing runtime ++/// handles or running module initialization. Every dynamic symbol remains ++/// available through store-aware lookup against the module's complete export ++/// metadata; it is intentionally not copied into the per-instance export map ++/// until the linker actually asks for it. ++const DYNAMIC_INSTANCE_INITIAL_EXPORTS: &[&str] = &[ ++ "__indirect_function_table", ++ "__stack_pointer", ++ "__data_end", ++ "__stack_low", ++ "__stack_high", ++ "__tls_base", ++ "_start", ++ "_initialize", ++ "wasi_thread_start", ++ "__wasm_signal", ++ "asyncify_start_unwind", ++ "asyncify_stop_unwind", ++ "asyncify_start_rewind", ++ "asyncify_stop_rewind", ++ "asyncify_get_state", ++ "__wasm_apply_data_relocs", ++ "__wasm_apply_tls_relocs", ++ "__wasm_call_ctors", ++ "__wasix_init_tls", ++]; ++ ++/// A dependency-free, single-slot broadcast channel for linker rendezvous. ++/// ++/// Dynamic-link operations cannot overlap: the linker write lock serializes the ++/// producer and the two barriers serialize every participating instance group. ++/// A general-purpose ring-buffer bus is therefore unnecessary here. In ++/// particular, `bus::Bus` creates a permanent helper thread for each channel, ++/// even though these channels are almost always idle. ++/// ++/// The slot remains occupied until every receiver that existed at broadcast ++/// time has either received the value or disconnected. Receivers added while a ++/// value is pending only observe future broadcasts. All state transitions and ++/// condition-variable waits use the same mutex, so broadcasts and sender ++/// closure cannot be lost between checking the state and going to sleep. ++struct SingleSlotBroadcast { ++ inner: Arc>, ++} ++ ++struct SingleSlotReceiver { ++ inner: Arc>, ++ reader_id: u64, ++} ++ ++struct SingleSlotBroadcastInner { ++ state: Mutex>, ++ changed: Condvar, ++} ++ ++struct SingleSlotBroadcastState { ++ slot: Option, ++ readers: HashMap, ++ pending_readers: usize, ++ next_reader_id: u64, ++ closed: bool, ++ #[cfg(test)] ++ waiting_readers: usize, ++} ++ ++impl SingleSlotBroadcastInner { ++ fn lock(&self) -> MutexGuard<'_, SingleSlotBroadcastState> { ++ self.state ++ .lock() ++ .unwrap_or_else(std::sync::PoisonError::into_inner) ++ } ++} ++ ++impl SingleSlotBroadcast { ++ fn new() -> Self { ++ Self { ++ inner: Arc::new(SingleSlotBroadcastInner { ++ state: Mutex::new(SingleSlotBroadcastState { ++ slot: None, ++ readers: HashMap::new(), ++ pending_readers: 0, ++ next_reader_id: 0, ++ closed: false, ++ #[cfg(test)] ++ waiting_readers: 0, ++ }), ++ changed: Condvar::new(), ++ }), ++ } ++ } ++ ++ fn add_rx(&mut self) -> SingleSlotReceiver { ++ let mut state = self.inner.lock(); ++ let reader_id = state.next_reader_id; ++ state.next_reader_id = state ++ .next_reader_id ++ .checked_add(1) ++ .expect("single-slot broadcast reader ID space exhausted"); ++ assert!( ++ state.readers.insert(reader_id, false).is_none(), ++ "single-slot broadcast reader ID reused" ++ ); ++ drop(state); ++ ++ SingleSlotReceiver { ++ inner: Arc::clone(&self.inner), ++ reader_id, ++ } ++ } ++ ++ fn rx_count(&self) -> usize { ++ self.inner.lock().readers.len() ++ } ++ ++ /// Broadcast without blocking. On failure, no receiver observes `value`. ++ fn try_broadcast(&mut self, value: T) -> Result<(), T> { ++ let mut state = self.inner.lock(); ++ debug_assert!(!state.closed, "live sender marked closed"); ++ ++ if state.slot.is_some() { ++ return Err(value); ++ } ++ ++ if state.readers.is_empty() { ++ return Ok(()); ++ } ++ ++ state.pending_readers = state.readers.len(); ++ for pending in state.readers.values_mut() { ++ *pending = true; ++ } ++ state.slot = Some(value); ++ drop(state); ++ self.inner.changed.notify_all(); ++ Ok(()) ++ } ++} ++ ++impl Drop for SingleSlotBroadcast { ++ fn drop(&mut self) { ++ let mut state = self.inner.lock(); ++ state.closed = true; ++ drop(state); ++ self.inner.changed.notify_all(); ++ } ++} ++ ++impl SingleSlotReceiver { ++ fn recv(&mut self) -> Result { ++ let mut state = self.inner.lock(); ++ ++ loop { ++ let Some(is_pending) = state.readers.get(&self.reader_id).copied() else { ++ return Err(std::sync::mpsc::RecvError); ++ }; ++ ++ if is_pending { ++ if state.pending_readers == 1 { ++ state.readers.insert(self.reader_id, false); ++ state.pending_readers = 0; ++ return Ok(state ++ .slot ++ .take() ++ .expect("pending broadcast has no stored value")); ++ } ++ ++ let value = state ++ .slot ++ .as_ref() ++ .expect("pending broadcast has no stored value") ++ .clone(); ++ state.readers.insert(self.reader_id, false); ++ state.pending_readers -= 1; ++ return Ok(value); ++ } ++ ++ // Values already in the slot at sender shutdown remain readable by ++ // their original receivers. Only report disconnect after this ++ // receiver has drained everything it was entitled to receive. ++ if state.closed { ++ return Err(std::sync::mpsc::RecvError); ++ } ++ ++ #[cfg(test)] ++ { ++ state.waiting_readers += 1; ++ } ++ state = self ++ .inner ++ .changed ++ .wait(state) ++ .unwrap_or_else(std::sync::PoisonError::into_inner); ++ #[cfg(test)] ++ { ++ state.waiting_readers -= 1; ++ } ++ } ++ } ++} ++ ++impl Drop for SingleSlotReceiver { ++ fn drop(&mut self) { ++ let mut state = self.inner.lock(); ++ let Some(was_pending) = state.readers.remove(&self.reader_id) else { ++ return; ++ }; ++ ++ if was_pending { ++ state.pending_readers -= 1; ++ if state.pending_readers == 0 { ++ state.slot = None; ++ } ++ } ++ } ++} ++ ++#[derive(Clone)] + struct AllocatedPage { + // The base_ptr is mutable, and will move forward as memory is allocated from the page. + base_ptr: u32, +@@ -336,6 +561,7 @@ struct AllocatedPage { + // out, since each module may request a specific amount of memory to be allocated + // for it before starting it up. + // TODO: Only supports Memory32, should implement proper Memory64 support ++#[derive(Clone)] + struct MemoryAllocator { + allocated_pages: Vec, + } +@@ -525,6 +751,12 @@ pub enum LinkError { + #[error("Bad __tls_base export, expected a global of type I32 or I64")] + BadTlsBaseExport, + ++ #[error("Preinitialized memory image rejected: {0}")] ++ PreinitializedMemoryImage(String), ++ ++ #[error("Shared-memory exec reservation rejected: {0}")] ++ SharedMemoryExecReservation(String), ++ + #[error( + "TLS symbol {0} cannot be resolved from module {1} because it does not export its __tls_base" + )] +@@ -737,7 +969,7 @@ pub enum SymbolResolutionKey { + }, + } + +-#[derive(Debug)] ++#[derive(Debug, Clone)] + pub enum SymbolResolutionResult { + // The symbol was resolved to a global address. We don't resolve again because + // the value of globals and the memory_base for each module and all of its instances +@@ -788,6 +1020,7 @@ enum DlOperation { + }, + } + ++#[derive(Clone)] + struct DlModule { + module: Module, + dylink_info: DylinkInfo, +@@ -816,10 +1049,10 @@ struct InstanceGroupState { + + // Once the dl_operation_pending flag is set, a barrier is created and broadcast + // by the instigating group, which others must use to rendezvous with it. +- recv_pending_operation_barrier: bus::BusReader>, ++ recv_pending_operation_barrier: SingleSlotReceiver>, + // The corresponding sender is stored in the shared linker state, and is used + // by the instigating instance group to broadcast the results. +- recv_pending_operation: bus::BusReader, ++ recv_pending_operation: SingleSlotReceiver, + } + + // There is only one LinkerState for all instance groups +@@ -854,8 +1087,8 @@ struct LinkerState { + + symbol_resolution_records: HashMap, + +- send_pending_operation_barrier: bus::Bus>, +- send_pending_operation: bus::Bus, ++ send_pending_operation_barrier: SingleSlotBroadcast>, ++ send_pending_operation: SingleSlotBroadcast, + } + + /// The linker is responsible for loading and linking dynamic modules at runtime, +@@ -964,12 +1197,15 @@ impl Linker { + func_env: &mut WasiFunctionEnv, + stack_size: u64, + ld_library_path: &[&Path], ++ preinitialized_memory_image: Option, ++ fresh_zeroed_memory: bool, + ) -> Result<(Self, LinkedMainModule), LinkError> { +- let dylink_section = parse_dylink0_section(main_module)?; ++ let dylink_section = { parse_dylink0_section(main_module) }?; + + trace!(?dylink_section, "Loading main module"); + +- let mut imports = import_object_for_all_wasi_versions(main_module, store, &func_env.env); ++ let mut imports = ++ { import_object_for_all_wasi_versions(main_module, store, &func_env.env) }; + + let function_table_type = main_module + .imports() +@@ -993,8 +1229,9 @@ impl Linker { + minimum_size = ?function_table_type.minimum, + "Creating indirect function table" + ); +- let indirect_function_table = Table::new(store, function_table_type, Value::FuncRef(None)) +- .map_err(LinkError::TableAllocationError)?; ++ let indirect_function_table = ++ { Table::new(store, function_table_type, Value::FuncRef(None)) } ++ .map_err(LinkError::TableAllocationError)?; + + let expected_table_length = + dylink_section.mem_info.table_size + MAIN_MODULE_TABLE_BASE as u32; +@@ -1003,8 +1240,7 @@ impl Linker { + let current_size = indirect_function_table.size(store); + let delta = expected_table_length - current_size; + trace!(?current_size, ?delta, "Growing indirect function table"); +- indirect_function_table +- .grow(store, delta, Value::FuncRef(None)) ++ { indirect_function_table.grow(store, delta, Value::FuncRef(None)) } + .map_err(LinkError::TableAllocationError)?; + } + +@@ -1050,15 +1286,60 @@ impl Linker { + + let stack_high = stack_low + stack_size; + +- // Allocate memory for the stack. This does not need to go through the memory allocator +- // because it's always placed directly after the main module's data +- memory.grow_at_least(store, stack_high)?; ++ let exec_reservations = func_env.data(store).state.shared_memory_exec_reservations(); ++ let mut exec_reservation_end = None; ++ for reservation in &exec_reservations { ++ let end = reservation ++ .start ++ .checked_add(reservation.len) ++ .ok_or_else(|| { ++ LinkError::SharedMemoryExecReservation( ++ "reservation range overflowed the guest address space".to_string(), ++ ) ++ })?; ++ if reservation.len == 0 || reservation.start < stack_high { ++ return Err(LinkError::SharedMemoryExecReservation(format!( ++ "reservation [{:#x}, {end:#x}) overlaps main data/stack ending at {stack_high:#x}", ++ reservation.start ++ ))); ++ } ++ exec_reservation_end = Some(exec_reservation_end.map_or(end, |old: u64| old.max(end))); ++ } ++ let initial_memory_end = exec_reservation_end.map_or(stack_high, |end| end.max(stack_high)); ++ ++ // Growing before Instance::new is the actual address reservation. ++ // Patched allocators claim only exact positive-sbrk return intervals, ++ // so this runtime-owned growth can never become malloc-owned space even ++ // when module start code allocates before PostgreSQL reattaches files. ++ // Growth reserves virtual address space; it does not fault every ++ // intervening page into RSS. ++ { ++ reserve_exec_memory_window( ++ &memory, ++ store, ++ initial_memory_end, ++ !exec_reservations.is_empty(), ++ ) ++ }?; ++ if !exec_reservations.is_empty() { ++ func_env ++ .data(store) ++ .state ++ .mark_shared_memory_exec_reservations_installed() ++ .map_err(|errno| { ++ LinkError::SharedMemoryExecReservation(format!( ++ "could not install reserved address window after memory growth: {errno}" ++ )) ++ })?; ++ } + + trace!( + memory_pages = ?memory.grow(store, 0).unwrap(), + memory_base, + stack_low, + stack_high, ++ exec_reservation_end, ++ initial_memory_end, + "Memory layout" + ); + +@@ -1069,14 +1350,15 @@ impl Linker { + "__stack_pointer".to_string(), + ))?; + +- let stack_pointer = define_integer_global_import(store, &stack_pointer_import, stack_high)?; ++ let stack_pointer = ++ { define_integer_global_import(store, &stack_pointer_import, stack_high) }?; + + let c_longjmp = Tag::new(store, vec![Type::I32]); + let cpp_exception = Tag::new(store, vec![Type::I32]); + +- let mut barrier_tx = Bus::new(1); ++ let mut barrier_tx = SingleSlotBroadcast::new(); + let barrier_rx = barrier_tx.add_rx(); +- let mut operation_tx = Bus::new(1); ++ let mut operation_tx = SingleSlotBroadcast::new(); + let operation_rx = operation_tx.add_rx(); + + let mut instance_group = InstanceGroupState { +@@ -1085,7 +1367,7 @@ impl Linker { + // `__tls_base` global export from the instance after instantiation. + main_instance_tls_base: None, + side_instances: HashMap::new(), +- stack_pointer, ++ stack_pointer: stack_pointer.clone(), + memory: memory.clone(), + indirect_function_table: indirect_function_table.clone(), + c_longjmp, +@@ -1122,35 +1404,71 @@ impl Linker { + ]; + + trace!("Resolving main module's symbols"); +- linker_state.resolve_symbols( +- &instance_group, +- store, +- main_module, +- MAIN_MODULE_HANDLE, +- &mut link_state, +- &well_known_imports, +- )?; ++ { ++ linker_state.resolve_symbols( ++ &instance_group, ++ store, ++ main_module, ++ MAIN_MODULE_HANDLE, ++ &mut link_state, ++ &well_known_imports, ++ ) ++ }?; + + trace!("Populating main module's imports object"); +- instance_group.populate_imports_from_link_state( +- MAIN_MODULE_HANDLE, +- &mut linker_state, +- &mut link_state, +- store, +- main_module, +- &mut imports, +- &func_env.env, +- &well_known_imports, +- )?; ++ { ++ instance_group.populate_imports_from_link_state( ++ MAIN_MODULE_HANDLE, ++ &mut linker_state, ++ &mut link_state, ++ store, ++ main_module, ++ &mut imports, ++ &func_env.env, ++ &well_known_imports, ++ ) ++ }?; + + // TODO: figure out which way is faster (stubs in main or stubs in sides), + // use that ordering. My *guess* is that, since main exports all the libc + // functions and those are called frequently by basically any code, then giving + // stubs to main will be faster, but we need numbers before we decide this. +- let main_instance = Instance::new(store, main_module, &imports)?; ++ let main_instance = { ++ Instance::new_with_export_names( ++ store, ++ main_module, ++ &imports, ++ DYNAMIC_INSTANCE_INITIAL_EXPORTS, ++ ) ++ }?; ++ let ordinary_start_completed = OrdinaryModuleStartCompleted { _private: () }; ++ if let Some(image) = preinitialized_memory_image { ++ match image { ++ PreinitializedMemoryImageMode::Apply(image) => image.apply( ++ ordinary_start_completed, ++ main_module, ++ &memory, ++ store, ++ memory_type, ++ &linker_state.main_module_dylink_info, ++ memory_base, ++ stack_low, ++ fresh_zeroed_memory, ++ ), ++ PreinitializedMemoryImageMode::Capture(capture) => capture.capture( ++ main_module, ++ &memory, ++ store, ++ memory_type, ++ &linker_state.main_module_dylink_info, ++ memory_base, ++ stack_low, ++ ), ++ }?; ++ } + instance_group.main_instance = Some(main_instance.clone()); + +- let tls_base = get_tls_base_export(&main_instance, store)?; ++ let tls_base = { get_tls_base_export(&main_instance, store) }?; + instance_group.main_instance_tls_base = tls_base; + + let runtime_path = linker_state.main_module_dylink_info.runtime_path.clone(); +@@ -1160,23 +1478,25 @@ impl Linker { + // guard.resolve_imports. + trace!(name = needed, "Loading module needed by main"); + let wasi_env = func_env.data(store); +- linker_state.load_module_tree( +- DlModuleSpec::FileSystem { +- module_spec: Path::new(needed.as_str()), +- ld_library_path, +- }, +- &mut link_state, +- &wasi_env.runtime, +- &wasi_env.state, +- runtime_path.as_ref(), +- // HACK: The main module doesn't have to exist in the virtual FS at all; e.g. +- // if one runs `wasmer ../module.wasm --volume .`, we won't have access to the +- // main module's folder within the virtual FS. This is why we're picking PWD +- // as the $ORIGIN of the main module, which should at least be slightly +- // sensible. The `main.wasm` file name will be stripped and only the `./` +- // will be taken into account by `locate_module`. +- Some(Path::new("./main.wasm")), +- )?; ++ { ++ linker_state.load_module_tree( ++ DlModuleSpec::FileSystem { ++ module_spec: Path::new(needed.as_str()), ++ ld_library_path, ++ }, ++ &mut link_state, ++ &wasi_env.runtime, ++ &wasi_env.state, ++ runtime_path.as_ref(), ++ // HACK: The main module doesn't have to exist in the virtual FS at all; e.g. ++ // if one runs `wasmer ../module.wasm --volume .`, we won't have access to the ++ // main module's folder within the virtual FS. This is why we're picking PWD ++ // as the $ORIGIN of the main module, which should at least be slightly ++ // sensible. The `main.wasm` file name will be stripped and only the `./` ++ // will be taken into account by `locate_module`. ++ Some(Path::new("./main.wasm")), ++ ) ++ }?; + } + + for module_handle in link_state +@@ -1186,13 +1506,15 @@ impl Linker { + .collect::>() + { + trace!(?module_handle, "Instantiating module"); +- instance_group.instantiate_side_module_from_link_state( +- &mut linker_state, +- store, +- &func_env.env, +- &mut link_state, +- module_handle, +- )?; ++ { ++ instance_group.instantiate_side_module_from_link_state( ++ &mut linker_state, ++ store, ++ &func_env.env, ++ &mut link_state, ++ module_handle, ++ ) ++ }?; + } + + let linker = Self { +@@ -1208,25 +1530,28 @@ impl Linker { + guard_size: 0, + tls_base, + }; ++ let mut main_module_instance_handles = WasiModuleInstanceHandles::new( ++ memory.clone(), ++ store, ++ main_instance.clone(), ++ Some(indirect_function_table.clone()), ++ ); ++ main_module_instance_handles.stack_pointer = Some(stack_pointer); + let module_handles = WasiModuleTreeHandles::Dynamic { + linker: linker.clone(), +- main_module_instance_handles: WasiModuleInstanceHandles::new( +- memory.clone(), +- store, +- main_instance.clone(), +- Some(indirect_function_table.clone()), +- ), ++ main_module_instance_handles, + }; + +- func_env +- .initialize_handles_and_layout( ++ { ++ func_env.initialize_handles_and_layout( + store, + main_instance.clone(), + module_handles, + Some(stack_layout), + true, + ) +- .map_err(LinkError::MainModuleHandleInitFailed)?; ++ } ++ .map_err(LinkError::MainModuleHandleInitFailed)?; + + { + trace!(?link_state, "Finalizing linking of main module"); +@@ -1235,23 +1560,33 @@ impl Linker { + let mut linker_state = linker.linker_state.write().unwrap(); + + let group_state = group_guard.as_mut().unwrap(); +- group_state.finalize_pending_globals( +- &mut linker_state, +- store, +- &link_state.unresolved_globals, +- )?; ++ { ++ group_state.finalize_pending_globals( ++ &mut linker_state, ++ store, ++ &link_state.unresolved_globals, ++ ) ++ }?; + + // The main module isn't added to the link state's list of new modules, so we need to + // call its initialization functions separately + trace!("Calling data relocator function for main module"); +- call_initialization_function::<()>(&main_instance, store, "__wasm_apply_data_relocs")?; +- call_initialization_function::<()>(&main_instance, store, "__wasm_apply_tls_relocs")?; ++ { ++ call_initialization_function::<()>( ++ &main_instance, ++ store, ++ "__wasm_apply_data_relocs", ++ ) ++ }?; ++ { ++ call_initialization_function::<()>(&main_instance, store, "__wasm_apply_tls_relocs") ++ }?; + +- linker.initialize_new_modules(group_guard, store, link_state)?; ++ { linker.initialize_new_modules(group_guard, store, link_state) }?; + } + + trace!("Calling main module's _initialize function"); +- call_initialization_function::<()>(&main_instance, store, "_initialize")?; ++ { call_initialization_function::<()>(&main_instance, store, "_initialize") }?; + + trace!("Link complete"); + +@@ -1289,11 +1624,14 @@ impl Linker { + let parent_store = parent_ctx.as_store_mut(); + + let main_module = linker_state.main_module.clone(); +- let memory = parent_group_state +- .memory +- .share_in_store(&parent_store, store)?; ++ let memory = { ++ parent_group_state ++ .memory ++ .share_in_store(&parent_store, store) ++ }?; + +- let mut imports = import_object_for_all_wasi_versions(&main_module, store, &func_env.env); ++ let mut imports = ++ { import_object_for_all_wasi_versions(&main_module, store, &func_env.env) }; + + let indirect_function_table_type = + parent_group_state.indirect_function_table.ty(&parent_store); +@@ -1303,7 +1641,7 @@ impl Linker { + ); + + let indirect_function_table = +- Table::new(store, indirect_function_table_type, Value::FuncRef(None)) ++ { Table::new(store, indirect_function_table_type, Value::FuncRef(None)) } + .map_err(LinkError::TableAllocationError)?; + + let expected_table_length = parent_group_state +@@ -1314,8 +1652,7 @@ impl Linker { + let current_size = indirect_function_table.size(store); + let delta = expected_table_length - current_size; + trace!(?current_size, ?delta, "Growing indirect function table"); +- indirect_function_table +- .grow(store, delta, Value::FuncRef(None)) ++ { indirect_function_table.grow(store, delta, Value::FuncRef(None)) } + .map_err(LinkError::TableAllocationError)?; + } + +@@ -1329,13 +1666,7 @@ impl Linker { + // FIXME: this needs to become a parameter if we ever decouple the linker from WASIX + let (stack_low, stack_high, tls_base) = { + let layout = &func_env.env.as_ref(store).layout; +- ( +- layout.stack_lower, +- layout.stack_upper, +- layout.tls_base.expect( +- "tls_base must be set in memory layout of new instance group's main instance", +- ), +- ) ++ (layout.stack_lower, layout.stack_upper, layout.tls_base) + }; + + trace!(stack_low, stack_high, "Memory layout"); +@@ -1359,9 +1690,9 @@ impl Linker { + + let mut instance_group = InstanceGroupState { + main_instance: None, +- main_instance_tls_base: Some(tls_base), ++ main_instance_tls_base: tls_base, + side_instances: HashMap::new(), +- stack_pointer, ++ stack_pointer: stack_pointer.clone(), + memory: memory.clone(), + indirect_function_table: indirect_function_table.clone(), + c_longjmp, +@@ -1381,37 +1712,48 @@ impl Linker { + ]; + + trace!("Populating imports object for new instance group's main instance"); +- instance_group.populate_imports_from_linker( +- MAIN_MODULE_HANDLE, +- &linker_state, +- store, +- &main_module, +- &mut imports, +- &func_env.env, +- &well_known_imports, +- &mut pending_resolutions, +- )?; ++ { ++ instance_group.populate_imports_from_linker( ++ MAIN_MODULE_HANDLE, ++ &linker_state, ++ store, ++ &main_module, ++ &mut imports, ++ &func_env.env, ++ &well_known_imports, ++ &mut pending_resolutions, ++ ) ++ }?; + +- let main_instance = Instance::new(store, &main_module, &imports)?; ++ let main_instance = { ++ Instance::new_with_export_names( ++ store, ++ &main_module, ++ &imports, ++ DYNAMIC_INSTANCE_INITIAL_EXPORTS, ++ ) ++ }?; + + instance_group.main_instance = Some(main_instance.clone()); + + for side in &linker_state.side_modules { + trace!(module_handle = ?side.0, "Instantiating existing side module"); +- instance_group.instantiate_side_module_from_linker( +- &linker_state, +- store, +- &func_env.env, +- *side.0, +- &mut pending_resolutions, +- )?; ++ { ++ instance_group.instantiate_side_module_from_linker( ++ &linker_state, ++ store, ++ &func_env.env, ++ *side.0, ++ &mut pending_resolutions, ++ ) ++ }?; + } + + trace!("Finalizing pending functions"); +- instance_group.finalize_pending_resolutions_from_linker(&pending_resolutions, store)?; ++ { instance_group.finalize_pending_resolutions_from_linker(&pending_resolutions, store) }?; + + trace!("Applying externally-requested function table entries"); +- instance_group.apply_requested_symbols_from_linker(store, &linker_state)?; ++ { instance_group.apply_requested_symbols_from_linker(store, &linker_state) }?; + + let linker = Self { + linker_state: self.linker_state.clone(), +@@ -1419,25 +1761,28 @@ impl Linker { + dl_operation_pending: self.dl_operation_pending.clone(), + }; + ++ let mut main_module_instance_handles = WasiModuleInstanceHandles::new( ++ memory.clone(), ++ store, ++ main_instance.clone(), ++ Some(indirect_function_table.clone()), ++ ); ++ main_module_instance_handles.stack_pointer = Some(stack_pointer); + let module_handles = WasiModuleTreeHandles::Dynamic { + linker: linker.clone(), +- main_module_instance_handles: WasiModuleInstanceHandles::new( +- memory.clone(), +- store, +- main_instance.clone(), +- Some(indirect_function_table.clone()), +- ), ++ main_module_instance_handles, + }; + +- func_env +- .initialize_handles_and_layout( ++ { ++ func_env.initialize_handles_and_layout( + store, + main_instance.clone(), + module_handles, + None, + false, + ) +- .map_err(LinkError::MainModuleHandleInitFailed)?; ++ } ++ .map_err(LinkError::MainModuleHandleInitFailed)?; + + trace!("Instance group spawned successfully"); + +@@ -2056,6 +2401,26 @@ impl Linker { + } + } + ++/// Reserves a fresh exec image's initial address window only after proving ++/// that inherited shared mappings can remain attached for the memory's full ++/// lifetime. The caller still owns rollback authority when this runs. ++fn reserve_exec_memory_window( ++ memory: &Memory, ++ store: &mut impl AsStoreMut, ++ initial_memory_end: u64, ++ has_exec_reservations: bool, ++) -> Result<(), LinkError> { ++ if has_exec_reservations && !memory.supports_persistent_shared_fixed_remap(store) { ++ return Err(LinkError::SharedMemoryExecReservation( ++ "inherited shared mappings require a supported nonmoving static linear memory" ++ .to_string(), ++ )); ++ } ++ ++ memory.grow_at_least(store, initial_memory_end)?; ++ Ok(()) ++} ++ + impl LinkerState { + fn allocate_memory( + &mut self, +@@ -2203,7 +2568,7 @@ impl LinkerState { + &self, + group: &InstanceGroupState, + import: &ImportType, +- store: &impl AsStoreRef, ++ store: &mut impl AsStoreMut, + ) -> Result { + let ExternType::Function(import_func_ty) = import.ty() else { + return Err(LinkError::ImportMustBeFunction( +@@ -2212,16 +2577,17 @@ impl LinkerState { + )); + }; + +- let export = group.resolve_exported_symbol(import.name()); ++ let export = group.resolve_exported_symbol(store, import.name()); + + match export { + Some((module_handle, export)) => { ++ let export_ty = export.ty(store); + let Extern::Function(export_func) = export else { + return Err(LinkError::ImportTypeMismatch( + "env".to_string(), + import.name().to_string(), + ExternType::Function(import_func_ty.clone()), +- export.ty(store).clone(), ++ export_ty, + )); + }; + +@@ -2230,7 +2596,7 @@ impl LinkerState { + "env".to_string(), + import.name().to_string(), + ExternType::Function(import_func_ty.clone()), +- export.ty(store).clone(), ++ ExternType::Function(export_func.ty(store)), + )); + } + +@@ -2255,18 +2621,19 @@ impl LinkerState { + &self, + group: &InstanceGroupState, + import: &ImportType, +- store: &impl AsStoreRef, ++ store: &mut impl AsStoreMut, + ) -> Result { + let global_type = get_integer_global_type_from_import(import)?; + +- match group.resolve_exported_symbol(import.name()) { ++ match group.resolve_exported_symbol(store, import.name()) { + Some((module_handle, export)) => { +- let ExternType::Global(global_type) = export.ty(store) else { ++ let export_ty = export.ty(store); ++ let ExternType::Global(global_type) = export_ty else { + return Err(LinkError::ImportTypeMismatch( + "GOT.mem".to_string(), + import.name().to_string(), + ExternType::Global(global_type), +- export.ty(store).clone(), ++ export.ty(store), + )); + }; + +@@ -2275,7 +2642,7 @@ impl LinkerState { + "GOT.mem".to_string(), + import.name().to_string(), + ExternType::Global(global_type), +- export.ty(store).clone(), ++ export.ty(store), + )); + } + +@@ -2292,12 +2659,12 @@ impl LinkerState { + &self, + group: &InstanceGroupState, + import: &ImportType, +- store: &impl AsStoreRef, ++ store: &mut impl AsStoreMut, + ) -> Result { + // Ensure the global is the correct type (i32 or i64) + let _ = get_integer_global_type_from_import(import)?; + +- match group.resolve_exported_symbol(import.name()) { ++ match group.resolve_exported_symbol(store, import.name()) { + Some((module_handle, export)) => { + let ExternType::Function(_) = export.ty(store) else { + return Err(LinkError::ExportMustBeFunction( +@@ -2643,7 +3010,12 @@ impl InstanceGroupState { + &well_known_imports, + )?; + +- let instance = Instance::new(store, &module, &imports)?; ++ let instance = Instance::new_with_export_names( ++ store, ++ &module, ++ &imports, ++ DYNAMIC_INSTANCE_INITIAL_EXPORTS, ++ )?; + + let instance_handles = WasiModuleInstanceHandles::new( + self.memory.clone(), +@@ -2756,7 +3128,12 @@ impl InstanceGroupState { + pending_resolutions, + )?; + +- let instance = Instance::new(store, &dl_module.module, &imports)?; ++ let instance = Instance::new_with_export_names( ++ store, ++ &dl_module.module, ++ &imports, ++ DYNAMIC_INSTANCE_INITIAL_EXPORTS, ++ )?; + + // This is a non-main instance of a side module, so it needs a new TLS area + let tls_base = call_initialization_function::(&instance, store, "__wasix_init_tls")? +@@ -2794,18 +3171,19 @@ impl InstanceGroupState { + trace!("Finalizing pending functions"); + + for pending in &pending_resolutions.functions { +- let func = self +- .instance(pending.resolved_from) +- .exports +- .get_function(&pending.name) +- .unwrap_or_else(|e| { +- panic!( +- "Internal error: failed to resolve exported function {}: {e:?}", +- pending.name +- ) +- }); ++ let func = lookup_exported_function( ++ self.instance(pending.resolved_from), ++ store, ++ &pending.name, ++ ) ++ .unwrap_or_else(|e| { ++ panic!( ++ "Internal error: failed to resolve exported function {}: {e:?}", ++ pending.name ++ ) ++ }); + +- self.place_in_function_table_at(store, func.clone(), pending.function_table_index) ++ self.place_in_function_table_at(store, func, pending.function_table_index) + .map_err(LinkError::TableAllocationError)?; + + trace!(?pending, "Placed pending function in table"); +@@ -2849,11 +3227,11 @@ impl InstanceGroupState { + panic!("Internal error: module {resolved_from:?} not loaded by this group") + }); + +- let func = instance.exports.get_function(name).unwrap_or_else(|e| { ++ let func = lookup_exported_function(instance, store, name).unwrap_or_else(|e| { + panic!("Internal error: failed to resolve exported function {name}: {e:?}") + }); + +- self.place_in_function_table_at(store, func.clone(), function_table_index) ++ self.place_in_function_table_at(store, func, function_table_index) + .map_err(LinkError::TableAllocationError)?; + + Ok(()) +@@ -2936,24 +3314,27 @@ impl InstanceGroupState { + Ok(()) + } + +- fn resolve_exported_symbol(&self, symbol: &str) -> Option<(ModuleHandle, &Extern)> { +- if let Some(export) = self +- .main_instance() +- .and_then(|instance| instance.exports.get_extern(symbol)) ++ fn resolve_exported_symbol( ++ &self, ++ store: &mut impl AsStoreMut, ++ symbol: &str, ++ ) -> Option<(ModuleHandle, Extern)> { ++ if let Some(instance) = self.main_instance() ++ && let Some(export) = instance.lookup_export(store, symbol) + { + trace!(symbol, from = ?MAIN_MODULE_HANDLE, ?export, "Resolved exported symbol"); +- Some((MAIN_MODULE_HANDLE, export)) +- } else { +- for (handle, dl_instance) in &self.side_instances { +- if let Some(export) = dl_instance.instance.exports.get_extern(symbol) { +- trace!(symbol, from = ?handle, ?export, "Resolved exported symbol"); +- return Some((*handle, export)); +- } +- } ++ return Some((MAIN_MODULE_HANDLE, export)); ++ } + +- trace!(symbol, "Failed to resolve exported symbol"); +- None ++ for (handle, dl_instance) in &self.side_instances { ++ if let Some(export) = dl_instance.instance.lookup_export(store, symbol) { ++ trace!(symbol, from = ?handle, ?export, "Resolved exported symbol"); ++ return Some((*handle, export)); ++ } + } ++ ++ trace!(symbol, "Failed to resolve exported symbol"); ++ None + } + + // This function populates the imports object for a single module from the given +@@ -3108,11 +3489,12 @@ impl InstanceGroupState { + + match resolution { + InProgressSymbolResolution::Function(module_handle) => { +- let func = self +- .instance(*module_handle) +- .exports +- .get_function(import.name()) +- .expect("Internal error: bad in-progress symbol resolution"); ++ let func = lookup_exported_function( ++ self.instance(*module_handle), ++ store, ++ import.name(), ++ ) ++ .expect("Internal error: bad in-progress symbol resolution"); + imports.define(import.module(), import.name(), func.clone()); + linker_state.symbol_resolution_records.insert( + SymbolResolutionKey::Needed(key.clone()), +@@ -3207,14 +3589,15 @@ impl InstanceGroupState { + } + + InProgressSymbolResolution::FuncGlobal(module_handle) => { +- let func = self +- .instance(*module_handle) +- .exports +- .get_function(import.name()) +- .expect("Internal error: bad in-progress symbol resolution"); ++ let func = lookup_exported_function( ++ self.instance(*module_handle), ++ store, ++ import.name(), ++ ) ++ .expect("Internal error: bad in-progress symbol resolution"); + + let func_handle = self +- .append_to_function_table(store, func.clone()) ++ .append_to_function_table(store, func) + .map_err(LinkError::TableAllocationError)?; + trace!( + ?module_handle, +@@ -3408,11 +3791,8 @@ impl InstanceGroupState { + ?resolved_from, + "Already have instance to resolve from" + ); +- instance +- .exports +- .get_function(import.name()) ++ lookup_exported_function(instance, store, import.name()) + .expect("Internal error: failed to get exported function") +- .clone() + } + // We may be loading a module tree, and the instance from which + // we're supposed to import the function may not exist yet, so +@@ -3452,15 +3832,14 @@ impl InstanceGroupState { + function_table_index, + } => { + let func = self.try_instance(*resolved_from).map(|instance| { +- instance +- .exports +- .get_function(import.name()) +- .unwrap_or_else(|e| { ++ lookup_exported_function(instance, store, import.name()).unwrap_or_else( ++ |e| { + panic!( + "Internal error: failed to resolve function {}: {e:?}", + import.name() + ) +- }) ++ }, ++ ) + }); + match func { + Some(func) => { +@@ -3470,12 +3849,8 @@ impl InstanceGroupState { + function_table_index, + "Placing function pointer into table" + ); +- self.place_in_function_table_at( +- store, +- func.clone(), +- *function_table_index, +- ) +- .map_err(LinkError::TableAllocationError)?; ++ self.place_in_function_table_at(store, func, *function_table_index) ++ .map_err(LinkError::TableAllocationError)?; + } + None => { + trace!( +@@ -3620,7 +3995,7 @@ impl InstanceGroupState { + allow_hidden: bool, + ) -> Result { + trace!(from = ?module_handle, symbol, "Resolving export from instance"); +- let export = instance.exports.get_extern(symbol).ok_or_else(|| { ++ let export = instance.lookup_export(store, symbol).ok_or_else(|| { + trace!(from = ?module_handle, symbol, "Not found"); + ResolveError::MissingExport + })?; +@@ -3638,12 +4013,15 @@ impl InstanceGroupState { + match export.ty(store) { + ExternType::Function(_) => { + trace!(from = ?module_handle, symbol, "Found function"); +- Ok(PartiallyResolvedExport::Function( +- Function::get_self_from_extern(export).unwrap().clone(), +- )) ++ let Extern::Function(function) = export else { ++ unreachable!("export type and value disagreed") ++ }; ++ Ok(PartiallyResolvedExport::Function(function)) + } + ty @ ExternType::Global(_) => { +- let global = Global::get_self_from_extern(export).unwrap(); ++ let Extern::Global(global) = export else { ++ unreachable!("export type and value disagreed") ++ }; + let value = match global.get(store) { + Value::I32(value) => value as u64, + Value::I64(value) => value as u64, +@@ -3712,7 +4090,7 @@ impl InstanceGroupState { + None => { + trace!(?requesting_module, name, "Resolving stub function"); + +- let (data, store) = env.data_and_store_mut(); ++ let (data, mut store) = env.data_and_store_mut(); + let env_inner = data.inner(); + // Safe to unwrap since we already know we're doing DL + let linker = env_inner.linker().unwrap(); +@@ -3777,12 +4155,12 @@ impl InstanceGroupState { + return Err(mk_error()); + } + +- let func = group_state +- .instance(*resolved_from) +- .exports +- .get_function(&name) +- .unwrap() +- .clone(); ++ let func = lookup_exported_function( ++ group_state.instance(*resolved_from), ++ &mut store, ++ &name, ++ ) ++ .unwrap(); + *resolved_guard = Some(Some(func.clone())); + func + } +@@ -3790,7 +4168,7 @@ impl InstanceGroupState { + trace!(?requesting_module, name, "Resolving function"); + + let Some((resolved_from, export)) = +- group_state.resolve_exported_symbol(name.as_str()) ++ group_state.resolve_exported_symbol(&mut store, name.as_str()) + else { + trace!(?requesting_module, name, "Failed to resolve symbol"); + *resolved_guard = Some(None); +@@ -4257,13 +4635,44 @@ fn set_integer_global( + Ok(()) + } + ++/// Resolves a function from either the small eager set or the VM's complete ++/// export metadata. Returning an owned handle is important: deferred lookup ++/// may populate the instance's identity cache while the caller holds only an ++/// immutable `Instance` reference. ++fn lookup_exported_function( ++ instance: &Instance, ++ store: &mut impl AsStoreMut, ++ name: &str, ++) -> Result { ++ match instance.lookup_export(store, name) { ++ Some(Extern::Function(function)) => Ok(function), ++ Some(_) => Err(ExportError::IncompatibleType), ++ None => Err(ExportError::Missing(name.to_string())), ++ } ++} ++ ++fn lookup_exported_global( ++ instance: &Instance, ++ store: &mut impl AsStoreMut, ++ name: &str, ++) -> Result { ++ match instance.lookup_export(store, name) { ++ Some(Extern::Global(global)) => Ok(global), ++ Some(_) => Err(ExportError::IncompatibleType), ++ None => Err(ExportError::Missing(name.to_string())), ++ } ++} ++ + fn call_initialization_function( + instance: &Instance, + store: &mut impl AsStoreMut, + name: &str, + ) -> Result, LinkError> { +- match instance.exports.get_typed_function::<(), Ret>(store, name) { +- Ok(f) => { ++ match lookup_exported_function(instance, store, name) { ++ Ok(function) => { ++ let f = function ++ .typed::<(), Ret>(store) ++ .map_err(|_| LinkError::InitFuncWithInvalidSignature(name.to_string()))?; + let ret = f + .call(store) + .map_err(|e| LinkError::InitFunctionFailed(name.to_string(), e))?; +@@ -4280,7 +4689,7 @@ fn get_tls_base_export( + instance: &Instance, + store: &mut impl AsStoreMut, + ) -> Result, LinkError> { +- match instance.exports.get_global("__tls_base") { ++ match lookup_exported_global(instance, store, "__tls_base") { + Ok(global) => match global.get(store) { + Value::I32(x) => Ok(Some(x as u64)), + Value::I64(x) => Ok(Some(x as u64)), +@@ -4291,6 +4700,143 @@ fn get_tls_base_export( + } + } + ++#[cfg(test)] ++mod dynamic_instance_export_tests { ++ use std::collections::HashSet; ++ ++ use super::DYNAMIC_INSTANCE_INITIAL_EXPORTS; ++ ++ #[test] ++ fn eager_exports_cover_every_runtime_handle_lookup_without_duplicates() { ++ const HANDLE_EXPORTS: &[&str] = &[ ++ "_start", ++ "_initialize", ++ "wasi_thread_start", ++ "__wasm_signal", ++ "asyncify_start_unwind", ++ "asyncify_stop_unwind", ++ "asyncify_start_rewind", ++ "asyncify_stop_rewind", ++ "asyncify_get_state", ++ ]; ++ ++ let eager: HashSet<_> = DYNAMIC_INSTANCE_INITIAL_EXPORTS.iter().copied().collect(); ++ assert_eq!(eager.len(), DYNAMIC_INSTANCE_INITIAL_EXPORTS.len()); ++ for name in HANDLE_EXPORTS { ++ assert!( ++ eager.contains(name), ++ "runtime handle export {name} must be materialized eagerly" ++ ); ++ } ++ } ++} ++ ++#[cfg(test)] ++mod single_slot_broadcast_tests { ++ use std::{sync::Arc, thread}; ++ ++ use super::{SingleSlotBroadcast, SingleSlotBroadcastInner}; ++ ++ #[test] ++ fn delivers_every_broadcast_to_every_existing_reader() { ++ let mut sender = SingleSlotBroadcast::new(); ++ let mut first = sender.add_rx(); ++ let mut second = sender.add_rx(); ++ ++ assert_eq!(sender.try_broadcast(41), Ok(())); ++ assert_eq!(first.recv(), Ok(41)); ++ assert_eq!(second.recv(), Ok(41)); ++ ++ assert_eq!(sender.try_broadcast(42), Ok(())); ++ assert_eq!(second.recv(), Ok(42)); ++ assert_eq!(first.recv(), Ok(42)); ++ } ++ ++ #[test] ++ fn rejects_a_full_slot_without_partial_delivery() { ++ let mut sender = SingleSlotBroadcast::new(); ++ let mut first = sender.add_rx(); ++ let mut second = sender.add_rx(); ++ ++ assert_eq!(sender.try_broadcast("first"), Ok(())); ++ assert_eq!(first.recv(), Ok("first")); ++ assert_eq!(sender.try_broadcast("rejected"), Err("rejected")); ++ assert_eq!(second.recv(), Ok("first")); ++ ++ assert_eq!(sender.try_broadcast("next"), Ok(())); ++ assert_eq!(first.recv(), Ok("next")); ++ assert_eq!(second.recv(), Ok("next")); ++ } ++ ++ #[test] ++ fn reader_added_to_an_occupied_channel_only_sees_future_values() { ++ let mut sender = SingleSlotBroadcast::new(); ++ let mut existing = sender.add_rx(); ++ assert_eq!(sender.try_broadcast(1), Ok(())); ++ ++ let mut added_later = sender.add_rx(); ++ assert_eq!(existing.recv(), Ok(1)); ++ assert_eq!(sender.try_broadcast(2), Ok(())); ++ assert_eq!(added_later.recv(), Ok(2)); ++ assert_eq!(existing.recv(), Ok(2)); ++ } ++ ++ #[test] ++ fn dropping_readers_updates_count_and_releases_the_slot() { ++ let mut sender = SingleSlotBroadcast::new(); ++ let first = sender.add_rx(); ++ let second = sender.add_rx(); ++ assert_eq!(sender.rx_count(), 2); ++ ++ assert_eq!(sender.try_broadcast(1), Ok(())); ++ drop(first); ++ assert_eq!(sender.rx_count(), 1); ++ assert_eq!(sender.try_broadcast(2), Err(2)); ++ ++ drop(second); ++ assert_eq!(sender.rx_count(), 0); ++ assert_eq!(sender.try_broadcast(3), Ok(())); ++ ++ let mut replacement = sender.add_rx(); ++ assert_eq!(sender.rx_count(), 1); ++ assert_eq!(sender.try_broadcast(4), Ok(())); ++ assert_eq!(replacement.recv(), Ok(4)); ++ } ++ ++ #[test] ++ fn closing_sender_wakes_a_blocked_reader() { ++ let mut sender = SingleSlotBroadcast::::new(); ++ let mut receiver = sender.add_rx(); ++ let inner = Arc::clone(&receiver.inner); ++ ++ let waiter = thread::spawn(move || receiver.recv()); ++ wait_until_reader_blocks(&inner); ++ drop(sender); ++ ++ assert_eq!(waiter.join().unwrap(), Err(std::sync::mpsc::RecvError)); ++ } ++ ++ #[test] ++ fn closing_sender_preserves_an_outstanding_delivery() { ++ let mut sender = SingleSlotBroadcast::new(); ++ let mut receiver = sender.add_rx(); ++ assert_eq!(sender.try_broadcast(42), Ok(())); ++ drop(sender); ++ ++ assert_eq!(receiver.recv(), Ok(42)); ++ assert_eq!(receiver.recv(), Err(std::sync::mpsc::RecvError)); ++ } ++ ++ fn wait_until_reader_blocks(inner: &Arc>) { ++ loop { ++ if inner.lock().waiting_readers != 0 { ++ return; ++ } ++ thread::yield_now(); ++ } ++ } ++} ++ + #[cfg(test)] + mod memory_allocator_tests { + use wasmer::{Engine, Memory, Store}; +@@ -4349,3 +4895,32 @@ mod memory_allocator_tests { + assert_eq!(addr, 2 * WASM_PAGE_SIZE + 512); + } + } ++ ++#[cfg(all(test, feature = "sys"))] ++mod shared_exec_memory_capability_tests { ++ use wasmer::sys::{BaseTunables, NativeEngineExt}; ++ use wasmer::{Engine, Memory, MemoryType, Pages, Store, WASM_PAGE_SIZE}; ++ ++ use super::{LinkError, reserve_exec_memory_window}; ++ ++ #[test] ++ fn unsupported_remap_capability_rejects_before_exec_window_growth() { ++ let mut engine = Engine::default(); ++ engine.set_tunables(BaseTunables { ++ static_memory_bound: Pages(0), ++ static_memory_offset_guard_size: 0, ++ dynamic_memory_offset_guard_size: 0, ++ }); ++ let mut store = Store::new(engine); ++ let memory = ++ Memory::new(&mut store, MemoryType::new(Pages(1), Some(Pages(2)), true)).unwrap(); ++ let size_before = memory.size(&store); ++ ++ let error = ++ reserve_exec_memory_window(&memory, &mut store, (WASM_PAGE_SIZE * 2) as u64, true) ++ .expect_err("dynamic memory must be rejected before exec reservation growth"); ++ ++ assert!(matches!(error, LinkError::SharedMemoryExecReservation(_))); ++ assert_eq!(memory.size(&store), size_before); ++ } ++} +diff --git a/lib/wasix/src/state/mod.rs b/lib/wasix/src/state/mod.rs +index fc3463c..c74fda1 100644 +--- a/lib/wasix/src/state/mod.rs ++++ b/lib/wasix/src/state/mod.rs +@@ -21,12 +21,14 @@ mod env; + mod func_env; + mod handles; + mod linker; ++mod preinitialized_memory_image; + mod types; + + use std::{ + collections::{BTreeMap, HashMap}, ++ ops::Bound::{Excluded, Unbounded}, + path::Path, +- sync::Mutex, ++ sync::{Arc, Mutex, Weak}, + task::Waker, + time::Duration, + }; +@@ -42,6 +44,18 @@ pub use self::{ + builder::*, + env::{WasiEnv, WasiEnvInit, WasiModuleInstanceHandles, WasiModuleTreeHandles}, + func_env::WasiFunctionEnv, ++ preinitialized_memory_image::{ ++ DETERMINISTIC_START_ANALYZER_POLICY, DETERMINISTIC_START_GLOBAL_EFFECTS, ++ DETERMINISTIC_START_MEMORY_EFFECTS, DETERMINISTIC_START_MEMORY_READS, ++ DETERMINISTIC_START_PROOF_SCHEMA, DETERMINISTIC_START_TABLE_EFFECTS, ++ DeterministicStartProof, IntrinsicFileImmutability, PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, ++ PREINITIALIZED_MEMORY_IMAGE_PHASE, PreinitializedMemoryImage, ++ PreinitializedMemoryImageBacking, PreinitializedMemoryImageCapture, ++ PreinitializedMemoryImageHandle, PreinitializedMemoryImageLoadAudit, ++ PreinitializedMemoryImageLoader, PreinitializedMemoryImageMetadata, ++ PreinitializedMemoryImageMode, PreinitializedMemoryImageRuntimeAudit, ++ intrinsic_file_immutability, ++ }, + types::*, + }; + pub use crate::fs::{InodeGuard, InodeWeakGuard}; +@@ -121,6 +135,561 @@ pub(crate) struct WasiFutexState { + pub futexes: HashMap, + } + ++#[derive(Debug, Clone, Copy, Hash, Eq, Ord, PartialEq, PartialOrd)] ++pub(crate) struct WasiSharedFileIdentity { ++ device: u64, ++ file: u64, ++} ++ ++impl WasiSharedFileIdentity { ++ #[cfg(unix)] ++ fn for_file(file: &std::fs::File) -> Result { ++ use std::os::unix::fs::MetadataExt; ++ ++ let metadata = file.metadata().map_err(|_| Errno::Io)?; ++ Ok(Self { ++ device: metadata.dev(), ++ file: metadata.ino(), ++ }) ++ } ++ ++ #[cfg(windows)] ++ fn for_file(file: &std::fs::File) -> Result { ++ use std::os::windows::fs::MetadataExt; ++ ++ let metadata = file.metadata().map_err(|_| Errno::Io)?; ++ Ok(Self { ++ device: metadata.volume_serial_number().ok_or(Errno::Notsup)? as u64, ++ file: metadata.file_index().ok_or(Errno::Notsup)?, ++ }) ++ } ++ ++ #[cfg(not(any(unix, windows)))] ++ fn for_file(_file: &std::fs::File) -> Result { ++ Err(Errno::Notsup) ++ } ++} ++ ++/// Number of weak shared-futex registry slots inspected on each lookup. ++/// ++/// Registries normally remove their own slot when the last live mapping or ++/// waiter drops. This bounded sweep is a deterministic backstop for a lookup ++/// racing that final drop without making mmap latency proportional to the ++/// lifetime number of mapped files. ++const SHARED_FUTEX_REGISTRY_PRUNE_BUDGET: usize = 16; ++ ++#[derive(Debug)] ++pub(crate) struct WasiSharedFutexRegistry { ++ futexs: Mutex, ++ ++ // The mapping's live file description is the generation token for the ++ // device/inode identity above. Keeping the same Arc with the wait registry ++ // prevents the backing inode (or Windows file index) from being recycled ++ // while a mapping or an in-flight futex wait can still reach old wait state. ++ _file_anchor: Arc, ++ identity: WasiSharedFileIdentity, ++ generation: Arc<()>, ++ owner: Weak>, ++} ++ ++impl WasiSharedFutexRegistry { ++ #[cfg(test)] ++ fn detached(file_anchor: Arc) -> Arc { ++ Arc::new(Self { ++ futexs: Default::default(), ++ identity: WasiSharedFileIdentity::for_file(&file_anchor).unwrap(), ++ _file_anchor: file_anchor, ++ generation: Arc::new(()), ++ owner: Weak::new(), ++ }) ++ } ++} ++ ++impl Drop for WasiSharedFutexRegistry { ++ fn drop(&mut self) { ++ let Some(owner) = self.owner.upgrade() else { ++ return; ++ }; ++ ++ // Drop must not panic while unwinding. Recovering a poisoned table is ++ // safe here because removing this exact generation is self-contained. ++ let mut owner = owner.lock().unwrap_or_else(|err| err.into_inner()); ++ owner.remove_if_current(self.identity, &self.generation); ++ } ++} ++ ++#[derive(Debug)] ++struct WasiSharedFutexRegistryEntry { ++ generation: Arc<()>, ++ registry: Weak, ++} ++ ++#[derive(Debug, Default)] ++pub(crate) struct WasiSharedFutexRegistries { ++ entries: BTreeMap, ++ prune_cursor: Option, ++} ++ ++impl WasiSharedFutexRegistries { ++ fn registry_for_file( ++ owner: &Arc>, ++ file_anchor: Arc, ++ ) -> Result, Errno> { ++ let identity = WasiSharedFileIdentity::for_file(&file_anchor)?; ++ ++ let mut registries = owner.lock().unwrap_or_else(|err| err.into_inner()); ++ registries.prune_stale(SHARED_FUTEX_REGISTRY_PRUNE_BUDGET); ++ ++ if let Some(registry) = registries ++ .entries ++ .get(&identity) ++ .and_then(|entry| entry.registry.upgrade()) ++ { ++ return Ok(registry); ++ } ++ registries.entries.remove(&identity); ++ ++ let generation = Arc::new(()); ++ let registry = Arc::new(WasiSharedFutexRegistry { ++ futexs: Default::default(), ++ _file_anchor: file_anchor, ++ identity, ++ generation: generation.clone(), ++ owner: Arc::downgrade(owner), ++ }); ++ registries.entries.insert( ++ identity, ++ WasiSharedFutexRegistryEntry { ++ generation, ++ registry: Arc::downgrade(®istry), ++ }, ++ ); ++ Ok(registry) ++ } ++ ++ fn remove_if_current(&mut self, identity: WasiSharedFileIdentity, generation: &Arc<()>) { ++ let is_current = self ++ .entries ++ .get(&identity) ++ .is_some_and(|entry| Arc::ptr_eq(&entry.generation, generation)); ++ if is_current { ++ self.entries.remove(&identity); ++ if self.entries.is_empty() { ++ self.prune_cursor = None; ++ } ++ } ++ } ++ ++ fn prune_stale(&mut self, budget: usize) -> usize { ++ if budget == 0 || self.entries.is_empty() { ++ return 0; ++ } ++ ++ let mut keys = Vec::with_capacity(budget.min(self.entries.len())); ++ if let Some(cursor) = self.prune_cursor { ++ keys.extend( ++ self.entries ++ .range((Excluded(cursor), Unbounded)) ++ .take(budget) ++ .map(|(identity, _)| *identity), ++ ); ++ let remaining = budget.saturating_sub(keys.len()); ++ keys.extend( ++ self.entries ++ .range(..=cursor) ++ .take(remaining) ++ .map(|(identity, _)| *identity), ++ ); ++ } else { ++ keys.extend(self.entries.keys().take(budget).copied()); ++ } ++ ++ self.prune_cursor = keys.last().copied(); ++ let before = self.entries.len(); ++ for identity in keys { ++ let stale = self ++ .entries ++ .get(&identity) ++ .is_some_and(|entry| entry.registry.strong_count() == 0); ++ if stale { ++ self.entries.remove(&identity); ++ } ++ } ++ if self.entries.is_empty() { ++ self.prune_cursor = None; ++ } ++ before - self.entries.len() ++ } ++ ++ #[cfg(test)] ++ fn counts(&self) -> (usize, usize) { ++ self.entries ++ .values() ++ .fold((0usize, 0usize), |(active, stale), entry| { ++ if entry.registry.strong_count() == 0 { ++ (active, stale + 1) ++ } else { ++ (active + 1, stale) ++ } ++ }) ++ } ++} ++ ++#[derive(Debug, Clone)] ++pub(crate) enum WasiFutexRegistry { ++ Private(Arc), ++ Shared(Arc), ++} ++ ++impl WasiFutexRegistry { ++ pub(crate) fn resolve(state: &Arc, addr: u64) -> Result<(Self, u64), Errno> { ++ if let Some((futexs, key)) = state.shared_futex_registry(addr)? { ++ Ok((Self::Shared(futexs), key)) ++ } else { ++ Ok((Self::Private(state.clone()), addr)) ++ } ++ } ++ ++ pub(crate) fn with(&self, f: impl FnOnce(&mut WasiFutexState) -> R) -> R { ++ match self { ++ Self::Private(state) => { ++ let mut guard = state.futexs.lock().unwrap(); ++ f(&mut guard) ++ } ++ Self::Shared(futexs) => { ++ let mut guard = futexs.futexs.lock().unwrap(); ++ f(&mut guard) ++ } ++ } ++ } ++} ++ ++#[derive(Debug, Clone)] ++pub(crate) struct WasiSharedMemoryMapping { ++ pub start: u64, ++ pub len: u64, ++ pub file: Arc, ++ pub file_offset: u64, ++ pub futexs: Arc, ++} ++ ++impl WasiSharedMemoryMapping { ++ fn end(&self) -> Result { ++ self.start.checked_add(self.len).ok_or(Errno::Overflow) ++ } ++ ++ fn same_exec_reservation(&self, candidate: &Self) -> bool { ++ self.start == candidate.start ++ && self.same_exec_backing(candidate.len, candidate.file_offset, &candidate.futexs) ++ } ++ ++ fn same_exec_backing( ++ &self, ++ len: u64, ++ file_offset: u64, ++ futexs: &Arc, ++ ) -> bool { ++ self.len == len && self.file_offset == file_offset && Arc::ptr_eq(&self.futexs, futexs) ++ } ++} ++ ++#[derive(Debug, Default, Clone)] ++pub(crate) struct WasiSharedMemoryMappings { ++ mappings: Vec, ++} ++ ++impl WasiSharedMemoryMappings { ++ pub fn snapshot(&self) -> Vec { ++ self.mappings.clone() ++ } ++ ++ fn max_end(&self) -> Result, Errno> { ++ self.mappings.iter().try_fold(None, |maximum, mapping| { ++ let end = mapping.end()?; ++ Ok(Some(maximum.map_or(end, |old: u64| old.max(end)))) ++ }) ++ } ++ ++ fn overlaps(&self, start: u64, len: u64) -> Result { ++ if len == 0 { ++ return Err(Errno::Inval); ++ } ++ let end = start.checked_add(len).ok_or(Errno::Overflow)?; ++ for mapping in &self.mappings { ++ let mapping_end = mapping.end()?; ++ if mapping.start < end && start < mapping_end { ++ return Ok(true); ++ } ++ } ++ Ok(false) ++ } ++ ++ fn covers(&self, start: u64, len: u64) -> Result { ++ if len == 0 { ++ return Err(Errno::Inval); ++ } ++ let end = start.checked_add(len).ok_or(Errno::Overflow)?; ++ let mut covered_until = start; ++ for mapping in &self.mappings { ++ let mapping_end = mapping.end()?; ++ if mapping_end <= covered_until { ++ continue; ++ } ++ if mapping.start > covered_until { ++ return Ok(false); ++ } ++ covered_until = mapping_end; ++ if covered_until >= end { ++ return Ok(true); ++ } ++ } ++ Ok(false) ++ } ++ ++ fn exact_exec_reservation_index(&self, candidate: &WasiSharedMemoryMapping) -> Option { ++ self.mappings ++ .iter() ++ .position(|mapping| mapping.same_exec_reservation(candidate)) ++ } ++ ++ fn exec_backing_reservation_index( ++ &self, ++ len: u64, ++ file_offset: u64, ++ futexs: &Arc, ++ ) -> Option { ++ self.mappings ++ .iter() ++ .position(|mapping| mapping.same_exec_backing(len, file_offset, futexs)) ++ } ++ ++ fn replacing(&self, mapping: WasiSharedMemoryMapping) -> Result { ++ if mapping.len == 0 { ++ return Err(Errno::Inval); ++ } ++ ++ let start = mapping.start; ++ let end = mapping.end()?; ++ let mut updated = Vec::with_capacity(self.mappings.len() + 1); ++ ++ for old in &self.mappings { ++ let old_end = old.end()?; ++ if old_end <= start || old.start >= end { ++ updated.push(old.clone()); ++ continue; ++ } ++ ++ if old.start < start { ++ updated.push(WasiSharedMemoryMapping { ++ start: old.start, ++ len: start - old.start, ++ file: old.file.clone(), ++ file_offset: old.file_offset, ++ futexs: old.futexs.clone(), ++ }); ++ } ++ ++ if old_end > end { ++ updated.push(WasiSharedMemoryMapping { ++ start: end, ++ len: old_end - end, ++ file: old.file.clone(), ++ file_offset: old ++ .file_offset ++ .checked_add(end - old.start) ++ .ok_or(Errno::Overflow)?, ++ futexs: old.futexs.clone(), ++ }); ++ } ++ } ++ ++ updated.push(mapping); ++ updated.sort_by_key(|mapping| mapping.start); ++ Ok(Self { mappings: updated }) ++ } ++ ++ pub fn replace(&mut self, mapping: WasiSharedMemoryMapping) -> Result<(), Errno> { ++ *self = self.replacing(mapping)?; ++ Ok(()) ++ } ++ ++ fn removing(&self, start: u64, len: u64) -> Result { ++ if len == 0 { ++ return Err(Errno::Inval); ++ } ++ ++ let end = start.checked_add(len).ok_or(Errno::Overflow)?; ++ let mut updated = Vec::with_capacity(self.mappings.len()); ++ ++ for old in &self.mappings { ++ let old_end = old.end()?; ++ if old_end <= start || old.start >= end { ++ updated.push(old.clone()); ++ continue; ++ } ++ ++ if old.start < start { ++ updated.push(WasiSharedMemoryMapping { ++ start: old.start, ++ len: start - old.start, ++ file: old.file.clone(), ++ file_offset: old.file_offset, ++ futexs: old.futexs.clone(), ++ }); ++ } ++ ++ if old_end > end { ++ updated.push(WasiSharedMemoryMapping { ++ start: end, ++ len: old_end - end, ++ file: old.file.clone(), ++ file_offset: old ++ .file_offset ++ .checked_add(end - old.start) ++ .ok_or(Errno::Overflow)?, ++ futexs: old.futexs.clone(), ++ }); ++ } ++ } ++ ++ updated.sort_by_key(|mapping| mapping.start); ++ Ok(Self { mappings: updated }) ++ } ++ ++ #[cfg(test)] ++ pub fn remove(&mut self, start: u64, len: u64) -> Result<(), Errno> { ++ *self = self.removing(start, len)?; ++ Ok(()) ++ } ++ ++ fn shared_futex_registry(&self, addr: u64) -> Option<(Arc, u64)> { ++ let idx = self ++ .mappings ++ .partition_point(|mapping| mapping.start <= addr); ++ let mapping = self.mappings.get(idx.checked_sub(1)?)?; ++ let end = mapping.end().ok()?; ++ if addr >= end { ++ return None; ++ } ++ ++ let key = mapping.file_offset.checked_add(addr - mapping.start)?; ++ Some((mapping.futexs.clone(), key)) ++ } ++} ++ ++/// Identifies the fixed-address path. Runtime-selected mappings use a separate ++/// API that atomically reserves fresh WebAssembly pages and therefore never ++/// need permission to replace existing private memory. ++#[derive(Debug, Clone, Copy, Eq, PartialEq)] ++pub(crate) enum WasiSharedMemoryMapOrigin { ++ GuestFixed, ++} ++ ++#[derive(Debug, Clone, Copy, Eq, PartialEq)] ++pub(crate) enum WasiSharedMemoryRuntimePlacement { ++ Fresh { minimum_start: u64 }, ++ Inherited { start: u64 }, ++} ++ ++/// Serializes the address-space handoff from an old process image to a fresh ++/// EXEC_BACKEND image. The two independent completion conditions are recorded ++/// explicitly: the linker must install the reserved address window, and the ++/// task manager must commit successor admission. Exact reattachment alone can ++/// therefore never discard rollback authority. ++#[derive(Debug, Clone, Copy, Eq, PartialEq)] ++pub(crate) enum WasiSharedMemoryExecPhase { ++ Normal, ++ Transition { ++ reservations_installed: bool, ++ successor_committed: bool, ++ }, ++} ++ ++impl Default for WasiSharedMemoryExecPhase { ++ fn default() -> Self { ++ Self::Normal ++ } ++} ++ ++/// Rollback guard for the point at which an old process image gives its active ++/// shared mappings to a fresh exec image as address-only reservations. Spawn ++/// failures restore the old process-visible registry exactly; a successfully ++/// accepted successor commits the transition. ++#[derive(Debug)] ++pub(crate) struct WasiSharedMemoryExecTransition { ++ state: Arc, ++ original_mappings: Option>, ++} ++ ++impl WasiSharedMemoryExecTransition { ++ pub(crate) fn commit(mut self) { ++ if self.original_mappings.is_some() { ++ self.state.commit_shared_memory_exec_transition(); ++ } ++ self.original_mappings.take(); ++ } ++} ++ ++impl Drop for WasiSharedMemoryExecTransition { ++ fn drop(&mut self) { ++ let Some(original_mappings) = self.original_mappings.take() else { ++ return; ++ }; ++ ++ // Use the same lock order as every multi-registry mapping operation. ++ let mut phase = self ++ .state ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if !matches!(*phase, WasiSharedMemoryExecPhase::Transition { .. }) { ++ tracing::error!( ++ ?phase, ++ "shared-memory exec rollback refused after phase advanced" ++ ); ++ return; ++ } ++ let mut reservations = self ++ .state ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let mut active = self ++ .state ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ // Linker pre-growth marks reservations installed before synchronous ++ // Instance::new can fail. A start function may also have consumed ++ // exact reservations before trapping. Rollback is still safe when the ++ // reservation+active partition is exactly the original set: all host ++ // remaps occurred in the discarded fresh memory, while the old image's ++ // mapping never moved. ++ let mut current = reservations.snapshot(); ++ current.extend(active.snapshot()); ++ current.sort_by_key(|mapping| (mapping.start, mapping.len, mapping.file_offset)); ++ let mut original = original_mappings.clone(); ++ original.sort_by_key(|mapping| (mapping.start, mapping.len, mapping.file_offset)); ++ let exact_partition = current.len() == original.len() ++ && current ++ .iter() ++ .zip(&original) ++ .all(|(current, original)| original.same_exec_reservation(current)); ++ if !exact_partition { ++ tracing::error!( ++ reservations = reservations.mappings.len(), ++ active = active.mappings.len(), ++ original = original_mappings.len(), ++ "shared-memory exec rollback refused for a non-exact successor partition" ++ ); ++ return; ++ } ++ reservations.mappings.clear(); ++ active.mappings = original_mappings; ++ *phase = WasiSharedMemoryExecPhase::Normal; ++ } ++} ++ + /// Top level data type containing all* the state with which WASI can + /// interact. + /// +@@ -139,7 +708,14 @@ pub(crate) struct WasiState { + pub args: Mutex>, + pub envs: Mutex>>, + pub signals: Mutex>, +- ++ #[cfg_attr(feature = "enable-serde", serde(skip))] ++ pub(crate) shared_futex_registries: Arc>, ++ #[cfg_attr(feature = "enable-serde", serde(skip))] ++ pub(crate) shared_memory_mappings: Mutex, ++ #[cfg_attr(feature = "enable-serde", serde(skip))] ++ pub(crate) shared_memory_exec_phase: Mutex, ++ #[cfg_attr(feature = "enable-serde", serde(skip))] ++ pub(crate) shared_memory_exec_reservations: Mutex, + // TODO: should not be here, since this requires active work to resolve. + // State should only hold active runtime state that can be reproducibly re-created. + pub preopen: Vec, +@@ -210,6 +786,465 @@ impl WasiState { + self.fs.root_fs.new_open_options() + } + ++ #[cfg(test)] ++ pub(crate) fn shared_memory_mappings(&self) -> Vec { ++ self.shared_memory_mappings.lock().unwrap().snapshot() ++ } ++ ++ /// Converts the current image's active shared mappings into reservations ++ /// for a fresh exec image. Reservations deliberately do not participate in ++ /// futex address resolution until the child proves and installs the exact ++ /// backing object again. ++ pub(crate) fn prepare_shared_memory_for_exec( ++ self: &Arc, ++ ) -> Result { ++ // Global order: phase, reservations, active mappings. ++ let mut phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if *phase != WasiSharedMemoryExecPhase::Normal { ++ return Err(Errno::Busy); ++ } ++ let mut reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if !reservations.mappings.is_empty() { ++ return Err(Errno::Busy); ++ } ++ let mut active = self ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let original_mappings = active.snapshot(); ++ if original_mappings.is_empty() { ++ return Ok(WasiSharedMemoryExecTransition { ++ state: self.clone(), ++ original_mappings: None, ++ }); ++ } ++ *phase = WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: false, ++ successor_committed: false, ++ }; ++ reservations.mappings = std::mem::take(&mut active.mappings); ++ drop(active); ++ drop(reservations); ++ drop(phase); ++ ++ Ok(WasiSharedMemoryExecTransition { ++ state: self.clone(), ++ original_mappings: Some(original_mappings), ++ }) ++ } ++ ++ #[cfg(test)] ++ pub(crate) fn shared_memory_exec_reservation_end(&self) -> Result, Errno> { ++ self.shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()) ++ .max_end() ++ } ++ ++ pub(crate) fn shared_memory_exec_reservations(&self) -> Vec { ++ self.shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()) ++ .snapshot() ++ } ++ ++ pub(crate) fn has_shared_memory_exec_reservations(&self) -> bool { ++ !self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()) ++ .mappings ++ .is_empty() ++ } ++ ++ /// Advances a rollback-capable exec handoff only after the linker has ++ /// grown the fresh memory above every reserved range. Once installed, the ++ /// guest may consume reservations only through exact fixed reattachments. ++ pub(crate) fn mark_shared_memory_exec_reservations_installed(&self) -> Result<(), Errno> { ++ let mut phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if reservations.mappings.is_empty() { ++ return Err(Errno::Inval); ++ } ++ match *phase { ++ WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: false, ++ successor_committed, ++ } => { ++ *phase = WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: true, ++ successor_committed, ++ }; ++ } ++ _ => return Err(Errno::Busy), ++ } ++ Ok(()) ++ } ++ ++ fn commit_shared_memory_exec_transition(&self) { ++ let mut phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ match *phase { ++ WasiSharedMemoryExecPhase::Transition { ++ reservations_installed, ++ .. ++ } if reservations_installed && reservations.mappings.is_empty() => { ++ *phase = WasiSharedMemoryExecPhase::Normal; ++ } ++ WasiSharedMemoryExecPhase::Transition { ++ reservations_installed, ++ .. ++ } => { ++ *phase = WasiSharedMemoryExecPhase::Transition { ++ reservations_installed, ++ successor_committed: true, ++ }; ++ } ++ WasiSharedMemoryExecPhase::Normal => { ++ // A transition with inherited mappings cannot reach Normal ++ // before this commit. Keep the task live, but surface an ++ // internal invariant violation loudly in debug telemetry. ++ tracing::error!("shared-memory exec commit observed no active transition"); ++ } ++ } ++ } ++ ++ /// Runs the host remap and publishes its active mapping as one atomic state ++ /// transition. A reattach must match range, file offset, and the stable ++ /// shared-backing registry identity inherited across exec. Other fixed ++ /// remaps require full coverage by an already active mapping. ++ pub(crate) fn install_shared_memory_mapping( ++ &self, ++ mapping: WasiSharedMemoryMapping, ++ origin: WasiSharedMemoryMapOrigin, ++ remap: impl FnOnce() -> Result<(), Errno>, ++ ) -> Result<(), Errno> { ++ let mut phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let mut reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let mut active = self ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ ++ let reservation_index = match (*phase, origin) { ++ ( ++ WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: false, ++ .. ++ }, ++ _, ++ ) => return Err(Errno::Busy), ++ ( ++ WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: true, ++ .. ++ }, ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ ) => Some( ++ reservations ++ .exact_exec_reservation_index(&mapping) ++ .ok_or(Errno::Inval)?, ++ ), ++ (WasiSharedMemoryExecPhase::Normal, WasiSharedMemoryMapOrigin::GuestFixed) => { ++ if reservations.mappings.is_empty() && active.covers(mapping.start, mapping.len)? { ++ // POSIX also permits MAP_FIXED to replace a known mapping ++ // in the same image. Requiring full active coverage keeps ++ // that behavior without allowing a raw fixed request to ++ // overwrite an untracked heap allocation. ++ None ++ } else { ++ return Err(Errno::Inval); ++ } ++ } ++ }; ++ ++ // Compute every fallible metadata transformation before touching the ++ // host mapping. Publication after remap is an infallible assignment. ++ let updated_active = active.replacing(mapping)?; ++ remap()?; ++ ++ if let Some(index) = reservation_index { ++ reservations.mappings.remove(index); ++ } ++ *active = updated_active; ++ if reservations.mappings.is_empty() ++ && matches!( ++ *phase, ++ WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: true, ++ successor_committed: true, ++ } ++ ) ++ { ++ *phase = WasiSharedMemoryExecPhase::Normal; ++ } ++ Ok(()) ++ } ++ ++ /// Atomically installs an original non-fixed `mmap`. During exec, a ++ /// request for an inherited backing consumes its matching reservation and ++ /// reuses that address; otherwise `reserve_and_remap` must use WebAssembly ++ /// memory.grow and return its old end, which is unique even when another ++ /// guest thread grows the heap concurrently. Holding the mapping locks ++ /// across either operation serializes it with fixed remaps and keeps the ++ /// registry publication atomic. ++ pub(crate) fn install_runtime_selected_shared_memory_mapping( ++ &self, ++ len: u64, ++ file: Arc, ++ file_offset: u64, ++ futexs: Arc, ++ reserve_and_remap: impl FnOnce(WasiSharedMemoryRuntimePlacement) -> Result, ++ ) -> Result { ++ if len == 0 { ++ return Err(Errno::Inval); ++ } ++ ++ let mut phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let mut reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let mut active = self ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ ++ if let WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: true, ++ successor_committed, ++ } = *phase ++ { ++ let reservation_index = reservations ++ .exec_backing_reservation_index(len, file_offset, &futexs) ++ .ok_or(Errno::Busy)?; ++ let start = reservations.mappings[reservation_index].start; ++ let mapping = WasiSharedMemoryMapping { ++ start, ++ len, ++ file, ++ file_offset, ++ futexs, ++ }; ++ let updated_active = active.replacing(mapping)?; ++ let selected = ++ reserve_and_remap(WasiSharedMemoryRuntimePlacement::Inherited { start })?; ++ if selected != start { ++ return Err(Errno::Inval); ++ } ++ reservations.mappings.remove(reservation_index); ++ *active = updated_active; ++ if reservations.mappings.is_empty() && successor_committed { ++ *phase = WasiSharedMemoryExecPhase::Normal; ++ } ++ return Ok(start); ++ } ++ if *phase != WasiSharedMemoryExecPhase::Normal { ++ return Err(Errno::Busy); ++ } ++ ++ let minimum_start = reservations ++ .max_end()? ++ .into_iter() ++ .chain(active.max_end()?) ++ .max() ++ .unwrap_or(0); ++ let start = reserve_and_remap(WasiSharedMemoryRuntimePlacement::Fresh { minimum_start })?; ++ let _end = start.checked_add(len).ok_or(Errno::Overflow)?; ++ if start < minimum_start ++ || reservations.overlaps(start, len)? ++ || active.overlaps(start, len)? ++ { ++ return Err(Errno::Inval); ++ } ++ ++ active.replace(WasiSharedMemoryMapping { ++ start, ++ len, ++ file, ++ file_offset, ++ futexs, ++ })?; ++ Ok(start) ++ } ++ ++ pub(crate) fn futex_registry_for_shared_file( ++ &self, ++ file: Arc, ++ ) -> Result, Errno> { ++ WasiSharedFutexRegistries::registry_for_file(&self.shared_futex_registries, file) ++ } ++ ++ /// Serializes a host-file shrink against creation and lifetime of shared mappings for the ++ /// same backing inode. The returned guard must be kept alive until the shrinking operation ++ /// has completed. Growth is harmless and returns no guard. ++ #[cfg(feature = "host-fs")] ++ pub(crate) fn guard_shared_mapping_file_shrink( ++ &self, ++ file: &std::fs::File, ++ new_len: u64, ++ ) -> Result>, Errno> { ++ let current_len = file.metadata().map_err(|_| Errno::Io)?.len(); ++ if new_len >= current_len { ++ return Ok(None); ++ } ++ ++ let identity = WasiSharedFileIdentity::for_file(file)?; ++ let registries = self ++ .shared_futex_registries ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if registries ++ .entries ++ .get(&identity) ++ .and_then(|entry| entry.registry.upgrade()) ++ .is_some() ++ { ++ return Err(Errno::Busy); ++ } ++ Ok(Some(registries)) ++ } ++ ++ fn shared_futex_registry( ++ &self, ++ addr: u64, ++ ) -> Result, u64)>, Errno> { ++ let phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let active = self ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ ++ if let Some(resolved) = active.shared_futex_registry(addr) { ++ return Ok(Some(resolved)); ++ } ++ if matches!(*phase, WasiSharedMemoryExecPhase::Transition { .. }) ++ && reservations.shared_futex_registry(addr).is_some() ++ { ++ return Err(Errno::Busy); ++ } ++ Ok(None) ++ } ++ ++ #[cfg(test)] ++ pub(crate) fn replace_shared_memory_mapping( ++ &self, ++ mapping: WasiSharedMemoryMapping, ++ ) -> Result<(), Errno> { ++ self.shared_memory_mappings.lock().unwrap().replace(mapping) ++ } ++ ++ /// Runs an operation only while the requested range is fully backed by ++ /// active shared mappings. Exec reservations are deliberately excluded: ++ /// they describe addresses that a fresh image must not touch until it has ++ /// reattached the exact backing object. `Noent` has the ABI-significant ++ /// meaning that the raw requested range has zero runtime ownership, so ++ /// wasix-libc may dispatch to its legacy malloc-backed implementation. ++ pub(crate) fn with_active_shared_memory_mapping( ++ &self, ++ start: u64, ++ requested_len: u64, ++ operation_len: u64, ++ operation: impl FnOnce() -> Result<(), Errno>, ++ ) -> Result<(), Errno> { ++ let phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if *phase != WasiSharedMemoryExecPhase::Normal { ++ return Err(Errno::Busy); ++ } ++ let reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let active = self ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if !active.overlaps(start, requested_len)? { ++ return Err(Errno::Noent); ++ } ++ if reservations.overlaps(start, requested_len)? || !active.covers(start, operation_len)? { ++ return Err(Errno::Inval); ++ } ++ operation() ++ } ++ ++ /// Restores private memory and removes an active shared range as one state ++ /// transition. The host operation runs only after ownership validation and ++ /// while both mapping registries are locked, so a rejected or racing ++ /// `munmap` cannot mutate an exec reservation or unrelated guest memory. ++ /// `Noent` is returned only for zero runtime overlap; partial ownership is ++ /// `Inval` and must never fall through to guest legacy metadata. ++ pub(crate) fn remove_shared_memory_mapping( ++ &self, ++ start: u64, ++ requested_len: u64, ++ operation_len: u64, ++ restore_private: impl FnOnce() -> Result<(), Errno>, ++ ) -> Result<(), Errno> { ++ let phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if *phase != WasiSharedMemoryExecPhase::Normal { ++ return Err(Errno::Busy); ++ } ++ let reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let mut active = self ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if !active.overlaps(start, requested_len)? { ++ return Err(Errno::Noent); ++ } ++ if reservations.overlaps(start, requested_len)? || !active.covers(start, operation_len)? { ++ return Err(Errno::Inval); ++ } ++ let updated_active = active.removing(start, operation_len)?; ++ restore_private()?; ++ *active = updated_active; ++ Ok(()) ++ } ++ + /// Turn the WasiState into bytes + #[cfg(feature = "enable-serde")] + pub fn freeze(&self) -> Option> { +@@ -251,9 +1286,44 @@ impl WasiState { + Ok(ret) + } + +- /// Forking the WasiState is used when either fork or vfork is called +- pub fn fork(&self) -> Self { +- WasiState { ++ /// Creates an owned process-local fork and applies caller-specific state ++ /// before the new identity becomes observable. ++ pub(crate) fn fork_with(&self, prepare: impl FnOnce(&mut Self)) -> Result, Errno> { ++ let mut state = self.fork()?; ++ prepare(&mut state); ++ Ok(Arc::new(state)) ++ } ++ ++ /// Raw fork construction is private to the state module. Runtime callers ++ /// must use the owned fork boundary; unit tests in this module may inspect ++ /// the unwrapped value directly. ++ fn fork(&self) -> Result { ++ // Mapping state is one logical object. Capture it under the global ++ // phase -> reservations -> active lock order, and reject fork while an ++ // exec image handoff is in flight. ++ let phase = self ++ .shared_memory_exec_phase ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if *phase != WasiSharedMemoryExecPhase::Normal { ++ return Err(Errno::Busy); ++ } ++ let reservations = self ++ .shared_memory_exec_reservations ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ if !reservations.mappings.is_empty() { ++ return Err(Errno::Inval); ++ } ++ let active = self ++ .shared_memory_mappings ++ .lock() ++ .unwrap_or_else(|err| err.into_inner()); ++ let active = active.clone(); ++ let reservations = reservations.clone(); ++ drop(phase); ++ ++ Ok(WasiState { + fs: self.fs.fork(), + secret: self.secret, + inodes: self.inodes.clone(), +@@ -262,7 +1332,1169 @@ impl WasiState { + args: Mutex::new(self.args.lock().unwrap().clone()), + envs: Mutex::new(self.envs.lock().unwrap().clone()), + signals: Mutex::new(self.signals.lock().unwrap().clone()), ++ shared_futex_registries: self.shared_futex_registries.clone(), ++ shared_memory_mappings: Mutex::new(active), ++ shared_memory_exec_phase: Default::default(), ++ shared_memory_exec_reservations: Mutex::new(reservations), + preopen: self.preopen.clone(), ++ }) ++ } ++} ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use std::{ ++ sync::{Barrier, mpsc}, ++ thread, ++ time::Instant, ++ }; ++ ++ fn temporary_file() -> Arc { ++ Arc::new(tempfile::tempfile().unwrap()) ++ } ++ ++ fn mapping(start: u64, len: u64, file_offset: u64) -> WasiSharedMemoryMapping { ++ let file = temporary_file(); ++ let futexs = WasiSharedFutexRegistry::detached(file.clone()); ++ ++ WasiSharedMemoryMapping { ++ start, ++ len, ++ file, ++ file_offset, ++ futexs, + } + } ++ ++ #[test] ++ #[cfg(feature = "host-fs")] ++ fn live_shared_mapping_registry_blocks_backing_file_shrink() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-file-shrink-test"); ++ let file = temporary_file(); ++ file.set_len(0x20_000).unwrap(); ++ let registry = state.futex_registry_for_shared_file(file.clone()).unwrap(); ++ ++ assert!(matches!( ++ state.guard_shared_mapping_file_shrink(&file, 0x10_000), ++ Err(Errno::Busy) ++ )); ++ ++ drop(registry); ++ let guard = state ++ .guard_shared_mapping_file_shrink(&file, 0x10_000) ++ .unwrap(); ++ assert!(guard.is_some()); ++ file.set_len(0x10_000).unwrap(); ++ drop(guard); ++ assert_eq!(file.metadata().unwrap().len(), 0x10_000); ++ } ++ ++ fn test_state(program: &str) -> Arc { ++ Arc::new( ++ WasiEnv::builder(program) ++ .engine(wasmer::Engine::default()) ++ .build_init() ++ .unwrap() ++ .state, ++ ) ++ } ++ ++ #[test] ++ fn exec_transition_hides_active_mappings_and_rolls_back_exactly() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-exec-rollback-test"); ++ let file = temporary_file(); ++ let futexs = state.futex_registry_for_shared_file(file.clone()).unwrap(); ++ let original = WasiSharedMemoryMapping { ++ start: 0x20_000, ++ len: 0x30_000, ++ file, ++ file_offset: 0x10_000, ++ futexs: futexs.clone(), ++ }; ++ state ++ .replace_shared_memory_mapping(original.clone()) ++ .unwrap(); ++ assert!(matches!( ++ WasiFutexRegistry::resolve(&state, original.start), ++ Ok((WasiFutexRegistry::Shared(_), _)) ++ )); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ assert!(state.shared_memory_mappings().is_empty()); ++ assert_eq!( ++ state.shared_memory_exec_reservation_end().unwrap(), ++ Some(original.start + original.len) ++ ); ++ assert!(matches!( ++ WasiFutexRegistry::resolve(&state, original.start), ++ Err(Errno::Busy) ++ )); ++ ++ drop(transition); ++ let restored = state.shared_memory_mappings(); ++ assert_eq!(restored.len(), 1); ++ assert_eq!(restored[0].start, original.start); ++ assert_eq!(restored[0].len, original.len); ++ assert_eq!(restored[0].file_offset, original.file_offset); ++ assert!(Arc::ptr_eq(&restored[0].futexs, &futexs)); ++ assert_eq!(state.shared_memory_exec_reservation_end().unwrap(), None); ++ } ++ ++ #[test] ++ fn exec_transition_rolls_back_an_exact_partially_installed_successor() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-exec-installed-rollback-test"); ++ let first = mapping(0x20_000, 0x10_000, 0); ++ let second = mapping(0x40_000, 0x20_000, 0x10_000); ++ state.replace_shared_memory_mapping(first.clone()).unwrap(); ++ state.replace_shared_memory_mapping(second.clone()).unwrap(); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ state ++ .install_shared_memory_mapping( ++ first.clone(), ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || Ok(()), ++ ) ++ .unwrap(); ++ assert_eq!(state.shared_memory_mappings().len(), 1); ++ assert_eq!(state.shared_memory_exec_reservations().len(), 1); ++ ++ // Model synchronous Instance::new failure after linker pre-growth and ++ // a start function's first exact reattachment. ++ drop(transition); ++ let restored = state.shared_memory_mappings(); ++ assert_eq!(restored.len(), 2); ++ assert_eq!( ++ (restored[0].start, restored[0].len), ++ (first.start, first.len) ++ ); ++ assert_eq!( ++ (restored[1].start, restored[1].len), ++ (second.start, second.len) ++ ); ++ assert!(state.shared_memory_exec_reservations().is_empty()); ++ assert_eq!( ++ *state.shared_memory_exec_phase.lock().unwrap(), ++ WasiSharedMemoryExecPhase::Normal ++ ); ++ } ++ ++ #[test] ++ fn exec_transition_resolves_only_reattached_futex_mappings() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-exec-partial-futex-test"); ++ let active = mapping(0x20_000, 0x10_000, 0x1000); ++ let reserved = mapping(0x40_000, 0x10_000, 0x2000); ++ state.replace_shared_memory_mapping(active.clone()).unwrap(); ++ state ++ .replace_shared_memory_mapping(reserved.clone()) ++ .unwrap(); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ state ++ .install_shared_memory_mapping( ++ active.clone(), ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || Ok(()), ++ ) ++ .unwrap(); ++ transition.commit(); ++ ++ let (registry, key) = WasiFutexRegistry::resolve(&state, active.start + 4).unwrap(); ++ let WasiFutexRegistry::Shared(registry) = registry else { ++ panic!("reattached mapping resolved as a private futex"); ++ }; ++ assert!(Arc::ptr_eq(®istry, &active.futexs)); ++ assert_eq!(key, active.file_offset + 4); ++ assert!(matches!( ++ WasiFutexRegistry::resolve(&state, reserved.start + 4), ++ Err(Errno::Busy) ++ )); ++ ++ state ++ .install_shared_memory_mapping(reserved, WasiSharedMemoryMapOrigin::GuestFixed, || { ++ Ok(()) ++ }) ++ .unwrap(); ++ assert_eq!( ++ *state.shared_memory_exec_phase.lock().unwrap(), ++ WasiSharedMemoryExecPhase::Normal ++ ); ++ } ++ ++ #[test] ++ fn exact_reattach_does_not_relinquish_rollback_before_admission() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-exec-reattached-rollback-test"); ++ let original = mapping(0x20_000, 0x10_000, 0); ++ state ++ .replace_shared_memory_mapping(original.clone()) ++ .unwrap(); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ state ++ .install_shared_memory_mapping( ++ original.clone(), ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || Ok(()), ++ ) ++ .unwrap(); ++ ++ assert!(state.shared_memory_exec_reservations().is_empty()); ++ assert!(matches!( ++ *state.shared_memory_exec_phase.lock().unwrap(), ++ WasiSharedMemoryExecPhase::Transition { ++ reservations_installed: true, ++ successor_committed: false, ++ } ++ )); ++ assert!(matches!(state.fork(), Err(Errno::Busy))); ++ let new_file = temporary_file(); ++ let new_futexs = state ++ .futex_registry_for_shared_file(new_file.clone()) ++ .unwrap(); ++ assert_eq!( ++ state.install_runtime_selected_shared_memory_mapping( ++ 0x10_000, ++ new_file, ++ 0, ++ new_futexs, ++ |_| Ok(0x40_000), ++ ), ++ Err(Errno::Busy) ++ ); ++ ++ // Model a start function that reattached every range and then trapped ++ // before task admission. The old image is still restored exactly. ++ drop(transition); ++ assert_eq!(state.shared_memory_mappings().len(), 1); ++ assert!(state.shared_memory_exec_reservations().is_empty()); ++ assert_eq!( ++ *state.shared_memory_exec_phase.lock().unwrap(), ++ WasiSharedMemoryExecPhase::Normal ++ ); ++ } ++ ++ #[test] ++ fn admission_and_exact_reattach_jointly_finish_exec_in_either_order() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-exec-two-condition-commit-test"); ++ let original = mapping(0x20_000, 0x10_000, 0); ++ state ++ .replace_shared_memory_mapping(original.clone()) ++ .unwrap(); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ state ++ .install_shared_memory_mapping(original, WasiSharedMemoryMapOrigin::GuestFixed, || { ++ Ok(()) ++ }) ++ .unwrap(); ++ transition.commit(); ++ ++ assert_eq!( ++ *state.shared_memory_exec_phase.lock().unwrap(), ++ WasiSharedMemoryExecPhase::Normal ++ ); ++ assert!(state.fork().is_ok()); ++ } ++ ++ #[test] ++ fn exec_reattach_consumes_only_exact_range_offset_and_backing_identity() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-exec-identity-test"); ++ let file = temporary_file(); ++ file.set_len(0x80_000).unwrap(); ++ let futexs = state.futex_registry_for_shared_file(file.clone()).unwrap(); ++ let original = WasiSharedMemoryMapping { ++ start: 0x40_000, ++ len: 0x20_000, ++ file: file.clone(), ++ file_offset: 0x10_000, ++ futexs: futexs.clone(), ++ }; ++ state ++ .replace_shared_memory_mapping(original.clone()) ++ .unwrap(); ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ ++ let mut remap_called = false; ++ assert_eq!( ++ state.install_shared_memory_mapping( ++ original.clone(), ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || { ++ remap_called = true; ++ Ok(()) ++ }, ++ ), ++ Err(Errno::Busy) ++ ); ++ assert!(!remap_called); ++ transition.commit(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ ++ let wrong_file = temporary_file(); ++ wrong_file.set_len(0x80_000).unwrap(); ++ let wrong_futexs = state ++ .futex_registry_for_shared_file(wrong_file.clone()) ++ .unwrap(); ++ assert_eq!( ++ state.install_shared_memory_mapping( ++ WasiSharedMemoryMapping { ++ file: wrong_file, ++ futexs: wrong_futexs, ++ ..original.clone() ++ }, ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || { ++ remap_called = true; ++ Ok(()) ++ }, ++ ), ++ Err(Errno::Inval) ++ ); ++ assert!(!remap_called); ++ ++ assert_eq!( ++ state.install_shared_memory_mapping( ++ WasiSharedMemoryMapping { ++ file_offset: original.file_offset + 4096, ++ ..original.clone() ++ }, ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || { ++ remap_called = true; ++ Ok(()) ++ }, ++ ), ++ Err(Errno::Inval) ++ ); ++ assert!(!remap_called); ++ ++ let reopened = Arc::new(file.try_clone().unwrap()); ++ let reopened_futexs = state ++ .futex_registry_for_shared_file(reopened.clone()) ++ .unwrap(); ++ state ++ .install_shared_memory_mapping( ++ WasiSharedMemoryMapping { ++ file: reopened, ++ futexs: reopened_futexs, ++ ..original.clone() ++ }, ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || { ++ remap_called = true; ++ Ok(()) ++ }, ++ ) ++ .unwrap(); ++ assert!(remap_called); ++ assert_eq!(state.shared_memory_exec_reservation_end().unwrap(), None); ++ assert_eq!(state.shared_memory_mappings().len(), 1); ++ } ++ ++ #[test] ++ fn runtime_selected_mapping_starts_after_reservations_and_active_ranges() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-runtime-selection-test"); ++ let original = mapping(0x40_000, 0x20_000, 0); ++ state ++ .replace_shared_memory_mapping(original.clone()) ++ .unwrap(); ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ transition.commit(); ++ ++ let rejected = mapping(0, 0x10_000, 0); ++ let mut remap_called = false; ++ assert_eq!( ++ state.install_runtime_selected_shared_memory_mapping( ++ rejected.len, ++ rejected.file, ++ rejected.file_offset, ++ rejected.futexs, ++ |_| { ++ remap_called = true; ++ Ok(0x80_000) ++ }, ++ ), ++ Err(Errno::Busy) ++ ); ++ assert!(!remap_called); ++ ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ let installed_rejected = mapping(0, 0x10_000, 0); ++ assert_eq!( ++ state.install_runtime_selected_shared_memory_mapping( ++ installed_rejected.len, ++ installed_rejected.file, ++ installed_rejected.file_offset, ++ installed_rejected.futexs, ++ |_| { ++ remap_called = true; ++ Ok(0x80_000) ++ }, ++ ), ++ Err(Errno::Busy) ++ ); ++ assert!(!remap_called); ++ ++ state ++ .install_shared_memory_mapping( ++ original.clone(), ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || Ok(()), ++ ) ++ .unwrap(); ++ ++ let bad_selection = mapping(0, 0x10_000, 0); ++ assert_eq!( ++ state.install_runtime_selected_shared_memory_mapping( ++ bad_selection.len, ++ bad_selection.file, ++ bad_selection.file_offset, ++ bad_selection.futexs, ++ |placement| { ++ assert_eq!( ++ placement, ++ WasiSharedMemoryRuntimePlacement::Fresh { ++ minimum_start: original.start + original.len, ++ } ++ ); ++ Err(Errno::Inval) ++ }, ++ ), ++ Err(Errno::Inval) ++ ); ++ ++ let selected_mapping = mapping(0, 0x10_000, 0); ++ let selected = state ++ .install_runtime_selected_shared_memory_mapping( ++ selected_mapping.len, ++ selected_mapping.file, ++ selected_mapping.file_offset, ++ selected_mapping.futexs, ++ |placement| { ++ assert_eq!( ++ placement, ++ WasiSharedMemoryRuntimePlacement::Fresh { ++ minimum_start: original.start + original.len, ++ } ++ ); ++ remap_called = true; ++ Ok(0x80_000) ++ }, ++ ) ++ .unwrap(); ++ assert!(remap_called); ++ assert_eq!(selected, 0x80_000); ++ assert_eq!(state.shared_memory_mappings()[1].start, selected); ++ } ++ ++ #[test] ++ fn runtime_selected_exec_reattach_consumes_matching_inherited_backing() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-runtime-exec-reattach-test"); ++ let main = mapping(0x20_000, 0x10_000, 0); ++ let dsm = mapping(0x40_000, 0x20_000, 0); ++ state.replace_shared_memory_mapping(main.clone()).unwrap(); ++ state.replace_shared_memory_mapping(dsm.clone()).unwrap(); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ state ++ .install_shared_memory_mapping(main, WasiSharedMemoryMapOrigin::GuestFixed, || Ok(())) ++ .unwrap(); ++ transition.commit(); ++ ++ let mut observed_placement = None; ++ let selected = state ++ .install_runtime_selected_shared_memory_mapping( ++ dsm.len, ++ dsm.file.clone(), ++ dsm.file_offset, ++ dsm.futexs.clone(), ++ |placement| { ++ observed_placement = Some(placement); ++ match placement { ++ WasiSharedMemoryRuntimePlacement::Inherited { start } => Ok(start), ++ WasiSharedMemoryRuntimePlacement::Fresh { .. } => Err(Errno::Inval), ++ } ++ }, ++ ) ++ .unwrap(); ++ ++ assert_eq!(selected, dsm.start); ++ assert_eq!( ++ observed_placement, ++ Some(WasiSharedMemoryRuntimePlacement::Inherited { start: dsm.start }) ++ ); ++ assert!(state.shared_memory_exec_reservations().is_empty()); ++ assert_eq!(state.shared_memory_mappings().len(), 2); ++ assert_eq!( ++ *state.shared_memory_exec_phase.lock().unwrap(), ++ WasiSharedMemoryExecPhase::Normal ++ ); ++ } ++ ++ #[test] ++ fn runtime_selected_exec_reattach_rolls_back_before_admission() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-runtime-exec-rollback-test"); ++ let original = mapping(0x40_000, 0x20_000, 0x10_000); ++ state ++ .replace_shared_memory_mapping(original.clone()) ++ .unwrap(); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ state ++ .install_runtime_selected_shared_memory_mapping( ++ original.len, ++ original.file.clone(), ++ original.file_offset, ++ original.futexs.clone(), ++ |placement| match placement { ++ WasiSharedMemoryRuntimePlacement::Inherited { start } => Ok(start), ++ WasiSharedMemoryRuntimePlacement::Fresh { .. } => Err(Errno::Inval), ++ }, ++ ) ++ .unwrap(); ++ ++ drop(transition); ++ let restored = state.shared_memory_mappings(); ++ assert_eq!(restored.len(), 1); ++ assert!(restored[0].same_exec_reservation(&original)); ++ assert!(state.shared_memory_exec_reservations().is_empty()); ++ assert_eq!( ++ *state.shared_memory_exec_phase.lock().unwrap(), ++ WasiSharedMemoryExecPhase::Normal ++ ); ++ } ++ ++ #[test] ++ fn exec_phase_rejects_fork_until_exact_reattach_finishes() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-exec-fork-exclusion-test"); ++ let original = mapping(0x40_000, 0x20_000, 0); ++ state ++ .replace_shared_memory_mapping(original.clone()) ++ .unwrap(); ++ ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ assert!(matches!(state.fork(), Err(Errno::Busy))); ++ transition.commit(); ++ state ++ .mark_shared_memory_exec_reservations_installed() ++ .unwrap(); ++ assert!(matches!(state.fork(), Err(Errno::Busy))); ++ ++ state ++ .install_shared_memory_mapping(original, WasiSharedMemoryMapOrigin::GuestFixed, || { ++ Ok(()) ++ }) ++ .unwrap(); ++ assert!(state.fork().is_ok()); ++ } ++ ++ #[test] ++ fn guest_fixed_mapping_requires_full_known_coverage_without_exec_reservation() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-fixed-coverage-test"); ++ state ++ .replace_shared_memory_mapping(mapping(0x20_000, 0x20_000, 0)) ++ .unwrap(); ++ ++ let mut remap_called = false; ++ state ++ .install_shared_memory_mapping( ++ mapping(0x28_000, 0x8_000, 0), ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || { ++ remap_called = true; ++ Ok(()) ++ }, ++ ) ++ .unwrap(); ++ assert!(remap_called); ++ ++ remap_called = false; ++ assert_eq!( ++ state.install_shared_memory_mapping( ++ mapping(0x38_000, 0x10_000, 0), ++ WasiSharedMemoryMapOrigin::GuestFixed, ++ || { ++ remap_called = true; ++ Ok(()) ++ }, ++ ), ++ Err(Errno::Inval) ++ ); ++ assert!(!remap_called); ++ } ++ ++ #[test] ++ fn shared_mapping_operations_validate_ownership_before_host_mutation() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ let state = test_state("shared-memory-operation-ownership-test"); ++ state ++ .replace_shared_memory_mapping(mapping(0x20_000, 0x20_000, 0)) ++ .unwrap(); ++ ++ let mut operation_called = false; ++ assert_eq!( ++ state.remove_shared_memory_mapping(0x50_000, 0x10_000, 0x10_000, || { ++ operation_called = true; ++ Ok(()) ++ }), ++ Err(Errno::Noent) ++ ); ++ assert!(!operation_called); ++ ++ assert_eq!( ++ state.remove_shared_memory_mapping(0x18_000, 0x10_000, 0x10_000, || { ++ operation_called = true; ++ Ok(()) ++ }), ++ Err(Errno::Inval) ++ ); ++ assert!(!operation_called); ++ ++ assert_eq!( ++ state.remove_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || Err(Errno::Io)), ++ Err(Errno::Io) ++ ); ++ assert_eq!(state.shared_memory_mappings().len(), 1); ++ ++ state ++ .with_active_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || { ++ operation_called = true; ++ Ok(()) ++ }) ++ .unwrap(); ++ assert!(operation_called); ++ ++ operation_called = false; ++ let transition = state.prepare_shared_memory_for_exec().unwrap(); ++ assert_eq!( ++ state.with_active_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || { ++ operation_called = true; ++ Ok(()) ++ }), ++ Err(Errno::Busy) ++ ); ++ assert!(!operation_called); ++ drop(transition); ++ ++ state ++ .remove_shared_memory_mapping(0x28_000, 0x8_000, 0x8_000, || { ++ operation_called = true; ++ Ok(()) ++ }) ++ .unwrap(); ++ assert!(operation_called); ++ let remaining = state.shared_memory_mappings(); ++ assert_eq!(remaining.len(), 2); ++ assert_eq!((remaining[0].start, remaining[0].len), (0x20_000, 0x8_000)); ++ assert_eq!((remaining[1].start, remaining[1].len), (0x30_000, 0x10_000)); ++ } ++ ++ #[test] ++ fn replacing_shared_memory_mapping_splits_overlapping_ranges() { ++ let mut mappings = WasiSharedMemoryMappings::default(); ++ mappings.replace(mapping(0, 16, 128)).unwrap(); ++ mappings.replace(mapping(4, 4, 512)).unwrap(); ++ ++ let snapshot = mappings.snapshot(); ++ let ranges = snapshot ++ .iter() ++ .map(|mapping| (mapping.start, mapping.len, mapping.file_offset)) ++ .collect::>(); ++ ++ assert_eq!(ranges, vec![(0, 4, 128), (4, 4, 512), (8, 8, 136)]); ++ } ++ ++ #[test] ++ fn removing_shared_memory_mapping_splits_overlapping_ranges() { ++ let mut mappings = WasiSharedMemoryMappings::default(); ++ mappings.replace(mapping(0, 16, 128)).unwrap(); ++ mappings.remove(4, 4).unwrap(); ++ ++ let snapshot = mappings.snapshot(); ++ let ranges = snapshot ++ .iter() ++ .map(|mapping| (mapping.start, mapping.len, mapping.file_offset)) ++ .collect::>(); ++ ++ assert_eq!(ranges, vec![(0, 4, 128), (8, 8, 136)]); ++ } ++ ++ #[test] ++ fn shared_memory_mapping_splits_reuse_futex_registry() { ++ let mut mappings = WasiSharedMemoryMappings::default(); ++ let original = mapping(0, 16, 128); ++ let registry = original.futexs.clone(); ++ mappings.replace(original).unwrap(); ++ mappings.replace(mapping(4, 4, 512)).unwrap(); ++ ++ let snapshot = mappings.snapshot(); ++ assert!(Arc::ptr_eq(&snapshot[0].futexs, ®istry)); ++ assert!(!Arc::ptr_eq(&snapshot[1].futexs, ®istry)); ++ assert!(Arc::ptr_eq(&snapshot[2].futexs, ®istry)); ++ ++ let left = mappings.shared_futex_registry(2).unwrap(); ++ let right = mappings.shared_futex_registry(12).unwrap(); ++ assert!(Arc::ptr_eq(&left.0, &right.0)); ++ assert_eq!(left.1, 130); ++ assert_eq!(right.1, 140); ++ } ++ ++ #[test] ++ fn shared_futex_registry_uses_containing_mapping_only() { ++ let mut mappings = WasiSharedMemoryMappings::default(); ++ mappings.replace(mapping(16, 8, 128)).unwrap(); ++ mappings.replace(mapping(32, 8, 256)).unwrap(); ++ ++ assert!(mappings.shared_futex_registry(15).is_none()); ++ assert!(mappings.shared_futex_registry(24).is_none()); ++ assert!(mappings.shared_futex_registry(31).is_none()); ++ assert!(mappings.shared_futex_registry(40).is_none()); ++ ++ assert_eq!(mappings.shared_futex_registry(16).unwrap().1, 128); ++ assert_eq!(mappings.shared_futex_registry(23).unwrap().1, 135); ++ assert_eq!(mappings.shared_futex_registry(32).unwrap().1, 256); ++ assert_eq!(mappings.shared_futex_registry(39).unwrap().1, 263); ++ } ++ ++ #[test] ++ fn shared_file_identity_survives_file_clone() { ++ let mapping = mapping(0, 4096, 0); ++ let cloned = mapping.file.try_clone().unwrap(); ++ ++ assert_eq!( ++ WasiSharedFileIdentity::for_file(&mapping.file).unwrap(), ++ WasiSharedFileIdentity::for_file(&cloned).unwrap() ++ ); ++ } ++ ++ #[test] ++ fn shared_futex_registry_reuses_same_live_file() { ++ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); ++ let mapping = mapping(0, 4096, 0); ++ let cloned = mapping.file.try_clone().unwrap(); ++ ++ let first = WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) ++ .unwrap(); ++ let second = ++ WasiSharedFutexRegistries::registry_for_file(®istries, Arc::new(cloned)).unwrap(); ++ ++ assert!(Arc::ptr_eq(&first, &second)); ++ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); ++ } ++ ++ #[test] ++ fn shared_futex_registry_pins_file_identity_until_last_live_reference() { ++ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); ++ let mapping = mapping(0, 4096, 0); ++ let expected_identity = WasiSharedFileIdentity::for_file(&mapping.file).unwrap(); ++ ++ let registry = ++ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) ++ .unwrap(); ++ let wait_reference = registry.clone(); ++ assert!(Arc::ptr_eq(®istry._file_anchor, &mapping.file)); ++ assert_eq!( ++ WasiSharedFileIdentity::for_file(®istry._file_anchor).unwrap(), ++ expected_identity ++ ); ++ ++ drop(registry); ++ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); ++ assert_eq!( ++ WasiSharedFileIdentity::for_file(&wait_reference._file_anchor).unwrap(), ++ expected_identity ++ ); ++ } ++ ++ #[test] ++ fn shared_futex_registry_replaces_entry_after_last_live_drop() { ++ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); ++ let mapping = mapping(0, 4096, 0); ++ ++ let first = WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) ++ .unwrap(); ++ let first_weak = Arc::downgrade(&first); ++ drop(first); ++ assert_eq!(registries.lock().unwrap().counts(), (0, 0)); ++ ++ let replacement = ++ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) ++ .unwrap(); ++ assert!(!Weak::ptr_eq(&first_weak, &Arc::downgrade(&replacement))); ++ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); ++ } ++ ++ #[test] ++ fn shared_futex_registry_last_drop_racing_lookup_keeps_exact_replacement() { ++ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); ++ let file = temporary_file(); ++ let identity = WasiSharedFileIdentity::for_file(&file).unwrap(); ++ let old = WasiSharedFutexRegistries::registry_for_file(®istries, file.clone()).unwrap(); ++ let old_weak = Arc::downgrade(&old); ++ let old_generation = old.generation.clone(); ++ ++ // Hold the owner table while the final strong reference starts its ++ // destructor. This makes the old Drop and the replacement lookup ++ // contend on the exact production mutex after the Weak can no longer ++ // be upgraded. ++ let owner_guard = registries.lock().unwrap(); ++ let drop_barrier = Arc::new(Barrier::new(2)); ++ let drop_thread = { ++ let drop_barrier = drop_barrier.clone(); ++ thread::spawn(move || { ++ drop_barrier.wait(); ++ drop(old); ++ }) ++ }; ++ drop_barrier.wait(); ++ ++ let deadline = Instant::now() + Duration::from_secs(5); ++ while old_weak.strong_count() != 0 && Instant::now() < deadline { ++ thread::yield_now(); ++ } ++ assert_eq!( ++ old_weak.strong_count(), ++ 0, ++ "final registry Drop did not reach the owner-table boundary" ++ ); ++ ++ let (lookup_started_tx, lookup_started_rx) = mpsc::channel(); ++ let lookup_thread = { ++ let registries = registries.clone(); ++ let file = file.clone(); ++ thread::spawn(move || { ++ lookup_started_tx.send(()).unwrap(); ++ WasiSharedFutexRegistries::registry_for_file(®istries, file).unwrap() ++ }) ++ }; ++ lookup_started_rx ++ .recv_timeout(Duration::from_secs(5)) ++ .unwrap(); ++ ++ // Either waiter may acquire the table first. Both legal orderings must ++ // leave the replacement installed after the old destructor completes. ++ drop(owner_guard); ++ let replacement = lookup_thread.join().unwrap(); ++ drop_thread.join().unwrap(); ++ ++ assert!(old_weak.upgrade().is_none()); ++ assert!(!Weak::ptr_eq(&old_weak, &Arc::downgrade(&replacement))); ++ assert!(!Arc::ptr_eq(&old_generation, &replacement.generation)); ++ { ++ let registries = registries.lock().unwrap(); ++ let current = registries.entries.get(&identity).unwrap(); ++ assert!(Arc::ptr_eq(¤t.generation, &replacement.generation)); ++ assert!(Arc::ptr_eq( ++ ¤t.registry.upgrade().unwrap(), ++ &replacement ++ )); ++ assert_eq!(registries.counts(), (1, 0)); ++ assert_eq!(registries.entries.len(), 1); ++ } ++ ++ drop(replacement); ++ let registries = registries.lock().unwrap(); ++ assert_eq!(registries.counts(), (0, 0)); ++ assert!(registries.entries.is_empty()); ++ } ++ ++ #[test] ++ fn forked_states_share_mapping_registry_and_return_to_zero_plateau() { ++ #[cfg(not(target_arch = "wasm32"))] ++ let runtime = tokio::runtime::Builder::new_current_thread() ++ .enable_all() ++ .build() ++ .unwrap(); ++ #[cfg(not(target_arch = "wasm32"))] ++ let _runtime_guard = runtime.enter(); ++ ++ let parent = WasiEnv::builder("shared-futex-fork-ownership-test") ++ .engine(wasmer::Engine::default()) ++ .build_init() ++ .unwrap() ++ .state; ++ let file = temporary_file(); ++ let registry = parent.futex_registry_for_shared_file(file.clone()).unwrap(); ++ let start = 0x10_000; ++ let file_offset = 0x20_000; ++ parent ++ .replace_shared_memory_mapping(WasiSharedMemoryMapping { ++ start, ++ len: 4096, ++ file, ++ file_offset, ++ futexs: registry.clone(), ++ }) ++ .unwrap(); ++ ++ let first_fork = Arc::new(parent.fork().unwrap()); ++ let second_fork = Arc::new(parent.fork().unwrap()); ++ let owner = parent.shared_futex_registries.clone(); ++ assert!(Arc::ptr_eq(&owner, &first_fork.shared_futex_registries)); ++ assert!(Arc::ptr_eq(&owner, &second_fork.shared_futex_registries)); ++ ++ let address = start + 512; ++ let (first_registry, first_key) = WasiFutexRegistry::resolve(&first_fork, address).unwrap(); ++ let (second_registry, second_key) = ++ WasiFutexRegistry::resolve(&second_fork, address).unwrap(); ++ let WasiFutexRegistry::Shared(first_registry) = first_registry else { ++ panic!("first fork resolved a shared mapping as a private futex") ++ }; ++ let WasiFutexRegistry::Shared(second_registry) = second_registry else { ++ panic!("second fork resolved a shared mapping as a private futex") ++ }; ++ assert!(Arc::ptr_eq(&first_registry, ®istry)); ++ assert!(Arc::ptr_eq(&second_registry, ®istry)); ++ assert_eq!(first_key, file_offset + 512); ++ assert_eq!(second_key, first_key); ++ { ++ let registries = owner.lock().unwrap(); ++ assert_eq!(registries.counts(), (1, 0)); ++ assert_eq!(registries.entries.len(), 1); ++ } ++ ++ drop(first_registry); ++ drop(second_registry); ++ drop(registry); ++ drop(first_fork); ++ drop(second_fork); ++ drop(parent); ++ ++ let registries = owner.lock().unwrap(); ++ assert_eq!(registries.counts(), (0, 0)); ++ assert!(registries.entries.is_empty()); ++ } ++ ++ #[cfg(target_os = "linux")] ++ fn linux_fd_count_for(identity: WasiSharedFileIdentity) -> usize { ++ use std::os::unix::fs::MetadataExt; ++ ++ std::fs::read_dir("/proc/self/fd") ++ .unwrap() ++ .filter_map(Result::ok) ++ .filter_map(|entry| entry.path().metadata().ok()) ++ .filter(|metadata| metadata.dev() == identity.device && metadata.ino() == identity.file) ++ .count() ++ } ++ ++ #[test] ++ fn repeated_shared_futex_registry_churn_returns_to_slot_and_fd_plateau() { ++ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); ++ let file = temporary_file(); ++ ++ #[cfg(target_os = "linux")] ++ let (identity, baseline_fds) = { ++ let identity = WasiSharedFileIdentity::for_file(&file).unwrap(); ++ let baseline_fds = linux_fd_count_for(identity); ++ assert_eq!(baseline_fds, 1); ++ (identity, baseline_fds) ++ }; ++ ++ for _ in 0..512 { ++ let registry = ++ WasiSharedFutexRegistries::registry_for_file(®istries, file.clone()).unwrap(); ++ { ++ let registries = registries.lock().unwrap(); ++ assert_eq!(registries.counts(), (1, 0)); ++ assert_eq!(registries.entries.len(), 1); ++ } ++ drop(registry); ++ { ++ let registries = registries.lock().unwrap(); ++ assert_eq!(registries.counts(), (0, 0)); ++ assert!(registries.entries.is_empty()); ++ } ++ } ++ ++ #[cfg(target_os = "linux")] ++ assert_eq!(linux_fd_count_for(identity), baseline_fds); ++ } ++ ++ #[test] ++ fn shared_futex_registry_old_generation_cannot_remove_replacement() { ++ let identity = WasiSharedFileIdentity { device: 1, file: 2 }; ++ let old_generation = Arc::new(()); ++ let current_generation = Arc::new(()); ++ let mut registries = WasiSharedFutexRegistries::default(); ++ registries.entries.insert( ++ identity, ++ WasiSharedFutexRegistryEntry { ++ generation: current_generation.clone(), ++ registry: Weak::new(), ++ }, ++ ); ++ ++ registries.remove_if_current(identity, &old_generation); ++ assert!(registries.entries.contains_key(&identity)); ++ ++ registries.remove_if_current(identity, ¤t_generation); ++ assert!(!registries.entries.contains_key(&identity)); ++ } ++ ++ #[test] ++ fn shared_futex_registry_prunes_stale_slots_in_bounded_order() { ++ let registries = Arc::new(Mutex::new(WasiSharedFutexRegistries::default())); ++ let stale_slots = SHARED_FUTEX_REGISTRY_PRUNE_BUDGET * 3; ++ { ++ let mut registries = registries.lock().unwrap(); ++ for file in 0..stale_slots as u64 { ++ registries.entries.insert( ++ WasiSharedFileIdentity { ++ device: u64::MAX, ++ file, ++ }, ++ WasiSharedFutexRegistryEntry { ++ generation: Arc::new(()), ++ registry: Weak::new(), ++ }, ++ ); ++ } ++ } ++ ++ let mapping = mapping(0, 4096, 0); ++ let live = WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) ++ .unwrap(); ++ { ++ let registries = registries.lock().unwrap(); ++ assert_eq!( ++ registries.prune_cursor, ++ Some(WasiSharedFileIdentity { ++ device: u64::MAX, ++ file: SHARED_FUTEX_REGISTRY_PRUNE_BUDGET as u64 - 1, ++ }) ++ ); ++ assert_eq!( ++ registries.counts(), ++ (1, stale_slots - SHARED_FUTEX_REGISTRY_PRUNE_BUDGET) ++ ); ++ } ++ ++ let reused = ++ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) ++ .unwrap(); ++ assert!(Arc::ptr_eq(&live, &reused)); ++ assert_eq!( ++ registries.lock().unwrap().counts(), ++ (1, stale_slots - SHARED_FUTEX_REGISTRY_PRUNE_BUDGET * 2) ++ ); ++ ++ let reused = ++ WasiSharedFutexRegistries::registry_for_file(®istries, mapping.file.clone()) ++ .unwrap(); ++ assert!(Arc::ptr_eq(&live, &reused)); ++ assert_eq!(registries.lock().unwrap().counts(), (1, 0)); ++ } + } +diff --git a/lib/wasix/src/state/preinitialized_memory_image.rs b/lib/wasix/src/state/preinitialized_memory_image.rs +new file mode 100644 +index 0000000..86542d1 +--- /dev/null ++++ b/lib/wasix/src/state/preinitialized_memory_image.rs +@@ -0,0 +1,1828 @@ ++use std::{ ++ collections::BTreeMap, ++ fmt, ++ fs::{File, OpenOptions}, ++ io::{Read, Seek, SeekFrom, Write}, ++ path::PathBuf, ++ sync::{ ++ Arc, OnceLock, ++ atomic::{AtomicBool, AtomicU64, Ordering}, ++ }, ++}; ++ ++use serde::{Deserialize, Serialize}; ++use sha2::{Digest, Sha256}; ++use wasmer::{AsStoreMut, AsStoreRef, Memory, MemoryType, Module}; ++use wasmer_types::ModuleHash; ++ ++use super::linker::{DylinkInfo, LinkError, OrdinaryModuleStartCompleted}; ++use crate::runtime::sealed_loader_audit::{ ++ FileAdviceAudit, FileResidencyAudit, advise_file_away, file_residency, ++}; ++ ++/// Stable mapping granularity for sealed preinitialized memory images. ++/// ++/// One WebAssembly page is aligned on the native page sizes used by Linux and ++/// macOS, and on the allocation granularity required by the planned Windows ++/// placeholder-backed implementation. ++pub const PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT: u64 = 64 * 1024; ++ ++/// The exact lifecycle point represented by a preinitialized memory image. ++pub const PREINITIALIZED_MEMORY_IMAGE_PHASE: &str = "post-module-start-pre-link-relocations-v1"; ++ ++/// Static-proof contract that permits one process-wide differential byte ++/// validation instead of repeating it for every fresh instance. ++pub const DETERMINISTIC_START_PROOF_SCHEMA: &str = ++ "oliphaunt.wasix-postmaster.deterministic-start-proof.v1"; ++pub const DETERMINISTIC_START_ANALYZER_POLICY: &str = ++ "llvm-shared-memory-init-restricted-effects.v1"; ++pub const DETERMINISTIC_START_MEMORY_READS: &str = "fresh-zero-atomic-guard-only"; ++pub const DETERMINISTIC_START_MEMORY_EFFECTS: &str = ++ "passive-data-init-zero-fill-atomic-guard-only"; ++pub const DETERMINISTIC_START_GLOBAL_EFFECTS: &str = "local-numeric-relocations-only"; ++pub const DETERMINISTIC_START_TABLE_EFFECTS: &str = "none"; ++ ++/// Carrier-bound static evidence that ordinary module start is deterministic ++/// for a newly allocated, zeroed memory with the receipt-bound layout. ++/// ++/// This does not authorize skipping WebAssembly start. Every instance still ++/// executes ordinary Wasm instantiation. It only allows later instances to ++/// reuse the first instance's exact post-start image comparison. ++#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] ++#[serde(rename_all = "kebab-case", deny_unknown_fields)] ++pub struct DeterministicStartProof { ++ pub schema: String, ++ pub analyzer_policy: String, ++ pub module_sha256: String, ++ pub proof_sha256: String, ++ pub start_function_index: u32, ++ pub start_function_export: String, ++ pub transitive_function_indices: Vec, ++ pub imported_function_calls: u32, ++ pub memory_reads: String, ++ pub memory_effects: String, ++ pub global_effects: String, ++ pub table_effects: String, ++ pub requires_fresh_zeroed_memory: bool, ++ pub ordinary_start_execution_per_instance: bool, ++ pub first_instance_full_byte_validation: bool, ++} ++ ++/// Versioned, portable metadata bound into the sealed carrier manifest. ++#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] ++#[serde(rename_all = "kebab-case", deny_unknown_fields)] ++pub struct PreinitializedMemoryImageMetadata { ++ pub schema: String, ++ pub module_sha256: String, ++ pub runtime_abi_id: String, ++ pub phase: String, ++ pub mapping_alignment: u64, ++ pub mapped_size: u64, ++ pub memory_minimum_pages: u32, ++ pub memory_maximum_pages: Option, ++ pub memory_shared: bool, ++ pub memory_base: u64, ++ pub dylink_memory_size: u32, ++ pub dylink_memory_alignment: u32, ++ pub stack_low: u64, ++ #[serde(default, skip_serializing_if = "Option::is_none")] ++ pub deterministic_start_proof: Option, ++ #[serde(default, skip_serializing_if = "Option::is_none")] ++ pub deterministic_start_proof_output_sha256: Option, ++} ++ ++impl PreinitializedMemoryImageMetadata { ++ pub const SCHEMA: &'static str = "oliphaunt.wasix-postmaster.memory-image.v1"; ++ pub const ATTESTED_SCHEMA: &'static str = "oliphaunt.wasix-postmaster.memory-image.v2"; ++} ++ ++/// An immutable backing file and its carrier-attested initialization metadata. ++#[derive(Debug)] ++pub struct PreinitializedMemoryImage { ++ file: File, ++ module_hash: ModuleHash, ++ metadata: PreinitializedMemoryImageMetadata, ++ backing: PreinitializedMemoryImageBacking, ++ load_audit: PreinitializedMemoryImageLoadAudit, ++ attested_runtime_validation: OnceLock>>, ++ runtime_audit: PreinitializedMemoryImageRuntimeCounters, ++} ++ ++#[derive(Debug, Default)] ++struct PreinitializedMemoryImageRuntimeCounters { ++ ordinary_start_completed_instances: AtomicU64, ++ fresh_zeroed_instances: AtomicU64, ++ nonfresh_instances: AtomicU64, ++ validation_attempts: AtomicU64, ++ full_compare_attempts: AtomicU64, ++ full_compare_successes: AtomicU64, ++ full_compare_failures: AtomicU64, ++ compared_bytes: AtomicU64, ++ reuse_successes: AtomicU64, ++ reuse_failures: AtomicU64, ++ skipped_bytes: AtomicU64, ++ remap_successes: AtomicU64, ++ remap_failures: AtomicU64, ++ counter_overflow: AtomicBool, ++} ++ ++/// Terminal, non-forcing audit snapshot for one attested memory image. ++/// ++/// The product executor reads this only after the complete WASIX process tree ++/// has joined. It is intentionally bounded: no per-instance records, paths, ++/// timestamps, or general tracing state are retained. ++#[derive(Debug, Clone, PartialEq, Eq, Serialize)] ++pub struct PreinitializedMemoryImageRuntimeAudit { ++ pub module_sha256: String, ++ pub memory_image_schema: String, ++ pub proof_sha256: String, ++ pub proof_output_sha256: String, ++ pub mapped_size: u64, ++ pub ordinary_start_completed_instances: u64, ++ pub fresh_zeroed_instances: u64, ++ pub nonfresh_instances: u64, ++ pub validation_attempts: u64, ++ pub full_compare_attempts: u64, ++ pub full_compare_successes: u64, ++ pub full_compare_failures: u64, ++ pub compared_bytes: u64, ++ pub reuse_successes: u64, ++ pub reuse_failures: u64, ++ pub skipped_bytes: u64, ++ pub remap_successes: u64, ++ pub remap_failures: u64, ++ pub counter_overflow: bool, ++} ++ ++impl PreinitializedMemoryImageRuntimeCounters { ++ fn add(&self, counter: &AtomicU64, value: u64) { ++ if value == 0 { ++ return; ++ } ++ if counter ++ .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { ++ current.checked_add(value) ++ }) ++ .is_err() ++ { ++ self.counter_overflow.store(true, Ordering::Release); ++ } ++ } ++ ++ fn record_start_boundary(&self, fresh_zeroed_memory: bool) { ++ self.add(&self.ordinary_start_completed_instances, 1); ++ self.add(&self.validation_attempts, 1); ++ if fresh_zeroed_memory { ++ self.add(&self.fresh_zeroed_instances, 1); ++ } else { ++ self.add(&self.nonfresh_instances, 1); ++ } ++ } ++ ++ fn record_validation(&self, compared: bool, succeeded: bool, mapped_size: u64) { ++ match (compared, succeeded) { ++ (true, true) => { ++ self.add(&self.full_compare_attempts, 1); ++ self.add(&self.full_compare_successes, 1); ++ self.add(&self.compared_bytes, mapped_size); ++ } ++ (true, false) => { ++ self.add(&self.full_compare_attempts, 1); ++ self.add(&self.full_compare_failures, 1); ++ } ++ (false, true) => { ++ self.add(&self.reuse_successes, 1); ++ self.add(&self.skipped_bytes, mapped_size); ++ } ++ (false, false) => self.add(&self.reuse_failures, 1), ++ } ++ } ++ ++ fn record_remap(&self, succeeded: bool) { ++ if succeeded { ++ self.add(&self.remap_successes, 1); ++ } else { ++ self.add(&self.remap_failures, 1); ++ } ++ } ++} ++ ++/// Loader evidence captured while the verified source mapping is still live. ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub struct PreinitializedMemoryImageLoadAudit { ++ pub source_residency_after_hash_inspect: Option, ++ pub mapping_cache_eviction: FileAdviceAudit, ++} ++ ++/// Storage retained behind a preinitialized memory image. ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub enum PreinitializedMemoryImageBacking { ++ /// Runtime-owned, sealed anonymous storage populated from a mutable source. ++ SealedCopy, ++ /// The carrier inode itself, admitted only after kernel-backed immutability. ++ DirectIntrinsic(IntrinsicFileImmutability), ++} ++ ++impl PreinitializedMemoryImageBacking { ++ pub const fn audit_mode(self) -> &'static str { ++ match self { ++ Self::SealedCopy => "streamed-copy-sealed-backing", ++ Self::DirectIntrinsic(IntrinsicFileImmutability::ReadOnlyFilesystem) => { ++ "direct-read-only-filesystem" ++ } ++ Self::DirectIntrinsic(IntrinsicFileImmutability::ImmutableInode) => { ++ "direct-immutable-inode" ++ } ++ } ++ } ++} ++ ++/// Kernel property that makes a live file descriptor safe as immutable mmap ++/// backing. Unix permission bits are intentionally not sufficient. ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub enum IntrinsicFileImmutability { ++ /// The opened inode resides on SquashFS or EROFS. ++ ReadOnlyFilesystem, ++ /// `FS_IMMUTABLE_FL` is set and this process cannot clear it. ++ ImmutableInode, ++} ++ ++/// Deferred, single-flight source for a preinitialized memory image. ++/// ++/// Sealed carriers use this indirection so an exec-only image does not consume ++/// RSS or copy bytes until the corresponding executable is actually selected. ++/// Implementations must return the same immutable image (or the same failure) ++/// to every caller. ++pub trait PreinitializedMemoryImageLoader: fmt::Debug + Send + Sync + 'static { ++ fn load(&self) -> Result, String>; ++ ++ /// Return terminal audit counters only when this loader has already ++ /// materialized an attested image. This must never force image loading. ++ fn runtime_audit_if_loaded(&self) -> Option { ++ None ++ } ++} ++ ++/// Cloneable handle to a deferred preinitialized memory image. ++#[derive(Clone)] ++pub struct PreinitializedMemoryImageHandle { ++ loader: Arc, ++} ++ ++impl PreinitializedMemoryImageHandle { ++ pub fn new(loader: Arc) -> Self { ++ Self { loader } ++ } ++ ++ pub fn load(&self) -> Result, String> { ++ self.loader.load() ++ } ++ ++ /// Snapshot an already-loaded attested image without activating the ++ /// underlying loader or faulting image bytes. ++ pub fn runtime_audit_if_loaded(&self) -> Option { ++ self.loader.runtime_audit_if_loaded() ++ } ++} ++ ++impl fmt::Debug for PreinitializedMemoryImageHandle { ++ fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { ++ formatter ++ .debug_struct("PreinitializedMemoryImageHandle") ++ .finish_non_exhaustive() ++ } ++} ++ ++impl PreinitializedMemoryImage { ++ /// Validates carrier metadata without reading or materializing image bytes. ++ pub fn validate_metadata( ++ metadata: &PreinitializedMemoryImageMetadata, ++ module_hash: ModuleHash, ++ ) -> Result<(), String> { ++ validate_metadata_shape(metadata, module_hash) ++ } ++ ++ /// Copies an image into runtime-owned immutable storage while verifying its ++ /// carrier-attested digest. The source file is never used as mmap backing, ++ /// so a concurrent carrier-path replacement or later source write cannot ++ /// change the bytes between comparison and remapping. ++ pub fn from_verified_reader( ++ mut source: impl Read, ++ expected_image_sha256: [u8; 32], ++ module_hash: ModuleHash, ++ metadata: PreinitializedMemoryImageMetadata, ++ ) -> Result { ++ validate_metadata_shape(&metadata, module_hash)?; ++ ++ let mut file = new_immutable_snapshot_backing()?; ++ let mut hasher = Sha256::new(); ++ let mut copied = 0_u64; ++ let mut buffer = [0_u8; 128 * 1024]; ++ loop { ++ let count = source ++ .read(&mut buffer) ++ .map_err(|err| format!("read preinitialized memory image: {err}"))?; ++ if count == 0 { ++ break; ++ } ++ copied = copied ++ .checked_add(count as u64) ++ .ok_or_else(|| "preinitialized memory image size overflow".to_string())?; ++ if copied > metadata.mapped_size { ++ return Err(format!( ++ "preinitialized memory image exceeds metadata size: metadata={} copied={copied}", ++ metadata.mapped_size ++ )); ++ } ++ hasher.update(&buffer[..count]); ++ file.write_all(&buffer[..count]) ++ .map_err(|err| format!("copy preinitialized memory image: {err}"))?; ++ } ++ if copied != metadata.mapped_size { ++ return Err(format!( ++ "preinitialized memory image size mismatch: metadata={} copied={copied}", ++ metadata.mapped_size, ++ )); ++ } ++ let actual: [u8; 32] = hasher.finalize().into(); ++ if actual != expected_image_sha256 { ++ return Err(format!( ++ "preinitialized memory image SHA-256 mismatch: expected={} actual={}", ++ hex::encode(expected_image_sha256), ++ hex::encode(actual) ++ )); ++ } ++ // This is a private, live-FD runtime snapshot, not a durable ++ // publication. Completed writes are coherent with the later mapping; ++ // hashing plus sealing supplies integrity. A filesystem sync would add ++ // latency (and possibly flash traffic for an unlinked temp backing) ++ // without strengthening correctness. ++ seal_immutable_snapshot(&file)?; ++ let mapping = shared_buffer::OwnedBuffer::from_file(&file) ++ .map_err(|err| format!("map sealed preinitialized memory image: {err}"))?; ++ let mapped_len = u64::try_from(mapping.len()) ++ .map_err(|_| "sealed memory-image mapping length exceeds u64".to_string())?; ++ if mapped_len != metadata.mapped_size { ++ return Err(format!( ++ "sealed preinitialized memory image mapping size mismatch: metadata={} mapping={mapped_len}", ++ metadata.mapped_size ++ )); ++ } ++ let backing_digest: [u8; 32] = Sha256::digest(mapping.as_slice()).into(); ++ if backing_digest != expected_image_sha256 { ++ return Err(format!( ++ "sealed preinitialized memory image backing SHA-256 mismatch: expected={} actual={}", ++ hex::encode(expected_image_sha256), ++ hex::encode(backing_digest) ++ )); ++ } ++ let mapping_cache_eviction = advise_mapping_away(&mapping); ++ file.seek(SeekFrom::Start(0)) ++ .map_err(|err| format!("rewind preinitialized memory image: {err}"))?; ++ ++ Ok(Self { ++ file, ++ module_hash, ++ metadata, ++ backing: PreinitializedMemoryImageBacking::SealedCopy, ++ load_audit: PreinitializedMemoryImageLoadAudit { ++ source_residency_after_hash_inspect: None, ++ mapping_cache_eviction, ++ }, ++ attested_runtime_validation: OnceLock::new(), ++ runtime_audit: PreinitializedMemoryImageRuntimeCounters::default(), ++ }) ++ } ++ ++ /// Retains an intrinsically immutable carrier inode as the MAP_PRIVATE ++ /// backing after hashing the exact file mapping. This avoids an anonymous ++ /// copy while preserving a fresh private mapping for every instance. ++ #[cfg(target_os = "linux")] ++ pub fn from_verified_immutable_file( ++ file: File, ++ expected_image_sha256: [u8; 32], ++ module_hash: ModuleHash, ++ metadata: PreinitializedMemoryImageMetadata, ++ ) -> Result { ++ (move || { ++ validate_metadata_shape(&metadata, module_hash)?; ++ let immutable = intrinsic_file_immutability(&file)?; ++ let actual_size = file ++ .metadata() ++ .map_err(|err| format!("stat immutable preinitialized memory image: {err}"))? ++ .len(); ++ if actual_size != metadata.mapped_size { ++ return Err(format!( ++ "preinitialized memory image size mismatch: metadata={} actual={actual_size}", ++ metadata.mapped_size ++ )); ++ } ++ ++ let mapping = shared_buffer::OwnedBuffer::from_file(&file) ++ .map_err(|err| format!("map immutable preinitialized memory image: {err}"))?; ++ let mapped_len = u64::try_from(mapping.len()) ++ .map_err(|_| "immutable memory-image mapping length exceeds u64".to_string())?; ++ if mapped_len != metadata.mapped_size { ++ return Err(format!( ++ "immutable preinitialized memory image mapping size mismatch: metadata={} mapping={}", ++ metadata.mapped_size, mapped_len ++ )); ++ } ++ let actual: [u8; 32] = Sha256::digest(mapping.as_slice()).into(); ++ if actual != expected_image_sha256 { ++ return Err(format!( ++ "preinitialized memory image SHA-256 mismatch: expected={} actual={}", ++ hex::encode(expected_image_sha256), ++ hex::encode(actual) ++ )); ++ } ++ let source_residency_after_hash_inspect = file_residency(&file, metadata.mapped_size); ++ let mapping_cache_eviction = advise_mapping_away(&mapping); ++ ++ Ok(Self { ++ file, ++ module_hash, ++ metadata, ++ backing: PreinitializedMemoryImageBacking::DirectIntrinsic(immutable), ++ load_audit: PreinitializedMemoryImageLoadAudit { ++ source_residency_after_hash_inspect: Some(source_residency_after_hash_inspect), ++ mapping_cache_eviction, ++ }, ++ attested_runtime_validation: OnceLock::new(), ++ runtime_audit: PreinitializedMemoryImageRuntimeCounters::default(), ++ }) ++ })() ++ } ++ ++ #[cfg(not(target_os = "linux"))] ++ pub fn from_verified_immutable_file( ++ _file: File, ++ _expected_image_sha256: [u8; 32], ++ _module_hash: ModuleHash, ++ _metadata: PreinitializedMemoryImageMetadata, ++ ) -> Result { ++ Err("direct immutable memory-image backing is only implemented on Linux".to_string()) ++ } ++ ++ pub fn metadata(&self) -> &PreinitializedMemoryImageMetadata { ++ &self.metadata ++ } ++ ++ pub fn backing(&self) -> PreinitializedMemoryImageBacking { ++ self.backing ++ } ++ ++ pub fn load_audit(&self) -> PreinitializedMemoryImageLoadAudit { ++ self.load_audit ++ } ++ ++ /// Return the bounded process-lifetime counters for a v2 attested image. ++ /// Legacy v1 images deliberately have no summary in the attested schema. ++ pub fn runtime_audit(&self) -> Option { ++ let proof = self.metadata.deterministic_start_proof.as_ref()?; ++ let proof_output_sha256 = self ++ .metadata ++ .deterministic_start_proof_output_sha256 ++ .as_ref()?; ++ Some(PreinitializedMemoryImageRuntimeAudit { ++ module_sha256: self.module_hash.to_string().to_ascii_lowercase(), ++ memory_image_schema: self.metadata.schema.clone(), ++ proof_sha256: proof.proof_sha256.clone(), ++ proof_output_sha256: proof_output_sha256.clone(), ++ mapped_size: self.metadata.mapped_size, ++ ordinary_start_completed_instances: self ++ .runtime_audit ++ .ordinary_start_completed_instances ++ .load(Ordering::Acquire), ++ fresh_zeroed_instances: self ++ .runtime_audit ++ .fresh_zeroed_instances ++ .load(Ordering::Acquire), ++ nonfresh_instances: self ++ .runtime_audit ++ .nonfresh_instances ++ .load(Ordering::Acquire), ++ validation_attempts: self ++ .runtime_audit ++ .validation_attempts ++ .load(Ordering::Acquire), ++ full_compare_attempts: self ++ .runtime_audit ++ .full_compare_attempts ++ .load(Ordering::Acquire), ++ full_compare_successes: self ++ .runtime_audit ++ .full_compare_successes ++ .load(Ordering::Acquire), ++ full_compare_failures: self ++ .runtime_audit ++ .full_compare_failures ++ .load(Ordering::Acquire), ++ compared_bytes: self.runtime_audit.compared_bytes.load(Ordering::Acquire), ++ reuse_successes: self.runtime_audit.reuse_successes.load(Ordering::Acquire), ++ reuse_failures: self.runtime_audit.reuse_failures.load(Ordering::Acquire), ++ skipped_bytes: self.runtime_audit.skipped_bytes.load(Ordering::Acquire), ++ remap_successes: self.runtime_audit.remap_successes.load(Ordering::Acquire), ++ remap_failures: self.runtime_audit.remap_failures.load(Ordering::Acquire), ++ counter_overflow: self.runtime_audit.counter_overflow.load(Ordering::Acquire), ++ }) ++ } ++ ++ pub(crate) fn apply( ++ &self, ++ _ordinary_start_completed: OrdinaryModuleStartCompleted, ++ main_module: &Module, ++ memory: &Memory, ++ store: &mut impl AsStoreMut, ++ memory_type: MemoryType, ++ dylink: &DylinkInfo, ++ memory_base: u64, ++ stack_low: u64, ++ fresh_zeroed_memory: bool, ++ ) -> Result<(), LinkError> { ++ // `Linker::new` can call this method only after ++ // `Instance::new_with_export_names` has synchronously completed the ++ // module start function. This is therefore the exact ordinary-start ++ // completion boundary for every image-backed fresh backend. ++ self.runtime_audit ++ .record_start_boundary(fresh_zeroed_memory); ++ validate_runtime_layout( ++ &self.metadata, ++ self.module_hash, ++ main_module, ++ memory_type, ++ dylink, ++ memory_base, ++ stack_low, ++ fresh_zeroed_memory, ++ ) ++ .map_err(LinkError::PreinitializedMemoryImage)?; ++ ++ #[cfg(unix)] ++ { ++ let mapped_size = usize::try_from(self.metadata.mapped_size).map_err(|_| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "preinitialized memory image exceeds host address width: {}", ++ self.metadata.mapped_size ++ )) ++ })?; ++ let runtime_bytes = self.validate_runtime_bytes(fresh_zeroed_memory, || { ++ compare_memory_to_file(memory, store, &self.file, self.metadata.mapped_size) ++ }); ++ self.runtime_audit.record_validation( ++ runtime_bytes.compared_this_instance, ++ runtime_bytes.validation.is_ok(), ++ self.metadata.mapped_size, ++ ); ++ // Comparison necessarily faults the complete image source, but a ++ // backend may touch only a fraction of that prefix. Release those ++ // validation-only cache references before installing the private ++ // file mapping so its steady-state working set is demand-paged. ++ // This deliberately trades at most one later refault per used clean ++ // page for not pinning every compared page. DONTNEED does not ++ // modify the still-live anonymous guest memory, and correctness ++ // remains independent of whether the kernel accepts or acts on it. ++ // An attested cache hit faulted no validation-only pages and must ++ // not evict clean pages that active private mappings may share. ++ let source_cache_eviction = if runtime_bytes.compared_this_instance { ++ release_validation_faulted_source_pages(self.backing, &self.file) ++ } else { ++ FileAdviceAudit::not_applicable() ++ }; ++ let compared_bytes = if runtime_bytes.compared_this_instance { ++ self.metadata.mapped_size ++ } else { ++ 0 ++ }; ++ let skipped_bytes = ++ if runtime_bytes.validation.is_ok() && !runtime_bytes.compared_this_instance { ++ self.metadata.mapped_size ++ } else { ++ 0 ++ }; ++ let validation_state = if self.metadata.deterministic_start_proof.is_none() { ++ "compared-per-instance" ++ } else if runtime_bytes.compared_this_instance { ++ "attested-first-instance-compared" ++ } else if runtime_bytes.validation.is_ok() { ++ "attested-validation-reused" ++ } else { ++ "attested-failure-reused" ++ }; ++ tracing::debug!( ++ target: "wasmer_wasix::preinitialized_memory_image_audit", ++ audit_schema = "oliphaunt.wasix-postmaster.sealed-loader-audit.v2", ++ artifact_kind = "preinitialized-memory-runtime-validation", ++ module_sha256 = %self.module_hash, ++ validation_state, ++ validation_succeeded = runtime_bytes.validation.is_ok(), ++ compared_bytes, ++ skipped_bytes, ++ backing_mode = self.backing.audit_mode(), ++ source_cache_eviction_supported = source_cache_eviction.supported, ++ source_cache_eviction_calls = source_cache_eviction.calls, ++ source_cache_eviction_successes = source_cache_eviction.successes, ++ source_cache_eviction_errno = source_cache_eviction.first_errno, ++ "validated deterministic module-start bytes" ++ ); ++ runtime_bytes ++ .validation ++ .map_err(LinkError::PreinitializedMemoryImage)?; ++ // SAFETY: module start has returned, the linker exclusively owns ++ // the store, and no guest thread can run before linking returns. ++ // `self.file` is runtime-owned and immutable for this image's ++ // lifetime. ++ let remap = ++ unsafe { memory.remap_private_file_fixed(store, 0, mapped_size, &self.file, 0) }; ++ self.runtime_audit.record_remap(remap.is_ok()); ++ remap.map_err(|err| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "private file remapping failed: {err}" ++ )) ++ })?; ++ Ok(()) ++ } ++ ++ #[cfg(not(unix))] ++ { ++ let _ = (memory, store); ++ Err(LinkError::PreinitializedMemoryImage( ++ "private preinitialized memory-image mapping is not implemented on this host" ++ .to_string(), ++ )) ++ } ++ } ++ ++ /// Run ordinary per-instance validation for v1 images. For an analyzer- ++ /// proven v2 image, execute one process-wide differential validation and ++ /// reuse its immutable success or failure. The proof contract requires a ++ /// freshly allocated zeroed memory and ordinary start on every instance. ++ fn validate_runtime_bytes( ++ &self, ++ fresh_zeroed_memory: bool, ++ compare: impl FnOnce() -> Result<(), String>, ++ ) -> RuntimeByteValidation { ++ if self.metadata.deterministic_start_proof.is_none() { ++ return RuntimeByteValidation { ++ validation: compare(), ++ compared_this_instance: true, ++ }; ++ } ++ if !fresh_zeroed_memory { ++ return RuntimeByteValidation { ++ validation: Err( ++ "attested deterministic-start validation requires fresh zeroed memory" ++ .to_string(), ++ ), ++ compared_this_instance: false, ++ }; ++ } ++ ++ let mut compared_this_instance = false; ++ let validation = self.attested_runtime_validation.get_or_init(|| { ++ compared_this_instance = true; ++ compare().map_err(Arc::::from) ++ }); ++ RuntimeByteValidation { ++ validation: validation.clone().map_err(|message| message.to_string()), ++ compared_this_instance, ++ } ++ } ++} ++ ++#[derive(Debug)] ++struct RuntimeByteValidation { ++ validation: Result<(), String>, ++ compared_this_instance: bool, ++} ++ ++/// A builder-only request to capture the exact post-start memory prefix. ++#[derive(Debug, Clone)] ++pub struct PreinitializedMemoryImageCapture { ++ pub image_path: PathBuf, ++ pub receipt_path: PathBuf, ++ pub module_hash: ModuleHash, ++} ++ ++impl PreinitializedMemoryImageCapture { ++ pub(crate) fn capture( ++ &self, ++ main_module: &Module, ++ memory: &Memory, ++ store: &impl AsStoreRef, ++ memory_type: MemoryType, ++ dylink: &DylinkInfo, ++ memory_base: u64, ++ stack_low: u64, ++ ) -> Result<(), LinkError> { ++ let module_hash = main_module.info().hash.ok_or_else(|| { ++ LinkError::PreinitializedMemoryImage( ++ "captured module has no embedded raw module hash".to_string(), ++ ) ++ })?; ++ if module_hash != self.module_hash { ++ return Err(LinkError::PreinitializedMemoryImage(format!( ++ "capture module hash mismatch: request={} module={module_hash}", ++ self.module_hash ++ ))); ++ } ++ ++ let mapped_size = stack_low / PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT ++ * PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT; ++ if mapped_size == 0 { ++ return Err(LinkError::PreinitializedMemoryImage( ++ "capture layout has no full 64-KiB page before the guest stack".to_string(), ++ )); ++ } ++ let view = memory.view(store); ++ if mapped_size > view.data_size() { ++ return Err(LinkError::PreinitializedMemoryImage(format!( ++ "capture range exceeds current memory: range={mapped_size} memory={}", ++ view.data_size() ++ ))); ++ } ++ ++ let mut image = create_new_regular_file(&self.image_path) ++ .map_err(LinkError::PreinitializedMemoryImage)?; ++ // SAFETY: module start has returned and no guest code can execute while ++ // the linker owns the store. The view is dropped before any remapping. ++ let initialized = unsafe { view.data_unchecked() }; ++ let mapped_size_usize = usize::try_from(mapped_size).map_err(|_| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "captured memory image exceeds host address width: {mapped_size}" ++ )) ++ })?; ++ image ++ .write_all(&initialized[..mapped_size_usize]) ++ .map_err(|err| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "write captured memory image {}: {err}", ++ self.image_path.display() ++ )) ++ })?; ++ image.sync_all().map_err(|err| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "sync captured memory image {}: {err}", ++ self.image_path.display() ++ )) ++ })?; ++ drop(image); ++ ++ let metadata = PreinitializedMemoryImageMetadata { ++ schema: PreinitializedMemoryImageMetadata::SCHEMA.to_string(), ++ module_sha256: module_hash.to_string().to_ascii_lowercase(), ++ runtime_abi_id: option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID") ++ .unwrap_or("") ++ .to_string(), ++ phase: PREINITIALIZED_MEMORY_IMAGE_PHASE.to_string(), ++ mapping_alignment: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, ++ mapped_size, ++ memory_minimum_pages: memory_type.minimum.0, ++ memory_maximum_pages: memory_type.maximum.map(|pages| pages.0), ++ memory_shared: memory_type.shared, ++ memory_base, ++ dylink_memory_size: dylink.mem_info.memory_size, ++ dylink_memory_alignment: dylink.mem_info.memory_alignment, ++ stack_low, ++ deterministic_start_proof: None, ++ deterministic_start_proof_output_sha256: None, ++ }; ++ validate_metadata_shape(&metadata, module_hash) ++ .map_err(LinkError::PreinitializedMemoryImage)?; ++ ++ let mut receipt = create_new_regular_file(&self.receipt_path) ++ .map_err(LinkError::PreinitializedMemoryImage)?; ++ serde_json::to_writer_pretty(&mut receipt, &metadata).map_err(|err| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "write memory image receipt {}: {err}", ++ self.receipt_path.display() ++ )) ++ })?; ++ receipt.write_all(b"\n").map_err(|err| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "finish memory image receipt {}: {err}", ++ self.receipt_path.display() ++ )) ++ })?; ++ receipt.sync_all().map_err(|err| { ++ LinkError::PreinitializedMemoryImage(format!( ++ "sync memory image receipt {}: {err}", ++ self.receipt_path.display() ++ )) ++ })?; ++ Ok(()) ++ } ++} ++ ++/// Operation performed at the post-module-start linker boundary. ++#[derive(Debug, Clone)] ++pub enum PreinitializedMemoryImageMode { ++ Apply(Arc), ++ Capture(PreinitializedMemoryImageCapture), ++} ++ ++fn validate_metadata_shape( ++ metadata: &PreinitializedMemoryImageMetadata, ++ module_hash: ModuleHash, ++) -> Result<(), String> { ++ if metadata.schema != PreinitializedMemoryImageMetadata::SCHEMA ++ && metadata.schema != PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA ++ { ++ return Err(format!( ++ "preinitialized memory image schema mismatch: {}", ++ metadata.schema ++ )); ++ } ++ if metadata.phase != PREINITIALIZED_MEMORY_IMAGE_PHASE { ++ return Err(format!( ++ "preinitialized memory image phase mismatch: {}", ++ metadata.phase ++ )); ++ } ++ if metadata.module_sha256 != module_hash.to_string().to_ascii_lowercase() { ++ return Err(format!( ++ "preinitialized memory image module hash mismatch: metadata={} module={module_hash}", ++ metadata.module_sha256 ++ )); ++ } ++ if metadata.runtime_abi_id != option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID").unwrap_or("") { ++ return Err(format!( ++ "preinitialized memory image runtime ABI mismatch: metadata={} runtime={}", ++ metadata.runtime_abi_id, ++ option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID").unwrap_or("") ++ )); ++ } ++ if metadata.mapping_alignment != PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT ++ || metadata.mapped_size == 0 ++ || !metadata ++ .mapped_size ++ .is_multiple_of(PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT) ++ { ++ return Err("preinitialized memory image has an invalid 64-KiB mapping range".to_string()); ++ } ++ usize::try_from(metadata.mapped_size).map_err(|_| { ++ format!( ++ "preinitialized memory image exceeds host address width: {}", ++ metadata.mapped_size ++ ) ++ })?; ++ match ( ++ metadata.schema.as_str(), ++ metadata.deterministic_start_proof.as_ref(), ++ metadata.deterministic_start_proof_output_sha256.as_deref(), ++ ) { ++ (PreinitializedMemoryImageMetadata::SCHEMA, None, None) => {} ++ (PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA, Some(proof), Some(output_sha256)) => { ++ validate_deterministic_start_proof(proof, output_sha256, module_hash)?; ++ } ++ (PreinitializedMemoryImageMetadata::SCHEMA, _, _) => { ++ return Err( ++ "v1 preinitialized memory images cannot contain start-proof evidence".to_string(), ++ ); ++ } ++ (PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA, _, _) => { ++ return Err( ++ "v2 preinitialized memory images require a start proof and canonical output digest" ++ .to_string(), ++ ); ++ } ++ _ => unreachable!("memory image schema was checked above"), ++ } ++ Ok(()) ++} ++ ++fn validate_deterministic_start_proof( ++ proof: &DeterministicStartProof, ++ output_sha256: &str, ++ module_hash: ModuleHash, ++) -> Result<(), String> { ++ if proof.schema != DETERMINISTIC_START_PROOF_SCHEMA ++ || proof.analyzer_policy != DETERMINISTIC_START_ANALYZER_POLICY ++ || proof.module_sha256 != module_hash.to_string().to_ascii_lowercase() ++ || proof.imported_function_calls != 0 ++ || proof.memory_reads != DETERMINISTIC_START_MEMORY_READS ++ || proof.memory_effects != DETERMINISTIC_START_MEMORY_EFFECTS ++ || proof.global_effects != DETERMINISTIC_START_GLOBAL_EFFECTS ++ || proof.table_effects != DETERMINISTIC_START_TABLE_EFFECTS ++ || !proof.requires_fresh_zeroed_memory ++ || !proof.ordinary_start_execution_per_instance ++ || !proof.first_instance_full_byte_validation ++ { ++ return Err( ++ "preinitialized memory image has an invalid deterministic-start proof policy" ++ .to_string(), ++ ); ++ } ++ if proof.start_function_export.is_empty() ++ || proof.transitive_function_indices.is_empty() ++ || !proof ++ .transitive_function_indices ++ .contains(&proof.start_function_index) ++ || proof ++ .transitive_function_indices ++ .windows(2) ++ .any(|pair| pair[0] >= pair[1]) ++ { ++ return Err("deterministic-start proof has an invalid function closure".to_string()); ++ } ++ if proof.proof_sha256.len() != 64 ++ || !proof ++ .proof_sha256 ++ .bytes() ++ .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) ++ { ++ return Err("deterministic-start proof digest is not canonical SHA-256".to_string()); ++ } ++ if output_sha256.len() != 64 ++ || !output_sha256 ++ .bytes() ++ .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) ++ || output_sha256 != canonical_deterministic_start_proof_sha256(proof)? ++ { ++ return Err( ++ "deterministic-start proof canonical output digest does not match the proof" ++ .to_string(), ++ ); ++ } ++ Ok(()) ++} ++ ++fn canonical_deterministic_start_proof_sha256( ++ proof: &DeterministicStartProof, ++) -> Result { ++ let value = serde_json::to_value(proof) ++ .map_err(|err| format!("serialize deterministic-start proof: {err}"))?; ++ let object = value ++ .as_object() ++ .ok_or_else(|| "deterministic-start proof did not serialize as an object".to_string())?; ++ let canonical = object ++ .iter() ++ .map(|(key, value)| (key.clone(), value.clone())) ++ .collect::>(); ++ let bytes = serde_json::to_vec(&canonical) ++ .map_err(|err| format!("canonicalize deterministic-start proof: {err}"))?; ++ Ok(hex::encode(Sha256::digest(bytes))) ++} ++ ++#[allow(clippy::too_many_arguments)] ++fn validate_runtime_layout( ++ metadata: &PreinitializedMemoryImageMetadata, ++ module_hash: ModuleHash, ++ main_module: &Module, ++ memory_type: MemoryType, ++ dylink: &DylinkInfo, ++ memory_base: u64, ++ stack_low: u64, ++ fresh_zeroed_memory: bool, ++) -> Result<(), String> { ++ validate_metadata_shape(metadata, module_hash)?; ++ if main_module.info().hash != Some(module_hash) { ++ return Err("preinitialized memory image is paired with a different module".to_string()); ++ } ++ let expected_mapped_size = ++ stack_low / PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT * PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT; ++ if metadata.mapped_size != expected_mapped_size ++ || metadata.memory_minimum_pages != memory_type.minimum.0 ++ || metadata.memory_maximum_pages != memory_type.maximum.map(|pages| pages.0) ++ || metadata.memory_shared != memory_type.shared ++ || metadata.memory_base != memory_base ++ || metadata.dylink_memory_size != dylink.mem_info.memory_size ++ || metadata.dylink_memory_alignment != dylink.mem_info.memory_alignment ++ || metadata.stack_low != stack_low ++ { ++ return Err( ++ "preinitialized memory image layout does not match the linked module".to_string(), ++ ); ++ } ++ if let Some(proof) = &metadata.deterministic_start_proof { ++ if !fresh_zeroed_memory { ++ return Err( ++ "attested deterministic-start image requires fresh zeroed memory".to_string(), ++ ); ++ } ++ let expected_start = wasmer_types::FunctionIndex::from_u32(proof.start_function_index); ++ if main_module.info().start_function != Some(expected_start) ++ || main_module.info().exports.get(&proof.start_function_export) ++ != Some(&wasmer_types::ExportIndex::Function(expected_start)) ++ { ++ return Err( ++ "deterministic-start proof does not match the module start function".to_string(), ++ ); ++ } ++ } ++ Ok(()) ++} ++ ++#[cfg(unix)] ++fn compare_memory_to_file( ++ memory: &Memory, ++ store: &impl AsStoreRef, ++ file: &File, ++ len: u64, ++) -> Result<(), String> { ++ use std::os::unix::fs::FileExt; ++ ++ let view = memory.view(store); ++ if len > view.data_size() { ++ return Err(format!( ++ "preinitialized memory image exceeds current memory: image={len} memory={}", ++ view.data_size() ++ )); ++ } ++ // SAFETY: module start has returned and the linker exclusively owns the ++ // store. No guest code runs until this function returns. ++ let len = usize::try_from(len) ++ .map_err(|_| format!("preinitialized memory image exceeds host address width: {len}"))?; ++ let current = unsafe { &view.data_unchecked()[..len] }; ++ let mut buffer = vec![0_u8; PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT as usize]; ++ let mut offset = 0_usize; ++ while offset < current.len() { ++ let count = buffer.len().min(current.len() - offset); ++ let mut read = 0_usize; ++ while read < count { ++ let got = file ++ .read_at(&mut buffer[read..count], (offset + read) as u64) ++ .map_err(|err| format!("read preinitialized memory image: {err}"))?; ++ if got == 0 { ++ return Err("preinitialized memory image ended during comparison".to_string()); ++ } ++ read += got; ++ } ++ if current[offset..offset + count] != buffer[..count] { ++ let mismatch = current[offset..offset + count] ++ .iter() ++ .zip(&buffer[..count]) ++ .position(|(actual, expected)| actual != expected) ++ .unwrap_or(0); ++ return Err(format!( ++ "preinitialized memory image differs from module-start state at byte {}", ++ offset + mismatch ++ )); ++ } ++ offset += count; ++ } ++ Ok(()) ++} ++ ++fn create_new_regular_file(path: &PathBuf) -> Result { ++ let mut options = OpenOptions::new(); ++ options.create_new(true).read(true).write(true); ++ #[cfg(unix)] ++ { ++ use std::os::unix::fs::OpenOptionsExt; ++ options.mode(0o600).custom_flags(libc::O_CLOEXEC); ++ } ++ options ++ .open(path) ++ .map_err(|err| format!("create {}: {err}", path.display())) ++} ++ ++#[cfg(target_os = "linux")] ++fn new_immutable_snapshot_backing() -> Result { ++ use std::{ffi::CString, os::fd::FromRawFd}; ++ ++ let name = CString::new("oliphaunt-wasix-preinitialized-memory") ++ .map_err(|err| format!("construct memory-image backing name: {err}"))?; ++ // SAFETY: the name is NUL terminated and the flags are valid for ++ // memfd_create. A successful descriptor is transferred into File below. ++ let descriptor = unsafe { ++ libc::syscall( ++ libc::SYS_memfd_create, ++ name.as_ptr(), ++ libc::MFD_CLOEXEC | libc::MFD_ALLOW_SEALING, ++ ) ++ }; ++ if descriptor < 0 { ++ return Err(format!( ++ "create sealed memory-image backing: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ // SAFETY: memfd_create returned a new owned descriptor on success. ++ Ok(unsafe { File::from_raw_fd(descriptor as i32) }) ++} ++ ++#[cfg(not(target_os = "linux"))] ++fn new_immutable_snapshot_backing() -> Result { ++ tempfile::tempfile().map_err(|err| format!("create unlinked memory-image backing: {err}")) ++} ++ ++#[cfg(target_os = "linux")] ++fn seal_immutable_snapshot(file: &File) -> Result<(), String> { ++ use std::os::fd::AsRawFd; ++ ++ let seals = libc::F_SEAL_WRITE | libc::F_SEAL_GROW | libc::F_SEAL_SHRINK | libc::F_SEAL_SEAL; ++ // SAFETY: fcntl is called with F_ADD_SEALS on a live memfd descriptor. ++ let result = unsafe { libc::fcntl(file.as_raw_fd(), libc::F_ADD_SEALS, seals) }; ++ if result != 0 { ++ return Err(format!( ++ "seal memory-image backing: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ Ok(()) ++} ++ ++#[cfg(not(target_os = "linux"))] ++fn seal_immutable_snapshot(_file: &File) -> Result<(), String> { ++ // The backing is an unlinked file and its only writable handle is retained ++ // privately by this type. Mapping support still fails closed on non-Unix. ++ Ok(()) ++} ++ ++/// Proves that the opened Linux inode cannot be modified through ordinary ++/// runtime privileges. The decision is tied to the live descriptor, not its ++/// path or advisory Unix mode bits. ++#[cfg(target_os = "linux")] ++pub fn intrinsic_file_immutability(file: &File) -> Result { ++ use std::os::fd::AsRawFd; ++ ++ if !file ++ .metadata() ++ .map_err(|err| format!("stat immutable-file candidate: {err}"))? ++ .is_file() ++ { ++ return Err("immutable-file candidate is not a regular file".to_string()); ++ } ++ ++ let mut stat = std::mem::MaybeUninit::::uninit(); ++ // SAFETY: `stat` points to writable storage and the descriptor remains live. ++ let result = unsafe { libc::fstatfs(file.as_raw_fd(), stat.as_mut_ptr()) }; ++ if result != 0 { ++ return Err(format!( ++ "identify immutable-file filesystem: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ // SAFETY: successful fstatfs initialized the value. ++ let filesystem_type = unsafe { stat.assume_init() }.f_type as u64; ++ if let Some(read_only) = classify_intrinsic_immutability(filesystem_type, None, true) { ++ return Ok(read_only); ++ } ++ ++ let mut inode_flags: libc::c_long = 0; ++ // SAFETY: FS_IOC_GETFLAGS only reads flags from this live descriptor into ++ // the suitably sized output word. ++ let flags_result = ++ unsafe { libc::ioctl(file.as_raw_fd(), libc::FS_IOC_GETFLAGS, &mut inode_flags) }; ++ let flags_error = (flags_result != 0).then(std::io::Error::last_os_error); ++ let inode_flags = (flags_result == 0).then_some(inode_flags as u32); ++ let can_clear_immutable = process_has_permitted_linux_immutable()?; ++ ++ classify_intrinsic_immutability(filesystem_type, inode_flags, can_clear_immutable) ++ .ok_or_else(|| { ++ if flags_result == 0 { ++ "file is neither on SquashFS/EROFS nor protected by an immutable inode flag the runtime cannot clear".to_string() ++ } else { ++ format!( ++ "file is not on SquashFS/EROFS and inode flags are unavailable: {}", ++ flags_error.expect("failed ioctl must retain its error") ++ ) ++ } ++ }) ++} ++ ++#[cfg(not(target_os = "linux"))] ++pub fn intrinsic_file_immutability(_file: &File) -> Result { ++ Err("intrinsically immutable file detection is only implemented on Linux".to_string()) ++} ++ ++#[cfg(target_os = "linux")] ++fn classify_intrinsic_immutability( ++ filesystem_type: u64, ++ inode_flags: Option, ++ can_clear_immutable: bool, ++) -> Option { ++ const SQUASHFS_MAGIC: u64 = 0x7371_7368; ++ const EROFS_SUPER_MAGIC: u64 = 0xe0f5_e1e2; ++ const FS_IMMUTABLE_FL: u32 = 0x0000_0010; ++ ++ if matches!(filesystem_type, SQUASHFS_MAGIC | EROFS_SUPER_MAGIC) { ++ Some(IntrinsicFileImmutability::ReadOnlyFilesystem) ++ } else if inode_flags.is_some_and(|flags| flags & FS_IMMUTABLE_FL != 0) && !can_clear_immutable ++ { ++ Some(IntrinsicFileImmutability::ImmutableInode) ++ } else { ++ None ++ } ++} ++ ++#[cfg(target_os = "linux")] ++fn process_has_permitted_linux_immutable() -> Result { ++ #[repr(C)] ++ struct CapabilityHeader { ++ version: u32, ++ pid: i32, ++ } ++ #[repr(C)] ++ #[derive(Clone, Copy)] ++ struct CapabilityData { ++ effective: u32, ++ permitted: u32, ++ inheritable: u32, ++ } ++ ++ const LINUX_CAPABILITY_VERSION_3: u32 = 0x2008_0522; ++ const CAP_LINUX_IMMUTABLE: u32 = 9; ++ let mut header = CapabilityHeader { ++ version: LINUX_CAPABILITY_VERSION_3, ++ pid: 0, ++ }; ++ let mut data = [CapabilityData { ++ effective: 0, ++ permitted: 0, ++ inheritable: 0, ++ }; 2]; ++ // SAFETY: capget writes exactly two v3 data words for the calling thread. ++ let result = unsafe { ++ libc::syscall( ++ libc::SYS_capget, ++ &mut header as *mut CapabilityHeader, ++ data.as_mut_ptr(), ++ ) ++ }; ++ if result != 0 { ++ return Err(format!( ++ "read process capabilities for immutable-file admission: {}", ++ std::io::Error::last_os_error() ++ )); ++ } ++ let word = (CAP_LINUX_IMMUTABLE / 32) as usize; ++ let bit = 1_u32 << (CAP_LINUX_IMMUTABLE % 32); ++ Ok(data[word].permitted & bit != 0) ++} ++ ++#[cfg(unix)] ++fn advise_mapping_away(mapping: &shared_buffer::OwnedBuffer) -> FileAdviceAudit { ++ if mapping.is_empty() { ++ FileAdviceAudit::not_applicable() ++ } else { ++ // SAFETY: the read-only mapping remains live for this call and the hint ++ // does not affect file contents. ++ let result = unsafe { ++ libc::madvise( ++ mapping.as_ptr().cast_mut().cast(), ++ mapping.len(), ++ libc::MADV_DONTNEED, ++ ) ++ }; ++ let errno = (result != 0).then(|| { ++ std::io::Error::last_os_error() ++ .raw_os_error() ++ .unwrap_or(libc::EIO) ++ }); ++ FileAdviceAudit { ++ supported: true, ++ calls: 1, ++ successes: u64::from(result == 0), ++ first_errno: errno, ++ } ++ } ++} ++ ++fn release_validation_faulted_source_pages( ++ backing: PreinitializedMemoryImageBacking, ++ file: &File, ++) -> FileAdviceAudit { ++ if matches!( ++ backing, ++ PreinitializedMemoryImageBacking::DirectIntrinsic(_) ++ ) { ++ advise_file_away(file) ++ } else { ++ FileAdviceAudit::not_applicable() ++ } ++} ++ ++#[cfg(not(unix))] ++fn advise_mapping_away(_mapping: &shared_buffer::OwnedBuffer) -> FileAdviceAudit { ++ FileAdviceAudit::unsupported() ++} ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ use std::io::Cursor; ++ ++ fn metadata(module_hash: ModuleHash) -> PreinitializedMemoryImageMetadata { ++ PreinitializedMemoryImageMetadata { ++ schema: PreinitializedMemoryImageMetadata::SCHEMA.to_string(), ++ module_sha256: module_hash.to_string().to_ascii_lowercase(), ++ runtime_abi_id: option_env!("OLIPHAUNT_WASIX_RUNTIME_ABI_ID") ++ .unwrap_or("") ++ .to_string(), ++ phase: PREINITIALIZED_MEMORY_IMAGE_PHASE.to_string(), ++ mapping_alignment: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, ++ mapped_size: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, ++ memory_minimum_pages: 1, ++ memory_maximum_pages: None, ++ memory_shared: true, ++ memory_base: 4096, ++ dylink_memory_size: 1, ++ dylink_memory_alignment: 12, ++ stack_low: PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT, ++ deterministic_start_proof: None, ++ deterministic_start_proof_output_sha256: None, ++ } ++ } ++ ++ fn deterministic_start_proof(module_hash: ModuleHash) -> DeterministicStartProof { ++ DeterministicStartProof { ++ schema: DETERMINISTIC_START_PROOF_SCHEMA.to_string(), ++ analyzer_policy: DETERMINISTIC_START_ANALYZER_POLICY.to_string(), ++ module_sha256: module_hash.to_string().to_ascii_lowercase(), ++ proof_sha256: "ab".repeat(32), ++ start_function_index: 7, ++ start_function_export: "__wasm_init_memory".to_string(), ++ transitive_function_indices: vec![7, 8], ++ imported_function_calls: 0, ++ memory_reads: DETERMINISTIC_START_MEMORY_READS.to_string(), ++ memory_effects: DETERMINISTIC_START_MEMORY_EFFECTS.to_string(), ++ global_effects: DETERMINISTIC_START_GLOBAL_EFFECTS.to_string(), ++ table_effects: DETERMINISTIC_START_TABLE_EFFECTS.to_string(), ++ requires_fresh_zeroed_memory: true, ++ ordinary_start_execution_per_instance: true, ++ first_instance_full_byte_validation: true, ++ } ++ } ++ ++ #[test] ++ fn metadata_rejects_mapping_ranges_that_are_not_wasm_page_aligned() { ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let mut metadata = metadata(module_hash); ++ metadata.mapped_size -= 1; ++ assert!(validate_metadata_shape(&metadata, module_hash).is_err()); ++ } ++ ++ #[test] ++ fn verified_reader_rejects_short_long_and_corrupt_images() { ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let metadata = metadata(module_hash); ++ let bytes = vec![0x5a; metadata.mapped_size as usize]; ++ let digest: [u8; 32] = Sha256::digest(&bytes).into(); ++ ++ assert!( ++ PreinitializedMemoryImage::from_verified_reader( ++ Cursor::new(&bytes[..bytes.len() - 1]), ++ digest, ++ module_hash, ++ metadata.clone(), ++ ) ++ .is_err() ++ ); ++ ++ let mut long = bytes.clone(); ++ long.push(0); ++ assert!( ++ PreinitializedMemoryImage::from_verified_reader( ++ Cursor::new(long), ++ digest, ++ module_hash, ++ metadata.clone(), ++ ) ++ .is_err() ++ ); ++ ++ let mut corrupt = bytes.clone(); ++ corrupt[1] ^= 1; ++ assert!( ++ PreinitializedMemoryImage::from_verified_reader( ++ Cursor::new(corrupt), ++ digest, ++ module_hash, ++ metadata, ++ ) ++ .is_err() ++ ); ++ } ++ ++ fn verified_image() -> Arc { ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let metadata = metadata(module_hash); ++ let bytes = vec![0x5a; metadata.mapped_size as usize]; ++ let digest: [u8; 32] = Sha256::digest(&bytes).into(); ++ Arc::new( ++ PreinitializedMemoryImage::from_verified_reader( ++ Cursor::new(bytes), ++ digest, ++ module_hash, ++ metadata, ++ ) ++ .unwrap(), ++ ) ++ } ++ ++ fn verified_attested_image() -> Arc { ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let mut metadata = metadata(module_hash); ++ metadata.schema = PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA.to_string(); ++ let proof = deterministic_start_proof(module_hash); ++ metadata.deterministic_start_proof_output_sha256 = ++ Some(canonical_deterministic_start_proof_sha256(&proof).unwrap()); ++ metadata.deterministic_start_proof = Some(proof); ++ let bytes = vec![0x5a; metadata.mapped_size as usize]; ++ let digest: [u8; 32] = Sha256::digest(&bytes).into(); ++ Arc::new( ++ PreinitializedMemoryImage::from_verified_reader( ++ Cursor::new(bytes), ++ digest, ++ module_hash, ++ metadata, ++ ) ++ .unwrap(), ++ ) ++ } ++ ++ #[test] ++ fn attested_runtime_audit_has_exact_terminal_conservation() { ++ let image = verified_attested_image(); ++ let mapped_size = image.metadata.mapped_size; ++ ++ image.runtime_audit.record_start_boundary(true); ++ image ++ .runtime_audit ++ .record_validation(true, true, mapped_size); ++ image.runtime_audit.record_remap(true); ++ for _ in 0..3 { ++ image.runtime_audit.record_start_boundary(true); ++ image ++ .runtime_audit ++ .record_validation(false, true, mapped_size); ++ image.runtime_audit.record_remap(true); ++ } ++ ++ let audit = image.runtime_audit().unwrap(); ++ assert_eq!(audit.ordinary_start_completed_instances, 4); ++ assert_eq!(audit.fresh_zeroed_instances, 4); ++ assert_eq!(audit.nonfresh_instances, 0); ++ assert_eq!(audit.validation_attempts, 4); ++ assert_eq!(audit.full_compare_attempts, 1); ++ assert_eq!(audit.full_compare_successes, 1); ++ assert_eq!(audit.full_compare_failures, 0); ++ assert_eq!(audit.compared_bytes, mapped_size); ++ assert_eq!(audit.reuse_successes, 3); ++ assert_eq!(audit.reuse_failures, 0); ++ assert_eq!(audit.skipped_bytes, 3 * mapped_size); ++ assert_eq!(audit.remap_successes, 4); ++ assert_eq!(audit.remap_failures, 0); ++ assert!(!audit.counter_overflow); ++ } ++ ++ #[test] ++ fn attested_runtime_audit_reports_counter_overflow_without_wrapping() { ++ let image = verified_attested_image(); ++ image ++ .runtime_audit ++ .skipped_bytes ++ .store(u64::MAX, Ordering::Release); ++ image ++ .runtime_audit ++ .record_validation(false, true, image.metadata.mapped_size); ++ ++ let audit = image.runtime_audit().unwrap(); ++ assert_eq!(audit.skipped_bytes, u64::MAX); ++ assert_eq!(audit.reuse_successes, 1); ++ assert!(audit.counter_overflow); ++ } ++ ++ #[test] ++ fn metadata_requires_proof_and_attested_schema_as_a_pair() { ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let mut metadata = metadata(module_hash); ++ let proof = deterministic_start_proof(module_hash); ++ metadata.deterministic_start_proof_output_sha256 = ++ Some(canonical_deterministic_start_proof_sha256(&proof).unwrap()); ++ metadata.deterministic_start_proof = Some(proof); ++ assert!(validate_metadata_shape(&metadata, module_hash).is_err()); ++ ++ metadata.schema = PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA.to_string(); ++ metadata.deterministic_start_proof = None; ++ assert!(validate_metadata_shape(&metadata, module_hash).is_err()); ++ ++ metadata.deterministic_start_proof = Some(deterministic_start_proof(module_hash)); ++ validate_metadata_shape(&metadata, module_hash).unwrap(); ++ } ++ ++ #[test] ++ fn metadata_rejects_valid_proof_for_another_module_and_output_digest_drift() { ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let other_hash = ModuleHash::from_bytes([0x43; 32]); ++ let proof = deterministic_start_proof(other_hash); ++ let mut metadata = metadata(module_hash); ++ metadata.schema = PreinitializedMemoryImageMetadata::ATTESTED_SCHEMA.to_string(); ++ metadata.deterministic_start_proof_output_sha256 = ++ Some(canonical_deterministic_start_proof_sha256(&proof).unwrap()); ++ metadata.deterministic_start_proof = Some(proof); ++ let error = validate_metadata_shape(&metadata, module_hash).unwrap_err(); ++ assert!(error.contains("proof policy"), "unexpected error: {error}"); ++ ++ let proof = deterministic_start_proof(module_hash); ++ metadata.deterministic_start_proof_output_sha256 = Some("cd".repeat(32)); ++ metadata.deterministic_start_proof = Some(proof); ++ let error = validate_metadata_shape(&metadata, module_hash).unwrap_err(); ++ assert!( ++ error.contains("canonical output digest"), ++ "unexpected error: {error}" ++ ); ++ } ++ ++ #[test] ++ fn attested_runtime_validation_is_single_flight() { ++ use std::sync::{ ++ Barrier, ++ atomic::{AtomicUsize, Ordering}, ++ }; ++ ++ let image = verified_attested_image(); ++ let comparisons = Arc::new(AtomicUsize::new(0)); ++ let barrier = Arc::new(Barrier::new(8)); ++ let threads = (0..8) ++ .map(|_| { ++ let image = image.clone(); ++ let comparisons = comparisons.clone(); ++ let barrier = barrier.clone(); ++ std::thread::spawn(move || { ++ barrier.wait(); ++ image.validate_runtime_bytes(true, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Ok(()) ++ }) ++ }) ++ }) ++ .collect::>(); ++ let results = threads ++ .into_iter() ++ .map(|thread| thread.join().unwrap()) ++ .collect::>(); ++ ++ assert!(results.iter().all(|result| result.validation.is_ok())); ++ assert_eq!(comparisons.load(Ordering::Relaxed), 1); ++ assert_eq!( ++ results ++ .iter() ++ .filter(|result| result.compared_this_instance) ++ .count(), ++ 1 ++ ); ++ } ++ ++ #[test] ++ fn attested_runtime_validation_caches_failure_and_rejects_reused_memory() { ++ use std::sync::atomic::{AtomicUsize, Ordering}; ++ ++ let image = verified_attested_image(); ++ let comparisons = AtomicUsize::new(0); ++ let reused = image.validate_runtime_bytes(false, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Ok(()) ++ }); ++ assert!( ++ reused ++ .validation ++ .unwrap_err() ++ .contains("fresh zeroed memory") ++ ); ++ assert!(!reused.compared_this_instance); ++ assert_eq!(comparisons.load(Ordering::Relaxed), 0); ++ ++ let first = image.validate_runtime_bytes(true, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Err("captured image mismatch".to_string()) ++ }); ++ assert_eq!(first.validation.unwrap_err(), "captured image mismatch"); ++ assert!(first.compared_this_instance); ++ ++ let cached = image.validate_runtime_bytes(true, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Ok(()) ++ }); ++ assert_eq!(cached.validation.unwrap_err(), "captured image mismatch"); ++ assert!(!cached.compared_this_instance); ++ assert_eq!(comparisons.load(Ordering::Relaxed), 1); ++ } ++ ++ #[test] ++ fn runtime_byte_validation_runs_for_every_fresh_instance() { ++ use std::sync::{ ++ Barrier, ++ atomic::{AtomicUsize, Ordering}, ++ }; ++ ++ let image = verified_image(); ++ let comparisons = Arc::new(AtomicUsize::new(0)); ++ let barrier = Arc::new(Barrier::new(8)); ++ let threads = (0..8) ++ .map(|_| { ++ let image = image.clone(); ++ let comparisons = comparisons.clone(); ++ let barrier = barrier.clone(); ++ std::thread::spawn(move || { ++ barrier.wait(); ++ image ++ .validate_runtime_bytes(true, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Ok(()) ++ }) ++ .validation ++ }) ++ }) ++ .collect::>(); ++ let results = threads ++ .into_iter() ++ .map(|thread| thread.join().unwrap()) ++ .collect::>(); ++ ++ assert!(results.iter().all(Result::is_ok)); ++ assert_eq!(comparisons.load(Ordering::Relaxed), 8); ++ ++ image ++ .validate_runtime_bytes(true, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Ok(()) ++ }) ++ .validation ++ .unwrap(); ++ assert_eq!(comparisons.load(Ordering::Relaxed), 9); ++ } ++ ++ #[test] ++ fn runtime_byte_validation_does_not_cache_a_prior_mismatch() { ++ use std::sync::{ ++ Barrier, ++ atomic::{AtomicUsize, Ordering}, ++ }; ++ ++ let image = verified_image(); ++ let comparisons = Arc::new(AtomicUsize::new(0)); ++ let barrier = Arc::new(Barrier::new(8)); ++ let threads = (0..8) ++ .map(|_| { ++ let image = image.clone(); ++ let comparisons = comparisons.clone(); ++ let barrier = barrier.clone(); ++ std::thread::spawn(move || { ++ barrier.wait(); ++ image ++ .validate_runtime_bytes(true, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Err("deterministic module-start mismatch".to_string()) ++ }) ++ .validation ++ }) ++ }) ++ .collect::>(); ++ let results = threads ++ .into_iter() ++ .map(|thread| thread.join().unwrap()) ++ .collect::>(); ++ ++ assert!(results.iter().all(|result| { ++ result.as_ref().unwrap_err() == "deterministic module-start mismatch" ++ })); ++ assert_eq!(comparisons.load(Ordering::Relaxed), 8); ++ ++ image ++ .validate_runtime_bytes(true, || { ++ comparisons.fetch_add(1, Ordering::Relaxed); ++ Ok(()) ++ }) ++ .validation ++ .unwrap(); ++ assert_eq!(comparisons.load(Ordering::Relaxed), 9); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[test] ++ fn validation_cache_release_is_direct_only_and_preserves_file_bytes() { ++ use std::os::unix::fs::FileExt; ++ ++ let mut source = tempfile::tempfile().unwrap(); ++ let bytes = vec![0x5a; PREINITIALIZED_MEMORY_IMAGE_ALIGNMENT as usize]; ++ source.write_all(&bytes).unwrap(); ++ let direct = release_validation_faulted_source_pages( ++ PreinitializedMemoryImageBacking::DirectIntrinsic( ++ IntrinsicFileImmutability::ReadOnlyFilesystem, ++ ), ++ &source, ++ ); ++ assert!(direct.supported); ++ assert_eq!(direct.calls, 1); ++ assert_eq!(direct.successes, 1); ++ assert_eq!(direct.first_errno, None); ++ ++ let copied = release_validation_faulted_source_pages( ++ PreinitializedMemoryImageBacking::SealedCopy, ++ &source, ++ ); ++ assert_eq!(copied, FileAdviceAudit::not_applicable()); ++ ++ let mut retained = vec![0_u8; bytes.len()]; ++ source.read_at(&mut retained, 0).unwrap(); ++ assert_eq!(retained, bytes); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[test] ++ fn intrinsic_immutability_classifier_is_fail_closed() { ++ const EXT4_SUPER_MAGIC: u64 = 0xef53; ++ const SQUASHFS_MAGIC: u64 = 0x7371_7368; ++ const EROFS_SUPER_MAGIC: u64 = 0xe0f5_e1e2; ++ const FS_IMMUTABLE_FL: u32 = 0x10; ++ ++ assert_eq!( ++ classify_intrinsic_immutability(SQUASHFS_MAGIC, None, true), ++ Some(IntrinsicFileImmutability::ReadOnlyFilesystem) ++ ); ++ assert_eq!( ++ classify_intrinsic_immutability(EROFS_SUPER_MAGIC, None, true), ++ Some(IntrinsicFileImmutability::ReadOnlyFilesystem) ++ ); ++ assert_eq!( ++ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, Some(FS_IMMUTABLE_FL), false), ++ Some(IntrinsicFileImmutability::ImmutableInode) ++ ); ++ assert_eq!( ++ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, Some(FS_IMMUTABLE_FL), true), ++ None ++ ); ++ assert_eq!( ++ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, Some(0), false), ++ None ++ ); ++ assert_eq!( ++ classify_intrinsic_immutability(EXT4_SUPER_MAGIC, None, false), ++ None ++ ); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[test] ++ fn direct_image_constructor_rejects_a_mutable_inode() { ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let metadata = metadata(module_hash); ++ let bytes = vec![0x5a; metadata.mapped_size as usize]; ++ let digest: [u8; 32] = Sha256::digest(&bytes).into(); ++ let mut source = tempfile::tempfile().unwrap(); ++ source.write_all(&bytes).unwrap(); ++ ++ let error = PreinitializedMemoryImage::from_verified_immutable_file( ++ source, ++ digest, ++ module_hash, ++ metadata, ++ ) ++ .unwrap_err(); ++ assert!(error.contains("neither on SquashFS/EROFS nor protected")); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[test] ++ fn sealed_copy_is_isolated_from_later_source_mutation() { ++ use std::os::unix::fs::FileExt; ++ ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let metadata = metadata(module_hash); ++ let bytes = vec![0x5a; metadata.mapped_size as usize]; ++ let digest: [u8; 32] = Sha256::digest(&bytes).into(); ++ let mut source = tempfile::tempfile().unwrap(); ++ source.write_all(&bytes).unwrap(); ++ source.seek(SeekFrom::Start(0)).unwrap(); ++ let image = ++ PreinitializedMemoryImage::from_verified_reader(&source, digest, module_hash, metadata) ++ .unwrap(); ++ ++ source.write_at(&[0x33], 0).unwrap(); ++ let mut retained = [0_u8; 1]; ++ image.file.read_at(&mut retained, 0).unwrap(); ++ assert_eq!(retained, [0x5a]); ++ assert_eq!( ++ image.backing(), ++ PreinitializedMemoryImageBacking::SealedCopy ++ ); ++ } ++ ++ #[cfg(target_os = "linux")] ++ #[test] ++ fn verified_reader_owns_a_sealed_backing() { ++ use std::os::{fd::AsRawFd, unix::fs::FileExt}; ++ ++ let module_hash = ModuleHash::from_bytes([0x42; 32]); ++ let metadata = metadata(module_hash); ++ let bytes = vec![0x5a; metadata.mapped_size as usize]; ++ let digest: [u8; 32] = Sha256::digest(&bytes).into(); ++ let image = PreinitializedMemoryImage::from_verified_reader( ++ Cursor::new(bytes), ++ digest, ++ module_hash, ++ metadata, ++ ) ++ .unwrap(); ++ ++ // SAFETY: F_GET_SEALS does not mutate the live memfd descriptor. ++ let seals = unsafe { libc::fcntl(image.file.as_raw_fd(), libc::F_GET_SEALS) }; ++ let expected = ++ libc::F_SEAL_WRITE | libc::F_SEAL_GROW | libc::F_SEAL_SHRINK | libc::F_SEAL_SEAL; ++ assert_eq!(seals & expected, expected); ++ assert!(image.file.write_at(&[0], 0).is_err()); ++ } ++} +diff --git a/lib/wasix/src/syscalls/mod.rs b/lib/wasix/src/syscalls/mod.rs +index a3d58df..a9e55fe 100644 +--- a/lib/wasix/src/syscalls/mod.rs ++++ b/lib/wasix/src/syscalls/mod.rs +@@ -136,8 +136,8 @@ pub(crate) use crate::{ + }, + runtime::SpawnType, + state::{ +- self, InodeGuard, InodeWeakGuard, PollEvent, PollEventBuilder, WasiFutex, WasiState, +- iterate_poll_events, ++ self, InodeGuard, InodeWeakGuard, PollEvent, PollEventBuilder, WasiFutex, ++ WasiFutexRegistry, WasiSharedMemoryMapping, WasiState, iterate_poll_events, + }, + utils::{self, map_io_err}, + }; +@@ -341,10 +341,23 @@ where + return Poll::Ready(Ok(res)); + } + if let Some(signals) = self.ctx.data().thread.pop_signals_or_subscribe(cx.waker()) { +- if let Err(err) = WasiEnv::process_signals_internal(self.ctx, signals) { ++ let processed_guest_signal = ++ match WasiEnv::process_signals_internal(self.ctx, signals) { ++ Ok(processed) => processed, ++ Err(err) => { ++ return Poll::Ready(Err(err)); ++ } ++ }; ++ if let Err(err) = WasiEnv::do_pending_link_operations(self.ctx, false) { + return Poll::Ready(Err(err)); + } +- return Poll::Ready(Ok(Err(Errno::Intr))); ++ if let Poll::Ready(res) = Pin::new(&mut self.pinned_work).poll(cx) { ++ return Poll::Ready(Ok(res)); ++ } ++ if processed_guest_signal { ++ return Poll::Ready(Ok(Err(Errno::Intr))); ++ } ++ self.ctx.data().thread.signals_subscribe(cx.waker()); + } + Poll::Pending + } +@@ -461,12 +474,12 @@ pub(crate) fn maybe_backoff( + // Determine if we need to do a backoff, if so lets do one + if let Some(backoff) = env.process.acquire_cpu_backoff_token(env.tasks()) { + tracing::trace!("exponential CPU backoff {:?}", backoff.backoff_time()); +- if let AsyncifyAction::Finish(mut ctx, _) = +- __asyncify_with_deep_sleep::(ctx, backoff)? +- { +- Ok(Ok(ctx)) +- } else { +- Ok(Err(Errno::Success)) ++ match __asyncify(&mut ctx, None, async move { ++ backoff.await; ++ Ok(()) ++ })? { ++ Ok(()) => Ok(Ok(ctx)), ++ Err(err) => Ok(Err(err)), + } + } else { + Ok(Ok(ctx)) +@@ -612,8 +625,8 @@ where + let mut work = { + let inode = fd_entry.inode.clone(); + let tasks = env.tasks().clone(); +- let mut guard = inode.write(); +- match guard.deref_mut() { ++ let guard = inode.read(); ++ match guard.deref() { + Kind::Socket { socket } => { + // Clone the socket and release the lock + let socket = socket.clone(); +@@ -654,8 +667,8 @@ where + } + + let inode = fd_entry.inode.clone(); +- let mut guard = inode.write(); +- match guard.deref_mut() { ++ let guard = inode.read(); ++ match guard.deref() { + Kind::Socket { socket } => { + // Clone the socket and release the lock + let socket = socket.clone(); +@@ -672,43 +685,49 @@ where + } + } + +-/// Performs an immutable operation on the socket while running in an asynchronous runtime +-/// This has built in signal support +-pub(crate) fn __sock_actor( +- ctx: &mut FunctionEnvMut<'_, WasiEnv>, ++/// Performs a synchronous operation on the socket without entering asyncify. ++pub(crate) fn __sock_actor_env( ++ env: &WasiEnv, + sock: WasiFd, + rights: Rights, + actor: F, + ) -> Result + where +- T: 'static, + F: FnOnce(crate::net::socket::InodeSocket, Fd) -> Result, + { +- let env = ctx.data(); +- let tasks = env.tasks().clone(); +- + let fd_entry = env.state.fs.get_fd(sock)?; + if !rights.is_empty() && !fd_entry.inner.rights.contains(rights) { + return Err(Errno::Access); + } + + let inode = fd_entry.inode.clone(); +- +- let tasks = env.tasks().clone(); +- let mut guard = inode.write(); +- match guard.deref_mut() { ++ let guard = inode.read(); ++ match guard.deref() { + Kind::Socket { socket } => { +- // Clone the socket and release the lock + let socket = socket.clone(); + drop(guard); +- +- // Start the work using the socket + actor(socket, fd_entry) + } + _ => Err(Errno::Notsock), + } + } + ++/// Performs an immutable operation on the socket while running in an asynchronous runtime ++/// This has built in signal support ++pub(crate) fn __sock_actor( ++ ctx: &mut FunctionEnvMut<'_, WasiEnv>, ++ sock: WasiFd, ++ rights: Rights, ++ actor: F, ++) -> Result ++where ++ T: 'static, ++ F: FnOnce(crate::net::socket::InodeSocket, Fd) -> Result, ++{ ++ let env = ctx.data(); ++ __sock_actor_env(env, sock, rights, actor) ++} ++ + /// Performs mutable work on a socket under an asynchronous runtime with + /// built in signal processing + pub(crate) fn __sock_actor_mut( +@@ -730,8 +749,8 @@ where + } + + let inode = fd_entry.inode.clone(); +- let mut guard = inode.write(); +- match guard.deref_mut() { ++ let guard = inode.read(); ++ match guard.deref() { + Kind::Socket { socket } => { + // Clone the socket and release the lock + let socket = socket.clone(); +@@ -1017,7 +1036,7 @@ pub(crate) fn deep_sleep( + trigger: Pin>, + ) -> Result<(), WasiError> { + // Grab all the globals and serialize them +- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) ++ let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut())? + .serialize() + .unwrap(); + let store_data = Bytes::from(store_data); +@@ -1033,9 +1052,12 @@ pub(crate) fn deep_sleep( + // If journal'ing is enabled then we dump the stack into the journal + if ctx.data().enable_journal { + // Grab all the globals and serialize them +- let store_data = crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) +- .serialize() +- .unwrap(); ++ let snapshot = ++ match crate::utils::store::capture_store_snapshot(&mut ctx.as_store_mut()) { ++ Ok(snapshot) => snapshot, ++ Err(err) => return OnCalledAction::Trap(Box::new(WasiError::from(err))), ++ }; ++ let store_data = snapshot.serialize().unwrap(); + let store_data = Bytes::from(store_data); + + tracing::trace!( +@@ -1158,8 +1180,9 @@ where + let asyncify_data = wasi_try_ok!(unwind_pointer.try_into().map_err(|_| Errno::Overflow)); + if let Some(asyncify_start_unwind) = env + .inner() +- .static_module_instance_handles() +- .and_then(|handles| handles.asyncify_start_unwind.clone()) ++ .main_module_instance_handles() ++ .asyncify_start_unwind ++ .clone() + { + asyncify_start_unwind.call(&mut ctx, asyncify_data); + } else { +@@ -1222,10 +1245,8 @@ where + .map_err(|err| format!("failed to read stack: {err}"))?; + + // Notify asyncify that we are no longer unwinding +- if let Some(asyncify_stop_unwind) = env +- .inner() +- .static_module_instance_handles() +- .and_then(|i| i.asyncify_stop_unwind.clone()) ++ if let Some(asyncify_stop_unwind) = ++ env.inner().main_module_instance_handles().asyncify_stop_unwind.clone() + { + asyncify_stop_unwind.call(&mut ctx); + } else { +@@ -1283,11 +1304,14 @@ pub fn rewind_ext( + let store_snapshot = match StoreSnapshot::deserialize(&store_data[..]) { + Ok(a) => a, + Err(err) => { +- warn!("snapshot restore failed - the store snapshot could not be deserialized"); ++ warn!("snapshot restore failed - the store snapshot could not be deserialized: {err}"); + return Errno::Unknown; + } + }; +- crate::utils::store::restore_store_snapshot(ctx, &store_snapshot); ++ if let Err(err) = crate::utils::store::restore_store_snapshot(ctx, &store_snapshot) { ++ warn!("snapshot restore failed - destination store is incompatible: {err}"); ++ return Errno::Unknown; ++ } + let env = ctx.data(); + let memory = match env.try_memory_view(&ctx) { + Some(v) => v, +@@ -1340,8 +1364,9 @@ pub fn rewind_ext( + let asyncify_data = wasi_try!(rewind_pointer.try_into().map_err(|_| Errno::Overflow)); + if let Some(asyncify_start_rewind) = env + .inner() +- .static_module_instance_handles() +- .and_then(|a| a.asyncify_start_rewind.clone()) ++ .main_module_instance_handles() ++ .asyncify_start_rewind ++ .clone() + { + asyncify_start_rewind.call(ctx, asyncify_data); + } else { +@@ -1443,8 +1468,9 @@ where + let env = ctx.data(); + if let Some(asyncify_stop_rewind) = env + .inner() +- .static_module_instance_handles() +- .and_then(|handles| handles.asyncify_stop_rewind.clone()) ++ .main_module_instance_handles() ++ .asyncify_stop_rewind ++ .clone() + { + asyncify_stop_rewind.call(ctx); + } else { +diff --git a/lib/wasix/src/syscalls/wasix/mod.rs b/lib/wasix/src/syscalls/wasix/mod.rs +index bc2bf06..19bc10f 100644 +--- a/lib/wasix/src/syscalls/wasix/mod.rs ++++ b/lib/wasix/src/syscalls/wasix/mod.rs +@@ -17,10 +17,12 @@ mod fd_dup2; + mod fd_fdflags_get; + mod fd_fdflags_set; + mod fd_pipe; ++mod fd_sync_range; + mod futex_wait; + mod futex_wake; + mod futex_wake_all; + mod getcwd; ++mod mem_mmap; + mod path_open2; + mod port_addr_add; + mod port_addr_clear; +@@ -44,6 +46,7 @@ mod proc_fork_env; + mod proc_id; + mod proc_join; + mod proc_parent; ++mod proc_rlimit_get; + mod proc_signal; + mod proc_signals_get; + mod proc_signals_sizes_get; +@@ -109,10 +112,12 @@ pub use fd_dup2::*; + pub use fd_fdflags_get::*; + pub use fd_fdflags_set::*; + pub use fd_pipe::*; ++pub use fd_sync_range::*; + pub use futex_wait::*; + pub use futex_wake::*; + pub use futex_wake_all::*; + pub use getcwd::*; ++pub use mem_mmap::*; + pub use path_open2::*; + pub use port_addr_add::*; + pub use port_addr_clear::*; +@@ -136,6 +141,7 @@ pub use proc_fork_env::*; + pub use proc_id::*; + pub use proc_join::*; + pub use proc_parent::*; ++pub use proc_rlimit_get::*; + pub use proc_signal::*; + pub use proc_signals_get::*; + pub use proc_signals_sizes_get::*; +diff --git a/lib/wasix/src/utils/store.rs b/lib/wasix/src/utils/store.rs +index d956dba..a4b8db9 100644 +--- a/lib/wasix/src/utils/store.rs ++++ b/lib/wasix/src/utils/store.rs +@@ -1,32 +1,531 @@ + use bincode::config; ++use sha2::{Digest, Sha256}; ++use wasmer_types::Mutability; + +-/// A snapshot that captures the runtime state of an instance. +-#[derive(Default, serde::Serialize, serde::Deserialize, Clone, Debug)] ++const STORE_SNAPSHOT_MAGIC: &[u8; 8] = b"WXSSTORE"; ++const STORE_SNAPSHOT_VERSION_SPARSE: u8 = 2; ++const STORE_SHAPE_DOMAIN_V2: &[u8] = b"wasix-store-snapshot-shape-v2\0"; ++ ++/// Values of all store globals in the original, unversioned snapshot format. ++/// ++/// Keep this single-field shape byte-compatible with the historical `StoreSnapshot` serde ++/// representation. Snapshots can be journaled, so old dense bytes must remain readable. ++#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] ++struct DenseStoreSnapshotV1 { ++ globals: Vec, ++} ++ ++/// A stable description of the complete store-global layout. ++#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] ++struct StoreShapeV2 { ++ global_count: u64, ++ mutable_global_count: u64, ++ descriptor_sha256: [u8; 32], ++} ++ ++/// Version-two wire representation. Mutable values are ordered by their rank among all mutable ++/// globals, while `shape` binds that rank to exact store indexes and WebAssembly value types. ++#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] ++struct SparseStoreSnapshotV2 { ++ shape: StoreShapeV2, ++ mutable_globals: Vec, ++} ++ ++/// A snapshot that captures the runtime state of every store-owned mutable global. ++/// ++/// The sparse representation includes globals created for imports and globals appended by ++/// side-module instantiation because it walks the store, rather than one instance's local-global ++/// list. Dense V1 is retained only to decode already-persisted snapshots; new captures fail closed ++/// on backends that do not expose exact raw WebAssembly values. ++#[derive(Clone, Debug)] ++enum StoreSnapshotRepr { ++ DenseV1(DenseStoreSnapshotV1), ++ SparseV2(SparseStoreSnapshotV2), ++} ++ ++#[derive(Clone, Debug)] + pub struct StoreSnapshot { +- /// Values of all globals, indexed by the same index used in Webassembly. +- pub globals: Vec, ++ repr: StoreSnapshotRepr, ++} ++ ++impl Default for StoreSnapshot { ++ fn default() -> Self { ++ Self { ++ repr: StoreSnapshotRepr::DenseV1(DenseStoreSnapshotV1 { ++ globals: Vec::new(), ++ }), ++ } ++ } + } + + impl StoreSnapshot { + pub fn serialize(&self) -> Result, bincode::error::EncodeError> { +- bincode::serde::encode_to_vec(self, config::legacy()) ++ match &self.repr { ++ StoreSnapshotRepr::DenseV1(snapshot) => { ++ bincode::serde::encode_to_vec(snapshot, config::legacy()) ++ } ++ StoreSnapshotRepr::SparseV2(snapshot) => { ++ let payload = bincode::serde::encode_to_vec(snapshot, config::legacy())?; ++ let mut encoded = ++ Vec::with_capacity(STORE_SNAPSHOT_MAGIC.len() + 1 + payload.len()); ++ encoded.extend_from_slice(STORE_SNAPSHOT_MAGIC); ++ encoded.push(STORE_SNAPSHOT_VERSION_SPARSE); ++ encoded.extend_from_slice(&payload); ++ Ok(encoded) ++ } ++ } + } + + pub fn deserialize(data: &[u8]) -> Result { +- bincode::serde::decode_from_slice(data, config::legacy()).map(|(ret, _)| ret) ++ if data.starts_with(STORE_SNAPSHOT_MAGIC) { ++ let Some((&version, payload)) = data[STORE_SNAPSHOT_MAGIC.len()..].split_first() else { ++ return Err(bincode::error::DecodeError::UnexpectedEnd { additional: 1 }); ++ }; ++ if version != STORE_SNAPSHOT_VERSION_SPARSE { ++ return Err(bincode::error::DecodeError::Other( ++ "unsupported WASIX store snapshot version", ++ )); ++ } ++ return bincode::serde::decode_from_slice(payload, config::legacy()).map( ++ |(snapshot, _)| Self { ++ repr: StoreSnapshotRepr::SparseV2(snapshot), ++ }, ++ ); ++ } ++ ++ // No magic means the historical single-Vec serde structure. This fallback is deliberately ++ // permanent because stack snapshots may outlive the runtime that wrote them. ++ bincode::serde::decode_from_slice(data, config::legacy()).map(|(snapshot, _)| Self { ++ repr: StoreSnapshotRepr::DenseV1(snapshot), ++ }) ++ } ++} ++ ++#[derive(Clone, Debug, thiserror::Error, PartialEq, Eq)] ++pub enum StoreSnapshotCaptureError { ++ #[error("mutable reference global at store index {index} cannot be snapshotted as raw bits")] ++ MutableReferenceGlobal { index: usize }, ++ #[error("the active backend cannot capture an exact store snapshot")] ++ UnsupportedBackend, ++} ++ ++#[derive(Debug, thiserror::Error, PartialEq, Eq)] ++pub enum StoreSnapshotRestoreError { ++ #[error("dense store snapshot has {snapshot} globals, but destination has {store}")] ++ DenseShapeMismatch { snapshot: usize, store: usize }, ++ #[error("sparse store snapshot mutable-value count does not match its shape")] ++ SparseValueCountMismatch, ++ #[error("sparse store snapshot shape does not match the destination store")] ++ SparseShapeMismatch, ++ #[error("the active backend cannot restore an exact sparse store snapshot")] ++ UnsupportedBackend, ++ #[error("reference global at store index {index} cannot be restored from raw bits")] ++ ReferenceGlobal { index: usize }, ++} ++ ++#[cfg(any(feature = "sys", feature = "sys-minimal"))] ++fn sparse_sys_store_shape( ++ store: &wasmer::sys::store::StoreObjects, ++) -> Result { ++ let mut descriptor = Sha256::new(); ++ descriptor.update(STORE_SHAPE_DOMAIN_V2); ++ // Buffer compact (type, mutability) descriptors so a large module does not make one digest ++ // call per global. Sequence position is the store index, so explicit index bytes are redundant. ++ let mut descriptor_buffer = [0_u8; 1024]; ++ let mut descriptor_buffer_len = 0_usize; ++ let mut global_count = 0_u64; ++ let mut mutable_global_count = 0_u64; ++ ++ for (index, global) in store.iter_globals().enumerate() { ++ let global_type = *global.ty(); ++ if global_type.mutability == Mutability::Var && global_type.ty.is_ref() { ++ return Err(StoreSnapshotCaptureError::MutableReferenceGlobal { index }); ++ } ++ let index = u64::try_from(index).expect("store global index does not fit u64"); ++ debug_assert_eq!(index, global_count); ++ descriptor_buffer[descriptor_buffer_len] = global_type.ty as u8; ++ descriptor_buffer[descriptor_buffer_len + 1] = global_type.mutability as u8; ++ descriptor_buffer_len += 2; ++ if descriptor_buffer_len == descriptor_buffer.len() { ++ descriptor.update(descriptor_buffer); ++ descriptor_buffer_len = 0; ++ } ++ global_count += 1; ++ if global_type.mutability == Mutability::Var { ++ mutable_global_count += 1; ++ } ++ } ++ ++ descriptor.update(&descriptor_buffer[..descriptor_buffer_len]); ++ descriptor.update(global_count.to_le_bytes()); ++ descriptor.update(mutable_global_count.to_le_bytes()); ++ Ok(StoreShapeV2 { ++ global_count, ++ mutable_global_count, ++ descriptor_sha256: descriptor.finalize().into(), ++ }) ++} ++ ++fn sparse_store_shape( ++ objects: &wasmer::StoreObjects, ++) -> Result { ++ match objects { ++ #[cfg(any(feature = "sys", feature = "sys-minimal"))] ++ wasmer::StoreObjects::Sys(store) => sparse_sys_store_shape(store), ++ #[allow(unreachable_patterns)] ++ _ => Err(StoreSnapshotCaptureError::UnsupportedBackend), ++ } ++} ++ ++#[cfg(any(feature = "sys", feature = "sys-minimal"))] ++fn capture_raw_mutable_globals( ++ store: &wasmer::sys::store::StoreObjects, ++ capacity: usize, ++) -> Vec { ++ let mut values = Vec::with_capacity(capacity); ++ for global in store ++ .iter_globals() ++ .filter(|global| global.ty().mutability == Mutability::Var) ++ { ++ debug_assert!(!global.ty().ty.is_ref()); ++ // The complete store shape was validated before this second pass, so every mutable ++ // global read here has a numeric/vector representation valid for raw bit capture. ++ values.push(unsafe { global.vmglobal().as_ref().val.u128 }); ++ } ++ values ++} ++ ++pub fn capture_store_snapshot( ++ store: &mut impl wasmer::AsStoreMut, ++) -> Result { ++ let objects = store.objects_mut(); ++ let shape = sparse_store_shape(objects)?; ++ ++ let mutable_capacity = usize::try_from(shape.mutable_global_count) ++ .expect("mutable store-global count does not fit usize"); ++ let mutable_globals = match objects { ++ #[cfg(any(feature = "sys", feature = "sys-minimal"))] ++ wasmer::StoreObjects::Sys(store) => capture_raw_mutable_globals(store, mutable_capacity), ++ #[allow(unreachable_patterns)] ++ _ => return Err(StoreSnapshotCaptureError::UnsupportedBackend), ++ }; ++ debug_assert_eq!(mutable_globals.len(), mutable_capacity); ++ ++ Ok(StoreSnapshot { ++ repr: StoreSnapshotRepr::SparseV2(SparseStoreSnapshotV2 { ++ shape, ++ mutable_globals, ++ }), ++ }) ++} ++ ++#[cfg(any(feature = "sys", feature = "sys-minimal"))] ++fn validate_dense_destination( ++ store: &wasmer::sys::store::StoreObjects, ++ snapshot_len: usize, ++) -> Result<(), StoreSnapshotRestoreError> { ++ let store_len = store.iter_globals().len(); ++ if snapshot_len != store_len { ++ return Err(StoreSnapshotRestoreError::DenseShapeMismatch { ++ snapshot: snapshot_len, ++ store: store_len, ++ }); ++ } ++ if let Some((index, _)) = store ++ .iter_globals() ++ .enumerate() ++ .find(|(_, global)| global.ty().ty.is_ref()) ++ { ++ return Err(StoreSnapshotRestoreError::ReferenceGlobal { index }); ++ } ++ Ok(()) ++} ++ ++#[cfg(any(feature = "sys", feature = "sys-minimal"))] ++unsafe fn restore_dense_numeric_globals(store: &wasmer::sys::store::StoreObjects, values: &[u128]) { ++ for (global, value) in store.iter_globals().zip(values.iter().copied()) { ++ // SAFETY: the caller validated the complete destination shape, including the absence of ++ // reference globals, before performing any write. ++ unsafe { global.vmglobal().as_mut().val.u128 = value }; + } + } + +-pub fn capture_store_snapshot(store: &mut impl wasmer::AsStoreMut) -> StoreSnapshot { +- let objs = store.objects_mut(); +- let globals = objs.as_u128_globals(); +- StoreSnapshot { globals } ++#[cfg(any(feature = "sys", feature = "sys-minimal"))] ++unsafe fn restore_sparse_numeric_globals( ++ store: &wasmer::sys::store::StoreObjects, ++ values: &[u128], ++) { ++ for (global, value) in store ++ .iter_globals() ++ .filter(|global| global.ty().mutability == Mutability::Var) ++ .zip(values.iter().copied()) ++ { ++ // SAFETY: the caller validated the exact descriptor hash, mutable count, and absence of ++ // mutable reference globals before performing any write. ++ unsafe { global.vmglobal().as_mut().val.u128 = value }; ++ } ++} ++ ++pub fn restore_store_snapshot( ++ store: &mut impl wasmer::AsStoreMut, ++ snapshot: &StoreSnapshot, ++) -> Result<(), StoreSnapshotRestoreError> { ++ let objects = store.objects_mut(); ++ match &snapshot.repr { ++ StoreSnapshotRepr::DenseV1(snapshot) => { ++ match objects { ++ #[cfg(any(feature = "sys", feature = "sys-minimal"))] ++ wasmer::StoreObjects::Sys(store) => { ++ validate_dense_destination(store, snapshot.globals.len())?; ++ // SAFETY: validation above rejects all reference-bearing destinations and ++ // verifies the value count before the first raw write. ++ unsafe { restore_dense_numeric_globals(store, &snapshot.globals) }; ++ } ++ #[allow(unreachable_patterns)] ++ _ => return Err(StoreSnapshotRestoreError::UnsupportedBackend), ++ } ++ Ok(()) ++ } ++ StoreSnapshotRepr::SparseV2(snapshot) => { ++ if u64::try_from(snapshot.mutable_globals.len()).ok() ++ != Some(snapshot.shape.mutable_global_count) ++ { ++ return Err(StoreSnapshotRestoreError::SparseValueCountMismatch); ++ } ++ let actual_shape = sparse_store_shape(objects).map_err(|err| match err { ++ StoreSnapshotCaptureError::MutableReferenceGlobal { index } => { ++ StoreSnapshotRestoreError::ReferenceGlobal { index } ++ } ++ StoreSnapshotCaptureError::UnsupportedBackend => { ++ StoreSnapshotRestoreError::UnsupportedBackend ++ } ++ })?; ++ if snapshot.shape != actual_shape { ++ return Err(StoreSnapshotRestoreError::SparseShapeMismatch); ++ } ++ match objects { ++ #[cfg(any(feature = "sys", feature = "sys-minimal"))] ++ wasmer::StoreObjects::Sys(store) => { ++ // SAFETY: value count and exact destination shape were validated before the ++ // first write, including rejection of every mutable reference global. ++ unsafe { restore_sparse_numeric_globals(store, &snapshot.mutable_globals) }; ++ } ++ #[allow(unreachable_patterns)] ++ _ => return Err(StoreSnapshotRestoreError::UnsupportedBackend), ++ } ++ Ok(()) ++ } ++ } + } + +-pub fn restore_store_snapshot(store: &mut impl wasmer::AsStoreMut, snapshot: &StoreSnapshot) { +- let objs = store.objects_mut(); ++#[cfg(all(test, feature = "sys"))] ++mod tests { ++ use super::*; ++ use wasmer::{Global, Instance, Module, Store, Value, imports}; ++ ++ struct GlobalFixture { ++ store: Store, ++ imported: Global, ++ main: Global, ++ main_constant: Global, ++ side: Global, ++ } ++ ++ impl GlobalFixture { ++ fn new(imported: i32, main: i64, side: f64, extra_side_global: bool) -> Self { ++ let mut store = Store::default(); ++ let imported_global = Global::new_mut(&mut store, Value::I32(imported)); ++ let main_module = Module::new( ++ &store, ++ r#" ++ (module ++ (import "env" "imported" (global $imported (mut i32))) ++ (global (export "main") (mut i64) (i64.const 0)) ++ (global (export "main_constant") i32 (i32.const 77))) ++ "#, ++ ) ++ .unwrap(); ++ let main_instance = Instance::new( ++ &mut store, ++ &main_module, ++ &imports! { ++ "env" => { "imported" => imported_global.clone() } ++ }, ++ ) ++ .unwrap(); ++ let main_global = main_instance.exports.get_global("main").unwrap().clone(); ++ let main_constant = main_instance ++ .exports ++ .get_global("main_constant") ++ .unwrap() ++ .clone(); ++ main_global.set(&mut store, Value::I64(main)).unwrap(); ++ ++ let side_wat = if extra_side_global { ++ r#" ++ (module ++ (global (export "side") (mut f64) (f64.const 0)) ++ (global (mut i32) (i32.const 1))) ++ "# ++ } else { ++ r#" ++ (module ++ (global (export "side") (mut f64) (f64.const 0))) ++ "# ++ }; ++ let side_module = Module::new(&store, side_wat).unwrap(); ++ let side_instance = Instance::new(&mut store, &side_module, &imports! {}).unwrap(); ++ let side_global = side_instance.exports.get_global("side").unwrap().clone(); ++ side_global.set(&mut store, Value::F64(side)).unwrap(); ++ ++ Self { ++ store, ++ imported: imported_global, ++ main: main_global, ++ main_constant, ++ side: side_global, ++ } ++ } ++ ++ fn values(&mut self) -> (Value, Value, Value, Value) { ++ ( ++ self.imported.get(&mut self.store), ++ self.main.get(&mut self.store), ++ self.main_constant.get(&mut self.store), ++ self.side.get(&mut self.store), ++ ) ++ } ++ } ++ ++ #[test] ++ fn sparse_snapshot_restores_imported_main_and_side_module_globals_to_fresh_store() { ++ let mut source = GlobalFixture::new(11, 22, 33.5, false); ++ let snapshot = capture_store_snapshot(&mut source.store).unwrap(); ++ let StoreSnapshotRepr::SparseV2(sparse) = &snapshot.repr else { ++ panic!("sys snapshots must use the sparse V2 representation"); ++ }; ++ assert_eq!(sparse.shape.mutable_global_count, 3); ++ assert_eq!(sparse.mutable_globals.len(), 3); ++ assert_eq!( ++ sparse.mutable_globals.capacity() * std::mem::size_of::(), ++ 48 ++ ); ++ ++ let encoded = snapshot.serialize().unwrap(); ++ assert!(encoded.starts_with(STORE_SNAPSHOT_MAGIC)); ++ assert_eq!( ++ encoded[STORE_SNAPSHOT_MAGIC.len()], ++ STORE_SNAPSHOT_VERSION_SPARSE ++ ); ++ let decoded = StoreSnapshot::deserialize(&encoded).unwrap(); ++ ++ // A distinct Store and fresh instances model EXEC_BACKEND creation. The store-global ++ // order is the same, but every allocation and initial mutable value is different. ++ let mut destination = GlobalFixture::new(101, 202, 303.5, false); ++ restore_store_snapshot(&mut destination.store, &decoded).unwrap(); ++ assert_eq!( ++ destination.values(), ++ ( ++ Value::I32(11), ++ Value::I64(22), ++ Value::I32(77), ++ Value::F64(33.5) ++ ) ++ ); ++ } ++ ++ #[test] ++ fn sparse_snapshot_rejects_shape_mismatch_before_changing_any_global() { ++ let mut source = GlobalFixture::new(11, 22, 33.5, false); ++ let snapshot = capture_store_snapshot(&mut source.store).unwrap(); ++ let mut destination = GlobalFixture::new(101, 202, 303.5, true); ++ let before = destination.values(); ++ ++ assert_eq!( ++ restore_store_snapshot(&mut destination.store, &snapshot), ++ Err(StoreSnapshotRestoreError::SparseShapeMismatch) ++ ); ++ assert_eq!(destination.values(), before); ++ } ++ ++ #[test] ++ fn const_heavy_store_allocation_scales_with_mutable_globals_only() { ++ let mut store = Store::default(); ++ for value in 0..10_907_i32 { ++ let _ = Global::new(&mut store, Value::I32(value)); ++ } ++ for value in 0..4_i32 { ++ let _ = Global::new_mut(&mut store, Value::I32(value)); ++ } ++ ++ let snapshot = capture_store_snapshot(&mut store).unwrap(); ++ let StoreSnapshotRepr::SparseV2(sparse) = &snapshot.repr else { ++ panic!("sys snapshots must use the sparse V2 representation"); ++ }; ++ assert_eq!(sparse.shape.global_count, 10_911); ++ assert_eq!(sparse.shape.mutable_global_count, 4); ++ assert_eq!(sparse.mutable_globals.len(), 4); ++ assert_eq!( ++ sparse.mutable_globals.capacity() * std::mem::size_of::(), ++ 64 ++ ); ++ assert_eq!(snapshot.serialize().unwrap().len(), 129); ++ assert_eq!(10_911 * std::mem::size_of::(), 174_576); ++ } ++ ++ #[test] ++ fn persisted_unversioned_dense_snapshot_still_decodes_and_restores() { ++ let legacy = DenseStoreSnapshotV1 { ++ globals: vec![11_u128, 22_u128, 77_u128, 33.5_f64.to_bits().into()], ++ }; ++ let persisted_bytes = bincode::serde::encode_to_vec(&legacy, config::legacy()).unwrap(); ++ assert!(!persisted_bytes.starts_with(STORE_SNAPSHOT_MAGIC)); ++ ++ let decoded = StoreSnapshot::deserialize(&persisted_bytes).unwrap(); ++ assert!(matches!(decoded.repr, StoreSnapshotRepr::DenseV1(_))); ++ // A decode/encode cycle does not silently migrate or invalidate journal bytes. ++ assert_eq!(decoded.serialize().unwrap(), persisted_bytes); ++ ++ let mut destination = GlobalFixture::new(101, 202, 303.5, false); ++ restore_store_snapshot(&mut destination.store, &decoded).unwrap(); ++ assert_eq!( ++ destination.values(), ++ ( ++ Value::I32(11), ++ Value::I64(22), ++ Value::I32(77), ++ Value::F64(33.5) ++ ) ++ ); ++ } ++ ++ #[test] ++ fn capture_rejects_mutable_reference_global_before_raw_read() { ++ let mut store = Store::default(); ++ let _numeric = Global::new_mut(&mut store, Value::I32(7)); ++ let _reference = Global::new_mut(&mut store, Value::ExternRef(None)); ++ ++ assert!(matches!( ++ capture_store_snapshot(&mut store), ++ Err(StoreSnapshotCaptureError::MutableReferenceGlobal { index: 1 }) ++ )); ++ } ++ ++ #[test] ++ fn dense_restore_rejects_reference_shape_before_any_write() { ++ let mut store = Store::default(); ++ let numeric = Global::new_mut(&mut store, Value::I32(7)); ++ let _reference = Global::new_mut(&mut store, Value::ExternRef(None)); ++ let snapshot = StoreSnapshot { ++ repr: StoreSnapshotRepr::DenseV1(DenseStoreSnapshotV1 { ++ globals: vec![99, 0], ++ }), ++ }; + +- for (index, value) in snapshot.globals.iter().enumerate() { +- objs.set_global_unchecked(index, *value); ++ assert_eq!( ++ restore_store_snapshot(&mut store, &snapshot), ++ Err(StoreSnapshotRestoreError::ReferenceGlobal { index: 1 }) ++ ); ++ assert_eq!(numeric.get(&mut store), Value::I32(7)); + } + } diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/0008-postmaster-executor-and-build-closure.patch b/src/wasix/postmaster/wasmer/patches/wasmer/0008-postmaster-executor-and-build-closure.patch new file mode 100644 index 000000000..0f8c18b88 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/0008-postmaster-executor-and-build-closure.patch @@ -0,0 +1,849 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH 8/9] postmaster: supply the engine build closure + +Keep CLI/workspace dependency integration for the sealed carrier. The product +compiler/executor, memory profile and startup proof are now first-party sources +in src/wasix/postmaster/executor, not patched into Wasmer. This patch retains +the main branch engine build closure; Cargo locks and features remain pinned. + +Extracted without runtime hunk changes from the inherited integration bundle. +The complete ordered series is the build and qualification unit. + +diff --git a/Cargo.lock b/Cargo.lock +index 19464e5..8c36cb7 100644 +--- a/Cargo.lock ++++ b/Cargo.lock +@@ -535,17 +535,6 @@ dependencies = [ + "allocator-api2", + ] + +-[[package]] +-name = "bus" +-version = "2.4.1" +-source = "registry+https://github.com/rust-lang/crates.io-index" +-checksum = "4b7118d0221d84fada881b657c2ddb7cd55108db79c8764c9ee212c0c259b783" +-dependencies = [ +- "crossbeam-channel", +- "num_cpus", +- "parking_lot_core", +-] +- + [[package]] + name = "bytecheck" + version = "0.8.2" +@@ -3716,6 +3705,26 @@ dependencies = [ + "ruzstd", + ] + ++[[package]] ++name = "oliphaunt-wasix-postmaster-executor" ++version = "7.2.0-alpha.2" ++dependencies = [ ++ "anyhow", ++ "async-trait", ++ "dirs", ++ "hex", ++ "libc", ++ "serde", ++ "serde_json", ++ "sha2 0.11.0", ++ "tempfile", ++ "tracing", ++ "wasmer", ++ "wasmer-types", ++ "wasmer-vm", ++ "wasmer-wasix", ++] ++ + [[package]] + name = "once_cell" + version = "1.21.4" +@@ -7113,6 +7128,7 @@ dependencies = [ + "mio", + "normpath", + "object 0.39.1", ++ "oliphaunt-wasix-postmaster-executor", + "once_cell", + "opener", + "parking_lot", +@@ -7162,6 +7178,7 @@ dependencies = [ + "wasmer-wasix", + "wasmer-wast", + "webc", ++ "windows-sys 0.61.2", + "zip", + ] + +@@ -7578,7 +7595,6 @@ dependencies = [ + "base64 0.22.1", + "bincode 2.0.1", + "blake3", +- "bus", + "bytecheck", + "bytes", + "cfg-if", +diff --git a/Cargo.toml b/Cargo.toml +index 6ce7557..fa7d8dc 100644 +--- a/Cargo.toml ++++ b/Cargo.toml +@@ -110,7 +111,6 @@ bindgen = "0.72.1" + bitflags = "2.11.0" + blake3 = "1.0" + build-deps = "0.1.4" +-bus = "2.4.1" + bytecheck = { version = "0.8.2" } + byteorder = "1.3" + bytes = "1.11.1" +diff --git a/lib/cli/Cargo.toml b/lib/cli/Cargo.toml +index aceecf7..6e710d3 100644 +--- a/lib/cli/Cargo.toml ++++ b/lib/cli/Cargo.toml +@@ -34,7 +34,8 @@ default = [ + "wast", + "journal", + "wasmer-artifact-create", +- "static-artifact-create" ++ "static-artifact-create", ++ "cli-signal-watcher" + ] + + # # Tun-tap client for connecting to Wasmer Edge VPNs +@@ -49,7 +50,8 @@ default = [ + journal = ["wasmer-wasix/journal"] + backend = [] + coredump = ["wasm-coredump-builder"] +-sys = ["compiler", "dep:wasmer-vm"] ++native-runtime = ["dep:wasmer-vm", "wasmer/sys"] ++sys = ["compiler", "native-runtime"] + v8 = ["backend", "wasmer/v8"] + wast = ["wasmer-wast"] + host-net = ["virtual-net/host-net"] +@@ -92,10 +94,18 @@ disable-all-logging = [ + "wasmer-wasix/disable-all-logging", + "log/release_max_level_off", + ] +-headless = ["dep:wasmer-vm", "wasmer/sys"] +-headless-minimal = ["headless", "disable-all-logging"] ++headless = ["native-runtime"] ++headless-minimal = [ ++ "headless", ++ "disable-all-logging", ++ "oliphaunt-wasix-postmaster-executor/memory-profile-core", ++] + telemetry = [] + napi-v8 = ["dep:wasmer-napi"] ++# Preserve the interactive full CLI's historical Ctrl+C exit watcher without ++# pulling Tokio's process-global signal machinery into the compiler-free ++# headless carrier. ++cli-signal-watcher = ["tokio/signal"] + + # Optional + enable-serde = [ +@@ -134,6 +144,9 @@ wasmer-types = { version = "=7.2.0-alpha.2", path = "../types", features = [ + "detect-wasm-features", + ] } + wasmer-napi = { version = "0.702.0-alpha.2", path = "../napi", optional = true } ++oliphaunt-wasix-postmaster-executor = { version = "=7.2.0-alpha.2", path = "../oliphaunt-wasix-postmaster-executor", default-features = false, features = [ ++ "compat-cache-dir", ++] } + virtual-fs = { version = "0.702.0-alpha.2", path = "../virtual-fs", default-features = false, features = [ + "host-fs", + ] } +@@ -281,6 +294,10 @@ pretty_assertions.workspace = true + + [target.'cfg(target_os = "windows")'.dependencies] + colored.workspace = true ++windows-sys = { workspace = true, features = [ ++ "Win32_Foundation", ++ "Win32_Storage_FileSystem", ++] } + + [package.metadata.binstall] + pkg-fmt = "tgz" +diff --git a/lib/cli/src/backend.rs b/lib/cli/src/backend.rs +index 1c37513..17fe4c9 100644 +--- a/lib/cli/src/backend.rs ++++ b/lib/cli/src/backend.rs +@@ -12,7 +12,7 @@ use std::sync::Arc; + use std::{path::PathBuf, str::FromStr}; + + use anyhow::{Context, Result, bail}; +-#[cfg(feature = "sys")] ++#[cfg(feature = "native-runtime")] + use wasmer::sys::*; + use wasmer::*; + use wasmer_types::{Features, target::Target}; +@@ -297,7 +297,6 @@ impl RuntimeOptions { + filtered_backends.first().unwrap().get_engine(target, self) + } + +- #[cfg(feature = "compiler")] + /// Get the enabled Wasm features. + pub fn get_features(&self, default_features: &Features) -> Result { + if self.features.all { +@@ -323,10 +322,33 @@ impl RuntimeOptions { + if self.features.reference_types { + result.reference_types(true); + } ++ if self.features.tail_call { ++ result.tail_call(true); ++ } ++ if self.features.module_linking { ++ result.module_linking(true); ++ } ++ if self.features.multi_memory { ++ result.multi_memory(true); ++ } ++ if self.features.memory64 { ++ result.memory64(true); ++ } ++ if self.features.exceptions { ++ result.exceptions(true); ++ } ++ if self.features.relaxed_simd { ++ result.relaxed_simd(true); ++ } ++ if self.features.extended_const { ++ result.extended_const(true); ++ } ++ if self.features.wide_arithmetic { ++ result.wide_arithmetic(true); ++ } + Ok(result) + } + +- #[cfg(feature = "compiler")] + /// Get a copy of the default features with user-configured options + pub fn get_configured_features(&self) -> Result { + let features = Features::default(); +@@ -531,6 +553,8 @@ impl BackendType { + Self::Singlepass, + #[cfg(feature = "v8")] + Self::V8, ++ #[cfg(all(feature = "headless", not(feature = "compiler")))] ++ Self::Headless, + ] + } + +@@ -644,7 +668,11 @@ impl BackendType { + } + #[cfg(feature = "v8")] + Self::V8 => Ok(wasmer::v8::V8::new().into()), +- Self::Headless => bail!("Headless is not a valid runtime to instantiate directly"), ++ Self::Headless => Ok(EngineBuilder::headless() ++ .set_features(Some(runtime_opts.get_configured_features()?)) ++ .set_target(Some(target.clone())) ++ .engine() ++ .into()), + #[allow(unreachable_patterns)] + _ => bail!("Unsupported backend type"), + } +@@ -663,7 +691,10 @@ impl BackendType { + Self::LLVM => wasmer::BackendKind::LLVM, + #[cfg(feature = "v8")] + Self::V8 => wasmer::BackendKind::V8, +- Self::Headless => return false, // Headless can't compile ++ // A headless engine does not compile the input module. Feature ++ // compatibility is enforced while deserializing the selected AOT ++ // artifact, so feature detection must not filter it out here. ++ Self::Headless => return true, + #[allow(unreachable_patterns)] + _ => return false, + }; +diff --git a/lib/cli/src/commands/mod.rs b/lib/cli/src/commands/mod.rs +index 5896d6f..eeeaaa9 100644 +--- a/lib/cli/src/commands/mod.rs ++++ b/lib/cli/src/commands/mod.rs +@@ -32,6 +32,7 @@ mod validate; + #[cfg(feature = "wast")] + mod wast; + use itertools::Itertools; ++#[cfg(feature = "cli-signal-watcher")] + use std::io::IsTerminal as _; + use tokio::task::JoinHandle; + +@@ -78,29 +79,38 @@ pub(crate) trait AsyncCliCommand: Send + Sync { + &self, + done: tokio::sync::oneshot::Receiver<()>, + ) -> Option>> { +- if std::io::stdin().is_terminal() { +- return Some(tokio::task::spawn(async move { +- tokio::select! { +- _ = done => {} +- +- _ = tokio::signal::ctrl_c() => { +- let term = console::Term::stdout(); +- let _ = term.show_cursor(); +- // https://learn.microsoft.com/en-us/cpp/c-runtime-library/signal-constants +- #[cfg(target_os = "windows")] +- std::process::exit(3); +- +- // POSIX compliant OSs: 128 + SIGINT (2) +- #[cfg(not(target_os = "windows"))] +- std::process::exit(130); ++ #[cfg(not(feature = "cli-signal-watcher"))] ++ { ++ let _ = done; ++ return None; ++ } ++ ++ #[cfg(feature = "cli-signal-watcher")] ++ { ++ if std::io::stdin().is_terminal() { ++ return Some(tokio::task::spawn(async move { ++ tokio::select! { ++ _ = done => {} ++ ++ _ = tokio::signal::ctrl_c() => { ++ let term = console::Term::stdout(); ++ let _ = term.show_cursor(); ++ // https://learn.microsoft.com/en-us/cpp/c-runtime-library/signal-constants ++ #[cfg(target_os = "windows")] ++ std::process::exit(3); ++ ++ // POSIX compliant OSs: 128 + SIGINT (2) ++ #[cfg(not(target_os = "windows"))] ++ std::process::exit(130); ++ } + } +- } + +- Ok::<(), anyhow::Error>(()) +- })); +- } ++ Ok::<(), anyhow::Error>(()) ++ })); ++ } + +- None ++ None ++ } + } + } + +diff --git a/lib/cli/src/commands/run/mod.rs b/lib/cli/src/commands/run/mod.rs +index cc9a3b0..8ffcb13 100644 +--- a/lib/cli/src/commands/run/mod.rs ++++ b/lib/cli/src/commands/run/mod.rs +@@ -20,15 +20,16 @@ use std::{ + time::{Duration, SystemTime, UNIX_EPOCH}, + }; + +-use anyhow::{Context, Error, anyhow, bail}; ++use anyhow::{Context, Error, anyhow, bail, ensure}; + use clap::{Parser, ValueEnum}; + use colored::Colorize; + use futures::future::BoxFuture; + use indicatif::{MultiProgress, ProgressBar}; ++use oliphaunt_wasix_postmaster_executor::sealed; + use once_cell::sync::Lazy; + use tempfile::NamedTempFile; + use url::Url; +-#[cfg(feature = "sys")] ++#[cfg(feature = "native-runtime")] + use wasmer::sys::NativeEngineExt; + use wasmer::{ + AsStoreMut, DeserializeError, Engine, Function, Imports, Instance, Module, RuntimeError, Store, +@@ -45,8 +46,10 @@ use wasmer_types::ModuleHash; + + #[cfg(feature = "journal")] + use wasmer_wasix::journal::{LogFileJournal, SnapshotTrigger}; ++#[cfg(any(unix, windows))] ++use wasmer_wasix::os::task::HostLifecycleSupervisor; + use wasmer_wasix::{ +- Runtime, SpawnError, WasiError, ++ ResourceLimits, Runtime, SpawnError, WasiError, + bin_factory::{BinaryPackage, BinaryPackageCommand}, + journal::CompactingLogFileJournal, + runners::{ +@@ -57,11 +60,8 @@ use wasmer_wasix::{ + wcgi::{self, AbortHandle, NoOpWcgiCallbacks, WcgiRunner}, + }, + runtime::{ +- OverriddenRuntime, +- module_cache::{CacheError, HashedModuleData}, +- package_loader::PackageLoader, +- resolver::QueryError, +- task_manager::VirtualTaskManagerExt, ++ OverriddenRuntime, module_cache::ModuleCache, package_loader::PackageLoader, ++ resolver::QueryError, task_manager::VirtualTaskManagerExt, + }, + }; + use webc::Container; +@@ -76,10 +76,54 @@ use crate::{ + }; + + use self::{ +- package_source::CliPackageSource, runtime::MonitoringRuntime, target::ExecutableTarget, ++ package_source::CliPackageSource, ++ runtime::{CliTokioRuntimePolicy, MonitoringRuntime}, ++ target::ExecutableTarget, + }; + + const TICK: Duration = Duration::from_millis(250); ++const WASIX_STACK_RLIMIT_DIVISOR: u64 = 8; ++ ++#[cfg(any(unix, windows))] ++fn attach_cli_host_lifecycle(runner: &mut WasiRunner) -> Result<(), Error> { ++ let supervisor = HostLifecycleSupervisor::install() ++ .context("Unable to install exclusive CLI host lifecycle supervision")?; ++ runner.with_host_lifecycle_supervisor(Arc::new(supervisor)); ++ Ok(()) ++} ++ ++#[cfg(not(any(unix, windows)))] ++fn attach_cli_host_lifecycle(_runner: &mut WasiRunner) -> Result<(), Error> { ++ Ok(()) ++} ++ ++fn conservative_stack_resource_limit(stack_size: usize) -> u64 { ++ ((stack_size as u64) / WASIX_STACK_RLIMIT_DIVISOR).max(1) ++} ++ ++fn resource_limits_for_engine(engine: &Engine, stack_size: Option) -> ResourceLimits { ++ ResourceLimits { ++ stack: { ++ #[cfg(feature = "native-runtime")] ++ { ++ if engine.is_sys() { ++ if let Some(stack_size) = stack_size { ++ wasmer_vm::set_stack_size(stack_size); ++ } ++ Some(conservative_stack_resource_limit( ++ wasmer_vm::get_stack_size(), ++ )) ++ } else { ++ None ++ } ++ } ++ #[cfg(not(feature = "native-runtime"))] ++ { ++ None ++ } ++ }, ++ } ++} + + /// The unstable `wasmer run` subcommand. + #[derive(Debug, Parser)] +@@ -95,6 +139,10 @@ pub struct Run { + /// Set the default stack size (default is 1048576) + #[clap(long = "stack-size")] + stack_size: Option, ++ /// Run a verified, precompiled-only executable closure. ++ #[cfg(feature = "headless")] ++ #[clap(long = "sealed-module-manifest", value_name = "PATH")] ++ sealed_module_manifest: Option, + /// The entrypoint module for webc packages. + #[clap(short, long, aliases = &["command", "command-name"])] + entrypoint: Option, +@@ -115,6 +163,17 @@ pub struct Run { + } + + impl Run { ++ fn sealed_manifest_path(&self) -> Option<&Path> { ++ #[cfg(feature = "headless")] ++ { ++ self.sealed_module_manifest.as_deref() ++ } ++ #[cfg(not(feature = "headless"))] ++ { ++ None ++ } ++ } ++ + #[cfg(feature = "napi-v8")] + fn module_needs_napi(module: &Module) -> bool { + let (napi_version, napi_extension_version) = wasmer_napi::module_needs_napi(module); +@@ -187,9 +246,28 @@ impl Run { + + pb.set_message("Initializing the WebAssembly VM"); + +- let runtime = tokio::runtime::Builder::new_multi_thread() +- .enable_all() +- .build()?; ++ let sealed_manifest_path = self.sealed_manifest_path().map(Path::to_path_buf); ++ let sealed_manifest = sealed_manifest_path ++ .as_deref() ++ .map(|manifest| { ++ let input = match &self.input { ++ CliPackageSource::File(path) => path.as_path(), ++ _ => bail!("--sealed-module-manifest requires a local executable file input"), ++ }; ++ sealed::prepare(manifest, input) ++ }) ++ .transpose()?; ++ let sealed_runtime_identity = sealed_manifest ++ .as_ref() ++ .and_then(sealed::PreparedSealedManifest::runtime_identity); ++ let tokio_runtime_policy = CliTokioRuntimePolicy::select(sealed_runtime_identity); ++ tracing::info!( ++ policy_id = tokio_runtime_policy.id(), ++ workload_id = ?tokio_runtime_policy.workload_id(), ++ configured_worker_threads = ?tokio_runtime_policy.configured_worker_threads(), ++ "Selected CLI Tokio runtime policy" ++ ); ++ let runtime = tokio_runtime_policy.build()?; + let handle = runtime.handle().clone(); + + // Check for the preferred webc version. +@@ -210,7 +288,9 @@ impl Run { + + // Try to detect WebAssembly features before selecting a backend + tracing::info!("Input source: {:?}", self.input); +- if let CliPackageSource::File(path) = &self.input { ++ if sealed_manifest_path.is_none() ++ && let CliPackageSource::File(path) = &self.input ++ { + tracing::info!("Input file path: {}", path.display()); + + // Try to read and detect any file that exists, regardless of extension +@@ -275,20 +355,24 @@ impl Run { + let engine_kind = engine.deterministic_id(); + tracing::info!("Executing on backend {engine_kind:?}"); + +- #[cfg(feature = "sys")] +- if engine.is_sys() && self.stack_size.is_some() { +- wasmer_vm::set_stack_size(self.stack_size.unwrap()); +- } ++ let sealed_modules = sealed_manifest ++ .map(|manifest| sealed::load(manifest, &engine)) ++ .transpose()?; ++ let sealed_module_cache = sealed_modules ++ .as_ref() ++ .map(|sealed| sealed.module_cache.clone() as Arc); + +- let engine = engine.clone(); ++ let resource_limits = resource_limits_for_engine(&engine, self.stack_size); + + let runtime = self.wasi.prepare_runtime( +- engine, ++ engine.clone(), + &self.env, + &capabilities::get_capability_cache_path(&self.env, &self.input)?, + runtime, + preferred_webc_version, + self.rt.compiler_debug_dir.is_some(), ++ resource_limits, ++ sealed_module_cache, + )?; + + // This is a slow operation, so let's temporarily wrap the runtime with +@@ -301,7 +385,20 @@ impl Run { + let runtime: Arc = monitoring_runtime.runtime.clone(); + let monitoring_runtime: Arc = monitoring_runtime; + +- let target = self.input.resolve_target(&monitoring_runtime, &pb)?; ++ let (target, sealed_executables) = match sealed_modules { ++ Some(sealed) => ( ++ ExecutableTarget::WebAssembly { ++ module: sealed.module, ++ module_hash: sealed.module_hash, ++ path: sealed.path, ++ }, ++ sealed.executables, ++ ), ++ None => ( ++ self.input.resolve_target(&monitoring_runtime, &pb)?, ++ Vec::new(), ++ ), ++ }; + + if let ExecutableTarget::Package(ref pkg) = target { + self.wasi +@@ -320,7 +417,13 @@ impl Run { + module, + module_hash, + path, +- } => self.execute_wasm(&path, module, module_hash, runtime.clone()), ++ } => self.execute_wasm( ++ &path, ++ module, ++ module_hash, ++ runtime.clone(), ++ sealed_executables, ++ ), + ExecutableTarget::Package(pkg) => { + // Check if we should update the engine based on the WebC package features + if let Some(cmd) = pkg.get_entrypoint_command() +@@ -359,15 +462,17 @@ impl Run { + &self.env, + &self.input, + )?; ++ let new_resource_limits = ++ resource_limits_for_engine(&new_engine, self.stack_size); + let new_runtime = self.wasi.prepare_runtime( + new_engine, + &self.env, + &capability_cache_path, +- tokio::runtime::Builder::new_multi_thread() +- .enable_all() +- .build()?, ++ tokio_runtime_policy.build()?, + preferred_webc_version, + self.rt.compiler_debug_dir.is_some(), ++ new_resource_limits, ++ None, + )?; + + let new_runtime = Arc::new(MonitoringRuntime::new( +@@ -405,9 +510,10 @@ impl Run { + module: Module, + module_hash: ModuleHash, + runtime: Arc, ++ sealed_executables: Vec<(String, ModuleHash)>, + ) -> Result<(), Error> { + if wasmer_wasix::is_wasi_module(&module) || wasmer_wasix::is_wasix_module(&module) { +- self.execute_wasi_module(path, module, module_hash, runtime) ++ self.execute_wasi_module(path, module, module_hash, runtime, sealed_executables) + } else { + self.execute_pure_wasm_module(&module) + } +@@ -495,6 +601,7 @@ impl Run { + + // Assume webcs are always WASIX + let mut runner = self.build_wasi_runner(&runtime, true)?; ++ attach_cli_host_lifecycle(&mut runner)?; + #[cfg(feature = "napi-v8")] + self.configure_wasi_runner_for_napi(&module, &mut runner); + Runner::run_command(&mut runner, command_name, pkg, runtime) +@@ -696,12 +803,17 @@ impl Run { + module: Module, + module_hash: ModuleHash, + runtime: Arc, ++ sealed_executables: Vec<(String, ModuleHash)>, + ) -> Result<(), Error> { + let program_name = wasm_path.display().to_string(); + let runtime = self.maybe_wrap_runtime_with_napi(&module, runtime)?; + + let mut runner = + self.build_wasi_runner(&runtime, wasmer_wasix::is_wasix_module(&module))?; ++ attach_cli_host_lifecycle(&mut runner)?; ++ for (guest_path, sealed_hash) in sealed_executables { ++ runner.with_sealed_module(guest_path, sealed_hash, None); ++ } + self.configure_wasi_runner_for_napi(&module, &mut runner); + runner.run_wasm( + RuntimeOrEngine::Runtime(runtime), +@@ -905,3 +1017,34 @@ fn get_exit_code( + + None + } ++ ++#[cfg(all(test, feature = "headless", not(feature = "compiler")))] ++mod tests { ++ use super::*; ++ ++ struct StackSizeRestore(usize); ++ ++ impl Drop for StackSizeRestore { ++ fn drop(&mut self) { ++ wasmer_vm::set_stack_size(self.0); ++ } ++ } ++ ++ #[test] ++ fn headless_stack_size_sets_vm_and_guest_limit() { ++ let run = Run::try_parse_from(["run", "--stack-size", "33554432", "probe.wasm"]) ++ .expect("headless run arguments should parse"); ++ let engine = run ++ .rt ++ .get_engine(&Target::default()) ++ .expect("headless engine should be available"); ++ assert_eq!(engine.deterministic_id(), "engine-headless"); ++ ++ let previous_stack_size = wasmer_vm::get_stack_size(); ++ let _restore = StackSizeRestore(previous_stack_size); ++ let limits = resource_limits_for_engine(&engine, run.stack_size); ++ ++ assert_eq!(wasmer_vm::get_stack_size(), 33_554_432); ++ assert_eq!(limits.stack, Some(4_194_304)); ++ } ++} +diff --git a/lib/cli/src/commands/run/runtime.rs b/lib/cli/src/commands/run/runtime.rs +index c562c24..d06fcc8 100644 +--- a/lib/cli/src/commands/run/runtime.rs ++++ b/lib/cli/src/commands/run/runtime.rs +@@ -1,6 +1,6 @@ + //! Provides CLI-specific Wasix components. + +-use std::{sync::Arc, time::Duration}; ++use std::{io, sync::Arc, time::Duration}; + + use anyhow::Error; + use futures::future::BoxFuture; +@@ -23,6 +23,64 @@ use wasmer_wasix::{ + }; + use webc::Container; + ++use oliphaunt_wasix_postmaster_executor::sealed::SealedRuntimeIdentity; ++ ++const GENERIC_TOKIO_RUNTIME_POLICY_ID: &str = "wasmer.cli.tokio.default.v1"; ++pub(super) const SEALED_POSTMASTER_TOKIO_RUNTIME_POLICY_ID: &str = ++ "oliphaunt.wasix-postmaster.tokio.sealed-postmaster-2-worker.v1"; ++const SEALED_POSTMASTER_TOKIO_WORKER_THREADS: usize = 2; ++ ++#[derive(Debug, Clone, Copy, PartialEq, Eq)] ++pub(super) enum CliTokioRuntimePolicy { ++ Generic, ++ SealedWasixPostmaster(SealedRuntimeIdentity), ++} ++ ++impl CliTokioRuntimePolicy { ++ pub(super) const fn select(identity: Option) -> Self { ++ match identity { ++ Some(identity) => Self::SealedWasixPostmaster(identity), ++ None => Self::Generic, ++ } ++ } ++ ++ pub(super) const fn id(self) -> &'static str { ++ match self { ++ Self::Generic => GENERIC_TOKIO_RUNTIME_POLICY_ID, ++ Self::SealedWasixPostmaster(_) => SEALED_POSTMASTER_TOKIO_RUNTIME_POLICY_ID, ++ } ++ } ++ ++ pub(super) const fn workload_id(self) -> Option<&'static str> { ++ match self { ++ Self::Generic => None, ++ Self::SealedWasixPostmaster(identity) => Some(identity.workload_id()), ++ } ++ } ++ ++ pub(super) const fn configured_worker_threads(self) -> Option { ++ match self { ++ Self::Generic => None, ++ Self::SealedWasixPostmaster(_) => Some(SEALED_POSTMASTER_TOKIO_WORKER_THREADS), ++ } ++ } ++ ++ pub(super) fn build(self) -> io::Result { ++ match self { ++ // Keep generic Wasmer behavior byte-for-byte equivalent to the ++ // historical builder: Tokio owns its platform-default worker ++ // selection and environment override semantics. ++ Self::Generic => tokio::runtime::Builder::new_multi_thread() ++ .enable_all() ++ .build(), ++ Self::SealedWasixPostmaster(_) => tokio::runtime::Builder::new_multi_thread() ++ .worker_threads(SEALED_POSTMASTER_TOKIO_WORKER_THREADS) ++ .enable_all() ++ .build(), ++ } ++ } ++} ++ + /// Special wasix runtime implementation for the CLI. + /// + /// Wraps an undelrying runtime and adds progress monitoring for package +@@ -53,6 +111,10 @@ impl wasmer_wasix::Runtime for Monitorin + self.runtime.task_manager() + } + ++ fn resource_limits(&self) -> wasmer_wasix::ResourceLimits { ++ self.runtime.resource_limits() ++ } ++ + fn package_loader( + &self, + ) -> Arc { +@@ -264,3 +326,42 @@ impl wasmer_wasix::runtime::package_loader::PackageLoader for MonitoringPackageL + .await + } + } ++ ++#[cfg(test)] ++mod tests { ++ use super::*; ++ ++ #[test] ++ fn sealed_postmaster_policy_id_and_worker_configuration_are_stable() { ++ for identity in [ ++ SealedRuntimeIdentity::WasixPostmasterInitdb, ++ SealedRuntimeIdentity::WasixPostmasterPostgres, ++ ] { ++ let policy = CliTokioRuntimePolicy::select(Some(identity)); ++ assert_eq!( ++ policy.id(), ++ "oliphaunt.wasix-postmaster.tokio.sealed-postmaster-2-worker.v1" ++ ); ++ assert_eq!(policy.workload_id(), Some(identity.workload_id())); ++ assert_eq!(policy.configured_worker_threads(), Some(2)); ++ } ++ } ++ ++ #[test] ++ fn sealed_postmaster_runtime_has_exactly_two_tokio_workers() { ++ let runtime = ++ CliTokioRuntimePolicy::select(Some(SealedRuntimeIdentity::WasixPostmasterPostgres)) ++ .build() ++ .unwrap(); ++ ++ assert_eq!(runtime.metrics().num_workers(), 2); ++ } ++ ++ #[test] ++ fn generic_runtime_policy_retains_tokio_default_worker_selection() { ++ let policy = CliTokioRuntimePolicy::select(None); ++ assert_eq!(policy.id(), "wasmer.cli.tokio.default.v1"); ++ assert_eq!(policy.workload_id(), None); ++ assert_eq!(policy.configured_worker_threads(), None); ++ } ++} +diff --git a/lib/cli/src/commands/run/wasi.rs b/lib/cli/src/commands/run/wasi.rs +index 3297890..4cfac63 100644 +--- a/lib/cli/src/commands/run/wasi.rs ++++ b/lib/cli/src/commands/run/wasi.rs +@@ -23,8 +23,8 @@ use wasmer_types::ModuleHash; + #[cfg(feature = "journal")] + use wasmer_wasix::journal::{LogFileJournal, SnapshotTrigger}; + use wasmer_wasix::{ +- PluggableRuntime, RewindState, Runtime, WasiEnv, WasiEnvBuilder, WasiError, WasiFunctionEnv, +- WasiVersion, ++ PluggableRuntime, ResourceLimits, RewindState, Runtime, WasiEnv, WasiEnvBuilder, WasiError, ++ WasiFunctionEnv, WasiVersion, + bin_factory::BinaryPackage, + capabilities::Capabilities, + get_wasi_versions, +@@ -580,12 +580,25 @@ impl Wasi { + rt_or_handle: I, + preferred_webc_version: webc::Version, + compiler_debug_dir_used: bool, ++ resource_limits: ResourceLimits, ++ sealed_module_cache: Option>, + ) -> Result> + where + I: Into, + { + let tokio_task_manager = Arc::new(TokioTaskManager::new(rt_or_handle.into())); + let mut rt = PluggableRuntime::new(tokio_task_manager.clone()); ++ rt.set_resource_limits(resource_limits); ++ if let Some(module_cache) = sealed_module_cache { ++ // This authoritative cache is the sealed carrier's exact immutable ++ // closure. It is installed before any runtime consumers are built. ++ rt.set_module_cache(module_cache); ++ } else if !self.disable_cache && !compiler_debug_dir_used { ++ let cache_dir = env.cache_dir().join("compiled"); ++ let module_cache = wasmer_wasix::runtime::module_cache::in_memory() ++ .with_fallback(FileSystemCache::new(cache_dir, tokio_task_manager)); ++ rt.set_module_cache(module_cache); ++ } + + let has_networking = self.networking.is_some() + || capabilities::get_cached_capability(pkg_cache_path) +@@ -643,13 +656,6 @@ impl Wasi { + + let registry = self.prepare_source(env, client, preferred_webc_version)?; + +- if !self.disable_cache && !compiler_debug_dir_used { +- let cache_dir = env.cache_dir().join("compiled"); +- let module_cache = wasmer_wasix::runtime::module_cache::in_memory() +- .with_fallback(FileSystemCache::new(cache_dir, tokio_task_manager)); +- rt.set_module_cache(module_cache); +- } +- + rt.set_package_loader(package_loader) + .set_source(registry) + .set_engine(engine); diff --git a/src/wasix/postmaster/wasmer/patches/wasmer/series b/src/wasix/postmaster/wasmer/patches/wasmer/series new file mode 100644 index 000000000..af56dbf83 --- /dev/null +++ b/src/wasix/postmaster/wasmer/patches/wasmer/series @@ -0,0 +1,8 @@ +0001-engine-memory-and-exception-lifetimes.patch +0002-virtual-fs-file-description-and-writeback.patch +0003-network-readiness-and-interest-lifetimes.patch +0004-wasix-open-file-description-operations.patch +0005-wasix-process-thread-and-join-lifetimes.patch +0006-wasix-epoll-and-socket-readiness.patch +0007-wasix-instance-linker-and-sealed-runtime.patch +0008-postmaster-executor-and-build-closure.patch diff --git a/src/wasix/postmaster/wasmer/probes/wasm_eh_sjlj_probe.c b/src/wasix/postmaster/wasmer/probes/wasm_eh_sjlj_probe.c index e798a7f3a..4fdcae043 100644 --- a/src/wasix/postmaster/wasmer/probes/wasm_eh_sjlj_probe.c +++ b/src/wasix/postmaster/wasmer/probes/wasm_eh_sjlj_probe.c @@ -1,9 +1,12 @@ #include +#include #include static jmp_buf plain_jmp; static sigjmp_buf signal_jmp; static volatile int stage; +static volatile int buffer_evaluations; +static volatile int mask_evaluations; static void jump_plain(void) { @@ -11,36 +14,106 @@ static void jump_plain(void) longjmp(plain_jmp, 7); } -static void jump_signal_zero(void) +static sigjmp_buf *signal_buffer(void) { - stage = 2; - siglongjmp(signal_jmp, 0); + buffer_evaluations++; + return &signal_jmp; +} + +static int signal_mask_option(int save) +{ + mask_evaluations++; + return save; +} + +static int check_signal_jump(int save, int value) +{ + int observed; + +#if !defined(__wasi__) + sigset_t mask; + sigemptyset(&mask); + sigaddset(&mask, SIGUSR1); + if (pthread_sigmask(SIG_BLOCK, &mask, NULL) != 0) + return 20; +#endif + buffer_evaluations = mask_evaluations = 0; + + /* Keep setjmp in the controlling expression, in this live caller frame. */ + switch (sigsetjmp(*signal_buffer(), signal_mask_option(save))) { + case 0: + if (buffer_evaluations != 1 || mask_evaluations != 1) + return 21; +#if !defined(__wasi__) + if (pthread_sigmask(SIG_UNBLOCK, &mask, NULL) != 0) + return 22; +#endif + siglongjmp(signal_jmp, value); + case 1: + observed = 1; + break; + case 7: + observed = 7; + break; + default: + return 23; + } + if (observed != (value ? value : 1) || + buffer_evaluations != 1 || mask_evaluations != 1) + return 24; +#if !defined(__wasi__) + if (pthread_sigmask(SIG_SETMASK, NULL, &mask) != 0) + return 25; + if (sigismember(&mask, SIGUSR1) != !!save) + return 26; +#endif + return 0; } int main(void) { - int ret = setjmp(plain_jmp); - if (ret == 0) { +#if !defined(__wasi__) + sigset_t original; +#endif + int result; + + switch (setjmp(plain_jmp)) { + case 0: jump_plain(); - fprintf(stderr, "longjmp returned unexpectedly\n"); return 10; - } - if (ret != 7 || stage != 1) { - fprintf(stderr, "plain setjmp/longjmp mismatch: ret=%d stage=%d\n", ret, stage); - return 11; - } - - ret = sigsetjmp(signal_jmp, 1); - if (ret == 0) { - jump_signal_zero(); - fprintf(stderr, "siglongjmp returned unexpectedly\n"); + case 7: + if (stage != 1) + return 11; + break; + default: return 12; } - if (ret != 1 || stage != 2) { - fprintf(stderr, "signal setjmp/longjmp mismatch: ret=%d stage=%d\n", ret, stage); +#if !defined(__wasi__) + if (pthread_sigmask(SIG_SETMASK, NULL, &original) != 0) return 13; +#endif + /* Reuse the buffer: savemask=0 must clear a preceding saved-mask state. */ + for (int save = 1; save >= 0; save--) { + for (int value = 0; value <= 7; value += 7) { + result = check_signal_jump(save, value); + if (result != 0) { + fprintf(stderr, "signal jump failed: save=%d value=%d check=%d\n", + save, value, result); +#if !defined(__wasi__) + pthread_sigmask(SIG_SETMASK, &original, NULL); +#endif + return result; + } + } } - printf("sjlj-ok plain=%d signal=%d\n", 7, ret); +#if !defined(__wasi__) + if (pthread_sigmask(SIG_SETMASK, &original, NULL) != 0) + return 14; + printf("sjlj-ok plain=7 signal=1 single-evaluation mask-preserved\n"); +#else + /* The pinned WASIX pthread_sigmask succeeds without implementing masks. */ + printf("sjlj-ok plain=7 signal=1 single-evaluation signal-mask=unsupported\n"); +#endif return 0; } diff --git a/src/wasix/postmaster/wasmer/tests.sh b/src/wasix/postmaster/wasmer/tests.sh index 1f363834e..0a46c4489 100644 --- a/src/wasix/postmaster/wasmer/tests.sh +++ b/src/wasix/postmaster/wasmer/tests.sh @@ -223,3 +223,11 @@ run_tests \ --features "$FRESH_MEMORY_PROFILE_FEATURES" \ -- \ memory_profile::wasm_tool::tests + +# The product compiler must preserve main's policy and distinguish strict AOT. +run_tests --locked --target-dir "$POSTMASTER_COMPILER_TARGET_DIR" \ + --manifest-path "$FRESH_ROOT/executor/Cargo.toml" \ + --package "$FRESH_POSTMASTER_EXECUTOR_PACKAGE" \ + --bin "$FRESH_POSTMASTER_COMPILER_BINARY" \ + --no-default-features --features "$FRESH_POSTMASTER_COMPILER_FEATURES" \ + -- --exact product_compiler_uses_main_memory_identity diff --git a/src/wasix/runtime/assets/build/configure_wasix_dl.sh b/src/wasix/runtime/assets/build/configure_wasix_dl.sh index 2bba44790..c11a429ec 100755 --- a/src/wasix/runtime/assets/build/configure_wasix_dl.sh +++ b/src/wasix/runtime/assets/build/configure_wasix_dl.sh @@ -29,7 +29,7 @@ COMMON_CPPFLAGS="-I$PGSRC/src/include/port/wasix-dl" COMMON_CPPFLAGS="$COMMON_CPPFLAGS $ICU_CFLAGS" COMMON_CFLAGS="$OLIPHAUNT_WASM_PROFILE_CFLAGS -sWASM_EXCEPTIONS=yes -sPIC=yes -Wno-unused-command-line-argument" COMMON_LDFLAGS="$OLIPHAUNT_WASM_PROFILE_LDFLAGS -sWASM_EXCEPTIONS=yes -sPIC=yes -L$ICU_PREFIX/lib" -MAIN_LDFLAGS="-sMODULE_KIND=dynamic-main -sSTACK_SIZE=8MB -sINITIAL_MEMORY=128MB" +MAIN_LDFLAGS="-sMODULE_KIND=dynamic-main -sSTACK_SIZE=$OLIPHAUNT_WASM_GUEST_STACK_SIZE -sINITIAL_MEMORY=$OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE" SIDE_MODULE_LDFLAGS="-Wl,-shared" CONFIGURE_EXTRA_ARGS=() if [ "${OLIPHAUNT_WASM_PG18_DISABLE_SPINLOCKS:-0}" = "1" ]; then diff --git a/src/wasix/runtime/assets/build/docker_initdb.sh b/src/wasix/runtime/assets/build/docker_initdb.sh index 39e4aee0c..194d76caa 100755 --- a/src/wasix/runtime/assets/build/docker_initdb.sh +++ b/src/wasix/runtime/assets/build/docker_initdb.sh @@ -103,7 +103,7 @@ fi COMMON_CPPFLAGS="-I$PGSRC/src/include/port/wasix-dl $ICU_CFLAGS" COMMON_CFLAGS="$OLIPHAUNT_WASM_PROFILE_CFLAGS -sWASM_EXCEPTIONS=yes -sPIC=yes -Wno-unused-command-line-argument" COMMON_LDFLAGS="$OLIPHAUNT_WASM_PROFILE_LDFLAGS -sWASM_EXCEPTIONS=yes -sPIC=yes -L$ICU_PREFIX/lib" - MAIN_LDFLAGS="-sMODULE_KIND=dynamic-main -sSTACK_SIZE=8MB -sINITIAL_MEMORY=128MB -Wl,--wrap=system -Wl,--wrap=popen -Wl,--wrap=pclose" + MAIN_LDFLAGS="-sMODULE_KIND=dynamic-main -sSTACK_SIZE=$OLIPHAUNT_WASM_GUEST_STACK_SIZE -sINITIAL_MEMORY=$OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE -Wl,--wrap=system -Wl,--wrap=popen -Wl,--wrap=pclose" INITDB_BUILD_DIR="$(oliphaunt_wasix_scratch_build_dir "${OLIPHAUNT_WASM_SOURCE_LANE:-stable}" wasix-initdb)" mkdir -p "$INITDB_BUILD_DIR" diff --git a/src/wasix/runtime/assets/build/link_wasix_runtime.sh b/src/wasix/runtime/assets/build/link_wasix_runtime.sh index 54de75de4..580342a71 100755 --- a/src/wasix/runtime/assets/build/link_wasix_runtime.sh +++ b/src/wasix/runtime/assets/build/link_wasix_runtime.sh @@ -17,6 +17,7 @@ icu_prefix="$2" bridge_object="$3" exports_file="$4" profile="$5" +script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" [ -z "${WASIXCC_LINKER_FLAGS:-}" ] || fail "custom WASIXCC_LINKER_FLAGS are unsupported by the sealed runtime link" @@ -94,6 +95,11 @@ while IFS= read -r symbol; do printf '%s\n' "--export=$symbol" done < "$exports_file" > "$response" +# Recovery must target a live call-site SJLJ frame in the main module. +bash "$script_dir/verify_wasix_sjlj_artifact.sh" \ + "$build_dir/libpgcore.o" "$wasix_home/llvm/bin/llvm-nm" \ + "$wasix_home/binaryen/bin/wasm-dis" + raw="$stage/oliphaunt.unoptimized.wasm" "$linker" \ -L"$build_dir/src/port" \ diff --git a/src/wasix/runtime/assets/build/postgres/patches/0001-oliphaunt-wasix-add-wasix-dl-build-spine.patch b/src/wasix/runtime/assets/build/postgres/patches/0001-oliphaunt-wasix-add-wasix-dl-build-spine.patch index fc6a63ee6..59d2d09f3 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0001-oliphaunt-wasix-add-wasix-dl-build-spine.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0001-oliphaunt-wasix-add-wasix-dl-build-spine.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: add wasix-dl build spine @@ -33,13 +33,13 @@ to review on their own merits. create mode 100644 src/template/wasix-dl diff --git a/src/Makefile.shlib b/src/Makefile.shlib -index 1b5ddf2825..55129150fa 100644 +index 3825af5b22..27632c5398 100644 --- a/src/Makefile.shlib +++ b/src/Makefile.shlib -@@ -196,6 +196,16 @@ ifeq ($(PORTNAME), linux) +@@ -183,6 +183,16 @@ ifeq ($(PORTNAME), linux) endif endif - + +ifeq ($(PORTNAME), wasix-dl) + LINK.shared = $(COMPILER) -shared + ifdef SO_MAJOR_VERSION @@ -54,29 +54,29 @@ index 1b5ddf2825..55129150fa 100644 LINK.shared = $(COMPILER) -shared ifdef soname diff --git a/src/backend/Makefile b/src/backend/Makefile -index 4827ee361d..9292f77031 100644 +index c0be941038..536c28821a 100644 --- a/src/backend/Makefile +++ b/src/backend/Makefile -@@ -44,13 +44,13 @@ endif - +@@ -61,13 +61,13 @@ override LDFLAGS := $(LDFLAGS) $(LDFLAGS_EX) $(LDFLAGS_EX_BE) + all: submake-libpgport submake-catalog-headers submake-utils-headers postgres $(POSTGRES_IMP) - + -ifneq ($(PORTNAME), cygwin) +ifeq (,$(filter cygwin wasix-dl,$(PORTNAME))) ifneq ($(PORTNAME), win32) - + postgres: $(OBJS) $(CC) $(CFLAGS) $(call expand_subsys,$^) $(LDFLAGS) $(LIBS) -o $@ - + -endif +endif # win32 endif - + ifeq ($(PORTNAME), cygwin) -@@ -82,6 +82,22 @@ libpostgres.a: postgres - +@@ -95,6 +95,22 @@ libpostgres.a: postgres + endif # win32 - + +ifeq ($(PORTNAME), wasix-dl) +oliphaunt: $(OBJS) + $(AR) rcs libpgmain.a main/main.o @@ -94,11 +94,11 @@ index 4827ee361d..9292f77031 100644 +endif # wasix-dl + $(top_builddir)/src/port/libpgport_srv.a: | submake-libpgport - - + + diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h new file mode 100644 -index 0000000000..630687ec15 +index 0000000000..24eeecbf98 --- /dev/null +++ b/src/include/port/wasix-dl.h @@ -0,0 +1,8 @@ @@ -112,7 +112,7 @@ index 0000000000..630687ec15 +#endif diff --git a/src/template/wasix-dl b/src/template/wasix-dl new file mode 100644 -index 0000000000..f3792d3e83 +index 0000000000..d8dce3e11e --- /dev/null +++ b/src/template/wasix-dl @@ -0,0 +1,14 @@ diff --git a/src/wasix/runtime/assets/build/postgres/patches/0002-oliphaunt-wasix-add-backend-host-io-hooks.patch b/src/wasix/runtime/assets/build/postgres/patches/0002-oliphaunt-wasix-add-backend-host-io-hooks.patch index fb2ae9b53..d62c36a6e 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0002-oliphaunt-wasix-add-backend-host-io-hooks.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0002-oliphaunt-wasix-add-backend-host-io-hooks.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000002 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: add backend host I/O hooks @@ -19,13 +19,13 @@ lifecycle, authentication, or protocol-pump behavior. 3 files changed, 29 insertions(+) diff --git a/src/backend/libpq/be-secure.c b/src/backend/libpq/be-secure.c -index 2c9dfc28f0..d069849c9d 100644 +index d723e74e81..40eae70ab5 100644 --- a/src/backend/libpq/be-secure.c +++ b/src/backend/libpq/be-secure.c -@@ -288,6 +288,11 @@ secure_raw_read(Port *port, void *ptr, size_t len) +@@ -282,6 +282,11 @@ secure_raw_read(Port *port, void *ptr, size_t len) return len; } - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER + if (port->oliphaunt_wasix_io != NULL && port->oliphaunt_wasix_io->read != NULL) + return port->oliphaunt_wasix_io->read(port->oliphaunt_wasix_io->context, ptr, len); @@ -34,10 +34,10 @@ index 2c9dfc28f0..d069849c9d 100644 /* * Try to read from the socket without blocking. If it succeeds we're * done, otherwise we'll wait for the socket using the latch mechanism. -@@ -381,6 +386,11 @@ secure_raw_write(Port *port, const void *ptr, size_t len) +@@ -378,6 +383,11 @@ secure_raw_write(Port *port, const void *ptr, size_t len) { ssize_t n; - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER + if (port->oliphaunt_wasix_io != NULL && port->oliphaunt_wasix_io->write != NULL) + return port->oliphaunt_wasix_io->write(port->oliphaunt_wasix_io->context, ptr, len); @@ -47,7 +47,7 @@ index 2c9dfc28f0..d069849c9d 100644 pgwin32_noblock = true; #endif diff --git a/src/backend/libpq/pqcomm.c b/src/backend/libpq/pqcomm.c -index 6f7b4f62b6..742d1856e5 100644 +index e5171467de..673bcd9966 100644 --- a/src/backend/libpq/pqcomm.c +++ b/src/backend/libpq/pqcomm.c @@ -309,6 +309,14 @@ pq_init(ClientSocket *client_sock) @@ -64,15 +64,15 @@ index 6f7b4f62b6..742d1856e5 100644 +#endif AddWaitEventToSet(FeBeWaitSet, WL_POSTMASTER_DEATH, PGINVALID_SOCKET, NULL, NULL); - + diff --git a/src/include/libpq/libpq-be.h b/src/include/libpq/libpq-be.h -index 6aa52a64ff..6068a0cb8f 100644 +index d6e671a638..9925538d1a 100644 --- a/src/include/libpq/libpq-be.h +++ b/src/include/libpq/libpq-be.h -@@ -99,6 +99,15 @@ typedef struct ClientConnectionInfo +@@ -105,6 +105,15 @@ typedef struct ClientConnectionInfo UserAuth auth_method; } ClientConnectionInfo; - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER +typedef struct OliphauntWasmHostIO +{ @@ -85,7 +85,7 @@ index 6aa52a64ff..6068a0cb8f 100644 /* * The Port structure holds state information about a client connection in a * backend process. It is available in the global variable MyProcPort. The -@@ -244,6 +253,9 @@ typedef struct Port +@@ -238,6 +247,9 @@ typedef struct Port char *raw_buf; ssize_t raw_buf_consumed, raw_buf_remaining; @@ -93,7 +93,7 @@ index 6aa52a64ff..6068a0cb8f 100644 + OliphauntWasmHostIO *oliphaunt_wasix_io; +#endif } Port; - + /* -- 2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0003-oliphaunt-wasix-export-startup-packet-parser.patch b/src/wasix/runtime/assets/build/postgres/patches/0003-oliphaunt-wasix-export-startup-packet-parser.patch index d4d502df7..195961ffe 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0003-oliphaunt-wasix-export-startup-packet-parser.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0003-oliphaunt-wasix-export-startup-packet-parser.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000003 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: export startup packet parser @@ -19,13 +19,13 @@ it with the ABI name expected by the current host. 1 file changed, 18 insertions(+) diff --git a/src/backend/tcop/backend_startup.c b/src/backend/tcop/backend_startup.c -index dd3307cb76..f7b053e3b8 100644 +index c89fe12825..9acecd5e8c 100644 --- a/src/backend/tcop/backend_startup.c +++ b/src/backend/tcop/backend_startup.c -@@ -44,9 +44,19 @@ char *log_connections_string = NULL; +@@ -57,9 +57,19 @@ char *log_connections_string = NULL; */ ConnectionTiming conn_timing = {.ready_for_use = TIMESTAMP_MINUS_INFINITY}; - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER +#define OLIPHAUNT_WASM_HOST_EXPORT(name) __attribute__((export_name(name))) +#else @@ -42,7 +42,7 @@ index dd3307cb76..f7b053e3b8 100644 static void ProcessCancelRequestPacket(Port *port, void *pkt, int pktlen); static void SendNegotiateProtocolVersion(List *unrecognized_protocol_options); static void process_startup_packet_die(SIGNAL_ARGS); -@@ -489,7 +499,11 @@ reject: +@@ -488,7 +498,11 @@ reject: * should make no assumption here about the order in which the client may make * requests. */ diff --git a/src/wasix/runtime/assets/build/postgres/patches/0004-oliphaunt-wasix-add-host-lifecycle-exports.patch b/src/wasix/runtime/assets/build/postgres/patches/0004-oliphaunt-wasix-add-host-lifecycle-exports.patch index 60a992f8d..b3d385464 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0004-oliphaunt-wasix-add-host-lifecycle-exports.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0004-oliphaunt-wasix-add-host-lifecycle-exports.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000004 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: add host lifecycle exports @@ -12,20 +12,27 @@ The host ABI added here only attaches a direct FE/BE Port after that backend state exists, then sends the normal AuthenticationOk, ParameterStatus, BackendKeyData, and ReadyForQuery messages through libpq. +Buffered bridge output has a sticky first-failure status: zero means no failed +buffered write, otherwise the value is the first positive errno and only the +bridge output reset clears it. The flush export samples that status after +`pq_flush()`, so a send failure consumed by PostgreSQL still prevents the host +from accepting a partial response. Once set, the bridge rejects protocol +writes in every transport mode until that explicit reset. + The exported names intentionally keep the current `oliphaunt_wasix_*` spelling so the Rust release lane can be brought up before a host-side ABI rename. The code path is guarded by `OLIPHAUNT_WASM_SINGLE_USER`. --- - src/backend/tcop/postgres.c | 120 ++++++++++++++++++++++++++++++++++++++ + src/backend/tcop/postgres.c | 123 ++++++++++++++++++++++++++++++++++++++ src/backend/utils/init/postinit.c | 9 +++ - src/include/port/wasix-dl.h | 10 ++++ - 3 files changed, 139 insertions(+) + src/include/port/wasix-dl.h | 12 ++++ + 3 files changed, 144 insertions(+) diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index 39e4e3216d..674dd23641 100644 +index 115a634c64..6e74a7f43c 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -24,6 +24,9 @@ +@@ -25,6 +25,9 @@ #include #include #include @@ -33,9 +40,9 @@ index 39e4e3216d..674dd23641 100644 +#include +#endif #include - + #ifdef USE_VALGRIND -@@ -40,13 +43,20 @@ +@@ -41,6 +44,10 @@ #include "common/pg_prng.h" #include "jit/jit.h" #include "libpq/libpq.h" @@ -46,7 +53,7 @@ index 39e4e3216d..674dd23641 100644 #include "libpq/pqformat.h" #include "libpq/pqsignal.h" #include "mb/pg_wchar.h" - #include "mb/stringinfo_mb.h" +@@ -48,6 +55,9 @@ #include "miscadmin.h" #include "nodes/print.h" #include "optimizer/optimizer.h" @@ -56,10 +63,10 @@ index 39e4e3216d..674dd23641 100644 #include "parser/analyze.h" #include "parser/parser.h" #include "pg_getopt.h" -@@ -158,6 +168,111 @@ static volatile sig_atomic_t RecoveryConflictPendingReasons[NUM_PROCSIGNALS]; +@@ -163,6 +173,114 @@ static volatile sig_atomic_t RecoveryConflictPendingReasons[NUM_PROCSIGNALS]; static MemoryContext row_description_context = NULL; static StringInfoData row_description_buf; - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER +#define OLIPHAUNT_WASM_EXIT_ALIVE 99 +#define OLIPHAUNT_WASM_HOST_EXPORT(name) __attribute__((export_name(name))) @@ -111,10 +118,13 @@ index 39e4e3216d..674dd23641 100644 + MyBackendType = B_BACKEND; +} + -+OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_pq_flush") void ++OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_pq_flush") int +oliphaunt_wasix_pq_flush(void) +{ -+ pq_flush(); ++ int status = pq_flush(); ++ int bridge_status = oliphaunt_wasix_output_status(); ++ ++ return status != 0 ? status : bridge_status; +} + +OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_get_proc_port") struct Port * @@ -168,7 +178,7 @@ index 39e4e3216d..674dd23641 100644 /* ---------------------------------------------------------------- * decls for routines only used in this file * ---------------------------------------------------------------- -@@ -5109,6 +5220,11 @@ PostgresMain(const char *dbname, const char *username) +@@ -5003,6 +5121,11 @@ PostgresMain(const char *dbname, const char *username) * it will fail to be called during other backend-shutdown * scenarios. */ @@ -178,16 +188,16 @@ index 39e4e3216d..674dd23641 100644 +#endif + proc_exit(0); - + case PqMsg_CopyData: diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c -index 47925d6848..bcce4363fe 100644 +index c86ceefda9..53e65b73ec 100644 --- a/src/backend/utils/init/postinit.c +++ b/src/backend/utils/init/postinit.c -@@ -1290,6 +1290,15 @@ process_startup_options(Port *port, bool am_superuser) +@@ -1299,6 +1299,15 @@ process_startup_options(Port *port, bool am_superuser) } } - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER +void +oliphaunt_wasix_process_startup_options(Port *port) @@ -201,12 +211,10 @@ index 47925d6848..bcce4363fe 100644 * Load GUC settings from pg_db_role_setting. * diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index 630687ec15..dfae28b4cc 100644 +index 24eeecbf98..512c0a2ee6 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h -@@ -3,6 +3,17 @@ - * Port-specific declarations for the Oliphaunt WASIX side-module build. - *------------------------------------------------------------------------- +@@ -5,4 +5,16 @@ */ #ifndef PG_PORT_WASIX_DL_H #define PG_PORT_WASIX_DL_H @@ -218,6 +226,7 @@ index 630687ec15..dfae28b4cc 100644 + +extern ssize_t oliphaunt_wasix_host_read(void *context, void *ptr, size_t len); +extern ssize_t oliphaunt_wasix_host_write(void *context, const void *ptr, size_t len); ++extern int oliphaunt_wasix_output_status(void); +extern void oliphaunt_wasix_process_startup_options(struct Port *port); +#endif + diff --git a/src/wasix/runtime/assets/build/postgres/patches/0005-oliphaunt-wasix-add-loop-pumped-protocol-exports.patch b/src/wasix/runtime/assets/build/postgres/patches/0005-oliphaunt-wasix-add-loop-pumped-protocol-exports.patch index cadbadc9b..58b549875 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0005-oliphaunt-wasix-add-loop-pumped-protocol-exports.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0005-oliphaunt-wasix-add-loop-pumped-protocol-exports.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000005 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: add loop-pumped protocol exports @@ -7,19 +7,20 @@ Expose PostgreSQL main-loop control points for the embedded WASIX host. PostgreSQL still owns the normal backend initialization path and the normal per-message FE/BE protocol logic. This patch factors the existing loop body -and top-level error recovery into helper functions, exporting them only for -OLIPHAUNT_WASM_SINGLE_USER builds so the Rust host can pump one frontend -message at a time and recover a backend ERROR without starting another -backend lifecycle. +and top-level error recovery into helper functions. The loop-step export is +available only to OLIPHAUNT_WASM_SINGLE_USER builds and returns a small typed +outcome: processed, recovered ERROR, or input ended. PostgreSQL performs its +own recovery while the call-site exception boundary is live; the host never +has to infer guest control flow from a trap string or process-exit code. diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index b2e2753..da82a52 100644 +index 7e9c232e10..fa6e9cfaf7 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -153,6 +153,16 @@ static bool DoingCommandRead = false; +@@ -153,6 +153,24 @@ static bool DoingCommandRead = false; static bool doing_extended_query_message = false; static bool ignore_till_sync = false; - + +/* + * These are local to PostgresMain() upstream. Oliphaunt WASIX exposes the + * main loop to the host, so the loop and recovery exports need the state at @@ -29,14 +30,22 @@ index b2e2753..da82a52 100644 +static volatile bool send_ready_for_query = true; +static volatile bool idle_in_transaction_timeout_enabled = false; +static volatile bool idle_session_timeout_enabled = false; ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++enum ++{ ++ OLIPHAUNT_WASM_MAIN_LOOP_PROCESSED = 0, ++ OLIPHAUNT_WASM_MAIN_LOOP_RECOVERED = 1, ++ OLIPHAUNT_WASM_MAIN_LOOP_INPUT_ENDED = 2, ++}; ++#endif + /* * If an unnamed prepared statement exists, it's stored here. * We keep it separate from the hashtable kept by commands/prepare.c -@@ -4284,861 +4294,899 @@ PostgresSingleUserMain(int argc, char *argv[], +@@ -4291,861 +4309,937 @@ PostgresSingleUserMain(int argc, char *argv[], } - - + + -/* ---------------------------------------------------------------- - * PostgresMain - * postgres main loop -- all backends, interactive or otherwise loop here @@ -108,7 +117,7 @@ index b2e2753..da82a52 100644 + { + set_ps_display("idle in transaction (aborted)"); + pgstat_report_activity(STATE_IDLEINTRANSACTION_ABORTED, NULL); - + - /* - * In a postmaster child backend, replace SignalHandlerForCrashExit - * with quickdie, so we can tell the client we're dying. @@ -147,7 +156,7 @@ index b2e2753..da82a52 100644 - InitializeTimeouts(); /* establishes SIGALRM handler */ + { + long stats_timeout; - + - /* - * Ignore failure to write to frontend. Note: if frontend closes - * connection, we will notice it and exit cleanly when control next @@ -167,7 +176,7 @@ index b2e2753..da82a52 100644 + */ + if (notifyInterruptPending) + ProcessNotifyInterrupt(false); - + - /* - * Reset some signals that are accepted by postmaster but not by - * backend @@ -201,12 +210,12 @@ index b2e2753..da82a52 100644 + if (get_timeout_active(IDLE_STATS_UPDATE_TIMEOUT)) + disable_timeout(IDLE_STATS_UPDATE_TIMEOUT, false); + } - + - /* Early initialization */ - BaseInit(); + set_ps_display("idle"); + pgstat_report_activity(STATE_IDLE, NULL); - + - /* We need to allow SIGINT, etc during the initial transaction */ - sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); + /* Start the idle-session timer */ @@ -217,7 +226,7 @@ index b2e2753..da82a52 100644 + IdleSessionTimeout); + } + } - + - /* - * Generate a random cancel key, if this is a backend serving a - * connection. InitPostgres() will advertise it in shared memory. @@ -228,7 +237,7 @@ index b2e2753..da82a52 100644 - int len; + /* Report any recently-changed GUC options */ + ReportChangedGUCOptions(); - + - len = (MyProcPort == NULL || MyProcPort->proto >= PG_PROTOCOL(3, 2)) - ? MAX_CANCEL_KEY_LENGTH : 4; - if (!pg_strong_random(&MyCancelKey, len)) @@ -272,12 +281,8 @@ index b2e2753..da82a52 100644 + send_ready_for_query = false; } +} - -+#ifdef OLIPHAUNT_WASM_SINGLE_USER -+OLIPHAUNT_WASM_HOST_EXPORT("PostgresMainLongJmp") void -+#else + +static void -+#endif +PostgresMainLongJmp(void) +{ /* @@ -304,7 +309,7 @@ index b2e2753..da82a52 100644 + + /* Prevent interrupts while cleaning up */ + HOLD_INTERRUPTS(); - + /* - * If the PostmasterContext is still around, recycle the space; we don't - * need it anymore after InitPostgres completes. @@ -327,7 +332,7 @@ index b2e2753..da82a52 100644 + QueryCancelPending = false; + idle_in_transaction_timeout_enabled = false; + idle_session_timeout_enabled = false; - + - SetProcessingMode(NormalProcessing); + /* Not reading from the client anymore. */ + DoingCommandRead = false; @@ -337,7 +342,7 @@ index b2e2753..da82a52 100644 + + /* Report the error to the client and/or server log */ + EmitErrorReport(); - + /* - * Now all GUC states are fully set up. Report them to client if - * appropriate. @@ -346,7 +351,7 @@ index b2e2753..da82a52 100644 */ - BeginReportingGUCOptions(); + valgrind_report_error_query(debug_query_string); - + /* - * Also set up handler to log session end; we have to wait till now to be - * sure Log_disconnections has its final value. @@ -356,20 +361,20 @@ index b2e2753..da82a52 100644 - if (IsUnderPostmaster && Log_disconnections) - on_proc_exit(log_disconnections, 0); + debug_query_string = NULL; - + - pgstat_report_connect(MyDatabaseId); + /* + * Abort the current transaction in order to recover. + */ + AbortCurrentTransaction(); - + - /* Perform initialization specific to a WAL sender process. */ if (am_walsender) - InitWalSender(); + WalSndErrorCleanup(); + + PortalErrorCleanup(); - + /* - * Send this backend's cancellation info to the frontend. + * We can't release replication slots inside AbortTransaction() as we @@ -387,19 +392,19 @@ index b2e2753..da82a52 100644 - pq_sendint32(&buf, (int32) MyProcPid); + if (MyReplicationSlot != NULL) + ReplicationSlotRelease(); - + - pq_sendbytes(&buf, MyCancelKey, MyCancelKeyLength); - pq_endmessage(&buf); - /* Need not flush since ReadyForQuery will do it. */ - } + /* We also want to cleanup temporary slots on error. */ + ReplicationSlotCleanup(false); - + - /* Welcome banner for standalone case */ - if (whereToSendOutput == DestDebug) - printf("\nPostgreSQL stand-alone backend %s\n", PG_VERSION); + jit_reset_after_error(); - + /* - * Create the memory context we will use in the main loop. - * @@ -413,7 +418,7 @@ index b2e2753..da82a52 100644 - ALLOCSET_DEFAULT_SIZES); + MemoryContextSwitchTo(MessageContext); + FlushErrorState(); - + /* - * Create memory context and buffer used for RowDescription messages. As - * SendRowDescriptionMessage(), via exec_describe_statement_message(), is @@ -431,12 +436,12 @@ index b2e2753..da82a52 100644 - MemoryContextSwitchTo(TopMemoryContext); + if (doing_extended_query_message) + ignore_till_sync = true; - + - /* Fire any defined login event triggers, if appropriate */ - EventTriggerOnLogin(); + /* We don't have a transaction command open anymore */ + xact_started = false; - + /* - * POSTGRES main processing loop begins here - * @@ -468,7 +473,7 @@ index b2e2753..da82a52 100644 + ereport(FATAL, + (errcode(ERRCODE_PROTOCOL_VIOLATION), + errmsg("terminating connection because protocol synchronization was lost"))); - + - if (sigsetjmp(local_sigjmp_buf, 1) != 0) - { - /* @@ -481,11 +486,11 @@ index b2e2753..da82a52 100644 + /* Now we can allow interrupts again */ + RESUME_INTERRUPTS(); +} - + - /* Since not using PG_TRY, must reset error stack by hand */ - error_context_stack = NULL; +#ifdef OLIPHAUNT_WASM_SINGLE_USER -+OLIPHAUNT_WASM_HOST_EXPORT("PostgresMainLoopOnce") void ++OLIPHAUNT_WASM_HOST_EXPORT("PostgresMainLoopOnce") int +#else +static void +#endif @@ -493,7 +498,21 @@ index b2e2753..da82a52 100644 +{ + int firstchar; + StringInfoData input_message; - ++ ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ /* ++ * Every host call needs a live call-site boundary. Never retain this ++ * frame in PG_exception_stack after returning to the host. ++ */ ++ if (sigsetjmp(postgresmain_sigjmp_buf, 1) != 0) ++ { ++ PostgresMainLongJmp(); ++ PG_exception_stack = NULL; ++ return OLIPHAUNT_WASM_MAIN_LOOP_RECOVERED; ++ } ++ PG_exception_stack = &postgresmain_sigjmp_buf; ++#endif + - /* Prevent interrupts while cleaning up */ - HOLD_INTERRUPTS(); + /* @@ -501,7 +520,7 @@ index b2e2753..da82a52 100644 + * errors encountered in "idle" state don't provoke skip. + */ + doing_extended_query_message = false; - + - /* - * Forget any pending QueryCancel request, since we're returning to - * the idle loop anyway, and cancel any active timeout requests. (In @@ -523,7 +542,7 @@ index b2e2753..da82a52 100644 +#ifdef USE_VALGRIND + old_valgrind_error_count = VALGRIND_COUNT_ERRORS; +#endif - + - /* Not reading from the client anymore. */ - DoingCommandRead = false; + /* @@ -532,11 +551,11 @@ index b2e2753..da82a52 100644 + */ + MemoryContextSwitchTo(MessageContext); + MemoryContextReset(MessageContext); - + - /* Make sure libpq is in a good state */ - pq_comm_reset(); + initStringInfo(&input_message); - + - /* Report the error to the client and/or server log */ - EmitErrorReport(); + /* @@ -544,14 +563,14 @@ index b2e2753..da82a52 100644 + * not preventing advance of global xmin while we wait for the client. + */ + InvalidateCatalogSnapshotConditionally(); - + - /* - * If Valgrind noticed something during the erroneous query, print the - * query string, assuming we have one. - */ - valgrind_report_error_query(debug_query_string); + PostgresSendReadyForQueryIfNecessary(); - + - /* - * Make sure debug_query_string gets reset before we possibly clobber - * the storage it points at. @@ -564,7 +583,7 @@ index b2e2753..da82a52 100644 + * STDIN doing the same thing.) + */ + DoingCommandRead = true; - + - /* - * Abort the current transaction in order to recover. - */ @@ -624,7 +643,7 @@ index b2e2753..da82a52 100644 + * (3) read a command (loop blocks here) + */ + firstchar = ReadCommand(&input_message); - + - /* Now we can allow interrupts again */ - RESUME_INTERRUPTS(); + /* @@ -651,7 +670,7 @@ index b2e2753..da82a52 100644 - - if (!ignore_till_sync) - send_ready_for_query = true; /* initially, or after error */ - + /* - * Non-error queries loop here. + * (5) disable async signal conditions again. @@ -664,7 +683,7 @@ index b2e2753..da82a52 100644 */ + CHECK_FOR_INTERRUPTS(); + DoingCommandRead = false; - + - for (;;) + /* + * (6) check for any other interesting events that happened while we @@ -690,7 +709,7 @@ index b2e2753..da82a52 100644 + ConfigReloadPending = false; + ProcessConfigFile(PGC_SIGHUP); + } - + - /* - * Release storage left over from prior query cycle, and create a new - * query input buffer in the cleared MessageContext. @@ -702,15 +721,22 @@ index b2e2753..da82a52 100644 + * Sync. + */ + if (ignore_till_sync && firstchar != EOF) ++ { ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ PG_exception_stack = NULL; ++ return OLIPHAUNT_WASM_MAIN_LOOP_PROCESSED; ++#else + return; - ++#endif ++ } + - initStringInfo(&input_message); + switch (firstchar) + { + case PqMsg_Query: + { + const char *query_string; - + - /* - * Also consider releasing our catalog snapshot if any, so that it's - * not preventing advance of global xmin while we wait for the client. @@ -718,7 +744,7 @@ index b2e2753..da82a52 100644 - InvalidateCatalogSnapshotConditionally(); + /* Set statement_timestamp() */ + SetCurrentStatementStartTimestamp(); - + - /* - * (1) If we've reached idle state, tell the frontend we're ready for - * a new query. @@ -743,7 +769,7 @@ index b2e2753..da82a52 100644 - pgstat_report_activity(STATE_IDLEINTRANSACTION_ABORTED, NULL); + query_string = pq_getmsgstring(&input_message); + pq_getmsgend(&input_message); - + - /* Start the idle-in-transaction timer */ - if (IdleInTransactionSessionTimeout > 0 - && (IdleInTransactionSessionTimeout < TransactionTimeout || TransactionTimeout == 0)) @@ -762,7 +788,7 @@ index b2e2753..da82a52 100644 - pgstat_report_activity(STATE_IDLEINTRANSACTION, NULL); + else + exec_simple_query(query_string); - + - /* Start the idle-in-transaction timer */ - if (IdleInTransactionSessionTimeout > 0 - && (IdleInTransactionSessionTimeout < TransactionTimeout || TransactionTimeout == 0)) @@ -785,7 +811,7 @@ index b2e2753..da82a52 100644 + const char *query_string; + int numParams; + Oid *paramTypes = NULL; - + - /* - * Process incoming notifies (including self-notifies), if - * any, and send relevant messages to the client. Doing it @@ -796,7 +822,7 @@ index b2e2753..da82a52 100644 - if (notifyInterruptPending) - ProcessNotifyInterrupt(false); + forbidden_in_wal_sender(firstchar); - + - /* - * Check if we need to report stats. If pgstat_report_stat() - * decides it's too soon to flush out pending stats / lock @@ -834,12 +860,12 @@ index b2e2753..da82a52 100644 + paramTypes[i] = pq_getmsgint(&input_message, 4); } + pq_getmsgend(&input_message); - + - set_ps_display("idle"); - pgstat_report_activity(STATE_IDLE, NULL); + exec_parse_message(query_string, stmt_name, + paramTypes, numParams); - + - /* Start the idle-session timer */ - if (IdleSessionTimeout > 0) - { @@ -850,7 +876,7 @@ index b2e2753..da82a52 100644 + valgrind_report_error_query(query_string); } + break; - + - /* Report any recently-changed GUC options */ - ReportChangedGUCOptions(); + case PqMsg_Bind: @@ -858,7 +884,7 @@ index b2e2753..da82a52 100644 + + /* Set statement_timestamp() */ + SetCurrentStatementStartTimestamp(); - + /* - * The first time this backend is ready for query, log the - * durations of the different components of connection @@ -900,12 +926,12 @@ index b2e2753..da82a52 100644 - } + const char *portal_name; + int max_rows; - + - ReadyForQuery(whereToSendOutput); - send_ready_for_query = false; - } + forbidden_in_wal_sender(firstchar); - + - /* - * (2) Allow asynchronous signals to be executed immediately if they - * come in while we are waiting for client input. (This must be @@ -915,7 +941,7 @@ index b2e2753..da82a52 100644 - DoingCommandRead = true; + /* Set statement_timestamp() */ + SetCurrentStatementStartTimestamp(); - + - /* - * (3) read a command (loop blocks here) - */ @@ -923,7 +949,7 @@ index b2e2753..da82a52 100644 + portal_name = pq_getmsgstring(&input_message); + max_rows = pq_getmsgint(&input_message, 4); + pq_getmsgend(&input_message); - + - /* - * (4) turn off the idle-in-transaction and idle-session timeouts if - * active. We do this before step (5) so that any last-moment timeout @@ -943,7 +969,7 @@ index b2e2753..da82a52 100644 - idle_session_timeout_enabled = false; - } + exec_execute_message(portal_name, max_rows); - + - /* - * (5) disable async signal conditions again. - * @@ -958,7 +984,7 @@ index b2e2753..da82a52 100644 + /* exec_execute_message does valgrind_report_error_query */ + } + break; - + - /* - * (6) check for any other interesting events that happened while we - * slept. @@ -970,7 +996,7 @@ index b2e2753..da82a52 100644 - } + case PqMsg_FunctionCall: + forbidden_in_wal_sender(firstchar); - + - /* - * (7) process the command. But ignore it if we're skipping till - * Sync. @@ -979,7 +1005,7 @@ index b2e2753..da82a52 100644 - continue; + /* Set statement_timestamp() */ + SetCurrentStatementStartTimestamp(); - + - switch (firstchar) - { - case PqMsg_Query: @@ -988,12 +1014,12 @@ index b2e2753..da82a52 100644 + /* Report query to various monitoring facilities. */ + pgstat_report_activity(STATE_FASTPATH, NULL); + set_ps_display(""); - + - /* Set statement_timestamp() */ - SetCurrentStatementStartTimestamp(); + /* start an xact for this function invocation */ + start_xact_command(); - + - query_string = pq_getmsgstring(&input_message); - pq_getmsgend(&input_message); + /* @@ -1004,7 +1030,7 @@ index b2e2753..da82a52 100644 + * Be careful not to do anything that assumes we're inside a + * valid transaction here. + */ - + - if (am_walsender) - { - if (!exec_replication_command(query_string)) @@ -1014,16 +1040,16 @@ index b2e2753..da82a52 100644 - exec_simple_query(query_string); + /* switch back to message context */ + MemoryContextSwitchTo(MessageContext); - + - valgrind_report_error_query(query_string); + HandleFunctionRequest(&input_message); - + - send_ready_for_query = true; - } - break; + /* commit the function-invocation transaction */ + finish_xact_command(); - + - case PqMsg_Parse: - { - const char *stmt_name; @@ -1031,18 +1057,18 @@ index b2e2753..da82a52 100644 - int numParams; - Oid *paramTypes = NULL; + valgrind_report_error_query("fastpath function call"); - + - forbidden_in_wal_sender(firstchar); + send_ready_for_query = true; + break; - + - /* Set statement_timestamp() */ - SetCurrentStatementStartTimestamp(); + case PqMsg_Close: + { + int close_type; + const char *close_target; - + - stmt_name = pq_getmsgstring(&input_message); - query_string = pq_getmsgstring(&input_message); - numParams = pq_getmsgint(&input_message, 2); @@ -1054,7 +1080,7 @@ index b2e2753..da82a52 100644 - } - pq_getmsgend(&input_message); + forbidden_in_wal_sender(firstchar); - + - exec_parse_message(query_string, stmt_name, - paramTypes, numParams); + close_type = pq_getmsgbyte(&input_message); @@ -1075,7 +1101,7 @@ index b2e2753..da82a52 100644 + case 'P': + { + Portal portal; - + - valgrind_report_error_query(query_string); + portal = GetPortalByName(close_target); + if (PortalIsValid(portal)) @@ -1090,7 +1116,7 @@ index b2e2753..da82a52 100644 + break; } - break; - + - case PqMsg_Bind: + if (whereToSendOutput == DestRemote) + pq_putemptymessage(PqMsg_CloseComplete); @@ -1105,11 +1131,11 @@ index b2e2753..da82a52 100644 + const char *describe_target; + forbidden_in_wal_sender(firstchar); - + - /* Set statement_timestamp() */ + /* Set statement_timestamp() (needed for xact) */ SetCurrentStatementStartTimestamp(); - + - /* - * this message is complex enough that it seems best to put - * the field extraction out-of-line @@ -1121,7 +1147,7 @@ index b2e2753..da82a52 100644 + describe_type = pq_getmsgbyte(&input_message); + describe_target = pq_getmsgstring(&input_message); + pq_getmsgend(&input_message); - + - case PqMsg_Execute: + switch (describe_type) { @@ -1140,12 +1166,12 @@ index b2e2753..da82a52 100644 + describe_type))); + break; + } - + - forbidden_in_wal_sender(firstchar); + valgrind_report_error_query("DESCRIBE message"); + } + break; - + - /* Set statement_timestamp() */ - SetCurrentStatementStartTimestamp(); + case PqMsg_Flush: @@ -1153,13 +1179,13 @@ index b2e2753..da82a52 100644 + if (whereToSendOutput == DestRemote) + pq_flush(); + break; - + - portal_name = pq_getmsgstring(&input_message); - max_rows = pq_getmsgint(&input_message, 4); - pq_getmsgend(&input_message); + case PqMsg_Sync: + pq_getmsgend(&input_message); - + - exec_execute_message(portal_name, max_rows); + /* + * If pipelining was used, we may be in an implicit @@ -1171,7 +1197,7 @@ index b2e2753..da82a52 100644 + valgrind_report_error_query("SYNC message"); + send_ready_for_query = true; + break; - + - /* exec_execute_message does valgrind_report_error_query */ - } - break; @@ -1181,21 +1207,21 @@ index b2e2753..da82a52 100644 + * Either way, perform normal shutdown. + */ + case EOF: - + - case PqMsg_FunctionCall: - forbidden_in_wal_sender(firstchar); + /* for the cumulative statistics system */ + pgStatSessionEndCause = DISCONNECT_CLIENT_EOF; - + - /* Set statement_timestamp() */ - SetCurrentStatementStartTimestamp(); + /* FALLTHROUGH */ - + - /* Report query to various monitoring facilities. */ - pgstat_report_activity(STATE_FASTPATH, NULL); - set_ps_display(""); + case PqMsg_Terminate: - + - /* start an xact for this function invocation */ - start_xact_command(); + /* @@ -1204,7 +1230,7 @@ index b2e2753..da82a52 100644 + */ + if (whereToSendOutput == DestRemote) + whereToSendOutput = DestNone; - + - /* - * Note: we may at this point be inside an aborted - * transaction. We can't throw error for that until we've @@ -1222,18 +1248,21 @@ index b2e2753..da82a52 100644 + */ +#ifdef OLIPHAUNT_WASM_SINGLE_USER + if (is_oliphaunt_active != 0) -+ exit(OLIPHAUNT_WASM_EXIT_ALIVE); ++ { ++ PG_exception_stack = NULL; ++ return OLIPHAUNT_WASM_MAIN_LOOP_INPUT_ENDED; ++ } +#endif - + - /* switch back to message context */ - MemoryContextSwitchTo(MessageContext); + proc_exit(0); - + - HandleFunctionRequest(&input_message); + case PqMsg_CopyData: + case PqMsg_CopyDone: + case PqMsg_CopyFail: - + - /* commit the function-invocation transaction */ - finish_xact_command(); + /* @@ -1242,7 +1271,7 @@ index b2e2753..da82a52 100644 + * is still sending data. + */ + break; - + - valgrind_report_error_query("fastpath function call"); + default: + ereport(FATAL, @@ -1250,11 +1279,15 @@ index b2e2753..da82a52 100644 + errmsg("invalid frontend message type %d", + firstchar))); + } ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ PG_exception_stack = NULL; ++ return OLIPHAUNT_WASM_MAIN_LOOP_PROCESSED; ++#endif +} - send_ready_for_query = true; - break; - + - case PqMsg_Close: - { - int close_type; @@ -1276,18 +1309,18 @@ index b2e2753..da82a52 100644 +#ifndef OLIPHAUNT_WASM_SINGLE_USER + sigjmp_buf local_sigjmp_buf; +#endif - + - forbidden_in_wal_sender(firstchar); + Assert(dbname != NULL); + Assert(username != NULL); - + - close_type = pq_getmsgbyte(&input_message); - close_target = pq_getmsgstring(&input_message); - pq_getmsgend(&input_message); + send_ready_for_query = true; + idle_in_transaction_timeout_enabled = false; + idle_session_timeout_enabled = false; - + - switch (close_type) - { - case 'S': @@ -1316,7 +1349,7 @@ index b2e2753..da82a52 100644 - break; - } + Assert(GetProcessingMode() == InitProcessing); - + - if (whereToSendOutput == DestRemote) - pq_putemptymessage(PqMsg_CloseComplete); + /* @@ -1341,7 +1374,7 @@ index b2e2753..da82a52 100644 + pqsignal(SIGHUP, SignalHandlerForConfigReload); + pqsignal(SIGINT, StatementCancelHandler); /* cancel current query */ + pqsignal(SIGTERM, die); /* cancel current query and exit */ - + - valgrind_report_error_query("CLOSE message"); - } - break; @@ -1358,7 +1391,7 @@ index b2e2753..da82a52 100644 + else + pqsignal(SIGQUIT, die); /* cancel current query and exit */ + InitializeTimeouts(); /* establishes SIGALRM handler */ - + - case PqMsg_Describe: - { - int describe_type; @@ -1373,7 +1406,7 @@ index b2e2753..da82a52 100644 + pqsignal(SIGUSR1, procsignal_sigusr1_handler); + pqsignal(SIGUSR2, SIG_IGN); + pqsignal(SIGFPE, FloatExceptionHandler); - + - forbidden_in_wal_sender(firstchar); + /* + * Reset some signals that are accepted by postmaster but not by @@ -1382,18 +1415,18 @@ index b2e2753..da82a52 100644 + pqsignal(SIGCHLD, SIG_DFL); /* system() requires this on some + * platforms */ + } - + - /* Set statement_timestamp() (needed for xact) */ - SetCurrentStatementStartTimestamp(); + /* Early initialization */ + BaseInit(); - + - describe_type = pq_getmsgbyte(&input_message); - describe_target = pq_getmsgstring(&input_message); - pq_getmsgend(&input_message); + /* We need to allow SIGINT, etc during the initial transaction */ + sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); - + - switch (describe_type) - { - case 'S': @@ -1417,7 +1450,7 @@ index b2e2753..da82a52 100644 + if (whereToSendOutput == DestRemote) + { + int len; - + - valgrind_report_error_query("DESCRIBE message"); - } - break; @@ -1431,7 +1464,7 @@ index b2e2753..da82a52 100644 + } + MyCancelKeyLength = len; + } - + - case PqMsg_Flush: - pq_getmsgend(&input_message); - if (whereToSendOutput == DestRemote) @@ -1450,7 +1483,7 @@ index b2e2753..da82a52 100644 + username, InvalidOid, /* role to connect as */ + (!am_walsender) ? INIT_PG_LOAD_SESSION_LIBS : 0, + NULL); /* no out_dbname */ - + - case PqMsg_Sync: - pq_getmsgend(&input_message); + /* @@ -1462,7 +1495,7 @@ index b2e2753..da82a52 100644 + MemoryContextDelete(PostmasterContext); + PostmasterContext = NULL; + } - + - /* - * If pipelining was used, we may be in an implicit - * transaction block. Close it before calling @@ -1474,7 +1507,7 @@ index b2e2753..da82a52 100644 - send_ready_for_query = true; - break; + SetProcessingMode(NormalProcessing); - + - /* - * PqMsg_Terminate means that the frontend is closing down the - * socket. EOF means unexpected loss of frontend connection. @@ -1486,7 +1519,7 @@ index b2e2753..da82a52 100644 + * appropriate. + */ + BeginReportingGUCOptions(); - + - /* for the cumulative statistics system */ - pgStatSessionEndCause = DISCONNECT_CLIENT_EOF; + /* @@ -1495,15 +1528,15 @@ index b2e2753..da82a52 100644 + */ + if (IsUnderPostmaster && Log_disconnections) + on_proc_exit(log_disconnections, 0); - + - /* FALLTHROUGH */ + pgstat_report_connect(MyDatabaseId); - + - case PqMsg_Terminate: + /* Perform initialization specific to a WAL sender process. */ + if (am_walsender) + InitWalSender(); - + - /* - * Reset whereToSendOutput to prevent ereport from attempting - * to send any more messages to client. @@ -1577,7 +1610,7 @@ index b2e2753..da82a52 100644 + * unblock in AbortTransaction() because the latter is only called if we + * were inside a transaction. + */ - + - /* - * NOTE: if you are tempted to add more code here, DON'T! - * Whatever you had in mind to do should be set up as an @@ -1595,7 +1628,7 @@ index b2e2753..da82a52 100644 + { + PostgresMainLongJmp(); + } - + - proc_exit(0); + /* We can now handle ereport(ERROR) */ +#ifdef OLIPHAUNT_WASM_SINGLE_USER @@ -1603,13 +1636,13 @@ index b2e2753..da82a52 100644 +#else + PG_exception_stack = &local_sigjmp_buf; +#endif - + - case PqMsg_CopyData: - case PqMsg_CopyDone: - case PqMsg_CopyFail: + if (!ignore_till_sync) + send_ready_for_query = true; /* initially, or after error */ - + - /* - * Accept but ignore these messages, per protocol spec; we - * probably got here because a COPY failed, and the frontend @@ -1619,7 +1652,7 @@ index b2e2753..da82a52 100644 + /* + * Non-error queries loop here. + */ - + - default: - ereport(FATAL, - (errcode(ERRCODE_PROTOCOL_VIOLATION), @@ -1628,26 +1661,41 @@ index b2e2753..da82a52 100644 - } - } /* end of input-reading loop */ + for (;;) -+ PostgresMainLoopOnce(); /* end of input-reading loop */ ++ { ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ int loop_status = PostgresMainLoopOnce(); ++ ++ if (loop_status == OLIPHAUNT_WASM_MAIN_LOOP_INPUT_ENDED) ++ { ++ PG_exception_stack = NULL; ++ exit(OLIPHAUNT_WASM_EXIT_ALIVE); ++ } ++ Assert(loop_status == OLIPHAUNT_WASM_MAIN_LOOP_PROCESSED || ++ loop_status == OLIPHAUNT_WASM_MAIN_LOOP_RECOVERED); ++#else ++ PostgresMainLoopOnce(); ++#endif ++ } /* end of input-reading loop */ } - + /* diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index 569a589..b8d22ca 100644 +index 512c0a2ee6..e3a6243c93 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h -@@ -7,10 +7,12 @@ +@@ -7,11 +7,13 @@ #define PG_PORT_WASIX_DL_H - + #ifdef OLIPHAUNT_WASM_SINGLE_USER +#include #include - + struct Port; - + +extern sigjmp_buf postgresmain_sigjmp_buf; extern ssize_t oliphaunt_wasix_host_read(void *context, void *ptr, size_t len); extern ssize_t oliphaunt_wasix_host_write(void *context, const void *ptr, size_t len); + extern int oliphaunt_wasix_output_status(void); extern void oliphaunt_wasix_process_startup_options(struct Port *port); -- 2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0006-oliphaunt-wasix-report-copy-protocol-state.patch b/src/wasix/runtime/assets/build/postgres/patches/0006-oliphaunt-wasix-report-copy-protocol-state.patch index ef805760a..93e1bc885 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0006-oliphaunt-wasix-report-copy-protocol-state.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0006-oliphaunt-wasix-report-copy-protocol-state.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000006 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: report COPY protocol state @@ -22,10 +22,10 @@ wasix-dl port header. 4 files changed, 30 insertions(+) diff --git a/src/backend/commands/copyfromparse.c b/src/backend/commands/copyfromparse.c -index 97a4c387a3..0f0eb63c8c 100644 +index f5fc346e20..1f6770dd1c 100644 --- a/src/backend/commands/copyfromparse.c +++ b/src/backend/commands/copyfromparse.c -@@ -66,6 +66,9 @@ +@@ -69,6 +69,9 @@ #include "libpq/libpq.h" #include "libpq/pqformat.h" #include "mb/pg_wchar.h" @@ -38,7 +38,7 @@ index 97a4c387a3..0f0eb63c8c 100644 @@ -174,6 +177,9 @@ ReceiveCopyBegin(CopyFromState cstate) int16 format = (cstate->opts.binary ? 1 : 0); int i; - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_IN); +#endif @@ -46,10 +46,10 @@ index 97a4c387a3..0f0eb63c8c 100644 pq_sendbyte(&buf, format); /* overall format */ pq_sendint16(&buf, natts); diff --git a/src/backend/commands/copyto.c b/src/backend/commands/copyto.c -index 84dc465cba..705f42916f 100644 +index ea6f18f2c8..80545ecd48 100644 --- a/src/backend/commands/copyto.c +++ b/src/backend/commands/copyto.c -@@ -24,6 +24,9 @@ +@@ -27,6 +27,9 @@ #include "libpq/libpq.h" #include "libpq/pqformat.h" #include "mb/pg_wchar.h" @@ -59,10 +59,10 @@ index 84dc465cba..705f42916f 100644 #include "miscadmin.h" #include "pgstat.h" #include "storage/fd.h" -@@ -394,6 +397,9 @@ SendCopyBegin(CopyToState cstate) +@@ -395,6 +398,9 @@ SendCopyBegin(CopyToState cstate) int16 format = (cstate->opts.binary ? 1 : 0); int i; - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_OUT); +#endif @@ -70,10 +70,10 @@ index 84dc465cba..705f42916f 100644 pq_sendbyte(&buf, format); /* overall format */ pq_sendint16(&buf, natts); diff --git a/src/backend/replication/walsender.c b/src/backend/replication/walsender.c -index ff1c357870..a4608ae247 100644 +index 70d90699cc..5840e7238e 100644 --- a/src/backend/replication/walsender.c +++ b/src/backend/replication/walsender.c -@@ -61,6 +61,9 @@ +@@ -67,6 +67,9 @@ #include "libpq/pqformat.h" #include "miscadmin.h" #include "nodes/replnodes.h" @@ -83,9 +83,9 @@ index ff1c357870..a4608ae247 100644 #include "pgstat.h" #include "postmaster/interrupt.h" #include "replication/decode.h" -@@ -686,6 +689,9 @@ UploadManifest(void) +@@ -687,6 +690,9 @@ UploadManifest(void) ib = CreateIncrementalBackupInfo(mcxt); - + /* Send a CopyInResponse message */ +#ifdef OLIPHAUNT_WASM_SINGLE_USER + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_IN); @@ -95,7 +95,7 @@ index ff1c357870..a4608ae247 100644 pq_sendint16(&buf, 0); @@ -940,6 +946,9 @@ StartReplication(StartReplicationCmd *cmd) WalSndSetState(WALSNDSTATE_CATCHUP); - + /* Send a CopyBothResponse message, and start streaming */ +#ifdef OLIPHAUNT_WASM_SINGLE_USER + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_BOTH); @@ -105,7 +105,7 @@ index ff1c357870..a4608ae247 100644 pq_sendint16(&buf, 0); @@ -1487,6 +1496,9 @@ StartLogicalReplication(StartReplicationCmd *cmd) WalSndSetState(WALSNDSTATE_CATCHUP); - + /* Send a CopyBothResponse message, and start streaming */ +#ifdef OLIPHAUNT_WASM_SINGLE_USER + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_BOTH); @@ -114,25 +114,26 @@ index ff1c357870..a4608ae247 100644 pq_sendbyte(&buf, 0); pq_sendint16(&buf, 0); diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index 70e2debc5c..7754c1eef9 100644 +index e3a6243c93..f4fc290feb 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h -@@ -10,10 +10,16 @@ - +@@ -12,11 +12,17 @@ + struct Port; - -+#define OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE 0 -+#define OLIPHAUNT_WASIX_PROTOCOL_COPY_IN 1 -+#define OLIPHAUNT_WASIX_PROTOCOL_COPY_OUT 2 -+#define OLIPHAUNT_WASIX_PROTOCOL_COPY_BOTH 3 -+ + ++#include "port/wasix-dl/oliphaunt_wasix_protocol_contract.generated.h" ++/* ++ * Mode and COPY state numbers come only from the generated contract. ++ * Keep PostgreSQL call sites on that shared private ABI. ++ */ extern sigjmp_buf postgresmain_sigjmp_buf; extern ssize_t oliphaunt_wasix_host_read(void *context, void *ptr, size_t len); extern ssize_t oliphaunt_wasix_host_write(void *context, const void *ptr, size_t len); + extern int oliphaunt_wasix_output_status(void); extern void oliphaunt_wasix_process_startup_options(struct Port *port); +extern void oliphaunt_wasix_protocol_report_copy_response(int state); #endif - + #endif -- 2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0007-oliphaunt-wasix-add-wasix-pgxs-side-module-support.patch b/src/wasix/runtime/assets/build/postgres/patches/0007-oliphaunt-wasix-add-wasix-pgxs-side-module-support.patch index 9d93beb75..750e25c79 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0007-oliphaunt-wasix-add-wasix-pgxs-side-module-support.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0007-oliphaunt-wasix-add-wasix-pgxs-side-module-support.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: add wasix PGXS side-module support @@ -26,7 +26,7 @@ rather than inheriting Emscripten variable names. diff --git a/src/makefiles/Makefile.wasix-dl b/src/makefiles/Makefile.wasix-dl new file mode 100644 -index 0000000000..cb23b1ff25 +index 0000000000..17af45f9e8 --- /dev/null +++ b/src/makefiles/Makefile.wasix-dl @@ -0,0 +1,20 @@ @@ -51,10 +51,10 @@ index 0000000000..cb23b1ff25 +wasm_dl_extension_dir := $(wasm_dl_include_dir)/extension +wasm_dl_extension_imports_dir := $(wasm_dl_extension_dir)/imports diff --git a/src/makefiles/pgxs.mk b/src/makefiles/pgxs.mk -index b9cf0a3c3c..123f938adb 100644 +index 039cee3dfe..45819eb568 100644 --- a/src/makefiles/pgxs.mk +++ b/src/makefiles/pgxs.mk -@@ -251,6 +251,12 @@ ifeq ($(with_llvm), yes) +@@ -250,6 +250,12 @@ ifeq ($(with_llvm), yes) $(foreach mod, $(MODULES), $(call install_llvm_module,$(mod),$(mod).bc)) endif # with_llvm endif # MODULES @@ -67,7 +67,7 @@ index b9cf0a3c3c..123f938adb 100644 ifdef DOCS ifdef docdir $(INSTALL_DATA) $(addprefix $(srcdir)/, $(DOCS)) '$(DESTDIR)$(docdir)/$(docmoduledir)/' -@@ -276,6 +282,12 @@ ifdef MODULE_big +@@ -271,6 +277,12 @@ ifdef MODULE_big ifeq ($(with_llvm), yes) $(call install_llvm_module,$(MODULE_big),$(OBJS)) endif # with_llvm @@ -77,20 +77,20 @@ index b9cf0a3c3c..123f938adb 100644 + find . -name "*.o" -exec $(WASM_DL_NM) --undefined-only {} \; | awk '{print $$2}' | sed '/^$$/d' | sort -u > '$(MODULE_big).undef.txt'; find . -type f \( -name "*.o" -o -name "*$(DLSUFFIX)" \) -exec $(WASM_DL_NM) --defined-only {} \; | awk '$$2 ~ /^[TDB]$$/ {print $$3}' | sed '/^$$/d' | sort -u > '$(MODULE_big).defs.txt'; comm -23 '$(MODULE_big).undef.txt' '$(MODULE_big).defs.txt' > '$(DESTDIR)$(wasm_dl_extension_imports_dir)/$(MODULE_big).imports' +endif # MODULE_big +endif # PORTNAME=wasix-dl - + install: install-lib endif # MODULE_big -@@ -306,6 +318,9 @@ endif # DOCS +@@ -297,6 +309,9 @@ endif # DOCS ifneq (,$(PROGRAM)$(SCRIPTS)$(SCRIPTS_built)) $(MKDIR_P) '$(DESTDIR)$(bindir)' endif +ifeq ($(PORTNAME), wasix-dl) + $(MKDIR_P) '$(DESTDIR)$(wasm_dl_extension_imports_dir)' +endif # PORTNAME=wasix-dl - + ifdef MODULE_big installdirs: installdirs-lib -@@ -330,6 +345,9 @@ ifdef MODULES +@@ -318,6 +333,9 @@ ifdef MODULES ifeq ($(with_llvm), yes) $(foreach mod, $(MODULES), $(call uninstall_llvm_module,$(mod))) endif # with_llvm @@ -100,14 +100,14 @@ index b9cf0a3c3c..123f938adb 100644 endif # MODULES ifdef DOCS rm -f $(addprefix '$(DESTDIR)$(docdir)/$(docmoduledir)'/, $(DOCS)) -@@ -355,6 +374,9 @@ ifdef MODULE_big +@@ -339,6 +357,9 @@ ifdef MODULE_big ifeq ($(with_llvm), yes) $(call uninstall_llvm_module,$(MODULE_big)) endif # with_llvm +ifeq ($(PORTNAME), wasix-dl) + rm -f '$(DESTDIR)$(wasm_dl_extension_imports_dir)/$(MODULE_big).imports' +endif # PORTNAME=wasix-dl - + uninstall: uninstall-lib endif # MODULE_big -- diff --git a/src/wasix/runtime/assets/build/postgres/patches/0008-oliphaunt-wasix-reset-copy-state-on-error-recovery.patch b/src/wasix/runtime/assets/build/postgres/patches/0008-oliphaunt-wasix-reset-copy-state-on-error-recovery.patch index 05ffef76a..d4ef316fc 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0008-oliphaunt-wasix-reset-copy-state-on-error-recovery.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0008-oliphaunt-wasix-reset-copy-state-on-error-recovery.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: reset copy state on error recovery @@ -16,13 +16,13 @@ reports that its COPY state is no longer active. 1 file changed, 8 insertions(+) diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index 8017e5e97e..bb7f33f100 100644 +index fa6e9cfaf7..dc749b799e 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -4478,6 +4478,14 @@ PostgresMainLongJmp(void) +@@ -4494,6 +4494,14 @@ PostgresMainLongJmp(void) /* Make sure libpq is in a good state */ pq_comm_reset(); - + +#ifdef OLIPHAUNT_WASM_SINGLE_USER + /* + * A top-level ERROR aborts COPY. Keep the host-side protocol handoff @@ -33,6 +33,6 @@ index 8017e5e97e..bb7f33f100 100644 + /* Report the error to the client and/or server log */ EmitErrorReport(); - + -- 2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0009-oliphaunt-wasix-route-process-identity-through-port.patch b/src/wasix/runtime/assets/build/postgres/patches/0009-oliphaunt-wasix-route-process-identity-through-port.patch index 339d7e249..183fca656 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0009-oliphaunt-wasix-route-process-identity-through-port.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0009-oliphaunt-wasix-route-process-identity-through-port.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: route process identity through port @@ -16,7 +16,7 @@ stack while still preserving normal PostgreSQL call sites. 1 file changed, 15 insertions(+) diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index 6252eecbfd..6c1c7b895f 100644 +index def8f37c8b..31379d6302 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h @@ -8,6 +8,7 @@ @@ -27,9 +27,10 @@ index 6252eecbfd..6c1c7b895f 100644 #include struct Port; -@@ -21,7 +22,21 @@ extern sigjmp_buf postgresmain_sigjmp_buf; +@@ -21,8 +22,22 @@ extern sigjmp_buf postgresmain_sigjmp_buf; extern ssize_t oliphaunt_wasix_host_read(void *context, void *ptr, size_t len); extern ssize_t oliphaunt_wasix_host_write(void *context, const void *ptr, size_t len); + extern int oliphaunt_wasix_output_status(void); extern void oliphaunt_wasix_process_startup_options(struct Port *port); +extern uid_t oliphaunt_wasix_geteuid(void); +extern uid_t oliphaunt_wasix_getuid(void); diff --git a/src/wasix/runtime/assets/build/postgres/patches/0010-oliphaunt-wasix-route-sysv-shmem-through-port.patch b/src/wasix/runtime/assets/build/postgres/patches/0010-oliphaunt-wasix-route-sysv-shmem-through-port.patch index 20b0c3391..a305c4108 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0010-oliphaunt-wasix-route-sysv-shmem-through-port.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0010-oliphaunt-wasix-route-sysv-shmem-through-port.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: route sysv shmem through port @@ -20,7 +20,7 @@ port owns the mapping to the embedded runtime implementation. create mode 100644 src/include/port/wasix-dl/sys/shm.h diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index c37ccbb546..5cfa325b50 100644 +index 31379d6302..41e09bb3a9 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h @@ -10,6 +10,8 @@ @@ -32,7 +32,7 @@ index c37ccbb546..5cfa325b50 100644 struct Port; -@@ -29,6 +31,10 @@ extern gid_t oliphaunt_wasix_getgid(void); +@@ -30,6 +32,10 @@ extern gid_t oliphaunt_wasix_getgid(void); extern struct passwd *oliphaunt_wasix_getpwuid(uid_t uid); extern int oliphaunt_wasix_getpwuid_r(uid_t uid, struct passwd *pwd, char *buf, size_t buflen, struct passwd **result); @@ -43,7 +43,7 @@ index c37ccbb546..5cfa325b50 100644 extern void oliphaunt_wasix_protocol_report_copy_response(int state); #define geteuid oliphaunt_wasix_geteuid -@@ -37,6 +43,10 @@ extern void oliphaunt_wasix_protocol_report_copy_response(int state); +@@ -38,6 +44,10 @@ extern void oliphaunt_wasix_protocol_report_copy_response(int state); #define getgid oliphaunt_wasix_getgid #define getpwuid oliphaunt_wasix_getpwuid #define getpwuid_r oliphaunt_wasix_getpwuid_r diff --git a/src/wasix/runtime/assets/build/postgres/patches/0011-oliphaunt-wasix-prefer-posix-semaphores.patch b/src/wasix/runtime/assets/build/postgres/patches/0011-oliphaunt-wasix-prefer-posix-semaphores.patch index 104eb71d1..c5104c8eb 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0011-oliphaunt-wasix-prefer-posix-semaphores.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0011-oliphaunt-wasix-prefer-posix-semaphores.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: prefer POSIX semaphores diff --git a/src/wasix/runtime/assets/build/postgres/patches/0012-oliphaunt-wasix-capture-startup-errors.patch b/src/wasix/runtime/assets/build/postgres/patches/0012-oliphaunt-wasix-capture-startup-errors.patch index 5ab8b87cc..6822f9cbe 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0012-oliphaunt-wasix-capture-startup-errors.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0012-oliphaunt-wasix-capture-startup-errors.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: capture startup errors @@ -7,32 +7,49 @@ PostgreSQL can emit protocol ErrorResponse bytes from InitPostgres() before the exported main-loop recovery buffer is installed. In the embedded WASIX runtime that path still needs to be PostgreSQL-owned: database, role, and startup-option failures should reach the host as PostgreSQL ErrorResponse -frames instead of collapsing into a generic process exit. +frames instead of collapsing into an unknown process exit. Attach the host-backed protocol Port before InitPostgres() only while startup -error capture is active, route startup errors to DestRemote, and ask the bridge -to trap nonzero proc_exit() so the host can flush the captured protocol bytes. -Normal successful startup restores the previous output destination and leaves -the later oliphaunt_wasix_start()/oliphaunt_wasix_send_conn_data() lifecycle unchanged. +error capture is active and retain DestRemote for the duration. A narrowly +scoped before_shmem_exit callback flushes during orderly proc_exit while the +bridge-backed output remains available, then snapshots only complete buffered +protocol output containing an ErrorResponse. The bridge exposes those bytes through its +versioned startup-outcome descriptor before exiting with the dedicated status +98. Successful startup disables the callback and restores the previous output +destination; the later oliphaunt_wasix_start()/send lifecycle is unchanged. --- - src/backend/tcop/postgres.c | 35 +++++++++++++++++++++++++++++++++++ - src/include/port/wasix-dl.h | 1 + - 2 files changed, 36 insertions(+) + src/backend/tcop/postgres.c | 44 ++++++++++++++++++++++++++++++++++++++++++++ + src/include/port/wasix-dl.h | 4 ++++ + 2 files changed, 48 insertions(+) diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index b2e2753f7c..a835d2c6dd 100644 +index dc749b799e..7927470a0b 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -281,3 +281,23 @@ oliphaunt_wasix_send_conn_data(void) +@@ -295,6 +295,39 @@ oliphaunt_wasix_send_conn_data(void) + ReadyForQuery(DestRemote); } +static CommandDest oliphaunt_wasix_startup_error_saved_dest = DestDebug; + +static void ++oliphaunt_wasix_publish_startup_rejection(int code, Datum arg) ++{ ++ (void) code; ++ (void) arg; ++ ++ if (!oliphaunt_wasix_startup_error_capture_active) ++ return; ++ if (pq_flush() == 0) ++ (void) oliphaunt_wasix_startup_outcome_publish_rejected(); ++} ++ ++static void +oliphaunt_wasix_begin_startup_error_capture(void) +{ + if (MyProcPort == NULL) + oliphaunt_wasix_init_protocol_port(); ++ oliphaunt_wasix_startup_outcome_reset(); + oliphaunt_wasix_startup_error_saved_dest = whereToSendOutput; + oliphaunt_wasix_startup_error_capture_active = 1; + whereToSendOutput = DestRemote; @@ -47,18 +64,19 @@ index b2e2753f7c..a835d2c6dd 100644 +} + #else -@@ -5051,10 +5071,22 @@ PostgresMain(const char *dbname, const char *username) + #define OLIPHAUNT_WASM_HOST_EXPORT(name) + #endif +@@ -5105,10 +5138,21 @@ PostgresMain(const char *dbname, const char *username) * * Honor session_preload_libraries if not dealing with a WAL sender. */ +#ifdef OLIPHAUNT_WASM_SINGLE_USER + /* -+ * If InitPostgres() fails before the exported top-level recovery buffer is -+ * active, keep the failure on PostgreSQL's wire-protocol path. The bridge -+ * traps nonzero proc_exit() while this flag is set so the host can collect -+ * the ErrorResponse bytes already written to the host-backed Port. ++ * Flush and publish during proc_exit() while bridge-backed output remains ++ * available. Success leaves this callback registered but inactive. + */ + oliphaunt_wasix_begin_startup_error_capture(); ++ before_shmem_exit(oliphaunt_wasix_publish_startup_rejection, (Datum) 0); +#endif InitPostgres(dbname, InvalidOid, /* database to connect to */ username, InvalidOid, /* role to connect as */ @@ -67,17 +85,20 @@ index b2e2753f7c..a835d2c6dd 100644 +#ifdef OLIPHAUNT_WASM_SINGLE_USER + oliphaunt_wasix_end_startup_error_capture(); +#endif - + /* * If the PostmasterContext is still around, recycle the space; we don't diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index e0e6dbab31..3b934a7d0f 100644 +index 41e09bb3a9..ef5bf884eb 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h -@@ -23,6 +23,7 @@ extern sigjmp_buf postgresmain_sigjmp_buf; - extern ssize_t oliphaunt_wasix_host_read(void *context, void *ptr, size_t len); +@@ -25,6 +25,10 @@ extern ssize_t oliphaunt_wasix_host_read(void *context, void *ptr, size_t len); extern ssize_t oliphaunt_wasix_host_write(void *context, const void *ptr, size_t len); + extern int oliphaunt_wasix_output_status(void); extern void oliphaunt_wasix_process_startup_options(struct Port *port); ++extern const void *oliphaunt_wasix_startup_outcome_v1(void); ++extern void oliphaunt_wasix_startup_outcome_reset(void); ++extern int oliphaunt_wasix_startup_outcome_publish_rejected(void); +extern volatile int oliphaunt_wasix_startup_error_capture_active; extern uid_t oliphaunt_wasix_geteuid(void); extern uid_t oliphaunt_wasix_getuid(void); diff --git a/src/wasix/runtime/assets/build/postgres/patches/0013-oliphaunt-wasix-fail-active-portals-on-host-recovery.patch b/src/wasix/runtime/assets/build/postgres/patches/0013-oliphaunt-wasix-fail-active-portals-on-host-recovery.patch deleted file mode 100644 index 939b602f8..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0013-oliphaunt-wasix-fail-active-portals-on-host-recovery.patch +++ /dev/null @@ -1,60 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: fail active portals on host recovery - -Some WASIX hosts cannot preserve nested WebAssembly exception unwinds for every -PostgreSQL PG_TRY/PG_CATCH boundary. On those hosts, the bridge routes ERROR -longjmp through the exported top-level recovery path, which means an active -portal can reach transaction abort before its local PG_CATCH marks it failed. - -Keep that cleanup inside PostgreSQL: when the embedded WASIX backend is -active, treat active portals like the existing FATAL/shmem-exit path during -AtAbort_Portals(). Marking them failed prevents executor shutdown from running -against transaction-aborted state while preserving the normal portal abort and -drop lifecycle. ---- - src/backend/utils/mmgr/portalmem.c | 10 ++++++++++ - src/include/port/wasix-dl.h | 1 + - 2 files changed, 11 insertions(+) - -diff --git a/src/backend/utils/mmgr/portalmem.c b/src/backend/utils/mmgr/portalmem.c -index 6479012910..18e6155689 100644 ---- a/src/backend/utils/mmgr/portalmem.c -+++ b/src/backend/utils/mmgr/portalmem.c -@@ -25,6 +25,9 @@ - #include "utils/memutils.h" - #include "utils/snapmgr.h" - #include "utils/timestamp.h" -+#ifdef OLIPHAUNT_WASM_SINGLE_USER -+#include "port/wasix-dl.h" -+#endif - - /* - * Estimate of the maximum number of open portals a user would have, -@@ -795,6 +798,11 @@ AtAbort_Portals(void) - */ - if (portal->status == PORTAL_ACTIVE && shmem_exit_inprogress) - MarkPortalFailed(portal); -+#ifdef OLIPHAUNT_WASM_SINGLE_USER -+ else if (portal->status == PORTAL_ACTIVE && -+ is_oliphaunt_active != 0) -+ MarkPortalFailed(portal); -+#endif - - /* - * Do nothing else to cursors held over from a previous transaction. -diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index 3b934a7d0f..4e8bfb98b7 100644 ---- a/src/include/port/wasix-dl.h -+++ b/src/include/port/wasix-dl.h -@@ -21,6 +21,7 @@ struct Port; - #define OLIPHAUNT_WASIX_PROTOCOL_COPY_BOTH 3 - - extern sigjmp_buf postgresmain_sigjmp_buf; -+extern volatile int is_oliphaunt_active; - extern ssize_t oliphaunt_wasix_host_read(void *context, void *ptr, size_t len); - extern ssize_t oliphaunt_wasix_host_write(void *context, const void *ptr, size_t len); - extern void oliphaunt_wasix_process_startup_options(struct Port *port); --- -2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0017-oliphaunt-wasix-keep-btree-delete-scratch-on-stack.patch b/src/wasix/runtime/assets/build/postgres/patches/0017-oliphaunt-wasix-keep-btree-delete-scratch-on-stack.patch deleted file mode 100644 index c47147460..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0017-oliphaunt-wasix-keep-btree-delete-scratch-on-stack.patch +++ /dev/null @@ -1,111 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: keep btree delete scratch on stack - -PostgreSQL btree simple deletion and bottom-up deletion use two page-local -scratch arrays sized by MaxTIDsPerBTreePage. The arrays are bounded by BLCKSZ -and only live for a single deletion pass, but upstream allocates and frees both -arrays for each pass. - -Keep upstream allocation behavior everywhere except the embedded WASIX -embedded WASIX runtime. In that lane, place the scratch arrays on the C stack to -avoid allocator traffic during indexed-update churn. This does not change -which tuples are considered, which tableam callback runs, or when physical -deletion is attempted. ---- - src/backend/access/nbtree/nbtdedup.c | 20 ++++++++++++++++++++ - src/backend/access/nbtree/nbtinsert.c | 19 +++++++++++++++++++ - 2 files changed, 39 insertions(+) - -diff --git a/src/backend/access/nbtree/nbtdedup.c b/src/backend/access/nbtree/nbtdedup.c -index 53907b54a6..7b016cdb6e 100644 ---- a/src/backend/access/nbtree/nbtdedup.c -+++ b/src/backend/access/nbtree/nbtdedup.c -@@ -314,6 +314,10 @@ _bt_bottomupdel_pass(Relation rel, Buffer buf, Relation heapRel, - BTPageOpaque opaque = BTPageGetOpaque(page); - BTDedupState state; - TM_IndexDeleteOp delstate; -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) -+ TM_IndexDelete deltids[MaxTIDsPerBTreePage]; -+ TM_IndexStatus status[MaxTIDsPerBTreePage]; -+#endif - bool neverdedup; - int nkeyatts = IndexRelationGetNumberOfKeyAttributes(rel); - -@@ -355,8 +359,18 @@ _bt_bottomupdel_pass(Relation rel, Buffer buf, Relation heapRel, - delstate.bottomup = true; - delstate.bottomupfreespace = Max(BLCKSZ / 16, newitemsz); - delstate.ndeltids = 0; -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) -+ /* -+ * The delete state is page-local and bounded by BLCKSZ. Keep it off the -+ * allocator in the embedded WASIX backend, where these palloc/pfree pairs -+ * are visible during indexed-update churn. -+ */ -+ delstate.deltids = deltids; -+ delstate.status = status; -+#else - delstate.deltids = palloc(MaxTIDsPerBTreePage * sizeof(TM_IndexDelete)); - delstate.status = palloc(MaxTIDsPerBTreePage * sizeof(TM_IndexStatus)); -+#endif - - minoff = P_FIRSTDATAKEY(opaque); - maxoff = PageGetMaxOffsetNumber(page); -@@ -409,8 +423,10 @@ _bt_bottomupdel_pass(Relation rel, Buffer buf, Relation heapRel, - /* Ask tableam which TIDs are deletable, then physically delete them */ - _bt_delitems_delete_check(rel, buf, heapRel, &delstate); - -+#if !defined(__wasi__) || !defined(OLIPHAUNT_WASM_SINGLE_USER) - pfree(delstate.deltids); - pfree(delstate.status); -+#endif - - /* Report "success" to caller unconditionally to avoid deduplication */ - if (neverdedup) -diff --git a/src/backend/access/nbtree/nbtinsert.c b/src/backend/access/nbtree/nbtinsert.c -index 0da44e1779..c006987ca9 100644 ---- a/src/backend/access/nbtree/nbtinsert.c -+++ b/src/backend/access/nbtree/nbtinsert.c -@@ -2817,6 +2817,10 @@ _bt_simpledel_pass(Relation rel, Buffer buffer, Relation heapRel, - BlockNumber *deadblocks; - int ndeadblocks; - TM_IndexDeleteOp delstate; -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) -+ TM_IndexDelete deltids[MaxTIDsPerBTreePage]; -+ TM_IndexStatus status[MaxTIDsPerBTreePage]; -+#endif - OffsetNumber offnum; - - /* Get array of table blocks pointed to by LP_DEAD-set tuples */ -@@ -2829,8 +2833,17 @@ _bt_simpledel_pass(Relation rel, Buffer buffer, Relation heapRel, - delstate.bottomup = false; - delstate.bottomupfreespace = 0; - delstate.ndeltids = 0; -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) -+ /* -+ * The scratch arrays are page-size bounded and used for this deletion -+ * pass only. Avoid allocator traffic in the embedded WASIX backend. -+ */ -+ delstate.deltids = deltids; -+ delstate.status = status; -+#else - delstate.deltids = palloc(MaxTIDsPerBTreePage * sizeof(TM_IndexDelete)); - delstate.status = palloc(MaxTIDsPerBTreePage * sizeof(TM_IndexStatus)); -+#endif - - for (offnum = minoff; - offnum <= maxoff; -@@ -2911,8 +2924,10 @@ _bt_simpledel_pass(Relation rel, Buffer buffer, Relation heapRel, - /* Physically delete LP_DEAD tuples (plus any delete-safe extra TIDs) */ - _bt_delitems_delete_check(rel, buffer, heapRel, &delstate); - -+#if !defined(__wasi__) || !defined(OLIPHAUNT_WASM_SINGLE_USER) - pfree(delstate.deltids); - pfree(delstate.status); -+#endif - } - - /* --- -2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-host-recovery.patch b/src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-loop-step-recovery.patch similarity index 52% rename from src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-host-recovery.patch rename to src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-loop-step-recovery.patch index 4e9cfdcef..ee00725f9 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-host-recovery.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-loop-step-recovery.patch @@ -1,42 +1,43 @@ From 0000000000000000000000000000000000000019 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: schedule ready after host recovery +Subject: [PATCH] oliphaunt-wasix: schedule ready after loop-step recovery PostgreSQL's normal top-level ERROR recovery returns through the PostgresMain() sigsetjmp site. After PostgresMainLongJmp() performs cleanup, PostgresMain() re-arms the exception stack and sets send_ready_for_query when the backend is not in extended-protocol skip-till-Sync mode. -The embedded WASIX host can instead force ERROR recovery across the process -exit boundary and then invoke PostgresMainLongJmp() directly. Preserve the -same ReadyForQuery scheduling in that host-callable path, while still -withholding ReadyForQuery when extended-query recovery has set ignore_till_sync. +The embedded WASIX loop-step wrapper performs that same cleanup while its +per-call exception boundary is live, then returns a typed recovered outcome. +Preserve the same ReadyForQuery scheduling in the factored recovery helper, +while still withholding ReadyForQuery when extended-query recovery has set +ignore_till_sync. --- src/backend/tcop/postgres.c | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index 0dc4c9d1b3..b33221f794 100644 +index 7927470a0b..06f64145cc 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -4585,6 +4585,17 @@ PostgresMainLongJmp(void) - +@@ -4608,6 +4608,17 @@ PostgresMainLongJmp(void) + /* Now we can allow interrupts again */ RESUME_INTERRUPTS(); + +#ifdef OLIPHAUNT_WASM_SINGLE_USER + /* -+ * Host-forced ERROR recovery calls this export directly instead of -+ * returning through PostgresMain()'s sigsetjmp site. Preserve upstream's -+ * post-recovery ReadyForQuery scheduling for simple-query errors without -+ * breaking extended-protocol skip-till-Sync behavior. ++ * The loop-step wrapper performs top-level ERROR cleanup before returning ++ * its typed recovered outcome. Preserve upstream's post-recovery ++ * ReadyForQuery scheduling for simple-query errors without breaking ++ * extended-protocol skip-till-Sync behavior. + */ + if (!ignore_till_sync) + send_ready_for_query = true; +#endif } - + #ifdef OLIPHAUNT_WASM_SINGLE_USER -- 2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0020-oliphaunt-wasix-rearm-exception-stack-after-host-recovery.patch b/src/wasix/runtime/assets/build/postgres/patches/0020-oliphaunt-wasix-rearm-exception-stack-after-host-recovery.patch deleted file mode 100644 index 7515533e0..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0020-oliphaunt-wasix-rearm-exception-stack-after-host-recovery.patch +++ /dev/null @@ -1,43 +0,0 @@ -From 0000000000000000000000000000000000000020 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Thu, 28 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: rearm exception stack after host recovery - -PostgresMain() normally reassigns PG_exception_stack after top-level ERROR -cleanup returns through the sigsetjmp site. The WASIX host-forced recovery -path invokes PostgresMainLongJmp() directly, so it must restore the same -top-level exception boundary before the host pumps another frontend message. - -Keep this paired with the direct-call ReadyForQuery scheduling and only under -OLIPHAUNT_WASM_SINGLE_USER. Non-WASIX builds and the normal sigsetjmp return -path keep PostgreSQL's upstream assignment. ---- - src/backend/tcop/postgres.c | 10 ++++++++++ - 1 file changed, 10 insertions(+) - -diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c -index b33221f794..d87ef8ba37 100644 ---- a/src/backend/tcop/postgres.c -+++ b/src/backend/tcop/postgres.c -@@ -4588,11 +4588,19 @@ PostgresMainLongJmp(void) - - #ifdef OLIPHAUNT_WASM_SINGLE_USER - /* -+ * The normal path re-arms the top-level exception stack immediately after -+ * returning through PostgresMain()'s sigsetjmp site. Host-forced recovery -+ * calls this function directly, so it has to restore the same boundary -+ * before the next frontend message can report another ERROR. -+ */ -+ PG_exception_stack = &postgresmain_sigjmp_buf; -+ -+ /* - * Host-forced ERROR recovery calls this export directly instead of - * returning through PostgresMain()'s sigsetjmp site. Preserve upstream's - * post-recovery ReadyForQuery scheduling for simple-query errors without - * breaking extended-protocol skip-till-Sync behavior. - */ - if (!ignore_till_sync) - send_ready_for_query = true; - #endif --- -2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0021-oliphaunt-wasix-declare-wasix-fork.patch b/src/wasix/runtime/assets/build/postgres/patches/0021-oliphaunt-wasix-declare-wasix-fork.patch index ceb7886ca..346b2369d 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0021-oliphaunt-wasix-declare-wasix-fork.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0021-oliphaunt-wasix-declare-wasix-fork.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: stub fork_process in embedded WASIX runtime diff --git a/src/wasix/runtime/assets/build/postgres/patches/0022-oliphaunt-wasix-use-wasm-ld-for-backend-core.patch b/src/wasix/runtime/assets/build/postgres/patches/0022-oliphaunt-wasix-use-wasm-ld-for-backend-core.patch index f9d8d45f3..268e2aba6 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0022-oliphaunt-wasix-use-wasm-ld-for-backend-core.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0022-oliphaunt-wasix-use-wasm-ld-for-backend-core.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: use wasm-ld for backend core @@ -8,24 +8,35 @@ assembling the backend core archive. Leave the final dynamic-main link to the product build helper so its generated side-module export closure stays the one authoritative policy instead of wasixcc's unconditional export-all default. -This partial link lowers the backend's ThinLTO bitcode, so enable Wasm EH and -SJLJ here as well as at the final link. Otherwise the backend's inlined -sigsetjmp sites retain an unsupported setjmp import. Limit inlining during -this link to avoid multiplying SJLJ catch-state spills across large functions; -keep the normal frontend and final-link optimization settings. +Release builds feed ThinLTO bitcode into this raw relocatable link, making it +the backend core's WebAssembly code-generation boundary. Mirror wasixcc's +complete exception and SJLJ lowering flags here. In particular, +--exception-model=wasm is what emits the live try_table/catch and +__wasm_setjmp_test path; the other SJLJ flags alone can rewrite setjmp to helper +calls without installing the handler. Once the relocatable link has emitted a +native Wasm object, the sealed final link cannot repair a missed lowering. +Keep main's zero inline threshold at this partial link to avoid multiplying +SJLJ catch-state spills; frontend and final-link optimization are unchanged. --- - src/backend/Makefile | 17 ++++++++++++----- - 1 file changed, 12 insertions(+), 5 deletions(-) + src/backend/Makefile | 23 ++++++++++++++++++----- + 1 file changed, 18 insertions(+), 5 deletions(-) diff --git a/src/backend/Makefile b/src/backend/Makefile -index 9292f77031..89c356d090 100644 +index 9292f77031..9c46c0d33f 100644 --- a/src/backend/Makefile +++ b/src/backend/Makefile -@@ -96,12 +96,19 @@ libpostgres.a: postgres +@@ -96,12 +96,25 @@ libpostgres.a: postgres endif # win32 ifeq ($(PORTNAME), wasix-dl) +WASM_LD ?= $(shell $(CC) -print-prog-name=wasm-ld) ++# This relocatable ThinLTO link is a code-generation boundary. Keep all four ++# flags synchronized with wasixcc's final WebAssembly EH/SJLJ link. ++WASM_LTO_SJLJ_FLAGS = \ ++ -mllvm --wasm-enable-eh \ ++ -mllvm --wasm-enable-sjlj \ ++ -mllvm --wasm-use-legacy-eh=false \ ++ -mllvm --exception-model=wasm +LIBPGCORE ?= $(top_builddir)/libpgcore.a +LIBPG = $(top_builddir)/libpostgres.a +PGCORE = $(top_builddir)/src/common/libpgcommon_srv.a $(top_builddir)/src/port/libpgport_srv.a $(LIBPG) @@ -38,10 +49,9 @@ index 9292f77031..89c356d090 100644 - $(LD) -r -o libpgcore.o $(filter-out main/main.o,$(call expand_subsys,$^)) - $(AR) rcs libpgcore.a libpgcore.o - $(CC) $(CFLAGS) main/main.o libpgcore.a $(LDFLAGS) $(LIBS) -Wl,--no-entry -Wl,--export-dynamic -Wl,--export=_start -nostartfiles -o $@ -+ rm -f $(top_builddir)/libpgmain.a $(LIBPG) $(top_builddir)/libpgcore.o $(LIBPGCORE) + $(AR) rcs $(top_builddir)/libpgmain.a $(PGMAIN) + $(AR) rcs $(LIBPG) $(PGBACKEND) -+ $(WASM_LD) --relocatable -mllvm -inline-threshold=0 -mllvm --wasm-enable-eh -mllvm --wasm-enable-sjlj -mllvm --wasm-use-legacy-eh=false -mllvm --exception-model=wasm -o $(top_builddir)/libpgcore.o --whole-archive $(PGCORE) --no-whole-archive ++ $(WASM_LD) --relocatable -mllvm -inline-threshold=0 $(WASM_LTO_SJLJ_FLAGS) -o $(top_builddir)/libpgcore.o --whole-archive $(PGCORE) --no-whole-archive + $(AR) rcs $(LIBPGCORE) $(top_builddir)/libpgcore.o postgres: oliphaunt diff --git a/src/wasix/runtime/assets/build/postgres/patches/0023-oliphaunt-wasix-skip-data-dir-ownership-check-under-embedded-wasix.patch b/src/wasix/runtime/assets/build/postgres/patches/0023-oliphaunt-wasix-skip-data-dir-ownership-check-under-embedded-wasix.patch index ef3621a28..de3b7f6e7 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0023-oliphaunt-wasix-skip-data-dir-ownership-check-under-embedded-wasix.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0023-oliphaunt-wasix-skip-data-dir-ownership-check-under-embedded-wasix.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: skip data-dir ownership check under embedded WASIX diff --git a/src/wasix/runtime/assets/build/postgres/patches/0025-oliphaunt-wasix-stub-pg-dump-parallel-fork.patch b/src/wasix/runtime/assets/build/postgres/patches/0025-oliphaunt-wasix-stub-pg-dump-parallel-fork.patch index bb25690ad..d8330dc13 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0025-oliphaunt-wasix-stub-pg-dump-parallel-fork.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0025-oliphaunt-wasix-stub-pg-dump-parallel-fork.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000025 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Thu, 28 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: stub pg_dump parallel fork @@ -14,12 +14,13 @@ existing pg_dump error path. 1 file changed, 10 insertions(+) diff --git a/src/bin/pg_dump/parallel.c b/src/bin/pg_dump/parallel.c +index 086adcdc50..793142efd5 100644 --- a/src/bin/pg_dump/parallel.c +++ b/src/bin/pg_dump/parallel.c -@@ -61,6 +61,16 @@ +@@ -60,6 +60,16 @@ #include #endif - + +#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) && !defined(WIN32) +static pid_t +oliphaunt_wasix_pgdump_fork(void) diff --git a/src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-avoid-xlog-size-checkpoint-requests.patch b/src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-defer-xlog-size-checkpoint-requests.patch similarity index 65% rename from src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-avoid-xlog-size-checkpoint-requests.patch rename to src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-defer-xlog-size-checkpoint-requests.patch index f3fae3b20..651f5094d 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-avoid-xlog-size-checkpoint-requests.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-defer-xlog-size-checkpoint-requests.patch @@ -1,25 +1,29 @@ From 0000000000000000000000000000000000000027 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Fri, 29 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: avoid XLog-size checkpoint requests +Subject: [PATCH] oliphaunt-wasix: defer XLog-size checkpoint requests The embedded backend has no checkpointer process, and running a checkpoint inside XLogWrite is unsafe. Record the normal XLog-size request there and let the host-pumped traffic cop perform it at the next ReadyForQuery boundary when the backend is outside a transaction. This keeps PostgreSQL's WAL recycling policy without recursive checkpoints or an absent background process. + +The embedded topology keeps IsPostmasterEnvironment false, so upstream +RequestCheckpoint already performs the requested checkpoint locally. Do not +override that topology-sensitive branch here. --- - src/backend/access/transam/xlog.c | 29 +++++++++++++++++++++++++++-- - src/backend/postmaster/checkpointer.c | 2 ++ - src/backend/tcop/postgres.c | 6 ++++++ - src/include/access/xlog.h | 4 ++++ - 4 files changed, 39 insertions(+), 2 deletions(-) + src/backend/access/transam/xlog.c | 29 +++++++++++++++++++++++++++-- + src/backend/tcop/postgres.c | 6 ++++++ + src/include/access/xlog.h | 4 ++++ + 3 files changed, 37 insertions(+), 2 deletions(-) diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c +index 47c04f26d4..6b4a99ebd6 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c -@@ -164,6 +164,10 @@ static WalLevel wal_level = WAL_LEVEL_MINIMAL; - /* See check_wal_consistency_checking */ +@@ -166,6 +166,10 @@ static double PrevCheckPointDistance = 0; + */ static bool check_wal_consistency_checking_deferred = false; +#ifdef OLIPHAUNT_WASM_SINGLE_USER @@ -27,9 +31,9 @@ diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog +#endif + /* - * During recovery, we keep a copy of the latest checkpoint record here. - * lastCheckPointRecPtr points to start of checkpoint record and -@@ -2300,6 +2304,17 @@ XLogCheckpointNeeded(XLogSegNo new_segno) + * GUC support + */ +@@ -2288,6 +2292,17 @@ XLogCheckpointNeeded(XLogSegNo new_segno) return false; } @@ -47,7 +51,7 @@ diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog /* * Write and/or fsync the log at least as far as WriteRqst indicates. * -@@ -2501,15 +2516,24 @@ XLogWrite(XLogwrtRqst WriteRqst, TimeLineID tli, bool flexible) +@@ -2496,12 +2511,21 @@ XLogWrite(XLogwrtRqst WriteRqst, TimeLineID tli, bool flexible) * like a checkpoint is needed, forcibly update RedoRecPtr and * recheck. */ @@ -69,26 +73,11 @@ diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog } } - if (ispartialpage) - { - /* Only asked to write a partial page */ -diff --git a/src/backend/postmaster/checkpointer.c b/src/backend/postmaster/checkpointer.c ---- a/src/backend/postmaster/checkpointer.c -+++ b/src/backend/postmaster/checkpointer.c -@@ -1009,7 +1009,9 @@ RequestCheckpoint(int flags) - /* - * If in a standalone backend, just do it ourselves. - */ -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - if (!IsPostmasterEnvironment) -+#endif - { - /* - * There's no point in doing slow checkpoints in a standalone backend, diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c +index 06f64145cc..dc8bf8dfa6 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c -@@ -31,6 +31,9 @@ +@@ -37,6 +37,9 @@ #include "access/parallel.h" #include "access/printtup.h" #include "access/xact.h" @@ -98,7 +87,7 @@ diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c #include "catalog/pg_type.h" #include "commands/async.h" #include "commands/event_trigger.h" -@@ -4470,6 +4473,9 @@ PostgresSendReadyForQueryIfNecessary(void) +@@ -4398,6 +4401,9 @@ PostgresSendReadyForQueryIfNecessary(void) else { long stats_timeout; @@ -109,10 +98,10 @@ diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c /* * Process incoming notifies (including self-notifies), if diff --git a/src/include/access/xlog.h b/src/include/access/xlog.h +index f20f5edb43..b588eae759 100644 --- a/src/include/access/xlog.h +++ b/src/include/access/xlog.h -@@ -242,7 +242,11 @@ extern void LocalProcessControlFile(bool reset); - extern WalLevel GetActiveWalLevelOnStandby(void); +@@ -244,6 +244,10 @@ extern WalLevel GetActiveWalLevelOnStandby(void); extern void StartupXLOG(void); extern void ShutdownXLOG(int code, Datum arg); extern bool CreateCheckPoint(int flags); diff --git a/src/wasix/runtime/assets/build/postgres/patches/0028-oliphaunt-wasix-use-lightweight-embedded-runtime-paths.patch b/src/wasix/runtime/assets/build/postgres/patches/0028-oliphaunt-wasix-use-lightweight-embedded-runtime-paths.patch deleted file mode 100644 index d80b2db8b..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0028-oliphaunt-wasix-use-lightweight-embedded-runtime-paths.patch +++ /dev/null @@ -1,63 +0,0 @@ -From 0000000000000000000000000000000000000028 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Fri, 29 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: use lightweight embedded runtime paths - -Oliphaunt WASIX has one embedded backend and reports startup parameters -directly to the host. Use the cheaper local semaphore reset path and avoid -retaining duplicate GUC report strings after ParameterStatus is emitted. ---- - src/backend/port/posix_sema.c | 5 +++++ - src/backend/utils/misc/guc.c | 2 ++ - 2 files changed, 7 insertions(+) - -diff --git a/src/backend/port/posix_sema.c b/src/backend/port/posix_sema.c ---- a/src/backend/port/posix_sema.c -+++ b/src/backend/port/posix_sema.c -@@ -293,20 +293,25 @@ void - PGSemaphoreReset(PGSemaphore sema) - { -+#ifdef OLIPHAUNT_WASM_SINGLE_USER -+ sem_trywait(PG_SEM_REF(sema)); -+ return; -+#else - /* - * There's no direct API for this in POSIX, so we have to ratchet the - * semaphore down to 0 with repeated trywait's. - */ - for (;;) - { - if (sem_trywait(PG_SEM_REF(sema)) < 0) - { - if (errno == EAGAIN || errno == EDEADLK) - break; /* got it down to 0 */ - if (errno == EINTR) - continue; /* can this happen? */ - elog(FATAL, "sem_trywait failed: %m"); - } - } -+#endif - } - - /* -diff --git a/src/backend/utils/misc/guc.c b/src/backend/utils/misc/guc.c ---- a/src/backend/utils/misc/guc.c -+++ b/src/backend/utils/misc/guc.c -@@ -2645,13 +2645,15 @@ ReportGUCOption(struct config_generic *record) - pq_sendstring(&msgbuf, val); - pq_endmessage(&msgbuf); - -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - /* - * We need a long-lifespan copy. If guc_strdup() fails due to OOM, - * we'll set last_reported to NULL and thereby possibly make a - * duplicate report later. - */ - guc_free(record->last_reported); - record->last_reported = guc_strdup(LOG, val); -+#endif - } - - pfree(val); --- -2.39.5 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-model-trusted-embedded-session.patch b/src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-model-trusted-embedded-session.patch new file mode 100644 index 000000000..076551da5 --- /dev/null +++ b/src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-model-trusted-embedded-session.patch @@ -0,0 +1,486 @@ +From 0000000000000000000000000000000000000029 Mon Sep 17 00:00:00 2001 +From: Sid Jain +Date: Wed, 26 Aug 2026 00:00:00 +0000 +Subject: [PATCH] oliphaunt-wasix: model trusted embedded session + +Keep IsPostmasterEnvironment and IsUnderPostmaster truthful for the +single-process WASIX backend. A host-only preparation export selects a +private one-way UNSELECTED -> PREPARED -> ATTACHED lifecycle. Preparation is +inactive, and protocol attach consumes it exactly once after _start. + +Before shared-memory sizing, the guest pins synchronous I/O and zero worker, +parallel, and WAL-sender settings. Live GUC changes cannot restore parallel +worker counts. Validation of settings stored for a future process remains +available. Worker planning and signaling continue to fail closed through +upstream standalone gates, while async Append planning is explicitly disabled. + +Use one positive normal-user-session capability only for DDL event triggers, +text-search option validation, catalog-backed superuser checks, normal OID +allocation, and XID/MultiXact wraparound stops. Autovacuum signals remain +postmaster-only. Reject reload and promotion before direct PID signaling, and +make log rotation report false. + +The old late flag assignment never enabled login triggers, startup settings, +or authentication, and this patch deliberately does not add them. Statistics +never depended on those flags. Native builds inline the new predicate to the +original IsUnderPostmaster checks. Embedded branches are limited to cold DDL, +catalog-cache-miss, wraparound, and plan-construction paths, so the change is +performance-neutral for query execution. +--- + src/backend/access/transam/multixact.c | 5 +- + src/backend/access/transam/varsup.c | 6 +- + src/backend/access/transam/xlogfuncs.c | 5 ++ + src/backend/commands/event_trigger.c | 8 +- + src/backend/commands/tsearchcmds.c | 2 +- + src/backend/main/main.c | 99 +++++++++++++++++++++++++ + src/backend/optimizer/plan/createplan.c | 3 +- + src/backend/storage/ipc/signalfuncs.c | 12 +++ + src/backend/tcop/postgres.c | 9 +++ + src/backend/utils/misc/guc_tables.c | 35 +++++++++- + src/backend/utils/misc/superuser.c | 2 +- + src/include/miscadmin.h | 23 ++++++ + 12 files changed, 194 insertions(+), 15 deletions(-) + +diff --git a/src/backend/access/transam/multixact.c b/src/backend/access/transam/multixact.c +index da2a174..6d83697 100644 +--- a/src/backend/access/transam/multixact.c ++++ b/src/backend/access/transam/multixact.c +@@ -1248,7 +1248,7 @@ GetNewMultiXactId(int nmembers, MultiXactOffset *offset) + + LWLockRelease(MultiXactGenLock); + +- if (IsUnderPostmaster && ++ if (IsNormalUserSession() && + !MultiXactIdPrecedes(result, multiStopLimit)) + { + char *oldest_datname = get_database_name(oldest_datoid); +@@ -1257,7 +1257,8 @@ GetNewMultiXactId(int nmembers, MultiXactOffset *offset) + * Immediately kick autovacuum into action as we're already in + * ERROR territory. + */ +- SendPostmasterSignal(PMSIGNAL_START_AUTOVAC_LAUNCHER); ++ if (IsUnderPostmaster) ++ SendPostmasterSignal(PMSIGNAL_START_AUTOVAC_LAUNCHER); + + /* complain even if that DB has disappeared */ + if (oldest_datname) +diff --git a/src/backend/access/transam/varsup.c b/src/backend/access/transam/varsup.c +index fe89578..1cb34ad 100644 +--- a/src/backend/access/transam/varsup.c ++++ b/src/backend/access/transam/varsup.c +@@ -144,7 +144,7 @@ GetNewTransactionId(bool isSubXact) + if (IsUnderPostmaster && (xid % 65536) == 0) + SendPostmasterSignal(PMSIGNAL_START_AUTOVAC_LAUNCHER); + +- if (IsUnderPostmaster && ++ if (IsNormalUserSession() && + TransactionIdFollowsOrEquals(xid, xidStopLimit)) + { + char *oldest_datname = get_database_name(oldest_datoid); +@@ -578,7 +578,7 @@ GetNewObjectId(void) + */ + if (TransamVariables->nextOid < ((Oid) FirstNormalObjectId)) + { +- if (IsPostmasterEnvironment) ++ if (IsNormalUserSession()) + { + /* wraparound, or first post-initdb assignment, in normal mode */ + TransamVariables->nextOid = FirstNormalObjectId; +@@ -623,7 +623,7 @@ static void + SetNextObjectId(Oid nextOid) + { + /* Safety check, this is only allowable during initdb */ +- if (IsPostmasterEnvironment) ++ if (IsNormalUserSession()) + elog(ERROR, "cannot advance OID counter anymore"); + + /* Taking the lock is, therefore, just pro forma; but do it anyway */ +diff --git a/src/backend/access/transam/xlogfuncs.c b/src/backend/access/transam/xlogfuncs.c +index 8c30901..95bc950 100644 +--- a/src/backend/access/transam/xlogfuncs.c ++++ b/src/backend/access/transam/xlogfuncs.c +@@ -674,6 +674,11 @@ pg_promote(PG_FUNCTION_ARGS) + FILE *promote_file; + int i; + ++ if (IsTrustedEmbeddedSession()) ++ ereport(ERROR, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("standby promotion is not supported in a trusted embedded session"))); ++ + if (!RecoveryInProgress()) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), +diff --git a/src/backend/commands/event_trigger.c b/src/backend/commands/event_trigger.c +index 074c476..13e5f1e 100644 +--- a/src/backend/commands/event_trigger.c ++++ b/src/backend/commands/event_trigger.c +@@ -746,7 +746,7 @@ EventTriggerDDLCommandStart(Node *parsetree) + * Additionally, event triggers can be disabled with a superuser-only GUC + * to make fixing database easier as per 1 above. + */ +- if (!IsUnderPostmaster || !event_triggers) ++ if (!IsNormalUserSession() || !event_triggers) + return; + + runlist = EventTriggerCommonSetup(parsetree, +@@ -782,7 +782,7 @@ EventTriggerDDLCommandEnd(Node *parsetree) + * See EventTriggerDDLCommandStart for a discussion about why event + * triggers are disabled in single user mode or via GUC. + */ +- if (!IsUnderPostmaster || !event_triggers) ++ if (!IsNormalUserSession() || !event_triggers) + return; + + /* +@@ -830,7 +830,7 @@ EventTriggerSQLDrop(Node *parsetree) + * See EventTriggerDDLCommandStart for a discussion about why event + * triggers are disabled in single user mode or via a GUC. + */ +- if (!IsUnderPostmaster || !event_triggers) ++ if (!IsNormalUserSession() || !event_triggers) + return; + + /* +@@ -1009,7 +1009,7 @@ EventTriggerTableRewrite(Node *parsetree, Oid tableOid, int reason) + * See EventTriggerDDLCommandStart for a discussion about why event + * triggers are disabled in single user mode or via a GUC. + */ +- if (!IsUnderPostmaster || !event_triggers) ++ if (!IsNormalUserSession() || !event_triggers) + return; + + /* +diff --git a/src/backend/commands/tsearchcmds.c b/src/backend/commands/tsearchcmds.c +index ab16d42..97a3ec6 100644 +--- a/src/backend/commands/tsearchcmds.c ++++ b/src/backend/commands/tsearchcmds.c +@@ -352,7 +352,7 @@ verify_dictoptions(Oid tmplId, List *dictoptions) + * that can't be translated into template1's encoding). We want to create + * them anyway, since they might be usable later in other databases. + */ +- if (!IsUnderPostmaster) ++ if (!IsNormalUserSession()) + return; + + tup = SearchSysCache1(TSTEMPLATEOID, ObjectIdGetDatum(tmplId)); +diff --git a/src/backend/main/main.c b/src/backend/main/main.c +index 32e0e97..a3513ec 100644 +--- a/src/backend/main/main.c ++++ b/src/backend/main/main.c +@@ -33,8 +33,16 @@ + #include "bootstrap/bootstrap.h" + #include "common/username.h" + #include "miscadmin.h" ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++#include "optimizer/cost.h" ++#include "replication/walsender.h" ++#include "storage/aio.h" ++#endif + #include "postmaster/postmaster.h" + #include "tcop/tcopprot.h" ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++#include "utils/guc.h" ++#endif + #include "utils/help_config.h" + #include "utils/memutils.h" + #include "utils/pg_locale.h" +@@ -44,6 +52,97 @@ + const char *progname; + static bool reached_main = false; + ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++#define OLIPHAUNT_WASM_HOST_EXPORT(name) __attribute__((export_name(name))) ++ ++typedef enum OliphauntTrustedEmbeddedLifecycle ++{ ++ OLIPHAUNT_TRUSTED_EMBEDDED_UNSELECTED = 0, ++ OLIPHAUNT_TRUSTED_EMBEDDED_PREPARED, ++ OLIPHAUNT_TRUSTED_EMBEDDED_ATTACHED ++} OliphauntTrustedEmbeddedLifecycle; ++ ++static OliphauntTrustedEmbeddedLifecycle trusted_embedded_lifecycle = ++ OLIPHAUNT_TRUSTED_EMBEDDED_UNSELECTED; ++ ++static bool ++trusted_embedded_settings_are_safe(void) ++{ ++ return io_method == IOMETHOD_SYNC && ++ max_worker_processes == 0 && ++ max_parallel_workers == 0 && ++ max_parallel_workers_per_gather == 0 && ++ max_parallel_maintenance_workers == 0 && ++ max_wal_senders == 0; ++} ++ ++bool ++IsTrustedEmbeddedSession(void) ++{ ++ return trusted_embedded_lifecycle == ++ OLIPHAUNT_TRUSTED_EMBEDDED_ATTACHED; ++} ++ ++/* ++ * Select the embedded topology exactly once and before main(). PREPARED is ++ * intentionally not an active session: PostgreSQL remains a truthful ++ * standalone backend until the host attaches after _start. ++ */ ++OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_prepare_trusted_embedded_session") ++int ++oliphaunt_wasix_prepare_trusted_embedded_session(void) ++{ ++ if (reached_main || ++ trusted_embedded_lifecycle != ++ OLIPHAUNT_TRUSTED_EMBEDDED_UNSELECTED) ++ return -1; ++ ++ trusted_embedded_lifecycle = OLIPHAUNT_TRUSTED_EMBEDDED_PREPARED; ++ return 0; ++} ++ ++/* ++ * Apply non-negotiable single-backend limits after reading configuration and ++ * before sizing shared memory. PGC_S_OVERRIDE also prevents database and ++ * role defaults from restoring worker-capable settings later in startup. ++ */ ++void ++ConfigurePreparedTrustedEmbeddedSession(void) ++{ ++ if (trusted_embedded_lifecycle != OLIPHAUNT_TRUSTED_EMBEDDED_PREPARED) ++ return; ++ ++ Assert(reached_main); ++ SetConfigOption("io_method", "sync", PGC_POSTMASTER, PGC_S_OVERRIDE); ++ SetConfigOption("max_worker_processes", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_parallel_workers", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_parallel_workers_per_gather", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_parallel_maintenance_workers", "0", PGC_POSTMASTER, ++ PGC_S_OVERRIDE); ++ SetConfigOption("max_wal_senders", "0", PGC_POSTMASTER, PGC_S_OVERRIDE); ++ ++ if (!trusted_embedded_settings_are_safe()) ++ ereport(FATAL, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("trusted embedded sessions require synchronous I/O and zero PostgreSQL workers"))); ++} ++ ++bool ++AttachPreparedTrustedEmbeddedSession(void) ++{ ++ if (!reached_main || ++ trusted_embedded_lifecycle != OLIPHAUNT_TRUSTED_EMBEDDED_PREPARED || ++ !trusted_embedded_settings_are_safe()) ++ return false; ++ ++ trusted_embedded_lifecycle = OLIPHAUNT_TRUSTED_EMBEDDED_ATTACHED; ++ return true; ++} ++#endif ++ + /* names of special must-be-first options for dispatching to subprograms */ + static const char *const DispatchOptionNames[] = + { +diff --git a/src/backend/optimizer/plan/createplan.c b/src/backend/optimizer/plan/createplan.c +index 10b7358..716082f 100644 +--- a/src/backend/optimizer/plan/createplan.c ++++ b/src/backend/optimizer/plan/createplan.c +@@ -1294,7 +1294,8 @@ create_append_plan(PlannerInfo *root, AppendPath *best_path, int flags) + } + + /* If appropriate, consider async append */ +- consider_async = (enable_async_append && pathkeys == NIL && ++ consider_async = (enable_async_append && ++ !IsTrustedEmbeddedSession() && pathkeys == NIL && + !best_path->path.parallel_safe && + list_length(best_path->subpaths) > 1); + +diff --git a/src/backend/storage/ipc/signalfuncs.c b/src/backend/storage/ipc/signalfuncs.c +index a3a670b..563bc04 100644 +--- a/src/backend/storage/ipc/signalfuncs.c ++++ b/src/backend/storage/ipc/signalfuncs.c +@@ -287,6 +287,11 @@ pg_terminate_backend(PG_FUNCTION_ARGS) + Datum + pg_reload_conf(PG_FUNCTION_ARGS) + { ++ if (IsTrustedEmbeddedSession()) ++ ereport(ERROR, ++ (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), ++ errmsg("configuration reload is not supported in a trusted embedded session"))); ++ + if (kill(PostmasterPid, SIGHUP)) + { + ereport(WARNING, +@@ -307,6 +312,13 @@ pg_reload_conf(PG_FUNCTION_ARGS) + Datum + pg_rotate_logfile(PG_FUNCTION_ARGS) + { ++ if (IsTrustedEmbeddedSession()) ++ { ++ ereport(WARNING, ++ (errmsg("log rotation is not supported in a trusted embedded session"))); ++ PG_RETURN_BOOL(false); ++ } ++ + if (!Logging_collector) + { + ereport(WARNING, +diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c +index dc8bf8d..298918f 100644 +--- a/src/backend/tcop/postgres.c ++++ b/src/backend/tcop/postgres.c +@@ -239,6 +239,10 @@ oliphaunt_wasix_init_protocol_port(void) + OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_start") void + oliphaunt_wasix_start(void) + { ++ if (!AttachPreparedTrustedEmbeddedSession()) ++ elog(FATAL, ++ "trusted embedded session attach requires exactly one pre-start preparation and safe single-backend settings"); ++ + oliphaunt_wasix_init_protocol_port(); + whereToSendOutput = DestRemote; + ExitOnAnyError = false; +@@ -2986,6 +2990,7 @@ start_xact_command(void) + enable_statement_timeout(); + + /* Start timeout for checking if the client has gone away if necessary. */ ++ /* IsUnderPostmaster excludes the trusted embedded session's synthetic fd. */ + if (client_connection_check_interval > 0 && + IsUnderPostmaster && + MyProcPort && +@@ -4263,6 +4268,10 @@ PostgresSingleUserMain(int argc, char *argv[], + if (!SelectConfigFiles(userDoption, progname)) + proc_exit(1); + ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ ConfigurePreparedTrustedEmbeddedSession(); ++#endif ++ + /* + * Validate we have been given a reasonable-looking DataDir and change + * into it. +diff --git a/src/backend/utils/misc/guc_tables.c b/src/backend/utils/misc/guc_tables.c +index 6b82a23..4e99dc7 100644 +--- a/src/backend/utils/misc/guc_tables.c ++++ b/src/backend/utils/misc/guc_tables.c +@@ -55,6 +55,7 @@ + #include "libpq/libpq.h" + #include "libpq/oauth.h" + #include "libpq/scram.h" ++#include "miscadmin.h" + #include "nodes/queryjumble.h" + #include "optimizer/cost.h" + #include "optimizer/geqo.h" +@@ -112,6 +113,34 @@ + #define PG_KRB_SRVTAB "" + #endif + ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++static bool ++check_oliphaunt_single_backend_worker_limit(int *newval, void **extra, ++ GucSource source) ++{ ++ (void) extra; ++ ++ /* ++ * PGC_S_OVERRIDE pins current and reset values. Lower-priority sources ++ * cannot displace it, and PGC_S_TEST validates without assigning. ++ */ ++ if (IsTrustedEmbeddedSession() && ++ source != PGC_S_TEST && ++ source >= PGC_S_OVERRIDE && ++ *newval != 0) ++ { ++ GUC_check_errdetail("Trusted embedded sessions cannot launch PostgreSQL workers."); ++ return false; ++ } ++ ++ return true; ++} ++#define CHECK_OLIPHAUNT_SINGLE_BACKEND_WORKER_LIMIT \ ++ check_oliphaunt_single_backend_worker_limit ++#else ++#define CHECK_OLIPHAUNT_SINGLE_BACKEND_WORKER_LIMIT NULL ++#endif ++ + /* + * Options for enum values defined in this module. + * +@@ -3620,7 +3643,7 @@ struct config_int ConfigureNamesInt[] = + }, + &max_parallel_maintenance_workers, + 2, 0, MAX_PARALLEL_WORKER_LIMIT, +- NULL, NULL, NULL ++ CHECK_OLIPHAUNT_SINGLE_BACKEND_WORKER_LIMIT, NULL, NULL + }, + + { +@@ -3631,7 +3654,7 @@ struct config_int ConfigureNamesInt[] = + }, + &max_parallel_workers_per_gather, + 2, 0, MAX_PARALLEL_WORKER_LIMIT, +- NULL, NULL, NULL ++ CHECK_OLIPHAUNT_SINGLE_BACKEND_WORKER_LIMIT, NULL, NULL + }, + + { +@@ -3642,7 +3665,7 @@ struct config_int ConfigureNamesInt[] = + }, + &max_parallel_workers, + 8, 0, MAX_PARALLEL_WORKER_LIMIT, +- NULL, NULL, NULL ++ CHECK_OLIPHAUNT_SINGLE_BACKEND_WORKER_LIMIT, NULL, NULL + }, + + { +diff --git a/src/backend/utils/misc/superuser.c b/src/backend/utils/misc/superuser.c +index 5858b0a..eb896e2 100644 +--- a/src/backend/utils/misc/superuser.c ++++ b/src/backend/utils/misc/superuser.c +@@ -63,7 +63,7 @@ superuser_arg(Oid roleid) + return last_roleid_is_super; + + /* Special escape path in case you deleted all your users. */ +- if (!IsUnderPostmaster && roleid == BOOTSTRAP_SUPERUSERID) ++ if (!IsNormalUserSession() && roleid == BOOTSTRAP_SUPERUSERID) + return true; + + /* OK, look up the information in pg_authid */ +diff --git a/src/include/miscadmin.h b/src/include/miscadmin.h +index 9a7d733..b8ce6be 100644 +--- a/src/include/miscadmin.h ++++ b/src/include/miscadmin.h +@@ -168,6 +168,29 @@ extern PGDLLIMPORT bool IsPostmasterEnvironment; + extern PGDLLIMPORT bool IsUnderPostmaster; + extern PGDLLIMPORT bool IsBinaryUpgrade; + ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++extern bool IsTrustedEmbeddedSession(void); ++extern void ConfigurePreparedTrustedEmbeddedSession(void); ++extern bool AttachPreparedTrustedEmbeddedSession(void); ++#else ++static inline bool ++IsTrustedEmbeddedSession(void) ++{ ++ return false; ++} ++#endif ++ ++/* ++ * A normal user session is either a real postmaster child or the explicitly ++ * attached single-backend session. This is a positive semantic capability, ++ * not a claim that the embedded backend has a postmaster supervisor. ++ */ ++static inline bool ++IsNormalUserSession(void) ++{ ++ return IsUnderPostmaster || IsTrustedEmbeddedSession(); ++} ++ + extern PGDLLIMPORT bool ExitOnAnyError; + + extern PGDLLIMPORT char *DataDir; +-- +2.43.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-set-embedded-postmaster-environment.patch b/src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-set-embedded-postmaster-environment.patch deleted file mode 100644 index ac4f65cec..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-set-embedded-postmaster-environment.patch +++ /dev/null @@ -1,31 +0,0 @@ -From 0000000000000000000000000000000000000029 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Fri, 29 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: set embedded postmaster environment - -The long-lived embedded backend should run through PostgreSQL's normal backend -conditionals where possible. Mark the WASIX backend as a postmaster environment -after startup while keeping checkpoint requests local under -OLIPHAUNT_WASM_SINGLE_USER. - -This is paired with the prior checkpoint patch: that patch keeps single-user -checkpoint behavior explicit, while this one restores the embedded backend's -postmaster-style runtime identity for the rest of PostgreSQL. ---- - src/backend/tcop/postgres.c | 2 ++ - 1 file changed, 2 insertions(+) - -diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c ---- a/src/backend/tcop/postgres.c -+++ b/src/backend/tcop/postgres.c -@@ -232,6 +232,8 @@ oliphaunt_wasix_start(void) - whereToSendOutput = DestRemote; - ExitOnAnyError = false; - MyBackendType = B_BACKEND; -+ IsPostmasterEnvironment = true; -+ IsUnderPostmaster = true; - } - - OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_pq_flush") void --- -2.39.5 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0030-oliphaunt-wasix-avoid-xlogwrite-prevseg-division.patch b/src/wasix/runtime/assets/build/postgres/patches/0030-oliphaunt-wasix-avoid-xlogwrite-prevseg-division.patch deleted file mode 100644 index 9e39ab486..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0030-oliphaunt-wasix-avoid-xlogwrite-prevseg-division.patch +++ /dev/null @@ -1,74 +0,0 @@ -From 0000000000000000000000000000000000000030 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Fri, 29 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: avoid xlogwrite prevseg division - -The PG18 embedded WASIX Test 11 COMMIT gap is dominated by the -per-WAL-page scan in XLogWrite(), not raw pwrite, fsync, pgstat accounting, or -WALWriteLock waiting. XLogWrite() calls XLByteInPrevSeg() for every WAL page -it scans, and that macro computes segment membership with a dynamic 64-bit -division by wal_segment_size. - -PostgreSQL validates WAL segment sizes as powers of two before normal -operation. In the XLogWrite() hot loop, keep the current open segment's byte -bounds and compare against those bounds instead of repeatedly dividing the LSN. -The segment bounds are refreshed whenever XLogWrite() changes openLogSegNo, so -the semantics match XLByteInPrevSeg() while avoiding the expensive per-page -divide in the WASIX codegen path. ---- - src/backend/access/transam/xlog.c | 15 +++++++++++++-- - 1 file changed, 13 insertions(+), 2 deletions(-) - -diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c ---- a/src/backend/access/transam/xlog.c -+++ b/src/backend/access/transam/xlog.c -@@ -2315,6 +2315,8 @@ XLogWrite(XLogwrtRqst WriteRqst, TimeLineID tli, bool flexible) - int npages; - int startidx; - uint32 startoffset; -+ XLogRecPtr openLogSegStart; -+ XLogRecPtr openLogSegEnd; - - /* We should always be inside a critical section here */ - Assert(CritSectionCount > 0); -@@ -2339,6 +2341,9 @@ XLogWrite(XLogwrtRqst WriteRqst, TimeLineID tli, bool flexible) - startidx = 0; - startoffset = 0; - -+ openLogSegStart = openLogSegNo * (uint64) wal_segment_size; -+ openLogSegEnd = openLogSegStart + (uint64) wal_segment_size; -+ - /* - * Within the loop, curridx is the cache block index of the page to - * consider writing. Begin at the buffer containing the next unwritten -@@ -2367,8 +2372,8 @@ XLogWrite(XLogwrtRqst WriteRqst, TimeLineID tli, bool flexible) - LogwrtResult.Write = EndPtr; - ispartialpage = WriteRqst.Write < LogwrtResult.Write; - -- if (!XLByteInPrevSeg(LogwrtResult.Write, openLogSegNo, -- wal_segment_size)) -+ if (!((LogwrtResult.Write - 1) >= openLogSegStart && -+ (LogwrtResult.Write - 1) < openLogSegEnd)) - { - /* - * Switch to new logfile segment. We cannot have any pending -@@ -2381,6 +2387,8 @@ XLogWrite(XLogwrtRqst WriteRqst, TimeLineID tli, bool flexible) - XLByteToPrevSeg(LogwrtResult.Write, openLogSegNo, - wal_segment_size); - openLogTLI = tli; -+ openLogSegStart = openLogSegNo * (uint64) wal_segment_size; -+ openLogSegEnd = openLogSegStart + (uint64) wal_segment_size; - - /* create/use new log file */ - openLogFile = XLogFileInit(openLogSegNo, tli); -@@ -2393,6 +2401,8 @@ XLogWrite(XLogwrtRqst WriteRqst, TimeLineID tli, bool flexible) - XLByteToPrevSeg(LogwrtResult.Write, openLogSegNo, - wal_segment_size); - openLogTLI = tli; -+ openLogSegStart = openLogSegNo * (uint64) wal_segment_size; -+ openLogSegEnd = openLogSegStart + (uint64) wal_segment_size; - openLogFile = XLogFileOpen(openLogSegNo, tli); - ReserveExternalFD(); - } --- -2.39.5 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0031-oliphaunt-wasix-skip-activity-id-reporting.patch b/src/wasix/runtime/assets/build/postgres/patches/0031-oliphaunt-wasix-skip-activity-id-reporting.patch deleted file mode 100644 index 5536b3786..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0031-oliphaunt-wasix-skip-activity-id-reporting.patch +++ /dev/null @@ -1,125 +0,0 @@ -From 0000000000000000000000000000000000000031 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Fri, 29 May 2026 00:00:00 +0530 -Subject: [PATCH] oliphaunt-wasix: skip activity id reporting - -PostgreSQL 18 reports plan IDs to activity statistics from the simple-query, -extended-query, and planner paths. The embedded embedded WASIX runtime does -not expose query or plan IDs as an observability surface, and the activity -tracking GUC is disabled for this lane. - -Guard query-id and plan-id activity reporting behind the existing single-user -build define. This keeps SQL behavior unchanged while avoiding extra reporting -calls and portal/query statement scans on the hot embedded path. ---- - src/backend/optimizer/plan/planner.c | 2 ++ - src/backend/tcop/postgres.c | 20 ++++++++++++++++++++ - 2 files changed, 22 insertions(+) - -diff --git a/src/backend/optimizer/plan/planner.c b/src/backend/optimizer/plan/planner.c ---- a/src/backend/optimizer/plan/planner.c -+++ b/src/backend/optimizer/plan/planner.c -@@ -309,7 +309,9 @@ planner(Query *parse, const char *query_string, int cursorOptions, - else - result = standard_planner(parse, query_string, cursorOptions, boundParams); - -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - pgstat_report_plan_id(result->planId, false); -+#endif - - return result; - } -diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c ---- a/src/backend/tcop/postgres.c -+++ b/src/backend/tcop/postgres.c -@@ -1259,8 +1259,10 @@ exec_simple_query(const char *query_string) - const char *cmdtagname; - size_t cmdtaglen; - -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - pgstat_report_query_id(0, true); - pgstat_report_plan_id(0, true); -+#endif - - /* - * Get the command name for use in status display (it also becomes the -@@ -1822,7 +1824,9 @@ exec_bind_message(StringInfo input_message) - char msec_str[32]; - ParamsErrorCbData params_data; - ErrorContextCallback params_errcxt; -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - ListCell *lc; -+#endif - - /* Get the fixed part of the message */ - portal_name = pq_getmsgstring(input_message); -@@ -1858,6 +1860,7 @@ exec_bind_message(StringInfo input_message) - - pgstat_report_activity(STATE_RUNNING, psrc->query_string); - -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - foreach(lc, psrc->query_list) - { - Query *query = lfirst_node(Query, lc); -@@ -1870,6 +1873,7 @@ exec_bind_message(StringInfo input_message) - break; - } - } -+#endif - - set_ps_display("BIND"); - -@@ -2211,6 +2215,7 @@ exec_bind_message(StringInfo input_message) - cplan->stmt_list, - cplan); - -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - /* Portal is defined, set the plan ID based on its contents. */ - foreach(lc, portal->stmts) - { -@@ -2223,6 +2228,7 @@ exec_bind_message(StringInfo input_message) - break; - } - } -+#endif - - /* Done with the snapshot used for parameter I/O and parsing/planning */ - if (snapshot_set) -@@ -2307,7 +2316,9 @@ exec_execute_message(const char *portal_name, long max_rows) - ErrorContextCallback params_errcxt; - const char *cmdtagname; - size_t cmdtaglen; -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - ListCell *lc; -+#endif - - /* Adjust destination to tell printtup.c what to do */ - dest = whereToSendOutput; -@@ -2347,6 +2353,7 @@ PortalRun(QueryCompletion *qc, DestReceiver *dest) - - pgstat_report_activity(STATE_RUNNING, sourceText); - -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - foreach(lc, portal->stmts) - { - PlannedStmt *stmt = lfirst_node(PlannedStmt, lc); -@@ -2359,7 +2366,9 @@ PortalRun(QueryCompletion *qc, DestReceiver *dest) - break; - } - } -+#endif - -+#ifndef OLIPHAUNT_WASM_SINGLE_USER - foreach(lc, portal->stmts) - { - PlannedStmt *stmt = lfirst_node(PlannedStmt, lc); -@@ -2372,6 +2381,7 @@ PortalRun(QueryCompletion *qc, DestReceiver *dest) - break; - } - } -+#endif - - cmdtagname = GetCommandTagNameAndLen(portal->commandTag, &cmdtaglen); - --- -2.39.5 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0032-oliphaunt-wasix-treat-directory-fsync-eisdir-as-unsupported.patch b/src/wasix/runtime/assets/build/postgres/patches/0032-oliphaunt-wasix-treat-directory-fsync-eisdir-as-unsupported.patch index 9dad7d950..29bd9d70a 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0032-oliphaunt-wasix-treat-directory-fsync-eisdir-as-unsupported.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0032-oliphaunt-wasix-treat-directory-fsync-eisdir-as-unsupported.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000032 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Sat, 30 May 2026 00:00:00 +0530 Subject: [PATCH] oliphaunt-wasix: treat directory fsync EISDIR as unsupported @@ -16,14 +16,28 @@ preserving fatal error handling for regular files and non-directory errors. 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/backend/storage/file/fd.c b/src/backend/storage/file/fd.c +index 378c078..48453fe 100644 --- a/src/backend/storage/file/fd.c +++ b/src/backend/storage/file/fd.c -@@ -3845,1 +3845,1 @@ fsync_fname_ext(const char *fname, bool isdir, bool ignore_perm, int elevel) +@@ -3902,7 +3902,7 @@ fsync_fname_ext(const char *fname, bool isdir, bool ignore_perm, int elevel) + * Some OSes don't allow us to fsync directories at all, so we can ignore + * those errors. Anything else needs to be logged. + */ - if (returncode != 0 && !(isdir && (errno == EBADF || errno == EINVAL))) + if (returncode != 0 && !(isdir && (errno == EBADF || errno == EINVAL || errno == EISDIR))) + { + int save_errno; + diff --git a/src/common/file_utils.c b/src/common/file_utils.c +index 7b62687..dc474fb 100644 --- a/src/common/file_utils.c +++ b/src/common/file_utils.c -@@ -416,1 +416,1 @@ fsync_fname(const char *fname, bool isdir) +@@ -435,7 +435,7 @@ fsync_fname(const char *fname, bool isdir) + * Some OSes don't allow us to fsync directories at all, so we can ignore + * those errors. Anything else needs to be reported. + */ - if (returncode != 0 && !(isdir && (errno == EBADF || errno == EINVAL))) + if (returncode != 0 && !(isdir && (errno == EBADF || errno == EINVAL || errno == EISDIR))) + { + pg_log_error("could not fsync file \"%s\": %m", fname); + (void) close(fd); diff --git a/src/wasix/runtime/assets/build/postgres/patches/0034-oliphaunt-wasix-declare-hybrid-protocol-transport.patch b/src/wasix/runtime/assets/build/postgres/patches/0034-oliphaunt-wasix-declare-hybrid-protocol-transport.patch index bf0f2f8e0..85de2e13e 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0034-oliphaunt-wasix-declare-hybrid-protocol-transport.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0034-oliphaunt-wasix-declare-hybrid-protocol-transport.patch @@ -1,22 +1,29 @@ From 0000000000000000000000000000000000000034 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Tue, 18 Aug 2026 00:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: declare hybrid protocol transport -The Rust proxy switches from buffered host I/O to a hybrid stream only while -PostgreSQL is inside COPY. Keep that bridge contract visible in the WASIX port -header without retaining the retired process-level stdio lifecycle. +The Rust and TypeScript proxies switch from buffered host I/O to a hybrid +stream only while PostgreSQL is inside COPY. Ordinary protocol execution can +instead keep its finite request bytes buffered while streaming every response +byte. Declare the selector beside the canonical generated transport contract +without retaining the retired process-level stdio lifecycle. --- - src/include/port/wasix-dl.h | 2 ++ - 1 file changed, 2 insertions(+) + src/include/port/wasix-dl.h | 6 ++++++ + 1 file changed, 6 insertions(+) diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h +index 73c79f3a54..b15c379d0b 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h -@@ -38,6 +38,8 @@ extern void *oliphaunt_wasix_shmat(int shmid, const void *shmaddr, int shmflg); +@@ -41,6 +41,12 @@ extern void *oliphaunt_wasix_shmat(int shmid, const void *shmaddr, int shmflg); extern int oliphaunt_wasix_shmdt(const void *shmaddr); extern int oliphaunt_wasix_shmctl(int shmid, int cmd, struct shmid_ds *buf); extern void oliphaunt_wasix_protocol_report_copy_response(int state); ++/* ++ * These declarations consume the generated mode numbers included above. ++ * Do not redeclare numeric transport values in this patch. ++ */ +extern int oliphaunt_wasix_set_protocol_transport(int mode); +extern int oliphaunt_wasix_protocol_stream_active(void); diff --git a/src/wasix/runtime/assets/build/postgres/patches/0035-oliphaunt-wasix-use-single-backend-spinlocks.patch b/src/wasix/runtime/assets/build/postgres/patches/0035-oliphaunt-wasix-use-single-backend-spinlocks.patch deleted file mode 100644 index 1d5846378..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0035-oliphaunt-wasix-use-single-backend-spinlocks.patch +++ /dev/null @@ -1,64 +0,0 @@ -From 0000000000000000000000000000000000000035 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Wed, 12 Aug 2026 00:00:00 +0000 -Subject: [PATCH] oliphaunt-wasix: use single-backend spinlocks - -The embedded WASIX runtime executes one PostgreSQL backend per WebAssembly -instance. Host workers may run concurrently, but each owns a distinct -instance and PostgreSQL shared-memory arena. Atomic read-modify-write -instructions therefore add cost without protecting a concurrent PostgreSQL -access in this lane. - -Keep the spinlock state machine and its stuck-lock detection, but use volatile -loads and stores plus compiler barriers for the explicitly single-backend -WASIX build. Preserve the existing int-sized slock_t ABI. Concurrent native -and non-embedded WASIX builds continue through PostgreSQL's normal atomic -implementation. ---- - src/include/storage/s_lock.h | 35 +++++++++++++++++++++++++++++++++++ - 1 file changed, 35 insertions(+) - -diff --git a/src/include/storage/s_lock.h b/src/include/storage/s_lock.h ---- a/src/include/storage/s_lock.h -+++ b/src/include/storage/s_lock.h -@@ -93,7 +93,40 @@ - #error "s_lock.h may not be included from frontend code" - #endif - -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) && defined(OLIPHAUNT_WASM_SINGLE_BACKEND_ATOMICS) -+/* -+ * The embedded WASIX lane has one PostgreSQL execution thread per Wasm -+ * instance. Separate host workers own separate instances, so no other -+ * thread can observe this PostgreSQL shared-memory arena. -+ * -+ * Keep volatile lock state and compiler barriers: they preserve ordering and -+ * retain the normal stuck-lock behavior if backend code accidentally attempts -+ * recursive acquisition. Hardware atomics remain mandatory everywhere that -+ * can execute PostgreSQL code concurrently. -+ */ -+#define HAS_TEST_AND_SET -+ -+typedef int slock_t; -+ -+static inline int -+oliphaunt_wasix_single_user_tas(volatile slock_t *lock) -+{ -+ int previous = *lock; -+ -+ *lock = 1; -+ __asm__ __volatile__("" : : : "memory"); -+ return previous; -+} -+ -+#define TAS(lock) oliphaunt_wasix_single_user_tas(lock) -+#define S_UNLOCK(lock) \ -+ do { \ -+ __asm__ __volatile__("" : : : "memory"); \ -+ *(lock) = 0; \ -+ } while (0) -+#endif -+ - #if defined(__GNUC__) || defined(__INTEL_COMPILER) - /************************************************************************* - * All the gcc inlines - * Gcc consistently defines the CPU as __cpu__. diff --git a/src/wasix/runtime/assets/build/postgres/patches/0036-oliphaunt-wasix-specialize-single-backend-atomics.patch b/src/wasix/runtime/assets/build/postgres/patches/0036-oliphaunt-wasix-specialize-single-backend-atomics.patch deleted file mode 100644 index 08289bf42..000000000 --- a/src/wasix/runtime/assets/build/postgres/patches/0036-oliphaunt-wasix-specialize-single-backend-atomics.patch +++ /dev/null @@ -1,290 +0,0 @@ -From 0000000000000000000000000000000000000036 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers -Date: Wed, 12 Aug 2026 00:00:00 +0000 -Subject: [PATCH] oliphaunt-wasix: specialize single-backend atomics - -The embedded WASIX runtime executes one PostgreSQL backend thread per -WebAssembly instance. Host workers own distinct instances and startup policy -disables PostgreSQL workers, so atomic read-modify-write and fence instructions -inside the backend add cost without protecting a concurrent observer. - -Select a dedicated implementation only while compiling PostgreSQL backend -objects. Preserve the configured flag, uint32, and aligned uint64 layouts, -compiler ordering, and compare-exchange contracts. Frontends, PGXS extensions, -and every concurrent PostgreSQL build retain the normal atomic implementation. ---- - src/backend/common.mk | 5 + - src/backend/main/main.c | 14 ++ - src/include/port/atomics.h | 6 +- - src/include/port/atomics/arch-wasix-single.h | 185 +++++++++++++++++++ - 4 files changed, 209 insertions(+), 3 deletions(-) - create mode 100644 src/include/port/atomics/arch-wasix-single.h - -diff --git a/src/backend/common.mk b/src/backend/common.mk -index 3bdd017213..25a9ea2f04 100644 ---- a/src/backend/common.mk -+++ b/src/backend/common.mk -@@ -7,7 +7,12 @@ - # When including this file, set OBJS to the object files created in - # this directory and SUBDIRS to subdirectories containing more things - # to build. - -+ifeq ($(PORTNAME), wasix-dl) -+# PGXS side modules deliberately retain PostgreSQL's normal atomic ABI. -+override CPPFLAGS += -DOLIPHAUNT_WASM_SINGLE_BACKEND_ATOMICS -+endif -+ - subsysfilename = objfiles.txt - - SUBDIROBJS = $(SUBDIRS:%=%/$(subsysfilename)) -diff --git a/src/backend/main/main.c b/src/backend/main/main.c -index 625f573c88..f85272b9ef 100644 ---- a/src/backend/main/main.c -+++ b/src/backend/main/main.c -@@ -210,7 +210,11 @@ main(int argc, char *argv[]) - BootstrapModeMain(argc, argv, false); - break; - case DISPATCH_FORKCHILD: --#ifdef EXEC_BACKEND -+#if defined(OLIPHAUNT_WASM_SINGLE_BACKEND_ATOMICS) -+ write_stderr("%s: fork-child mode is unavailable in the single-backend WASIX runtime\n", -+ progname); -+ exit(1); -+#elif defined(EXEC_BACKEND) - SubPostmasterMain(argc, argv); - #else - Assert(false); /* should never happen */ -@@ -224,6 +229,14 @@ main(int argc, char *argv[]) - strdup(get_user_name_or_exit(progname))); - break; - case DISPATCH_POSTMASTER: -+#if defined(OLIPHAUNT_WASM_SINGLE_BACKEND_ATOMICS) -+ if (argc <= 2 || strcmp(argv[1], "-C") != 0) -+ { -+ write_stderr("%s: postmaster mode is unavailable in the single-backend WASIX runtime\n", -+ progname); -+ exit(1); -+ } -+#endif - PostmasterMain(argc, argv); - break; - } -diff --git a/src/include/port/atomics.h b/src/include/port/atomics.h -index e849de7a4d..43d2f9e6a7 100644 ---- a/src/include/port/atomics.h -+++ b/src/include/port/atomics.h -@@ -63,7 +63,9 @@ - * compiler barrier. - * - */ --#if defined(__arm__) || defined(__arm) || defined(__aarch64__) -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) && defined(OLIPHAUNT_WASM_SINGLE_BACKEND_ATOMICS) -+#include "port/atomics/arch-wasix-single.h" -+#elif defined(__arm__) || defined(__arm) || defined(__aarch64__) - #include "port/atomics/arch-arm.h" - #elif defined(__i386__) || defined(__i386) || defined(__x86_64__) - #include "port/atomics/arch-x86.h" -@@ -84,7 +86,9 @@ - /* - * gcc or compatible, including clang and icc. - */ --#if defined(__GNUC__) || defined(__INTEL_COMPILER) -+#if defined(__wasi__) && defined(OLIPHAUNT_WASM_SINGLE_USER) && defined(OLIPHAUNT_WASM_SINGLE_BACKEND_ATOMICS) -+/* The backend-only WASIX header supplies the complete implementation. */ -+#elif defined(__GNUC__) || defined(__INTEL_COMPILER) - #include "port/atomics/generic-gcc.h" - #elif defined(_MSC_VER) - #include "port/atomics/generic-msvc.h" -diff --git a/src/include/port/atomics/arch-wasix-single.h b/src/include/port/atomics/arch-wasix-single.h -new file mode 100644 -index 0000000000..5548925c11 ---- /dev/null -+++ b/src/include/port/atomics/arch-wasix-single.h -@@ -0,0 +1,185 @@ -+/*------------------------------------------------------------------------- -+ * -+ * arch-wasix-single.h -+ * PostgreSQL atomics for the single-backend WASIX runtime. -+ * -+ * Portions Copyright (c) 1996-2025, PostgreSQL Global Development Group -+ * -+ * src/include/port/atomics/arch-wasix-single.h -+ * -+ *------------------------------------------------------------------------- -+ */ -+ -+/* intentionally no include guards, should only be included by atomics.h */ -+#ifndef INSIDE_ATOMICS_H -+#error "should be included via atomics.h" -+#endif -+ -+/* -+ * One host worker owns one WebAssembly instance and PostgreSQL backend. -+ * Compiler barriers preserve the public ordering contracts; no hardware fence -+ * or read-modify-write instruction is needed without a concurrent observer. -+ * Keep these layouts byte-for-byte compatible with generic-gcc.h. -+ */ -+#define pg_compiler_barrier_impl() __asm__ __volatile__("" ::: "memory") -+#define pg_memory_barrier_impl() pg_compiler_barrier_impl() -+#define pg_read_barrier_impl() pg_compiler_barrier_impl() -+#define pg_write_barrier_impl() pg_compiler_barrier_impl() -+ -+#ifndef HAVE_GCC__SYNC_INT32_TAS -+#error "single-backend WASIX requires the configured int-sized flag ABI" -+#endif -+ -+#define PG_HAVE_ATOMIC_FLAG_SUPPORT -+typedef struct pg_atomic_flag -+{ -+ volatile int value; -+} pg_atomic_flag; -+ -+#define PG_HAVE_ATOMIC_U32_SUPPORT -+typedef struct pg_atomic_uint32 -+{ -+ volatile uint32 value; -+} pg_atomic_uint32; -+ -+#define PG_HAVE_ATOMIC_U64_SUPPORT -+#define PG_HAVE_8BYTE_SINGLE_COPY_ATOMICITY -+typedef struct pg_atomic_uint64 -+{ -+ volatile uint64 value pg_attribute_aligned(8); -+} pg_atomic_uint64; -+ -+#define PG_HAVE_ATOMIC_INIT_FLAG -+static inline void -+pg_atomic_init_flag_impl(volatile pg_atomic_flag *ptr) -+{ -+ ptr->value = 0; -+} -+ -+#define PG_HAVE_ATOMIC_TEST_SET_FLAG -+static inline bool -+pg_atomic_test_set_flag_impl(volatile pg_atomic_flag *ptr) -+{ -+ int old = ptr->value; -+ -+ ptr->value = 1; -+ pg_compiler_barrier_impl(); -+ return old == 0; -+} -+ -+#define PG_HAVE_ATOMIC_UNLOCKED_TEST_FLAG -+static inline bool -+pg_atomic_unlocked_test_flag_impl(volatile pg_atomic_flag *ptr) -+{ -+ return ptr->value == 0; -+} -+ -+#define PG_HAVE_ATOMIC_CLEAR_FLAG -+static inline void -+pg_atomic_clear_flag_impl(volatile pg_atomic_flag *ptr) -+{ -+ pg_compiler_barrier_impl(); -+ ptr->value = 0; -+} -+ -+#define PG_HAVE_ATOMIC_COMPARE_EXCHANGE_U32 -+#define PG_HAVE_ATOMIC_EXCHANGE_U32 -+#define PG_HAVE_ATOMIC_FETCH_ADD_U32 -+#define PG_HAVE_ATOMIC_FETCH_SUB_U32 -+#define PG_HAVE_ATOMIC_FETCH_AND_U32 -+#define PG_HAVE_ATOMIC_FETCH_OR_U32 -+#define PG_HAVE_ATOMIC_COMPARE_EXCHANGE_U64 -+#define PG_HAVE_ATOMIC_EXCHANGE_U64 -+#define PG_HAVE_ATOMIC_FETCH_ADD_U64 -+#define PG_HAVE_ATOMIC_FETCH_SUB_U64 -+#define PG_HAVE_ATOMIC_FETCH_AND_U64 -+#define PG_HAVE_ATOMIC_FETCH_OR_U64 -+ -+/* -+ * Generate the same scalar operations for PostgreSQL's two unsigned widths. -+ * Compare-exchange always returns the observed value through expected, as the -+ * public API requires on both success and failure. -+ */ -+#define OLIPHAUNT_WASIX_SINGLE_BACKEND_UINT_IMPL(suffix, atomic_type, uint_type, int_type) \ -+static inline bool \ -+pg_atomic_compare_exchange_##suffix##_impl(volatile atomic_type *ptr, \ -+ uint_type *expected, uint_type newval) \ -+{ \ -+ uint_type current; \ -+ bool exchanged; \ -+ \ -+ AssertPointerAlignment(expected, sizeof(uint_type)); \ -+ pg_compiler_barrier_impl(); \ -+ current = ptr->value; \ -+ exchanged = current == *expected; \ -+ if (exchanged) \ -+ ptr->value = newval; \ -+ *expected = current; \ -+ pg_compiler_barrier_impl(); \ -+ return exchanged; \ -+} \ -+\ -+static inline uint_type \ -+pg_atomic_exchange_##suffix##_impl(volatile atomic_type *ptr, uint_type newval) \ -+{ \ -+ uint_type old; \ -+ \ -+ pg_compiler_barrier_impl(); \ -+ old = ptr->value; \ -+ ptr->value = newval; \ -+ pg_compiler_barrier_impl(); \ -+ return old; \ -+} \ -+\ -+static inline uint_type \ -+pg_atomic_fetch_add_##suffix##_impl(volatile atomic_type *ptr, int_type add_) \ -+{ \ -+ uint_type old; \ -+ \ -+ pg_compiler_barrier_impl(); \ -+ old = ptr->value; \ -+ ptr->value = old + add_; \ -+ pg_compiler_barrier_impl(); \ -+ return old; \ -+} \ -+\ -+static inline uint_type \ -+pg_atomic_fetch_sub_##suffix##_impl(volatile atomic_type *ptr, int_type sub_) \ -+{ \ -+ uint_type old; \ -+ \ -+ pg_compiler_barrier_impl(); \ -+ old = ptr->value; \ -+ ptr->value = old - sub_; \ -+ pg_compiler_barrier_impl(); \ -+ return old; \ -+} \ -+\ -+static inline uint_type \ -+pg_atomic_fetch_and_##suffix##_impl(volatile atomic_type *ptr, uint_type and_) \ -+{ \ -+ uint_type old; \ -+ \ -+ pg_compiler_barrier_impl(); \ -+ old = ptr->value; \ -+ ptr->value = old & and_; \ -+ pg_compiler_barrier_impl(); \ -+ return old; \ -+} \ -+\ -+static inline uint_type \ -+pg_atomic_fetch_or_##suffix##_impl(volatile atomic_type *ptr, uint_type or_) \ -+{ \ -+ uint_type old; \ -+ \ -+ pg_compiler_barrier_impl(); \ -+ old = ptr->value; \ -+ ptr->value = old | or_; \ -+ pg_compiler_barrier_impl(); \ -+ return old; \ -+} -+ -+OLIPHAUNT_WASIX_SINGLE_BACKEND_UINT_IMPL(u32, pg_atomic_uint32, uint32, int32) -+OLIPHAUNT_WASIX_SINGLE_BACKEND_UINT_IMPL(u64, pg_atomic_uint64, uint64, int64) -+ -+#undef OLIPHAUNT_WASIX_SINGLE_BACKEND_UINT_IMPL --- -2.49.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0038-oliphaunt-wasix-disable-unsupported-writeback-hints.patch b/src/wasix/runtime/assets/build/postgres/patches/0038-oliphaunt-wasix-disable-unsupported-writeback-hints.patch index df54aed61..fb81bbc91 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0038-oliphaunt-wasix-disable-unsupported-writeback-hints.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0038-oliphaunt-wasix-disable-unsupported-writeback-hints.patch @@ -1,5 +1,5 @@ From 0000000000000000000000000000000000000038 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Mon, 17 Aug 2026 00:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: disable unsupported writeback hints @@ -19,7 +19,7 @@ targets retain the upstream selection logic. 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/src/backend/storage/file/fd.c b/src/backend/storage/file/fd.c -index 739b1b9..b1f0a22 100644 +index 48453fe..80a2146 100644 --- a/src/backend/storage/file/fd.c +++ b/src/backend/storage/file/fd.c @@ -103,7 +103,12 @@ @@ -37,11 +37,11 @@ index 739b1b9..b1f0a22 100644 #elif !defined(WIN32) && defined(MS_ASYNC) #define PG_FLUSH_DATA_WORKS 1 diff --git a/src/common/file_utils.c b/src/common/file_utils.c -index d0ddfd9..066d455 100644 +index dc474fb..62be891 100644 --- a/src/common/file_utils.c +++ b/src/common/file_utils.c -@@ -34,7 +34,12 @@ - #include "common/logging.h" +@@ -34,8 +34,13 @@ + #ifdef FRONTEND /* Define PG_FLUSH_DATA_WORKS if we have an implementation for pg_flush_data */ -#if defined(HAVE_SYNC_FILE_RANGE) @@ -52,7 +52,8 @@ index d0ddfd9..066d455 100644 + */ +#elif defined(HAVE_SYNC_FILE_RANGE) #define PG_FLUSH_DATA_WORKS 1 - #elif !defined(WIN32) && defined(MS_ASYNC) + #elif defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_DONTNEED) #define PG_FLUSH_DATA_WORKS 1 + #endif -- 2.51.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0039-oliphaunt-wasix-inline-sigsetjmp.patch b/src/wasix/runtime/assets/build/postgres/patches/0039-oliphaunt-wasix-inline-sigsetjmp.patch index 97b4eceeb..b48cb23a0 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0039-oliphaunt-wasix-inline-sigsetjmp.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0039-oliphaunt-wasix-inline-sigsetjmp.patch @@ -1,33 +1,33 @@ From 0000000000000000000000000000000000000039 Mon Sep 17 00:00:00 2001 -From: Oliphaunt Maintainers +From: Sid Jain Date: Tue, 18 Aug 2026 00:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: inline sigsetjmp The released WASIX exception-handling sysroot exposes sigsetjmp as a normal function. That is not sufficient for WebAssembly SJLJ: the compiler must see setjmp at each call site so that it emits the exception handler in the module -whose frame is being protected. Core and side-module callers otherwise return -from the wrapper's handler frame before siglongjmp, turning a perfectly -catchable PostgreSQL ERROR into an uncaught WebAssembly exception. +whose frame is being protected. The main executable and PostgreSQL side +modules otherwise call an out-of-line sigsetjmp, return from its handler frame, +and later turn a perfectly catchable PostgreSQL ERROR into an uncaught +WebAssembly exception. Mark PostgreSQL's shared-library and PGXS module compilations explicitly, then define the POSIX spelling in the WASIX port header in terms of the -compiler-recognized setjmp call for the single-user core and those modules. -The pinned WASIX -implementation currently does not preserve a signal mask in sigsetjmp either, -so discarding savesigs retains its existing semantics while making -PG_TRY/PG_CATCH correct in the core and every PostgreSQL side module. +compiler-recognized setjmp call for the embedded main executable and those +modules. The pinned WASIX implementation currently does not preserve a signal +mask in sigsetjmp either, so discarding savesigs retains its existing semantics +while making PG_TRY/PG_CATCH protect the frame at every PostgreSQL call site. --- src/Makefile.shlib | 1 + - src/include/port/wasix-dl.h | 16 +++++++++++++++- + src/include/port/wasix-dl.h | 17 ++++++++++++++++- src/makefiles/pgxs.mk | 5 +++++ - 3 files changed, 22 insertions(+), 1 deletion(-) + 3 files changed, 23 insertions(+), 1 deletion(-) diff --git a/src/Makefile.shlib b/src/Makefile.shlib -index 55129150fa..8340cb90bd 100644 +index 27632c5398..dff1f41736 100644 --- a/src/Makefile.shlib +++ b/src/Makefile.shlib -@@ -197,6 +197,7 @@ ifeq ($(PORTNAME), linux) +@@ -184,6 +184,7 @@ ifeq ($(PORTNAME), linux) endif ifeq ($(PORTNAME), wasix-dl) @@ -36,20 +36,24 @@ index 55129150fa..8340cb90bd 100644 ifdef SO_MAJOR_VERSION shlib = $(shlib_bare) diff --git a/src/include/port/wasix-dl.h b/src/include/port/wasix-dl.h -index ba0d68bb70..aac596375e 100644 +index 8261180367..be319862c1 100644 --- a/src/include/port/wasix-dl.h +++ b/src/include/port/wasix-dl.h -@@ -8,6 +8,21 @@ +@@ -6,8 +6,24 @@ + #ifndef PG_PORT_WASIX_DL_H + #define PG_PORT_WASIX_DL_H + -#ifdef OLIPHAUNT_WASM_SINGLE_USER +#if defined(OLIPHAUNT_WASM_SINGLE_USER) || defined(OLIPHAUNT_WASM_SIDE_MODULE) #include +#endif + -+#if defined(__wasm_exception_handling__) && (defined(OLIPHAUNT_WASM_SINGLE_USER) || defined(OLIPHAUNT_WASM_SIDE_MODULE)) ++#if defined(__wasm_exception_handling__) && \ ++ (defined(OLIPHAUNT_WASM_SINGLE_USER) || defined(OLIPHAUNT_WASM_SIDE_MODULE)) +/* + * WebAssembly SJLJ requires setjmp to be visible at the protected call site. + * The released WASIX EH sysroot implements sigsetjmp as an out-of-line -+ * wrapper, whose exception handler has already returned by the time the ++ * wrapper, whose exception handler has already returned by the time its + * caller invokes siglongjmp. Its implementation ignores savesigs, so this + * call-site expansion preserves the same signal-mask semantics. + */ @@ -61,12 +65,11 @@ index ba0d68bb70..aac596375e 100644 #include #include #include - #include diff --git a/src/makefiles/pgxs.mk b/src/makefiles/pgxs.mk -index 4a1474db44..ab8037f33a 100644 +index 45819eb568..ee5ab1cba0 100644 --- a/src/makefiles/pgxs.mk +++ b/src/makefiles/pgxs.mk -@@ -101,6 +101,12 @@ +@@ -101,6 +101,12 @@ endif # PGXS override CPPFLAGS := -I. -I$(srcdir) $(CPPFLAGS) diff --git a/src/wasix/runtime/assets/build/postgres/patches/0040-oliphaunt-wasix-set-libpq-sockets-nonblocking.patch b/src/wasix/runtime/assets/build/postgres/patches/0040-oliphaunt-wasix-set-libpq-sockets-nonblocking.patch index 26e5891ef..343fbf04f 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0040-oliphaunt-wasix-set-libpq-sockets-nonblocking.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0040-oliphaunt-wasix-set-libpq-sockets-nonblocking.patch @@ -1,4 +1,4 @@ -From: Oliphaunt Maintainers +From: Sid Jain Date: Fri, 21 Aug 2026 13:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: set libpq sockets nonblocking diff --git a/src/wasix/runtime/assets/build/postgres/patches/0041-oliphaunt-wasix-honor-noninteractive-psql-invocations.patch b/src/wasix/runtime/assets/build/postgres/patches/0041-oliphaunt-wasix-honor-noninteractive-psql-invocations.patch index 7eaa2dcf5..194eada04 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0041-oliphaunt-wasix-honor-noninteractive-psql-invocations.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0041-oliphaunt-wasix-honor-noninteractive-psql-invocations.patch @@ -1,4 +1,4 @@ -From: Oliphaunt Maintainers +From: Sid Jain Date: Mon, 24 Aug 2026 00:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: honor noninteractive psql invocations @@ -14,10 +14,10 @@ psql invocations retain PostgreSQL's normal isatty-based behavior. 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/bin/psql/startup.c b/src/bin/psql/startup.c -index b0c07fb..49f7b99 100644 +index 249b6aa516..88e30fc6e2 100644 --- a/src/bin/psql/startup.c +++ b/src/bin/psql/startup.c -@@ -128,6 +128,7 @@ main(int argc, char *argv[]) +@@ -129,6 +129,7 @@ main(int argc, char *argv[]) int successResult; char *password = NULL; bool new_pass; @@ -25,7 +25,7 @@ index b0c07fb..49f7b99 100644 pg_logging_init(argv[0]); pg_logging_set_pre_callback(log_pre_callback); -@@ -184,7 +185,9 @@ main(int argc, char *argv[]) +@@ -183,7 +184,9 @@ main(int argc, char *argv[]) /* We must get COLUMNS here before readline() sets it */ pset.popt.topt.env_columns = getenv("COLUMNS") ? atoi(getenv("COLUMNS")) : 0; @@ -38,3 +38,5 @@ index b0c07fb..49f7b99 100644 -- 2.51.0 +-- +2.51.0 diff --git a/src/wasix/runtime/assets/build/postgres/patches/0042-oliphaunt-wasix-use-explicit-wal-sync-operations.patch b/src/wasix/runtime/assets/build/postgres/patches/0042-oliphaunt-wasix-use-explicit-wal-sync-operations.patch index f2e3a2e5e..a1061ec75 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0042-oliphaunt-wasix-use-explicit-wal-sync-operations.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0042-oliphaunt-wasix-use-explicit-wal-sync-operations.patch @@ -1,4 +1,4 @@ -From: Oliphaunt Maintainers +From: Sid Jain Date: Sun, 30 Aug 2026 00:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: use explicit WAL sync operations diff --git a/src/wasix/runtime/assets/build/postgres/patches/0043-oliphaunt-wasix-cache-jsonb-build-object-metadata.patch b/src/wasix/runtime/assets/build/postgres/patches/0043-oliphaunt-wasix-cache-jsonb-build-object-metadata.patch index f111e88c3..f19b10a06 100644 --- a/src/wasix/runtime/assets/build/postgres/patches/0043-oliphaunt-wasix-cache-jsonb-build-object-metadata.patch +++ b/src/wasix/runtime/assets/build/postgres/patches/0043-oliphaunt-wasix-cache-jsonb-build-object-metadata.patch @@ -1,4 +1,4 @@ -From: Oliphaunt Maintainers +From: Sid Jain Date: Sun, 30 Aug 2026 00:00:00 +0000 Subject: [PATCH] oliphaunt-wasix: cache jsonb_build_object metadata diff --git a/src/wasix/runtime/assets/build/postgres/patches/0044-oliphaunt-wasix-initialize-trusted-session-identity.patch b/src/wasix/runtime/assets/build/postgres/patches/0044-oliphaunt-wasix-initialize-trusted-session-identity.patch new file mode 100644 index 000000000..d0158912b --- /dev/null +++ b/src/wasix/runtime/assets/build/postgres/patches/0044-oliphaunt-wasix-initialize-trusted-session-identity.patch @@ -0,0 +1,271 @@ +From: Sid Jain +Date: Mon, 7 Sep 2026 00:00:00 +0000 +Subject: [PATCH] oliphaunt-wasix: initialize the trusted session principal + +W0029 activated normal-session policy only after standalone InitPostgres +had selected the bootstrap superuser. Consequently the configured PGUSER +was ignored, RESET ROLE returned to postgres, and role defaults and login +policy were bypassed. + +Consume the one-shot prepared capability in InitPostgres, using a regular +backend PGPROC and a protocol port initialized before session admission. +Use PostgreSQL's catalog-backed session authorization, role/database +settings, connection policy and login event triggers. Capture login-trigger +startup errors through the existing typed startup-error boundary. + +The host remains the trust boundary: this is not HBA authentication and no +authentication provider identity is fabricated. Unprepared initdb and +recovery retain genuine standalone behavior. IsUnderPostmaster and +IsPostmasterEnvironment remain truthful, and worker limits remain pinned +before shared-memory sizing. This is embedded lifecycle integration, not a +performance shortcut or a new runtime option. + +--- +diff --git a/src/backend/commands/event_trigger.c b/src/backend/commands/event_trigger.c +index 13e5f1e..3725525 100644 +--- a/src/backend/commands/event_trigger.c ++++ b/src/backend/commands/event_trigger.c +@@ -904,7 +904,7 @@ EventTriggerOnLogin(void) + * triggers are disabled in single user mode or via a GUC. We also need a + * database connection (some background workers don't have it). + */ +- if (!IsUnderPostmaster || !event_triggers || ++ if (!IsNormalUserSession() || !event_triggers || + !OidIsValid(MyDatabaseId) || !MyDatabaseHasLoginEventTriggers) + return; + +diff --git a/src/backend/main/main.c b/src/backend/main/main.c +index a3513ec..ba399b0 100644 +--- a/src/backend/main/main.c ++++ b/src/backend/main/main.c +@@ -83,10 +83,17 @@ IsTrustedEmbeddedSession(void) + OLIPHAUNT_TRUSTED_EMBEDDED_ATTACHED; + } + ++bool ++IsPreparedTrustedEmbeddedSession(void) ++{ ++ return trusted_embedded_lifecycle == ++ OLIPHAUNT_TRUSTED_EMBEDDED_PREPARED; ++} ++ + /* + * Select the embedded topology exactly once and before main(). PREPARED is + * intentionally not an active session: PostgreSQL remains a truthful +- * standalone backend until the host attaches after _start. ++ * standalone backend until InitPostgres consumes the prepared identity. + */ + OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_prepare_trusted_embedded_session") + int +diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c +index 120e4a4..63d7bfb 100644 +--- a/src/backend/tcop/postgres.c ++++ b/src/backend/tcop/postgres.c +@@ -239,9 +239,9 @@ oliphaunt_wasix_init_protocol_port(void) + OLIPHAUNT_WASM_HOST_EXPORT("oliphaunt_wasix_start") void + oliphaunt_wasix_start(void) + { +- if (!AttachPreparedTrustedEmbeddedSession()) ++ if (!IsTrustedEmbeddedSession() || whereToSendOutput == DestRemote) + elog(FATAL, +- "trusted embedded session attach requires exactly one pre-start preparation and safe single-backend settings"); ++ "trusted embedded protocol requires a successfully initialized session and one attachment"); + + oliphaunt_wasix_init_protocol_port(); + whereToSendOutput = DestRemote; +@@ -4237,12 +4237,29 @@ PostgresSingleUserMain(int argc, char *argv[], + const char *username) + { + const char *dbname = NULL; ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ bool trusted_client = IsPreparedTrustedEmbeddedSession(); ++#endif + + Assert(!IsUnderPostmaster); + + /* Initialize startup process environment. */ + InitStandaloneProcess(argv[0]); + ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ if (trusted_client) ++ { ++ /* PGUSER is supplied by the host; initdb/recovery keep the OS user. */ ++ username = getenv("PGUSER"); ++ if (username == NULL || username[0] == '\0') ++ ereport(FATAL, ++ (errcode(ERRCODE_INVALID_PARAMETER_VALUE), ++ errmsg("trusted embedded startup requires a host-selected role"))); ++ /* InitProcess must allocate a regular backend's PGPROC. */ ++ MyBackendType = B_BACKEND; ++ } ++#endif ++ + /* + * Set default values for command-line options. + */ +@@ -4350,6 +4367,14 @@ PostgresSingleUserMain(int argc, char *argv[], + * Now that sufficient infrastructure has been initialized, PostgresMain() + * can do the rest. + */ ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ if (trusted_client) ++ { ++ oliphaunt_wasix_init_protocol_port(); ++ MyProcPort->database_name = MemoryContextStrdup(TopMemoryContext, dbname); ++ MyProcPort->user_name = MemoryContextStrdup(TopMemoryContext, username); ++ } ++#endif + PostgresMain(dbname, username); + } + +@@ -5174,7 +5199,10 @@ PostgresMain(const char *dbname, const char *username) + #endif + InitPostgres(dbname, InvalidOid, /* database to connect to */ + username, InvalidOid, /* role to connect as */ +- (!am_walsender) ? INIT_PG_LOAD_SESSION_LIBS : 0, ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ (IsPreparedTrustedEmbeddedSession() ? INIT_PG_TRUSTED_CLIENT : 0) | ++#endif ++ ((!am_walsender) ? INIT_PG_LOAD_SESSION_LIBS : 0), + NULL); /* no out_dbname */ + #ifdef OLIPHAUNT_WASM_SINGLE_USER + oliphaunt_wasix_end_startup_error_capture(); +@@ -5255,7 +5283,13 @@ PostgresMain(const char *dbname, const char *username) + MemoryContextSwitchTo(TopMemoryContext); + + /* Fire any defined login event triggers, if appropriate */ ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ oliphaunt_wasix_begin_startup_error_capture(); ++#endif + EventTriggerOnLogin(); ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ oliphaunt_wasix_end_startup_error_capture(); ++#endif + + /* + * POSTGRES main processing loop begins here +diff --git a/src/backend/utils/init/miscinit.c b/src/backend/utils/init/miscinit.c +index 6941770..6e9e99a 100644 +--- a/src/backend/utils/init/miscinit.c ++++ b/src/backend/utils/init/miscinit.c +@@ -842,11 +842,10 @@ InitializeSessionUserId(const char *rolename, Oid roleid, + PGC_BACKEND, PGC_S_OVERRIDE); + + /* +- * These next checks are not enforced when in standalone mode, so that +- * there is a way to recover from sillinesses like "UPDATE pg_authid SET +- * rolcanlogin = false;". ++ * Trusted embedded clients obey login policy. Genuine recovery standalone ++ * mode uses InitializeSessionUserIdStandalone() instead. + */ +- if (IsUnderPostmaster) ++ if (IsNormalUserSession()) + { + /* + * Is role allowed to login at all? (But background workers can +diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c +index 53e65b7..e66464e 100644 +--- a/src/backend/utils/init/postinit.c ++++ b/src/backend/utils/init/postinit.c +@@ -351,7 +351,7 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect + * a way to recover from disabling all access to all databases, for + * example "UPDATE pg_database SET datallowconn = false;". + */ +- if (IsUnderPostmaster) ++ if (IsNormalUserSession()) + { + /* + * Check that the database is currently allowing connections. +@@ -677,6 +677,8 @@ BaseInit(void) + * - INIT_PG_LOAD_SESSION_LIBS to honor [session|local]_preload_libraries. + * - INIT_PG_OVERRIDE_ALLOW_CONNS to connect despite !datallowconn. + * - INIT_PG_OVERRIDE_ROLE_LOGIN to connect despite !rolcanlogin. ++ * - INIT_PG_TRUSTED_CLIENT to initialize a prepared WASIX host-selected ++ * identity without HBA authentication, retaining normal login policy. + * out_dbname: optional output parameter, see below; pass NULL if not used + * + * The database can be specified by name, using the in_dbname parameter, or by +@@ -716,6 +718,9 @@ InitPostgres(const char *in_dbname, Oid dboid, + { + bool bootstrap = IsBootstrapProcessingMode(); + bool am_superuser; ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ bool trusted_client = (flags & INIT_PG_TRUSTED_CLIENT) != 0; ++#endif + char *fullpath; + char dbname[NAMEDATALEN]; + int nfree = 0; +@@ -855,6 +860,18 @@ InitPostgres(const char *in_dbname, Oid dboid, + XactIsoLevel = XACT_READ_COMMITTED; + } + ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ if (trusted_client) ++ { ++ if (bootstrap || IsPostmasterEnvironment || IsUnderPostmaster || ++ !AmRegularBackendProcess() || MyProcPort == NULL || ++ !AttachPreparedTrustedEmbeddedSession()) ++ ereport(FATAL, ++ (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), ++ errmsg("trusted client requires a prepared standalone regular backend and protocol port"))); ++ } ++#endif ++ + /* + * Perform client authentication if necessary, then figure out our + * postgres user ID, and see if we are a superuser. +@@ -868,6 +885,14 @@ InitPostgres(const char *in_dbname, Oid dboid, + InitializeSessionUserIdStandalone(); + am_superuser = true; + } ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++ else if (trusted_client) ++ { ++ /* The host selects the principal; no authentication provider ran. */ ++ InitializeSessionUserId(username, useroid, false); ++ am_superuser = superuser(); ++ } ++#endif + else if (!IsUnderPostmaster) + { + InitializeSessionUserIdStandalone(); +@@ -1304,7 +1329,7 @@ void + oliphaunt_wasix_process_startup_options(Port *port) + { + if (port != NULL) +- process_startup_options(port, true); ++ process_startup_options(port, superuser()); + } + #endif + +@@ -1320,7 +1345,7 @@ process_settings(Oid databaseid, Oid roleid) + Relation relsetting; + Snapshot snapshot; + +- if (!IsUnderPostmaster) ++ if (!IsNormalUserSession()) + return; + + relsetting = table_open(DbRoleSettingRelationId, AccessShareLock); +diff --git a/src/include/miscadmin.h b/src/include/miscadmin.h +index b8ce6be..2b00a43 100644 +--- a/src/include/miscadmin.h ++++ b/src/include/miscadmin.h +@@ -170,6 +170,7 @@ extern PGDLLIMPORT bool IsBinaryUpgrade; + + #ifdef OLIPHAUNT_WASM_SINGLE_USER + extern bool IsTrustedEmbeddedSession(void); ++extern bool IsPreparedTrustedEmbeddedSession(void); + extern void ConfigurePreparedTrustedEmbeddedSession(void); + extern bool AttachPreparedTrustedEmbeddedSession(void); + #else +@@ -522,6 +523,9 @@ extern PGDLLIMPORT ProcessingMode Mode; + #define INIT_PG_LOAD_SESSION_LIBS 0x0001 + #define INIT_PG_OVERRIDE_ALLOW_CONNS 0x0002 + #define INIT_PG_OVERRIDE_ROLE_LOGIN 0x0004 ++#ifdef OLIPHAUNT_WASM_SINGLE_USER ++#define INIT_PG_TRUSTED_CLIENT 0x0008 ++#endif + extern void pg_split_opts(char **argv, int *argcp, const char *optstr); + extern void InitializeMaxBackends(void); + extern void InitializeFastPathLocks(void); diff --git a/src/wasix/runtime/assets/build/prepare_postgres_source.sh b/src/wasix/runtime/assets/build/prepare_postgres_source.sh index f215338da..bd220e2bd 100755 --- a/src/wasix/runtime/assets/build/prepare_postgres_source.sh +++ b/src/wasix/runtime/assets/build/prepare_postgres_source.sh @@ -9,6 +9,7 @@ REPO_ROOT="$(oliphaunt_wasix_repo_root "$SCRIPT_DIR")" SOURCE_TOML="$REPO_ROOT/src/third-party/postgres/source.toml" PATCH_DIR="$REPO_ROOT" PATCH_SERIES="$REPO_ROOT/src/wasix/runtime/postgres/series" +CONTRACT_HEADER="$SCRIPT_DIR/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h" read_toml_value() { local key="$1" @@ -78,8 +79,13 @@ fi series_hash="$( { - sha256_text_lf "$PATCH_SERIES" - while IFS= read -r patch_name; do + while IFS= read -r input || [[ -n "$input" ]]; do + input="${input%$'\r'}" + [[ -z "$input" || "$input" =~ ^# ]] && continue + sha256_text_lf "$REPO_ROOT/$input" + done < "$REPO_ROOT/src/wasix/runtime/postgres/source-inputs" + while IFS= read -r patch_name || [[ -n "$patch_name" ]]; do + patch_name="${patch_name%$'\r'}" [[ -z "$patch_name" || "$patch_name" =~ ^# ]] && continue sha256_text_lf "$PATCH_DIR/$patch_name" done < "$PATCH_SERIES" @@ -88,6 +94,7 @@ series_hash="$( new_fingerprint="$PG_VERSION:$PG_SHA256:$series_hash" if [[ -d "$PATCHED_PGSRC" && -f "$FINGERPRINT" && "$(cat "$FINGERPRINT")" == "$new_fingerprint" ]] && ! source_has_patch_artifacts "$PATCHED_PGSRC"; then + install -m 0644 "$CONTRACT_HEADER" "$PATCHED_PGSRC/src/include/port/wasix-dl/oliphaunt_wasix_protocol_contract.generated.h" if [[ ! -f "$SOURCE_FINGERPRINT_FILE" || "$(cat "$SOURCE_FINGERPRINT_FILE")" != "$new_fingerprint" ]]; then printf '%s\n' "$new_fingerprint" > "$SOURCE_FINGERPRINT_FILE" fi @@ -102,7 +109,8 @@ rm -rf "$PATCHED_PGSRC" tar -xjf "$TARBALL" -C "$WORK_ROOT/work" mv "$WORK_ROOT/work/postgresql-$PG_VERSION" "$PATCHED_PGSRC" -bash "$REPO_ROOT/src/third-party/postgres/apply-series.sh" "$PATCHED_PGSRC" "$PATCH_SERIES" --context-fuzz >&2 +bash "$REPO_ROOT/src/third-party/postgres/apply-series.sh" "$PATCHED_PGSRC" "$PATCH_SERIES" >&2 +install -m 0644 "$CONTRACT_HEADER" "$PATCHED_PGSRC/src/include/port/wasix-dl/oliphaunt_wasix_protocol_contract.generated.h" if source_has_patch_artifacts "$PATCHED_PGSRC"; then echo "prepare_postgres_source: patch backup/reject files were left in $PATCHED_PGSRC" >&2 diff --git a/src/wasix/runtime/assets/build/profile_flags.sh b/src/wasix/runtime/assets/build/profile_flags.sh index 85e81c321..06a5315cf 100644 --- a/src/wasix/runtime/assets/build/profile_flags.sh +++ b/src/wasix/runtime/assets/build/profile_flags.sh @@ -1,5 +1,12 @@ #!/usr/bin/env bash +# Link-time guest budgets, not environment overrides. The C stack occupies +# linear memory; it does NOT size Wasmer's native execution/coroutine stack. +# More stack reduces heap headroom; more initial memory increases instance +# footprint, not query throughput. See src/docs/maintainers/runtime-resource-budgets.md. +OLIPHAUNT_WASM_GUEST_STACK_SIZE=8MB +OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE=128MB + oliphaunt_wasix_wasix_profile="${OLIPHAUNT_WASM_BUILD_PROFILE:-release}" case "$oliphaunt_wasix_wasix_profile" in @@ -96,6 +103,8 @@ oliphaunt_wasix_wasix_profile_signature() { printf 'profile=%s\n' "$oliphaunt_wasix_wasix_profile" printf 'cflags=%s\n' "$OLIPHAUNT_WASM_PROFILE_CFLAGS" printf 'ldflags=%s\n' "$OLIPHAUNT_WASM_PROFILE_LDFLAGS" + printf 'guest_stack_size=%s\n' "$OLIPHAUNT_WASM_GUEST_STACK_SIZE" + printf 'initial_memory_size=%s\n' "$OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE" printf 'configure_wasm_opt=%s\n' "$OLIPHAUNT_WASM_WASIX_CONFIGURE_WASM_OPT" printf 'build_wasm_opt=%s\n' "$OLIPHAUNT_WASM_WASIX_BUILD_WASM_OPT" printf 'wasm_opt_flags=%s\n' "$OLIPHAUNT_WASM_WASM_OPT_FLAGS" diff --git a/src/wasix/runtime/assets/build/profile_flags.test.sh b/src/wasix/runtime/assets/build/profile_flags.test.sh new file mode 100644 index 000000000..bc3e19e19 --- /dev/null +++ b/src/wasix/runtime/assets/build/profile_flags.test.sh @@ -0,0 +1,22 @@ +#!/usr/bin/env bash +set -euo pipefail + +root="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +for profile in debug release release-o3 release-os release-oz; do + ( + export OLIPHAUNT_WASM_BUILD_PROFILE="$profile" + # These are owned build constants, not inherited consumer overrides. + export OLIPHAUNT_WASM_GUEST_STACK_SIZE=1MB + export OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE=1MB + . "$root/profile_flags.sh" + [ "$OLIPHAUNT_WASM_GUEST_STACK_SIZE" = 8MB ] + [ "$OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE" = 128MB ] + original="$(oliphaunt_wasix_wasix_profile_signature)" + OLIPHAUNT_WASM_GUEST_STACK_SIZE=16MB + [ "$original" != "$(oliphaunt_wasix_wasix_profile_signature)" ] + OLIPHAUNT_WASM_GUEST_STACK_SIZE=8MB + OLIPHAUNT_WASM_INITIAL_MEMORY_SIZE=256MB + [ "$original" != "$(oliphaunt_wasix_wasix_profile_signature)" ] + ) +done +echo 'WASIX guest memory budget/profile identity: PASS' diff --git a/src/wasix/runtime/assets/build/verify_wasix_sjlj_artifact.sh b/src/wasix/runtime/assets/build/verify_wasix_sjlj_artifact.sh new file mode 100644 index 000000000..7bf6bb588 --- /dev/null +++ b/src/wasix/runtime/assets/build/verify_wasix_sjlj_artifact.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +set -euo pipefail + +fail() { + echo "WASIX SJLJ artifact guard: $*" >&2 + exit 2 +} + +if [ "$#" -ne 3 ]; then + fail "usage: $0 ARTIFACT LLVM_NM WASM_DIS" +fi + +artifact="$1" +symbol_tool="$2" +disassembler="$3" + +[ -f "$artifact" ] && [ ! -L "$artifact" ] || fail "unsafe or missing artifact: $artifact" +[ -x "$symbol_tool" ] || fail "llvm-nm is not executable: $symbol_tool" +[ -x "$disassembler" ] || fail "wasm-dis is not executable: $disassembler" + +if ! undefined_symbols="$("$symbol_tool" -u "$artifact")"; then + fail "llvm-nm could not inspect artifact: $artifact" +fi + +plain_symbols="$({ + printf '%s\n' "$undefined_symbols" | + awk '$NF ~ /^_?(sig)?(setjmp|longjmp)$/ { print $NF }' | + LC_ALL=C sort -u | + paste -sd, - +})" +[ -z "$plain_symbols" ] || + fail "unlowered SJLJ symbols remain in $artifact: $plain_symbols" + +for required_symbol in __c_longjmp __wasm_setjmp_test; do + printf '%s\n' "$undefined_symbols" | + awk -v symbol="$required_symbol" '$NF == symbol { found = 1 } END { exit found ? 0 : 1 }' || + fail "lowered SJLJ symbol is missing from $artifact: $required_symbol" +done + +if ! "$disassembler" "$artifact" -o - 2>/dev/null | awk ' + /\(import "env" "__c_longjmp" \(tag / { tag_import = 1 } + /\(try_table \(catch / { catch_frame = 1 } + END { exit tag_import && catch_frame ? 0 : 1 } +'; then + fail "could not prove a lowered __c_longjmp try_table/catch in artifact: $artifact" +fi diff --git a/src/wasix/runtime/assets/build/verify_wasix_sjlj_artifact.test.sh b/src/wasix/runtime/assets/build/verify_wasix_sjlj_artifact.test.sh new file mode 100644 index 000000000..21e50d057 --- /dev/null +++ b/src/wasix/runtime/assets/build/verify_wasix_sjlj_artifact.test.sh @@ -0,0 +1,82 @@ +#!/usr/bin/env bash +set -euo pipefail + +script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +guard="$script_dir/verify_wasix_sjlj_artifact.sh" +scratch="$(mktemp -d "${TMPDIR:-/tmp}/oliphaunt-sjlj-guard.XXXXXX")" +cleanup() { + rm -rf -- "$scratch" +} +trap cleanup EXIT + +mkdir -p "$scratch/bin" +touch \ + "$scratch/good.wasm" \ + "$scratch/plain-setjmp.wasm" \ + "$scratch/missing-tag-symbol.wasm" \ + "$scratch/missing-test-symbol.wasm" \ + "$scratch/missing-catch.wasm" \ + "$scratch/disassembly-fails.wasm" + +cat >"$scratch/bin/llvm-nm" <<'EOF' +#!/usr/bin/env bash +set -euo pipefail +artifact="${2:?artifact is required}" +case "$artifact" in + *plain-setjmp*) + printf '%s\n' ' U __c_longjmp' ' U __wasm_setjmp_test' ' U setjmp' + ;; + *missing-tag-symbol*) + printf '%s\n' ' U __wasm_setjmp_test' + ;; + *missing-test-symbol*) + printf '%s\n' ' U __c_longjmp' + ;; + *) + printf '%s\n' ' U __c_longjmp' ' U __wasm_setjmp_test' + ;; +esac +EOF + +cat >"$scratch/bin/wasm-dis" <<'EOF' +#!/usr/bin/env bash +set -euo pipefail +artifact="${1:?artifact is required}" +case "$artifact" in + *disassembly-fails*) exit 7 ;; +esac +printf '%s\n' \ + '(module' \ + ' (import "env" "__c_longjmp" (tag $eimport$0 (param i32)))' +case "$artifact" in + *missing-catch*) ;; + *) printf '%s\n' ' (func (try_table (catch $eimport$0 0) (nop)))' ;; +esac +printf '%s\n' ')' +EOF +chmod 0755 "$scratch/bin/llvm-nm" "$scratch/bin/wasm-dis" + +bash "$guard" "$scratch/good.wasm" "$scratch/bin/llvm-nm" "$scratch/bin/wasm-dis" + +expect_failure() { + local artifact="$1" + local expected="$2" + if bash "$guard" "$scratch/$artifact" "$scratch/bin/llvm-nm" "$scratch/bin/wasm-dis" \ + >"$scratch/stdout" 2>"$scratch/stderr"; then + echo "SJLJ artifact guard accepted invalid fixture: $artifact" >&2 + exit 1 + fi + grep -F "$expected" "$scratch/stderr" >/dev/null || { + echo "SJLJ artifact guard emitted the wrong failure for $artifact" >&2 + cat "$scratch/stderr" >&2 + exit 1 + } +} + +expect_failure plain-setjmp.wasm 'unlowered SJLJ symbols remain' +expect_failure missing-tag-symbol.wasm '__c_longjmp' +expect_failure missing-test-symbol.wasm '__wasm_setjmp_test' +expect_failure missing-catch.wasm 'could not prove a lowered __c_longjmp try_table/catch' +expect_failure disassembly-fails.wasm 'could not prove a lowered __c_longjmp try_table/catch' + +echo "WASIX SJLJ artifact guard: PASS" diff --git a/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge.c b/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge.c index f1ab8c5ea..b9f215bf9 100644 --- a/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge.c +++ b/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge.c @@ -34,17 +34,44 @@ #include #include +#include "oliphaunt_wasix_protocol_contract.generated.h" + #ifndef EMSCRIPTEN_KEEPALIVE #define EMSCRIPTEN_KEEPALIVE __attribute__((used)) #endif #define OLIPHAUNT_UID 123 #define OLIPHAUNT_PROTOCOL_FD 1 -#define POSTGRES_MAIN_LONGJMP 100 #define MAX_ATEXIT_FUNCS 32 +#define OLIPHAUNT_WASIX_STARTUP_REJECTED_EXIT 98 +#define OLIPHAUNT_WASIX_STARTUP_OUTCOME_MAX_PROTOCOL_BYTES (1024U * 1024U) +_Static_assert(OLIPHAUNT_WASIX_BUFFERED_PROTOCOL_OUTPUT_LIMIT > 0, + "buffered protocol output limit must be positive"); +_Static_assert((size_t) OLIPHAUNT_WASIX_BUFFERED_PROTOCOL_OUTPUT_LIMIT <= + (size_t) SSIZE_MAX, + "buffered protocol output limit must fit in ssize_t"); + +enum +{ + OLIPHAUNT_WASIX_STARTUP_OUTCOME_ABI_VERSION = 1, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_BYTE_SIZE = 32, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PENDING = 0, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_REJECTED = 1, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_VERSION_OFFSET = 0, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_SIZE_OFFSET = 4, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_KIND_OFFSET = 8, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_RESERVED_OFFSET = 12, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_PTR_OFFSET = 16, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_LEN_OFFSET = 24, +}; + +_Static_assert(CHAR_BIT == 8, "startup outcome ABI requires 8-bit bytes"); +_Static_assert(sizeof(uintptr_t) <= sizeof(uint64_t), + "startup outcome ABI cannot encode this pointer width"); +_Static_assert(sizeof(size_t) <= sizeof(uint64_t), + "startup outcome ABI cannot encode this length width"); volatile int is_oliphaunt_active = 0; -volatile int force_host_error_recovery = 0; volatile int oliphaunt_wasix_startup_error_capture_active = 0; sigjmp_buf postgresmain_sigjmp_buf; volatile bool ignore_till_sync = false; @@ -85,20 +112,22 @@ static size_t oliphaunt_wasix_output_len_value; static size_t oliphaunt_wasix_output_cap; static size_t oliphaunt_wasix_output_scan_off; static bool oliphaunt_wasix_output_contains_error_value; -enum -{ - OLIPHAUNT_WASIX_PROTOCOL_BUFFERED = 0, - OLIPHAUNT_WASIX_PROTOCOL_STREAM = 1, - OLIPHAUNT_WASIX_PROTOCOL_HYBRID = 2, -}; -enum -{ - OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE = 0, - OLIPHAUNT_WASIX_PROTOCOL_COPY_IN = 1, - OLIPHAUNT_WASIX_PROTOCOL_COPY_OUT = 2, - OLIPHAUNT_WASIX_PROTOCOL_COPY_BOTH = 3, -}; - +static int oliphaunt_wasix_output_failure_status; +/* + * This descriptor is a wire ABI, not a native C struct: keeping its storage as + * bytes makes the 32-byte little-endian layout independent of host alignment. + * The address is stable for the lifetime of the module. Rejected protocol + * bytes live in a separate bridge-owned allocation and remain immutable until + * the next startup attempt resets the descriptor. + */ +static volatile unsigned char + oliphaunt_wasix_startup_outcome_descriptor[OLIPHAUNT_WASIX_STARTUP_OUTCOME_BYTE_SIZE] = { + 1, 0, 0, 0, + 32, 0, 0, 0, + }; +static unsigned char *oliphaunt_wasix_startup_outcome_protocol; +static size_t oliphaunt_wasix_startup_outcome_protocol_len; +static size_t oliphaunt_wasix_startup_outcome_protocol_cap; static int oliphaunt_wasix_protocol_transport; static int oliphaunt_wasix_protocol_fd = OLIPHAUNT_PROTOCOL_FD; static int oliphaunt_wasix_protocol_status_flags; @@ -110,15 +139,79 @@ static bool oliphaunt_wasix_protocol_stream_active_value; static void (*atexit_funcs[MAX_ATEXIT_FUNCS])(void); static int atexit_func_count; -int oliphaunt_wasix_set_protocol_transport(int mode); int oliphaunt_wasix_socket(int domain, int type, int protocol); ssize_t oliphaunt_wasix_recv(int fd, void *buf, size_t n, int flags); ssize_t oliphaunt_wasix_send(int fd, const void *buf, size_t n, int flags); +static void +oliphaunt_wasix_startup_outcome_store_u32(size_t offset, uint32_t value) +{ + for (size_t i = 0; i < sizeof(value); i++) + oliphaunt_wasix_startup_outcome_descriptor[offset + i] = + (unsigned char) (value >> (i * CHAR_BIT)); +} + +static void +oliphaunt_wasix_startup_outcome_store_u64(size_t offset, uint64_t value) +{ + for (size_t i = 0; i < sizeof(value); i++) + oliphaunt_wasix_startup_outcome_descriptor[offset + i] = + (unsigned char) (value >> (i * CHAR_BIT)); +} + +static uint32_t +oliphaunt_wasix_startup_outcome_load_u32(size_t offset) +{ + uint32_t value = 0; + for (size_t i = 0; i < sizeof(value); i++) + value |= (uint32_t) oliphaunt_wasix_startup_outcome_descriptor[offset + i] + << (i * CHAR_BIT); + return value; +} + +static uint64_t +oliphaunt_wasix_startup_outcome_load_u64(size_t offset) +{ + uint64_t value = 0; + for (size_t i = 0; i < sizeof(value); i++) + value |= (uint64_t) oliphaunt_wasix_startup_outcome_descriptor[offset + i] + << (i * CHAR_BIT); + return value; +} + +const void *EMSCRIPTEN_KEEPALIVE +oliphaunt_wasix_startup_outcome_v1(void) +{ + return (const void *) (uintptr_t) oliphaunt_wasix_startup_outcome_descriptor; +} + +void +oliphaunt_wasix_startup_outcome_reset(void) +{ + /* Invalidate a prior result before clearing or replacing its payload. */ + oliphaunt_wasix_startup_outcome_store_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_KIND_OFFSET, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PENDING); + oliphaunt_wasix_startup_outcome_store_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_VERSION_OFFSET, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_ABI_VERSION); + oliphaunt_wasix_startup_outcome_store_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_SIZE_OFFSET, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_BYTE_SIZE); + oliphaunt_wasix_startup_outcome_store_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_RESERVED_OFFSET, 0); + oliphaunt_wasix_startup_outcome_store_u64( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_PTR_OFFSET, 0); + oliphaunt_wasix_startup_outcome_store_u64( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_LEN_OFFSET, 0); + oliphaunt_wasix_startup_outcome_protocol_len = 0; +} + int EMSCRIPTEN_KEEPALIVE oliphaunt_wasix_set_protocol_transport(int mode) { - if (mode < OLIPHAUNT_WASIX_PROTOCOL_BUFFERED || mode > OLIPHAUNT_WASIX_PROTOCOL_HYBRID) + if (mode < OLIPHAUNT_WASIX_PROTOCOL_BUFFERED || + mode > OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT) { errno = EINVAL; return -1; @@ -162,14 +255,6 @@ oliphaunt_wasix_protocol_copy_state(void) return oliphaunt_wasix_protocol_copy_state_value; } -int EMSCRIPTEN_KEEPALIVE -oliphaunt_wasix_set_force_host_error_recovery(int new_value) -{ - int current = force_host_error_recovery; - force_host_error_recovery = new_value != 0; - return current; -} - int EMSCRIPTEN_KEEPALIVE oliphaunt_wasix_set_active(int new_value) { @@ -187,18 +272,10 @@ void EMSCRIPTEN_KEEPALIVE oliphaunt_wasix_longjmp(jmp_buf env, int val) { /* - * Some hosts can run nested WebAssembly exception unwinds and can preserve - * PostgreSQL's normal PG_TRY/PG_CATCH behavior. Hosts without that support - * must route every PostgreSQL ERROR longjmp through the existing - * single-user process-exit boundary; Rust then invokes PostgresMainLongJmp() - * to perform the same top-level cleanup and emit the backend ErrorResponse. + * PostgreSQL owns every nested and top-level error boundary. With + * sigsetjmp expanded at each WebAssembly call site, the jump must remain + * inside the guest so PG_CATCH cleanup cannot be skipped by the host. */ - if (is_oliphaunt_active && - (force_host_error_recovery || - env == (void *) postgresmain_sigjmp_buf)) - { - exit(POSTGRES_MAIN_LONGJMP); - } longjmp(env, val); } @@ -313,6 +390,39 @@ oliphaunt_wasix_buffer_read(void *buffer, size_t max_length) return (ssize_t) to_copy; } +static int +oliphaunt_wasix_record_output_failure(int status) +{ + if (status <= 0) + status = EIO; + if (oliphaunt_wasix_output_failure_status == 0) + oliphaunt_wasix_output_failure_status = status; + errno = oliphaunt_wasix_output_failure_status; + return -1; +} + +static ssize_t +oliphaunt_wasix_stream_write(int fd, const void *buffer, size_t length) +{ + if (length == 0) + return 0; + if (buffer == NULL) + return oliphaunt_wasix_record_output_failure(EINVAL); + + ssize_t written = write(fd, buffer, length); + if (written < 0) + return oliphaunt_wasix_record_output_failure(errno); + if (written == 0) + return oliphaunt_wasix_record_output_failure(EIO); + return written; +} + +int +oliphaunt_wasix_output_status(void) +{ + return oliphaunt_wasix_output_failure_status; +} + static int oliphaunt_wasix_flush_output_to_stdio(void) { @@ -323,12 +433,9 @@ oliphaunt_wasix_flush_output_to_stdio(void) oliphaunt_wasix_output_buf + off, oliphaunt_wasix_output_len_value - off); if (written < 0) - return -1; + return oliphaunt_wasix_record_output_failure(errno); if (written == 0) - { - errno = EIO; - return -1; - } + return oliphaunt_wasix_record_output_failure(EIO); off += (size_t) written; } oliphaunt_wasix_output_len_value = 0; @@ -341,6 +448,7 @@ oliphaunt_wasix_output_reset(void) oliphaunt_wasix_output_len_value = 0; oliphaunt_wasix_output_scan_off = 0; oliphaunt_wasix_output_contains_error_value = false; + oliphaunt_wasix_output_failure_status = 0; oliphaunt_wasix_protocol_copy_state_value = OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE; oliphaunt_wasix_protocol_stream_requested = false; return 0; @@ -386,42 +494,201 @@ oliphaunt_wasix_scan_buffered_output(void) } } -static ssize_t -oliphaunt_wasix_buffer_write(const void *buffer, size_t length) +static bool +oliphaunt_wasix_error_response_has_valid_sqlstate(const unsigned char *fields, size_t length) { - if (length == 0) - return 0; - if (buffer == NULL) + size_t offset = 0; + bool contains_sqlstate = false; + + if (fields == NULL || length == 0 || fields[length - 1] != 0) + return false; + while (offset < length - 1) { - errno = EINVAL; - return -1; + unsigned char field_type = fields[offset++]; + const unsigned char *terminator; + size_t value_len; + + if (field_type == 0) + return false; + terminator = memchr(fields + offset, 0, length - offset); + if (terminator == NULL) + return false; + value_len = (size_t) (terminator - (fields + offset)); + if (field_type == 'C') + { + if (value_len != 5) + return false; + contains_sqlstate = true; + } + offset += value_len + 1; + } + return offset == length - 1 && contains_sqlstate; +} + +static bool +oliphaunt_wasix_protocol_is_complete_with_error(const unsigned char *buffer, size_t length) +{ + size_t offset = 0; + bool contains_error = false; + + if (buffer == NULL || length == 0) + return false; + while (offset + 5 <= length) + { + const unsigned char *message = buffer + offset; + size_t body_len = ((size_t) message[1] << 24) | + ((size_t) message[2] << 16) | + ((size_t) message[3] << 8) | + (size_t) message[4]; + if (body_len < 4 || body_len > SIZE_MAX - 1) + return false; + size_t message_len = body_len + 1; + if (message_len > length - offset) + return false; + if (message[0] == 'E') + { + if (!oliphaunt_wasix_error_response_has_valid_sqlstate( + message + 5, body_len - 4)) + return false; + contains_error = true; + } + offset += message_len; } + return offset == length && contains_error; +} - if (length > INT_MAX || oliphaunt_wasix_output_len_value > (size_t) INT_MAX - length) +int +oliphaunt_wasix_startup_outcome_publish_rejected(void) +{ + if (!oliphaunt_wasix_startup_error_capture_active) + { + errno = EPERM; + return -1; + } + if (oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_VERSION_OFFSET) != + OLIPHAUNT_WASIX_STARTUP_OUTCOME_ABI_VERSION || + oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_SIZE_OFFSET) != + OLIPHAUNT_WASIX_STARTUP_OUTCOME_BYTE_SIZE || + oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_RESERVED_OFFSET) != 0) + { + errno = EPROTO; + return -1; + } + if (oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_KIND_OFFSET) != + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PENDING) + { + errno = EALREADY; + return -1; + } + if (oliphaunt_wasix_output_len_value == 0 || + oliphaunt_wasix_output_scan_off != oliphaunt_wasix_output_len_value || + !oliphaunt_wasix_output_contains_error_value || + !oliphaunt_wasix_protocol_is_complete_with_error( + oliphaunt_wasix_output_buf, oliphaunt_wasix_output_len_value)) + { + errno = EPROTO; + return -1; + } + if (oliphaunt_wasix_output_len_value > + OLIPHAUNT_WASIX_STARTUP_OUTCOME_MAX_PROTOCOL_BYTES) { errno = EOVERFLOW; return -1; } + if (oliphaunt_wasix_output_len_value > oliphaunt_wasix_startup_outcome_protocol_cap) + { + unsigned char *protocol = realloc(oliphaunt_wasix_startup_outcome_protocol, + oliphaunt_wasix_output_len_value); + if (protocol == NULL) + return -1; + oliphaunt_wasix_startup_outcome_protocol = protocol; + oliphaunt_wasix_startup_outcome_protocol_cap = oliphaunt_wasix_output_len_value; + } + memcpy(oliphaunt_wasix_startup_outcome_protocol, + oliphaunt_wasix_output_buf, + oliphaunt_wasix_output_len_value); + oliphaunt_wasix_startup_outcome_protocol_len = oliphaunt_wasix_output_len_value; + + /* Publish owned payload metadata before making the result observable. */ + oliphaunt_wasix_startup_outcome_store_u64( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_PTR_OFFSET, + (uint64_t) (uintptr_t) oliphaunt_wasix_startup_outcome_protocol); + oliphaunt_wasix_startup_outcome_store_u64( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_LEN_OFFSET, + (uint64_t) oliphaunt_wasix_startup_outcome_protocol_len); + __atomic_signal_fence(__ATOMIC_RELEASE); + oliphaunt_wasix_startup_outcome_store_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_KIND_OFFSET, + OLIPHAUNT_WASIX_STARTUP_OUTCOME_REJECTED); + return 0; +} + +static bool +oliphaunt_wasix_startup_outcome_is_valid_rejection(void) +{ + uint64_t protocol_ptr = oliphaunt_wasix_startup_outcome_load_u64( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_PTR_OFFSET); + uint64_t protocol_len = oliphaunt_wasix_startup_outcome_load_u64( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_PROTOCOL_LEN_OFFSET); + + return oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_VERSION_OFFSET) == + OLIPHAUNT_WASIX_STARTUP_OUTCOME_ABI_VERSION && + oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_SIZE_OFFSET) == + OLIPHAUNT_WASIX_STARTUP_OUTCOME_BYTE_SIZE && + oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_KIND_OFFSET) == + OLIPHAUNT_WASIX_STARTUP_OUTCOME_REJECTED && + oliphaunt_wasix_startup_outcome_load_u32( + OLIPHAUNT_WASIX_STARTUP_OUTCOME_RESERVED_OFFSET) == 0 && + protocol_ptr == (uint64_t) (uintptr_t) oliphaunt_wasix_startup_outcome_protocol && + protocol_len == (uint64_t) oliphaunt_wasix_startup_outcome_protocol_len && + oliphaunt_wasix_protocol_is_complete_with_error( + oliphaunt_wasix_startup_outcome_protocol, + oliphaunt_wasix_startup_outcome_protocol_len); +} + +static ssize_t +oliphaunt_wasix_buffer_write(const void *buffer, size_t length) +{ + if (oliphaunt_wasix_output_failure_status != 0) + return oliphaunt_wasix_record_output_failure( + oliphaunt_wasix_output_failure_status); + if (length == 0) + return 0; + if (buffer == NULL) + return oliphaunt_wasix_record_output_failure(EINVAL); + + const size_t output_limit = + (size_t) OLIPHAUNT_WASIX_BUFFERED_PROTOCOL_OUTPUT_LIMIT; + if (oliphaunt_wasix_output_len_value > output_limit || + length > output_limit - oliphaunt_wasix_output_len_value) + return oliphaunt_wasix_record_output_failure(EFBIG); + size_t required = oliphaunt_wasix_output_len_value + length; if (required > oliphaunt_wasix_output_cap) { size_t next_cap = oliphaunt_wasix_output_cap ? oliphaunt_wasix_output_cap : 8192; + if (next_cap > output_limit) + next_cap = output_limit; while (next_cap < required) { - if (next_cap > SIZE_MAX / 2) + if (next_cap > output_limit / 2) { - next_cap = required; + next_cap = output_limit; break; } next_cap *= 2; } unsigned char *new_buf = realloc(oliphaunt_wasix_output_buf, next_cap); if (new_buf == NULL) - { - errno = ENOMEM; - return -1; - } + return oliphaunt_wasix_record_output_failure(ENOMEM); oliphaunt_wasix_output_buf = new_buf; oliphaunt_wasix_output_cap = next_cap; } @@ -679,6 +946,8 @@ oliphaunt_wasix_exit(int status) if (oliphaunt_wasix_startup_error_capture_active && status != 0) { oliphaunt_wasix_startup_error_capture_active = 0; + if (oliphaunt_wasix_startup_outcome_is_valid_rejection()) + exit(OLIPHAUNT_WASIX_STARTUP_REJECTED_EXIT); __builtin_trap(); } exit(status); @@ -942,11 +1211,21 @@ oliphaunt_wasix_send(int fd, const void *buf, size_t n, int flags) { if (fd != oliphaunt_wasix_protocol_fd) return send(fd, buf, n, flags); + if (oliphaunt_wasix_output_failure_status != 0) + return oliphaunt_wasix_record_output_failure( + oliphaunt_wasix_output_failure_status); + if (oliphaunt_wasix_protocol_transport == + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT) + { + (void) flags; + return oliphaunt_wasix_stream_write(STDOUT_FILENO, buf, n); + } if (oliphaunt_wasix_protocol_transport == OLIPHAUNT_WASIX_PROTOCOL_STREAM || oliphaunt_wasix_protocol_stream_active_value) { (void) flags; - return write(oliphaunt_wasix_direct_tool_active ? fd : STDOUT_FILENO, buf, n); + return oliphaunt_wasix_stream_write( + oliphaunt_wasix_direct_tool_active ? fd : STDOUT_FILENO, buf, n); } return oliphaunt_wasix_buffer_write(buf, n); } diff --git a/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge_abi_test.c b/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge_abi_test.c index f0eefe5bf..d33bf9fa3 100644 --- a/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge_abi_test.c +++ b/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_bridge_abi_test.c @@ -30,6 +30,8 @@ #include #include +#include "oliphaunt_wasix_protocol_contract.generated.h" + #define CHECK(condition) \ do \ { \ @@ -43,7 +45,6 @@ FILE *oliphaunt_wasix_popen(const char *command, const char *mode); int oliphaunt_wasix_system(const char *command); -int oliphaunt_wasix_set_force_host_error_recovery(int new_value); int oliphaunt_wasix_set_active(int new_value); int oliphaunt_wasix_atexit(void (*function)(void)); void oliphaunt_wasix_run_atexit_funcs(void); @@ -54,22 +55,15 @@ gid_t oliphaunt_wasix_getgid(void); struct passwd *oliphaunt_wasix_getpwuid(uid_t uid); int oliphaunt_wasix_getpwuid_r(uid_t uid, struct passwd *pwd, char *buf, size_t buflen, struct passwd **result); -int oliphaunt_wasix_input_reset(void); -void *oliphaunt_wasix_input_reserve(size_t length); -int oliphaunt_wasix_input_commit(size_t length); -size_t oliphaunt_wasix_input_available(void); -int oliphaunt_wasix_output_reset(void); -size_t oliphaunt_wasix_output_len(void); -const void *oliphaunt_wasix_output_data(void); int oliphaunt_wasix_output_contains_error(void); +const void *oliphaunt_wasix_startup_outcome_v1(void); +void oliphaunt_wasix_startup_outcome_reset(void); +int oliphaunt_wasix_startup_outcome_publish_rejected(void); +extern volatile int oliphaunt_wasix_startup_error_capture_active; int oliphaunt_wasix_fcntl(int fd, int cmd, ...); int oliphaunt_wasix_setsockopt(int fd, int level, int optname, const void *optval, socklen_t optlen); int oliphaunt_wasix_getsockopt(int fd, int level, int optname, void *optval, socklen_t *optlen); int oliphaunt_wasix_getsockname(int fd, struct sockaddr *addr, socklen_t *len); -int oliphaunt_wasix_set_protocol_transport(int mode); -int oliphaunt_wasix_protocol_stream_active(void); -void oliphaunt_wasix_protocol_report_copy_response(int state); -int oliphaunt_wasix_protocol_copy_state(void); ssize_t oliphaunt_wasix_recv(int fd, void *buf, size_t n, int flags); ssize_t oliphaunt_wasix_send(int fd, const void *buf, size_t n, int flags); int oliphaunt_wasix_socket(int domain, int type, int protocol); @@ -95,12 +89,187 @@ pg_encoding_to_char_private(int encoding) static int atexit_counter; +typedef struct OliphauntWasixStartupOutcomeV1 +{ + uint32_t abi_version; + uint32_t byte_size; + uint32_t kind; + uint32_t reserved; + uint64_t protocol_ptr; + uint64_t protocol_len; +} OliphauntWasixStartupOutcomeV1; + +_Static_assert(sizeof(OliphauntWasixStartupOutcomeV1) == 32, + "startup outcome ABI must remain 32 bytes"); +_Static_assert(offsetof(OliphauntWasixStartupOutcomeV1, abi_version) == 0, + "startup outcome ABI version offset changed"); +_Static_assert(offsetof(OliphauntWasixStartupOutcomeV1, byte_size) == 4, + "startup outcome ABI size offset changed"); +_Static_assert(offsetof(OliphauntWasixStartupOutcomeV1, kind) == 8, + "startup outcome ABI kind offset changed"); +_Static_assert(offsetof(OliphauntWasixStartupOutcomeV1, reserved) == 12, + "startup outcome ABI reserved offset changed"); +_Static_assert(offsetof(OliphauntWasixStartupOutcomeV1, protocol_ptr) == 16, + "startup outcome ABI protocol pointer offset changed"); +_Static_assert(offsetof(OliphauntWasixStartupOutcomeV1, protocol_len) == 24, + "startup outcome ABI protocol length offset changed"); + +static uint32_t +load_le32(const unsigned char *bytes) +{ + return (uint32_t) bytes[0] | + ((uint32_t) bytes[1] << 8) | + ((uint32_t) bytes[2] << 16) | + ((uint32_t) bytes[3] << 24); +} + +static uint64_t +load_le64(const unsigned char *bytes) +{ + uint64_t value = 0; + for (size_t i = 0; i < sizeof(value); i++) + value |= (uint64_t) bytes[i] << (i * 8); + return value; +} + static void increment_atexit_counter(void) { atexit_counter++; } +static int +check_send_reaches_stdout(const void *sent_bytes, size_t sent_len, + const void *expected, size_t expected_len) +{ + int capture_fds[2] = {-1, -1}; + int saved_stdout = -1; + unsigned char actual[64]; + size_t actual_len = 0; + ssize_t sent = -1; + int result = 1; + + if (expected_len > sizeof(actual) || pipe(capture_fds) != 0) + goto cleanup; + saved_stdout = dup(STDOUT_FILENO); + if (saved_stdout < 0 || dup2(capture_fds[1], STDOUT_FILENO) < 0) + goto cleanup; + if (close(capture_fds[1]) != 0) + goto cleanup; + capture_fds[1] = -1; + + sent = oliphaunt_wasix_send(1, sent_bytes, sent_len, 0); + if (dup2(saved_stdout, STDOUT_FILENO) < 0) + goto cleanup; + if (close(saved_stdout) != 0) + goto cleanup; + saved_stdout = -1; + + while (actual_len < expected_len) + { + ssize_t count = read(capture_fds[0], actual + actual_len, + expected_len - actual_len); + if (count <= 0) + goto cleanup; + actual_len += (size_t) count; + } + unsigned char trailing; + ssize_t trailing_len = read(capture_fds[0], &trailing, 1); + result = sent == (ssize_t) sent_len && + actual_len == expected_len && + trailing_len == 0 && + memcmp(actual, expected, expected_len) == 0 ? 0 : 1; + +cleanup: + if (saved_stdout >= 0) + { + (void) dup2(saved_stdout, STDOUT_FILENO); + (void) close(saved_stdout); + } + if (capture_fds[0] >= 0) + (void) close(capture_fds[0]); + if (capture_fds[1] >= 0) + (void) close(capture_fds[1]); + return result; +} + +static int +check_send_rejected_without_stdout(const void *sent_bytes, size_t sent_len, + int expected_errno) +{ + int capture_fds[2] = {-1, -1}; + int saved_stdout = -1; + ssize_t sent = -1; + int send_errno = 0; + unsigned char unexpected; + ssize_t captured_len = -1; + int result = 1; + + if (pipe(capture_fds) != 0) + goto cleanup; + saved_stdout = dup(STDOUT_FILENO); + if (saved_stdout < 0 || dup2(capture_fds[1], STDOUT_FILENO) < 0) + goto cleanup; + if (close(capture_fds[1]) != 0) + goto cleanup; + capture_fds[1] = -1; + + errno = 0; + sent = oliphaunt_wasix_send(1, sent_bytes, sent_len, 0); + send_errno = errno; + if (dup2(saved_stdout, STDOUT_FILENO) < 0) + goto cleanup; + if (close(saved_stdout) != 0) + goto cleanup; + saved_stdout = -1; + captured_len = read(capture_fds[0], &unexpected, 1); + result = sent == -1 && send_errno == expected_errno && captured_len == 0 ? 0 : 1; + +cleanup: + if (saved_stdout >= 0) + { + (void) dup2(saved_stdout, STDOUT_FILENO); + (void) close(saved_stdout); + } + if (capture_fds[0] >= 0) + (void) close(capture_fds[0]); + if (capture_fds[1] >= 0) + (void) close(capture_fds[1]); + return result; +} + +static int +check_stream_write_failure_is_sticky(int mode) +{ + int saved_stdout = -1; + ssize_t failed_send = -1; + int failed_errno = 0; + + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport(mode) >= 0); + saved_stdout = dup(STDOUT_FILENO); + CHECK(saved_stdout >= 0); + CHECK(close(STDOUT_FILENO) == 0); + errno = 0; + failed_send = oliphaunt_wasix_send(1, "x", 1, 0); + failed_errno = errno; + CHECK(dup2(saved_stdout, STDOUT_FILENO) == STDOUT_FILENO); + CHECK(close(saved_stdout) == 0); + + CHECK(failed_send == -1); + CHECK(failed_errno == EBADF); + CHECK(oliphaunt_wasix_output_status() == EBADF); + CHECK(check_send_rejected_without_stdout("y", 1, EBADF) == 0); + CHECK(oliphaunt_wasix_output_status() == EBADF); + + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); + CHECK(check_send_reaches_stdout("z", 1, "z", 1) == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == mode); + return 0; +} + static int check_locale_pipe(void) { @@ -169,8 +338,6 @@ check_identity_and_fail_closed_calls(void) CHECK(oliphaunt_wasix_system("echo unsafe") == -1); CHECK(errno == ENOSYS); - CHECK(oliphaunt_wasix_set_force_host_error_recovery(1) == 0); - CHECK(oliphaunt_wasix_set_force_host_error_recovery(0) == 1); CHECK(oliphaunt_wasix_set_active(1) == 0); CHECK(oliphaunt_wasix_set_active(0) == 1); CHECK(oliphaunt_wasix_atexit(increment_atexit_counter) == 0); @@ -198,6 +365,7 @@ check_protocol_socket(void) CHECK(oliphaunt_wasix_input_reset() == 0); CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); CHECK(oliphaunt_wasix_recv(1, buf, sizeof(buf), 0) == 0); void *input_buffer = oliphaunt_wasix_input_reserve(sizeof(input) - 1); CHECK(input_buffer != NULL); @@ -238,28 +406,119 @@ check_protocol_socket(void) CHECK(errno == EOVERFLOW); errno = 0; CHECK(oliphaunt_wasix_send(1, output, (size_t) INT_MAX + 1, 0) == -1); - CHECK(errno == EOVERFLOW); + CHECK(errno == EFBIG); + CHECK(oliphaunt_wasix_output_len() == 0); + CHECK(oliphaunt_wasix_output_status() == EFBIG); + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); - CHECK(oliphaunt_wasix_set_protocol_transport(0) == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); CHECK(oliphaunt_wasix_protocol_stream_active() == 0); - CHECK(oliphaunt_wasix_set_protocol_transport(1) == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_STREAM) == OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); CHECK(oliphaunt_wasix_protocol_stream_active() == 1); - CHECK(oliphaunt_wasix_set_protocol_transport(0) == 1); + CHECK(check_send_reaches_stdout(output, sizeof(output) - 1, + output, sizeof(output) - 1) == 0); + CHECK(oliphaunt_wasix_output_len() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == OLIPHAUNT_WASIX_PROTOCOL_STREAM); CHECK(oliphaunt_wasix_protocol_stream_active() == 0); - CHECK(oliphaunt_wasix_set_protocol_transport(2) == 0); + + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_HYBRID) == OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); CHECK(oliphaunt_wasix_protocol_stream_active() == 0); - CHECK(oliphaunt_wasix_protocol_copy_state() == 0); - oliphaunt_wasix_protocol_report_copy_response(1); - CHECK(oliphaunt_wasix_protocol_copy_state() == 1); - CHECK(oliphaunt_wasix_send(1, output, sizeof(output) - 1, 0) == (ssize_t) (sizeof(output) - 1)); + CHECK(oliphaunt_wasix_protocol_copy_state() == OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE); + CHECK(oliphaunt_wasix_send(1, "a", 1, 0) == 1); + CHECK(oliphaunt_wasix_output_len() == 1); + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_IN); + CHECK(oliphaunt_wasix_protocol_copy_state() == OLIPHAUNT_WASIX_PROTOCOL_COPY_IN); + const char hybrid_transition_output[] = "axyz"; + CHECK(check_send_reaches_stdout(output, sizeof(output) - 1, + hybrid_transition_output, + sizeof(hybrid_transition_output) - 1) == 0); CHECK(oliphaunt_wasix_protocol_stream_active() == 1); - CHECK(oliphaunt_wasix_set_protocol_transport(0) == 2); + CHECK(oliphaunt_wasix_output_len() == 0); + CHECK(check_send_reaches_stdout(output, sizeof(output) - 1, + output, sizeof(output) - 1) == 0); + CHECK(oliphaunt_wasix_output_len() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == OLIPHAUNT_WASIX_PROTOCOL_HYBRID); CHECK(oliphaunt_wasix_protocol_stream_active() == 0); - CHECK(oliphaunt_wasix_protocol_copy_state() == 0); - CHECK(oliphaunt_wasix_set_protocol_transport(2) == 0); - oliphaunt_wasix_protocol_report_copy_response(0); - CHECK(oliphaunt_wasix_protocol_copy_state() == 0); - CHECK(oliphaunt_wasix_set_protocol_transport(0) == 2); + CHECK(oliphaunt_wasix_protocol_copy_state() == OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_HYBRID) == OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE); + CHECK(oliphaunt_wasix_protocol_copy_state() == OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == OLIPHAUNT_WASIX_PROTOCOL_HYBRID); + + /* A failed hybrid buffer-to-stdio transition is sticky and preserves bytes. */ + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_HYBRID) == OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); + CHECK(oliphaunt_wasix_send(1, "a", 1, 0) == 1); + oliphaunt_wasix_protocol_report_copy_response(OLIPHAUNT_WASIX_PROTOCOL_COPY_IN); + int saved_stdout = dup(STDOUT_FILENO); + CHECK(saved_stdout >= 0); + CHECK(close(STDOUT_FILENO) == 0); + errno = 0; + ssize_t hybrid_failed_send = oliphaunt_wasix_send(1, "b", 1, 0); + int hybrid_failure_errno = errno; + int restore_stdout_result = dup2(saved_stdout, STDOUT_FILENO); + int close_saved_stdout_result = close(saved_stdout); + CHECK(restore_stdout_result == STDOUT_FILENO); + CHECK(close_saved_stdout_result == 0); + CHECK(hybrid_failed_send == -1); + CHECK(hybrid_failure_errno == EBADF); + CHECK(oliphaunt_wasix_output_status() == EBADF); + CHECK(oliphaunt_wasix_output_len() == 2); + CHECK(memcmp(oliphaunt_wasix_output_data(), "ab", 2) == 0); + errno = 0; + CHECK(oliphaunt_wasix_send(1, "c", 1, 0) == -1); + CHECK(errno == EBADF); + CHECK(oliphaunt_wasix_output_len() == 2); + CHECK(memcmp(oliphaunt_wasix_output_data(), "ab", 2) == 0); + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == OLIPHAUNT_WASIX_PROTOCOL_HYBRID); + + /* Mode 3 keeps request input buffered while streaming every response byte. */ + CHECK(oliphaunt_wasix_input_reset() == 0); + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT) == + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); + CHECK(oliphaunt_wasix_protocol_stream_active() == 0); + input_buffer = oliphaunt_wasix_input_reserve(1); + CHECK(input_buffer != NULL); + memcpy(input_buffer, "q", 1); + CHECK(oliphaunt_wasix_input_commit(1) == 1); + struct pollfd mode3_poll = { + .fd = 1, + .events = POLLIN | POLLOUT, + .revents = 0, + }; + CHECK(oliphaunt_wasix_poll(&mode3_poll, 1, 0) == 1); + CHECK((mode3_poll.revents & POLLIN) != 0); + CHECK((mode3_poll.revents & POLLOUT) != 0); + memset(buf, 0, sizeof(buf)); + CHECK(oliphaunt_wasix_recv(1, buf, 1, 0) == 1); + CHECK(buf[0] == 'q'); + CHECK(check_send_reaches_stdout(output, sizeof(output) - 1, + output, sizeof(output) - 1) == 0); + CHECK(oliphaunt_wasix_output_len() == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT); + CHECK(oliphaunt_wasix_protocol_stream_active() == 0); + CHECK(check_stream_write_failure_is_sticky( + OLIPHAUNT_WASIX_PROTOCOL_STREAM) == 0); + CHECK(check_stream_write_failure_is_sticky( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT) == 0); + errno = 0; CHECK(oliphaunt_wasix_set_protocol_transport(99) == -1); CHECK(errno == EINVAL); @@ -332,6 +591,224 @@ check_protocol_socket(void) return 0; } +static int +check_startup_outcome(void) +{ + const unsigned char notice_then_error[] = { + 'N', 0, 0, 0, 5, 0, + 'E', 0, 0, 0, 12, 'C', '3', 'D', '0', '0', '0', 0, 0, + }; + const unsigned char ready[] = {'Z', 0, 0, 0, 5, 'I'}; + const unsigned char malformed_error[] = {'E', 0, 0, 0, 5, 0}; + const unsigned char incomplete_error[] = { + 'E', 0, 0, 0, 12, 'C', '3', 'D', '0', '0', '0', 0, + }; + const unsigned char complete_error_with_incomplete_tail[] = { + 'E', 0, 0, 0, 12, 'C', '3', 'D', '0', '0', '0', 0, 0, + 'N', 0, 0, 0, 5, + }; + + const unsigned char *descriptor = oliphaunt_wasix_startup_outcome_v1(); + CHECK(descriptor != NULL); + CHECK(oliphaunt_wasix_startup_outcome_v1() == descriptor); + oliphaunt_wasix_startup_outcome_reset(); + CHECK(load_le32(descriptor + 0) == 1); + CHECK(load_le32(descriptor + 4) == 32); + CHECK(load_le32(descriptor + 8) == 0); + CHECK(load_le32(descriptor + 12) == 0); + CHECK(load_le64(descriptor + 16) == 0); + CHECK(load_le64(descriptor + 24) == 0); + + /* Publication is impossible outside the narrow InitPostgres capture. */ + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_send(1, notice_then_error, sizeof(notice_then_error), 0) == + (ssize_t) sizeof(notice_then_error)); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EPERM); + CHECK(load_le32(descriptor + 8) == 0); + + oliphaunt_wasix_startup_error_capture_active = 1; + CHECK(oliphaunt_wasix_output_reset() == 0); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EPROTO); + + CHECK(oliphaunt_wasix_send(1, ready, sizeof(ready), 0) == (ssize_t) sizeof(ready)); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EPROTO); + + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_send(1, malformed_error, sizeof(malformed_error), 0) == + (ssize_t) sizeof(malformed_error)); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EPROTO); + + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_send(1, incomplete_error, sizeof(incomplete_error), 0) == + (ssize_t) sizeof(incomplete_error)); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EPROTO); + + /* A complete ErrorResponse is insufficient when trailing output is partial. */ + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_send(1, + complete_error_with_incomplete_tail, + sizeof(complete_error_with_incomplete_tail), + 0) == (ssize_t) sizeof(complete_error_with_incomplete_tail)); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EPROTO); + + /* Complete preceding frames are retained with the ErrorResponse. */ + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_send(1, notice_then_error, sizeof(notice_then_error), 0) == + (ssize_t) sizeof(notice_then_error)); + const void *bridge_output = oliphaunt_wasix_output_data(); + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == 0); + CHECK(load_le32(descriptor + 0) == 1); + CHECK(load_le32(descriptor + 4) == 32); + CHECK(load_le32(descriptor + 8) == 1); + CHECK(load_le32(descriptor + 12) == 0); + CHECK(load_le64(descriptor + 24) == sizeof(notice_then_error)); + const unsigned char *owned_protocol = + (const unsigned char *) (uintptr_t) load_le64(descriptor + 16); + CHECK(owned_protocol != NULL); + CHECK(owned_protocol != bridge_output); + CHECK(memcmp(owned_protocol, notice_then_error, sizeof(notice_then_error)) == 0); + + /* Later bridge output mutations cannot invalidate the published snapshot. */ + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_send(1, ready, sizeof(ready), 0) == (ssize_t) sizeof(ready)); + CHECK(memcmp(owned_protocol, notice_then_error, sizeof(notice_then_error)) == 0); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EALREADY); + + oliphaunt_wasix_startup_outcome_reset(); + CHECK(oliphaunt_wasix_startup_outcome_v1() == descriptor); + CHECK(load_le32(descriptor + 8) == 0); + CHECK(load_le64(descriptor + 16) == 0); + CHECK(load_le64(descriptor + 24) == 0); + + /* A corrupt or unexpectedly huge startup error cannot force a giant snapshot. */ + const size_t oversized_len = (1024U * 1024U) + 1U; + if ((size_t) OLIPHAUNT_WASIX_BUFFERED_PROTOCOL_OUTPUT_LIMIT >= oversized_len) + { + unsigned char *oversized = malloc(oversized_len); + CHECK(oversized != NULL); + memset(oversized, 'x', oversized_len); + oversized[0] = 'E'; + const uint32_t oversized_body_len = (uint32_t) (oversized_len - 1); + oversized[1] = (unsigned char) (oversized_body_len >> 24); + oversized[2] = (unsigned char) (oversized_body_len >> 16); + oversized[3] = (unsigned char) (oversized_body_len >> 8); + oversized[4] = (unsigned char) oversized_body_len; + oversized[5] = 'C'; + memcpy(oversized + 6, "3D000", 5); + oversized[11] = 0; + oversized[12] = 'M'; + oversized[oversized_len - 2] = 0; + oversized[oversized_len - 1] = 0; + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_send(1, oversized, oversized_len, 0) == + (ssize_t) oversized_len); + errno = 0; + CHECK(oliphaunt_wasix_startup_outcome_publish_rejected() == -1); + CHECK(errno == EOVERFLOW); + CHECK(load_le32(descriptor + 8) == 0); + free(oversized); + } + + oliphaunt_wasix_startup_error_capture_active = 0; + CHECK(oliphaunt_wasix_output_reset() == 0); + return 0; +} + +static int +check_buffered_protocol_output_limit(void) +{ + const size_t limit = + (size_t) OLIPHAUNT_WASIX_BUFFERED_PROTOCOL_OUTPUT_LIMIT; + /* Prove ordinary large responses without allocating the whole i32 range. */ + const size_t response_size = 80U * 1024U * 1024U; + unsigned char chunk[4096]; + size_t written = 0; + + memset(chunk, 'x', sizeof(chunk)); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); + errno = 0; + CHECK(oliphaunt_wasix_send(1, NULL, 1, 0) == -1); + CHECK(errno == EINVAL); + CHECK(oliphaunt_wasix_output_status() == EINVAL); + CHECK(oliphaunt_wasix_output_len() == 0); + errno = 0; + CHECK(oliphaunt_wasix_send(1, "x", 1, 0) == -1); + CHECK(errno == EINVAL); + CHECK(oliphaunt_wasix_output_len() == 0); + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); + CHECK(oliphaunt_wasix_send(1, "x", 1, 0) == 1); + const unsigned char *single_byte_output = oliphaunt_wasix_output_data(); + errno = 0; + CHECK(oliphaunt_wasix_send(1, "y", SIZE_MAX, 0) == -1); + CHECK(errno == EFBIG); + CHECK(oliphaunt_wasix_output_status() == EFBIG); + CHECK(oliphaunt_wasix_output_len() == 1); + CHECK(oliphaunt_wasix_output_data() == single_byte_output); + CHECK(single_byte_output[0] == 'x'); + errno = 0; + CHECK(oliphaunt_wasix_send(1, NULL, 1, 0) == -1); + CHECK(errno == EFBIG); + CHECK(oliphaunt_wasix_output_len() == 1); + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); + + CHECK(response_size < limit); + while (written < response_size) + { + size_t remaining = response_size - written; + size_t count = remaining < sizeof(chunk) ? remaining : sizeof(chunk); + CHECK(oliphaunt_wasix_send(1, chunk, count, 0) == (ssize_t) count); + written += count; + } + CHECK(oliphaunt_wasix_output_len() == response_size); + const unsigned char *output = oliphaunt_wasix_output_data(); + CHECK(output != NULL); + CHECK(output[0] == 'x'); + CHECK(output[response_size - 1] == 'x'); + + errno = 0; + CHECK(oliphaunt_wasix_send(1, "y", limit - response_size + 1, 0) == -1); + CHECK(errno == EFBIG); + CHECK(oliphaunt_wasix_output_status() == EFBIG); + CHECK(oliphaunt_wasix_output_len() == response_size); + CHECK(oliphaunt_wasix_output_data() == output); + CHECK(output[0] == 'x'); + CHECK(output[response_size - 1] == 'x'); + + /* Sticky failure is terminal across modes; a reset restores true mode-3 streaming. */ + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT) == + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED); + CHECK(check_send_rejected_without_stdout("z", 1, EFBIG) == 0); + CHECK(oliphaunt_wasix_output_status() == EFBIG); + CHECK(oliphaunt_wasix_output_len() == response_size); + CHECK(oliphaunt_wasix_output_reset() == 0); + CHECK(oliphaunt_wasix_output_status() == 0); + CHECK(check_send_reaches_stdout("z", 1, "z", 1) == 0); + CHECK(oliphaunt_wasix_set_protocol_transport( + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED) == + OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT); + return 0; +} + static int check_memory_and_shared_memory(void) { @@ -424,6 +901,8 @@ main(void) CHECK(check_locale_pipe() == 0); CHECK(check_identity_and_fail_closed_calls() == 0); CHECK(check_protocol_socket() == 0); + CHECK(check_buffered_protocol_output_limit() == 0); + CHECK(check_startup_outcome() == 0); CHECK(check_memory_and_shared_memory() == 0); CHECK(check_direct_tool_transport() == 0); return 0; diff --git a/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h b/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h new file mode 100644 index 000000000..ab55935af --- /dev/null +++ b/src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h @@ -0,0 +1,34 @@ +/* Generated by src/wasix/runtime/protocol-contract/generate.mjs. */ +#ifndef OLIPHAUNT_WASIX_PROTOCOL_TRANSPORT_CONTRACT_GENERATED_H +#define OLIPHAUNT_WASIX_PROTOCOL_TRANSPORT_CONTRACT_GENERATED_H + +#include + +#define OLIPHAUNT_WASIX_BUFFERED_PROTOCOL_OUTPUT_LIMIT 2147483647U +#define OLIPHAUNT_WASIX_PROTOCOL_CALLBACK_CHUNK_MAX 65536U + +#define OLIPHAUNT_WASIX_PROTOCOL_BUFFERED 0 +#define OLIPHAUNT_WASIX_PROTOCOL_STREAM 1 +#define OLIPHAUNT_WASIX_PROTOCOL_HYBRID 2 +#define OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT 3 + +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE 0 +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_IN 1 +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_OUT 2 +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_BOTH 3 + +int oliphaunt_wasix_set_protocol_transport(int mode); +int oliphaunt_wasix_protocol_stream_active(void); +void oliphaunt_wasix_protocol_report_copy_response(int state); +int oliphaunt_wasix_protocol_copy_state(void); +int oliphaunt_wasix_input_reset(void); +void *oliphaunt_wasix_input_reserve(size_t length); +int oliphaunt_wasix_input_commit(size_t length); +size_t oliphaunt_wasix_input_available(void); +int oliphaunt_wasix_output_reset(void); +size_t oliphaunt_wasix_output_len(void); +const void *oliphaunt_wasix_output_data(void); +int oliphaunt_wasix_output_status(void); +int oliphaunt_wasix_pq_flush(void); + +#endif diff --git a/src/wasix/runtime/assets/generated/wasix-dl.exports b/src/wasix/runtime/assets/generated/wasix-dl.exports index 456df7e2b..84d4bf303 100644 --- a/src/wasix/runtime/assets/generated/wasix-dl.exports +++ b/src/wasix/runtime/assets/generated/wasix-dl.exports @@ -245,7 +245,6 @@ ParseDateTime ParseFuncOrColumn PinPortal PopActiveSnapshot -PostgresMainLongJmp PostgresMainLoopOnce PostgresSendReadyForQueryIfNecessary ProcDiePending @@ -907,13 +906,15 @@ oliphaunt_wasix_output_contains_error oliphaunt_wasix_output_data oliphaunt_wasix_output_len oliphaunt_wasix_output_reset +oliphaunt_wasix_output_status oliphaunt_wasix_pq_flush +oliphaunt_wasix_prepare_trusted_embedded_session oliphaunt_wasix_protocol_stream_active oliphaunt_wasix_send_conn_data oliphaunt_wasix_set_active -oliphaunt_wasix_set_force_host_error_recovery oliphaunt_wasix_set_protocol_transport oliphaunt_wasix_start +oliphaunt_wasix_startup_outcome_v1 op_hashjoinable op_mergejoinable open diff --git a/src/wasix/runtime/crates/aot/aarch64-apple-darwin/build-support.rs b/src/wasix/runtime/crates/aot/aarch64-apple-darwin/build-support.rs index 97c8715fa..7de9cfe54 100644 --- a/src/wasix/runtime/crates/aot/aarch64-apple-darwin/build-support.rs +++ b/src/wasix/runtime/crates/aot/aarch64-apple-darwin/build-support.rs @@ -225,6 +225,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/runtime/crates/aot/aarch64-unknown-linux-gnu/build-support.rs b/src/wasix/runtime/crates/aot/aarch64-unknown-linux-gnu/build-support.rs index 97c8715fa..7de9cfe54 100644 --- a/src/wasix/runtime/crates/aot/aarch64-unknown-linux-gnu/build-support.rs +++ b/src/wasix/runtime/crates/aot/aarch64-unknown-linux-gnu/build-support.rs @@ -225,6 +225,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/runtime/crates/aot/x86_64-pc-windows-msvc/build-support.rs b/src/wasix/runtime/crates/aot/x86_64-pc-windows-msvc/build-support.rs index 97c8715fa..7de9cfe54 100644 --- a/src/wasix/runtime/crates/aot/x86_64-pc-windows-msvc/build-support.rs +++ b/src/wasix/runtime/crates/aot/x86_64-pc-windows-msvc/build-support.rs @@ -225,6 +225,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/runtime/crates/aot/x86_64-unknown-linux-gnu/build-support.rs b/src/wasix/runtime/crates/aot/x86_64-unknown-linux-gnu/build-support.rs index 97c8715fa..7de9cfe54 100644 --- a/src/wasix/runtime/crates/aot/x86_64-unknown-linux-gnu/build-support.rs +++ b/src/wasix/runtime/crates/aot/x86_64-unknown-linux-gnu/build-support.rs @@ -225,6 +225,11 @@ fn write_core_aot_manifest(source: &Path, destination: &Path) -> Vec { let text = fs::read_to_string(source).expect("read generated WASIX AOT manifest"); let mut manifest: serde_json::Value = serde_json::from_str(&text).expect("parse generated WASIX AOT manifest"); + assert_eq!( + manifest.get("engine").and_then(serde_json::Value::as_str), + Some("llvm-opta"), + "stale WASIX AOT profile; rebuild artifacts before compiling the carrier" + ); let artifacts = manifest .get_mut("artifacts") .and_then(|value| value.as_array_mut()) diff --git a/src/wasix/runtime/moon.yml b/src/wasix/runtime/moon.yml index e06430673..d1b68546f 100644 --- a/src/wasix/runtime/moon.yml +++ b/src/wasix/runtime/moon.yml @@ -107,12 +107,30 @@ tasks: tags: - quality - unit - script: "set -e\nbash src/wasix/runtime/tools/check-shim-abi.sh\nbash src/wasix/runtime/assets/build/docker/install-pinned-apt-packages.test.sh\nbash src/wasix/runtime/assets/build/docker/install-pinned-wasixcc.test.sh\n" + script: | + set -e + node src/wasix/runtime/protocol-contract/generate.mjs --check + node --test src/wasix/runtime/protocol-contract/generate.test.mjs + bash src/wasix/runtime/tools/check-shim-abi.sh + bash src/wasix/runtime/assets/build/profile_flags.test.sh + bash src/wasix/runtime/assets/build/verify_wasix_sjlj_artifact.test.sh + node --test src/wasix/runtime/tools/strong-random.test.mjs src/third-party/postgres/apply-series.test.mjs + bash src/wasix/runtime/assets/build/docker/install-pinned-apt-packages.test.sh + bash src/wasix/runtime/assets/build/docker/install-pinned-wasixcc.test.sh inputs: - /tools/dev/acquisition.sh - assets/build/docker/**/* - assets/build/wasix_shim/**/* - tools/check-shim-abi.sh + - tools/strong-random.test.mjs + - protocol-contract/**/* + - /src/wasix/sdks/ts/src/protocol-limits.generated.ts + - /src/wasix/sdks/rust/src/oliphaunt/protocol_limits_generated.rs + - /src/wasix/browser-host/protocol-contract.generated.rs + - assets/build/profile_flags* + - assets/build/verify_wasix_sjlj_artifact* + - /src/third-party/postgres/apply-series* + - /src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-use-checked-getrandom.patch options: cache: true runFromWorkspaceRoot: true diff --git a/src/wasix/runtime/postgres/series b/src/wasix/runtime/postgres/series index aa37de82a..2810d4f23 100644 --- a/src/wasix/runtime/postgres/series +++ b/src/wasix/runtime/postgres/series @@ -11,34 +11,22 @@ src/wasix/runtime/assets/build/postgres/patches/0009-oliphaunt-wasix-route-proce src/wasix/runtime/assets/build/postgres/patches/0010-oliphaunt-wasix-route-sysv-shmem-through-port.patch src/wasix/runtime/assets/build/postgres/patches/0011-oliphaunt-wasix-prefer-posix-semaphores.patch src/wasix/runtime/assets/build/postgres/patches/0012-oliphaunt-wasix-capture-startup-errors.patch -src/wasix/runtime/assets/build/postgres/patches/0013-oliphaunt-wasix-fail-active-portals-on-host-recovery.patch -src/third-party/postgres/patches/wasix/0014-oliphaunt-wasix-speed-up-hash-bytes-unaligned-loads.patch -src/third-party/postgres/patches/wasix/0015-oliphaunt-wasix-add-top-xid-current-transaction-fast-path.patch -src/third-party/postgres/patches/wasix/0016-oliphaunt-wasix-add-btree-int4-compare-fast-path.patch -src/wasix/runtime/assets/build/postgres/patches/0017-oliphaunt-wasix-keep-btree-delete-scratch-on-stack.patch src/third-party/postgres/patches/wasix/0018-oliphaunt-wasix-avoid-pg-dump-executequery-lto-collision.patch -src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-host-recovery.patch -src/wasix/runtime/assets/build/postgres/patches/0020-oliphaunt-wasix-rearm-exception-stack-after-host-recovery.patch +src/wasix/runtime/assets/build/postgres/patches/0019-oliphaunt-wasix-schedule-ready-after-loop-step-recovery.patch src/wasix/runtime/assets/build/postgres/patches/0021-oliphaunt-wasix-declare-wasix-fork.patch src/wasix/runtime/assets/build/postgres/patches/0022-oliphaunt-wasix-use-wasm-ld-for-backend-core.patch src/wasix/runtime/assets/build/postgres/patches/0023-oliphaunt-wasix-skip-data-dir-ownership-check-under-embedded-wasix.patch -src/third-party/postgres/patches/wasix/0024-oliphaunt-wasix-add-like-literal-substring-fast-path.patch src/wasix/runtime/assets/build/postgres/patches/0025-oliphaunt-wasix-stub-pg-dump-parallel-fork.patch -src/third-party/postgres/patches/wasix/0026-oliphaunt-wasix-add-first-int4-leaf-compare-fast-path.patch -src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-avoid-xlog-size-checkpoint-requests.patch -src/wasix/runtime/assets/build/postgres/patches/0028-oliphaunt-wasix-use-lightweight-embedded-runtime-paths.patch -src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-set-embedded-postmaster-environment.patch -src/wasix/runtime/assets/build/postgres/patches/0030-oliphaunt-wasix-avoid-xlogwrite-prevseg-division.patch -src/wasix/runtime/assets/build/postgres/patches/0031-oliphaunt-wasix-skip-activity-id-reporting.patch +src/wasix/runtime/assets/build/postgres/patches/0027-oliphaunt-wasix-defer-xlog-size-checkpoint-requests.patch +src/wasix/runtime/assets/build/postgres/patches/0029-oliphaunt-wasix-model-trusted-embedded-session.patch src/wasix/runtime/assets/build/postgres/patches/0032-oliphaunt-wasix-treat-directory-fsync-eisdir-as-unsupported.patch src/third-party/postgres/patches/common/control-initdb-collation-discovery.patch src/wasix/runtime/assets/build/postgres/patches/0034-oliphaunt-wasix-declare-hybrid-protocol-transport.patch -src/wasix/runtime/assets/build/postgres/patches/0035-oliphaunt-wasix-use-single-backend-spinlocks.patch -src/wasix/runtime/assets/build/postgres/patches/0036-oliphaunt-wasix-specialize-single-backend-atomics.patch -src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-buffer-strong-random.patch +src/third-party/postgres/patches/wasix/0037-oliphaunt-wasix-use-checked-getrandom.patch src/wasix/runtime/assets/build/postgres/patches/0038-oliphaunt-wasix-disable-unsupported-writeback-hints.patch src/wasix/runtime/assets/build/postgres/patches/0039-oliphaunt-wasix-inline-sigsetjmp.patch src/wasix/runtime/assets/build/postgres/patches/0040-oliphaunt-wasix-set-libpq-sockets-nonblocking.patch src/wasix/runtime/assets/build/postgres/patches/0041-oliphaunt-wasix-honor-noninteractive-psql-invocations.patch src/wasix/runtime/assets/build/postgres/patches/0042-oliphaunt-wasix-use-explicit-wal-sync-operations.patch src/wasix/runtime/assets/build/postgres/patches/0043-oliphaunt-wasix-cache-jsonb-build-object-metadata.patch +src/wasix/runtime/assets/build/postgres/patches/0044-oliphaunt-wasix-initialize-trusted-session-identity.patch diff --git a/src/wasix/runtime/postgres/source-inputs b/src/wasix/runtime/postgres/source-inputs new file mode 100644 index 000000000..8b7f8fc60 --- /dev/null +++ b/src/wasix/runtime/postgres/source-inputs @@ -0,0 +1,3 @@ +# Repository-relative text inputs, in hash order; then each patch from series. +src/wasix/runtime/postgres/series +src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h diff --git a/src/wasix/runtime/protocol-contract/README.md b/src/wasix/runtime/protocol-contract/README.md new file mode 100644 index 000000000..bed4cff11 --- /dev/null +++ b/src/wasix/runtime/protocol-contract/README.md @@ -0,0 +1,65 @@ +# PostgreSQL protocol transport contract + +The Oliphaunt WASIX PostgreSQL guest, Wasmer host, Rust binding, TypeScript +facade share one private transport ABI. This contract keeps +its mode values, size bounds, ownership, flush status, and failure lifecycle in +one reviewable place. Implementations retain compile-time constants; +`generate.mjs --check` verifies the generated views, while the compiled bridge ABI +test exercises bounds, failure propagation, reset and transport transitions. +Published packages do not depend on repository-relative JSON at runtime. + +Mode 0 buffers both directions. Mode 1 streams both directions. Mode 2 begins +buffered and switches both directions to streaming only after PostgreSQL reports +a COPY response and the complete buffered response prefix has flushed +successfully. Mode 3 buffers the finite frontend request and streams ordinary +non-COPY output. COPY must keep mode 2's duplex behavior; ordinary responses do +not need to become duplex merely to avoid complete-output buffering. + +Buffered output grows on demand, up to the signed-i32 bridge length (2 GiB minus +one byte) or available memory. There is no application-specific 64 MiB quota. +Materializing results necessarily retains the complete response; use streaming +when it should not remain in memory. Output reset reuses guest buffer capacity +for later queries; it does not return that capacity to the allocator. +Arithmetic/allocation failures still fail +closed without publishing an incomplete response; callers must reopen it. +The bound applies to mode 0 +and mode 2 before its transition. Buffered bytes remain guest-owned until the +host validates the size, copies them into host-owned storage, and resets output +in that order. The host callback boundary, +not the C guest, caps stream chunks at 64 KiB. JavaScript callbacks receive a +fresh owned copy; Rust callbacks borrow a slice only for the synchronous +callback invocation. A stream can already have delivered a prefix before a +later write fails, so failure is terminal and stops further delivery but cannot +retract earlier callbacks. + +The guest flush export returns a signed status. It calls PostgreSQL's +`pq_flush()` first, then samples the bridge's first sticky positive errno, and +preserves PostgreSQL's nonzero status when both failed. Buffered writes, hybrid +prefix flushes, and direct mode-1/mode-3 writes retain the first positive errno; +once recorded it gates writes in every mode and is cleared only by output reset. +This prevents PostgreSQL from publishing more stream bytes after a failed write +while still allowing the host to observe the failure through the signed flush +status. A host may publish buffered output only after a zero flush status. + +Output reset clears buffered bytes and scan state, the buffered ErrorResponse +marker, sticky buffered-output failure, COPY state, and a pending hybrid +transition. It does not silently select another transport mode or retract an +already-active duplex stream. The stream-active export reports full-duplex mode +1 or an activated mode 2; output-only mode 3 deliberately leaves it false. + +This is a downstream correctness and maintainability contract, not a claim that +the branded ABI should be upstreamed unchanged. + +`contract.json` supplies the shared numeric constants. Run +`node src/wasix/runtime/protocol-contract/generate.mjs` after a +contract change. The checked-in generated C header is consumed by the bridge +and its native ABI test; Rust, TypeScript and the browser Wasmer host consume +generated constants. Function signatures remain in the C declarations and typed +host bindings; the JSON is not a separate signature-validation framework. +Build provenance hashes the browser's generated module along with its sources. + +PostgreSQL source preparation, Rust validation and release packaging read the +ordered `../postgres/source-inputs` list, then the patches in `../postgres/series`. +Each input is hashed with CRLF normalized to LF. The version, upstream archive +checksum and hash of those newline-separated hashes identify the prepared source. +Adding an injected header requires updating this list, not three implementations. diff --git a/src/wasix/runtime/protocol-contract/contract.json b/src/wasix/runtime/protocol-contract/contract.json new file mode 100644 index 000000000..71ce9e7db --- /dev/null +++ b/src/wasix/runtime/protocol-contract/contract.json @@ -0,0 +1,118 @@ +{ + "schema": "oliphaunt-wasix-postgres-protocol-transport-contract-v1", + "transportSelector": { + "name": "oliphaunt_wasix_set_protocol_transport", + "parameters": ["i32"], + "result": "i32", + "success": "previous-mode", + "invalid": "minus-one-with-einval" + }, + "streamActive": { + "name": "oliphaunt_wasix_protocol_stream_active", + "parameters": [], + "result": "i32", + "meaning": "duplex-stream-active-not-output-only-mode" + }, + "modes": [ + { + "name": "buffered", + "value": 0, + "input": "buffered", + "output": "buffered", + "streamActive": "false" + }, + { + "name": "stream", + "value": 1, + "input": "streamed", + "output": "streamed", + "streamActive": "true" + }, + { + "name": "hybrid", + "value": 2, + "input": "buffered-until-copy-then-streamed", + "output": "buffered-until-copy-then-streamed", + "transition": "reported-copy-response", + "activation": "after-full-buffered-prefix-flush-succeeds", + "streamActive": "false-until-transition-then-true" + }, + { + "name": "buffered-input-streamed-output", + "value": 3, + "input": "buffered", + "output": "streamed", + "streamActive": "false-output-only-does-not-claim-duplex" + } + ], + "copyStates": [ + { + "name": "none", + "value": 0 + }, + { + "name": "in", + "value": 1 + }, + { + "name": "out", + "value": 2 + }, + { + "name": "both", + "value": 3 + } + ], + "bufferedOutput": { + "limitBytes": 2147483647, + "limitReason": "signed-i32-host-bridge-length-not-an-application-response-quota", + "boundary": "inclusive", + "appliesTo": ["buffered-mode", "hybrid-mode-before-transition"], + "overflow": "efbig-fail-closed-without-partial-publication", + "sessionAfterOverflow": "terminal-reopen-required", + "ownership": "guest-owned-until-host-copy-before-output-reset", + "hostReadOrder": ["validate-limit", "copy-to-host-owned-buffer", "output-reset"] + }, + "streamedOutput": { + "callbackChunkMaxBytes": 65536, + "chunkLimitOwner": "host-callback-boundary-not-c-guest", + "javascriptOwnership": "fresh-owned-copy-per-callback", + "rustOwnership": "borrowed-only-for-synchronous-callback", + "failure": "terminal-no-further-delivery-already-delivered-prefix-is-not-retracted" + }, + "flush": { + "name": "oliphaunt_wasix_pq_flush", + "parameters": [], + "cResult": "int", + "wasmResult": "i32", + "success": 0, + "failure": "nonzero", + "evaluationOrder": ["postgres-pq-flush", "sticky-output-status"], + "precedence": "postgres-pq-flush-first" + }, + "outputReset": { + "name": "oliphaunt_wasix_output_reset", + "parameters": [], + "result": "i32", + "success": 0, + "clears": [ + "buffered-output-length-and-scan", + "buffered-error-marker", + "sticky-buffered-output-failure", + "copy-state-and-pending-transition" + ], + "doesNotClear": ["transport-mode", "already-active-duplex-stream-flag"] + }, + "stickyBufferedOutputFailure": { + "status": "first-positive-errno", + "recordedBy": "buffered-write-and-hybrid-prefix-flush-failures", + "sendGate": "all-modes-before-dispatch-once-recorded", + "clear": "output-reset-only" + }, + "directStreamWriteFailure": { + "status": "first-positive-errno-visible-to-postgres-pq-flush", + "effect": "terminal-no-further-delivery", + "alreadyDeliveredPrefix": "not-retracted", + "clear": "output-reset-only" + } +} diff --git a/src/wasix/runtime/protocol-contract/generate.mjs b/src/wasix/runtime/protocol-contract/generate.mjs new file mode 100644 index 000000000..30d17e54a --- /dev/null +++ b/src/wasix/runtime/protocol-contract/generate.mjs @@ -0,0 +1,98 @@ +#!/usr/bin/env node + +import assert from 'node:assert/strict'; +import { readFile, writeFile } from 'node:fs/promises'; +import { dirname, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const contractRoot = dirname(fileURLToPath(import.meta.url)); +const repositoryRoot = resolve(contractRoot, '../../../..'); +const contract = JSON.parse(await readFile(resolve(contractRoot, 'contract.json'), 'utf8')); + +if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + const generated = renderGeneratedArtifacts(contract); + await Promise.all( + Object.entries(generated).map(async ([path, contents]) => { + const destination = resolve(repositoryRoot, path); + if (process.argv.includes('--check')) { + assert.equal( + await readFile(destination, 'utf8'), + contents, + `${path} is stale; run node src/wasix/runtime/protocol-contract/generate.mjs`, + ); + } else { + await writeFile(destination, contents, 'utf8'); + } + }), + ); +} + +export function renderGeneratedArtifacts(value) { + const modes = Object.fromEntries(value.modes.map(({ name, value: mode }) => [name, mode])); + const copyStates = Object.fromEntries( + value.copyStates.map(({ name, value: state }) => [name, state]), + ); + const rustModes = (names) => + names + .map( + (name) => + `pub(crate) const PROTOCOL_${name.toUpperCase().replaceAll('-', '_')}: i32 = ${modes[name]};\n`, + ) + .join(''); + const header = `/* Generated by src/wasix/runtime/protocol-contract/generate.mjs. */ +#ifndef OLIPHAUNT_WASIX_PROTOCOL_TRANSPORT_CONTRACT_GENERATED_H +#define OLIPHAUNT_WASIX_PROTOCOL_TRANSPORT_CONTRACT_GENERATED_H + +#include + +#define OLIPHAUNT_WASIX_BUFFERED_PROTOCOL_OUTPUT_LIMIT ${value.bufferedOutput.limitBytes}U +#define OLIPHAUNT_WASIX_PROTOCOL_CALLBACK_CHUNK_MAX ${value.streamedOutput.callbackChunkMaxBytes}U + +#define OLIPHAUNT_WASIX_PROTOCOL_BUFFERED ${modes.buffered} +#define OLIPHAUNT_WASIX_PROTOCOL_STREAM ${modes.stream} +#define OLIPHAUNT_WASIX_PROTOCOL_HYBRID ${modes.hybrid} +#define OLIPHAUNT_WASIX_PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT ${modes['buffered-input-streamed-output']} + +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_NONE ${copyStates.none} +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_IN ${copyStates.in} +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_OUT ${copyStates.out} +#define OLIPHAUNT_WASIX_PROTOCOL_COPY_BOTH ${copyStates.both} + +int oliphaunt_wasix_set_protocol_transport(int mode); +int oliphaunt_wasix_protocol_stream_active(void); +void oliphaunt_wasix_protocol_report_copy_response(int state); +int oliphaunt_wasix_protocol_copy_state(void); +int oliphaunt_wasix_input_reset(void); +void *oliphaunt_wasix_input_reserve(size_t length); +int oliphaunt_wasix_input_commit(size_t length); +size_t oliphaunt_wasix_input_available(void); +int oliphaunt_wasix_output_reset(void); +size_t oliphaunt_wasix_output_len(void); +const void *oliphaunt_wasix_output_data(void); +int oliphaunt_wasix_output_status(void); +int oliphaunt_wasix_pq_flush(void); + +#endif +`; + return { + 'src/wasix/sdks/ts/src/protocol-limits.generated.ts': + '// Generated by src/wasix/runtime/protocol-contract/generate.mjs.\n' + + '// Edit contract.json, not this callback limit.\n' + + `export const PROTOCOL_CALLBACK_CHUNK_BYTES = ${value.streamedOutput.callbackChunkMaxBytes};\n`, + 'src/wasix/sdks/rust/src/oliphaunt/protocol_limits_generated.rs': + '// Generated by src/wasix/runtime/protocol-contract/generate.mjs.\n' + + '// Edit contract.json, not these ABI limits. See src/docs/maintainers/runtime-resource-budgets.md.\n' + + '// Signed i32 bridge length ceiling, not an application quota or eager allocation.\n' + + `pub(crate) const BUFFERED_PROTOCOL_OUTPUT_LIMIT_BYTES: usize = ${value.bufferedOutput.limitBytes};\n` + + '// Maximum callback slice; independent of socket read and COPY frame sizes.\n' + + `pub(crate) const PROTOCOL_CALLBACK_CHUNK_BYTES: usize = ${value.streamedOutput.callbackChunkMaxBytes};\n` + + rustModes(Object.keys(modes)), + 'src/wasix/browser-host/protocol-contract.generated.rs': + '// Generated by src/wasix/runtime/protocol-contract/generate.mjs.\n' + + '// Staged as src/protocol_contract.rs in the patched Wasmer JS host.\n' + + `pub(crate) const PROTOCOL_CALLBACK_CHUNK_BYTES: usize = ${value.streamedOutput.callbackChunkMaxBytes};\n` + + rustModes(['buffered', 'hybrid', 'buffered-input-streamed-output']), + 'src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h': + header, + }; +} diff --git a/src/wasix/runtime/protocol-contract/generate.test.mjs b/src/wasix/runtime/protocol-contract/generate.test.mjs new file mode 100644 index 000000000..05b45e034 --- /dev/null +++ b/src/wasix/runtime/protocol-contract/generate.test.mjs @@ -0,0 +1,26 @@ +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import test from 'node:test'; +import { renderGeneratedArtifacts } from './generate.mjs'; + +test('shared numeric edits reach each generated consumer', () => { + const contract = JSON.parse(readFileSync(new URL('./contract.json', import.meta.url), 'utf8')); + contract.modes.find(({ name }) => name === 'hybrid').value = 7; + contract.streamedOutput.callbackChunkMaxBytes = 12345; + const files = renderGeneratedArtifacts(contract); + assert.match( + files['src/wasix/sdks/rust/src/oliphaunt/protocol_limits_generated.rs'], + /PROTOCOL_HYBRID: i32 = 7;/, + ); + assert.match( + files['src/wasix/browser-host/protocol-contract.generated.rs'], + /PROTOCOL_HYBRID: i32 = 7;/, + ); + for (const contents of Object.values(files)) assert.match(contents, /12345/); + assert.match( + files[ + 'src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h' + ], + /OLIPHAUNT_WASIX_PROTOCOL_HYBRID 7/, + ); +}); diff --git a/src/wasix/runtime/tools/package-release-assets.mts b/src/wasix/runtime/tools/package-release-assets.mts index cd7342f5b..0c161334b 100644 --- a/src/wasix/runtime/tools/package-release-assets.mts +++ b/src/wasix/runtime/tools/package-release-assets.mts @@ -61,11 +61,11 @@ function copyTree(source, destination) { }); } -export function postgresSourceFingerprint() { +export function postgresSourceFingerprint(root = ROOT) { const { postgresql } = Bun.TOML.parse( - readFileSync(path.join(ROOT, 'src/third-party/postgres/source.toml'), 'utf8'), + readFileSync(path.join(root, 'src/third-party/postgres/source.toml'), 'utf8'), ); - const patches = ROOT; + const patches = root; const series = 'src/wasix/runtime/postgres/series'; const names = readFileSync(path.join(patches, series), 'utf8') .split(/\r?\n/) @@ -83,7 +83,10 @@ export function postgresSourceFingerprint() { ), 'invalid PostgreSQL patch series', ); - const hashes = [series, ...names] + const inputs = readFileSync(path.join(root, 'src/wasix/runtime/postgres/source-inputs'), 'utf8') + .split(/\r?\n/) + .filter((line) => line && !line.startsWith('#')); + const hashes = [...inputs, ...names] .map( (name) => sha256(readFileSync(path.join(patches, name), 'utf8').replaceAll('\r\n', '\n')) + '\n', diff --git a/src/wasix/runtime/tools/package-release-assets.test.mts b/src/wasix/runtime/tools/package-release-assets.test.mts index f6e253b09..a3c680c0c 100644 --- a/src/wasix/runtime/tools/package-release-assets.test.mts +++ b/src/wasix/runtime/tools/package-release-assets.test.mts @@ -11,9 +11,43 @@ import { import { tmpdir } from 'node:os'; import path from 'node:path'; import test from 'node:test'; -import { stageAotAssets, stagePortableAssets } from './package-release-assets.mts'; +import { + postgresSourceFingerprint, + stageAotAssets, + stagePortableAssets, +} from './package-release-assets.mts'; import { canonicalWasixAotMetadata } from './wasix-aot-manifest.mts'; +test('PostgreSQL source identity includes the generated protocol header and normalizes line endings', (t) => { + const root = mkdtempSync(path.join(tmpdir(), 'wasix-source-identity-')); + t.after(() => rmSync(root, { recursive: true, force: true })); + const write = (name, bytes) => { + const file = path.join(root, name); + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, bytes); + }; + write( + 'src/third-party/postgres/source.toml', + '[postgresql]\nversion = "18.4"\nsha256 = "source"\n', + ); + write('src/wasix/runtime/postgres/series', 'fixture.patch\n'); + write('fixture.patch', 'patch\n'); + const header = + 'src/wasix/runtime/assets/build/wasix_shim/oliphaunt_wasix_protocol_contract.generated.h'; + const inputs = 'src/wasix/runtime/postgres/source-inputs'; + write(inputs, `# ordered inputs\nsrc/wasix/runtime/postgres/series\n${header}\n`); + write(header, '#define LIMIT 1\n'); + const original = postgresSourceFingerprint(root); + write(header, '#define LIMIT 2\n'); + assert.notEqual(postgresSourceFingerprint(root), original); + write(header, '#define LIMIT 1\r\n'); + write(inputs, `# ordered inputs\r\nsrc/wasix/runtime/postgres/series\r\n${header}\r\n`); + assert.equal(postgresSourceFingerprint(root), original); + write('extra.h', 'extra\n'); + write(inputs, `src/wasix/runtime/postgres/series\n${header}\nextra.h\n`); + assert.notEqual(postgresSourceFingerprint(root), original); +}); + test('release staging excludes independent tools and extensions, and rejects stale or unsafe inputs', (t) => { const root = mkdtempSync(path.join(tmpdir(), 'wasix-release-stage-')); t.after(() => rmSync(root, { recursive: true, force: true })); diff --git a/src/wasix/runtime/tools/strong-random.test.mjs b/src/wasix/runtime/tools/strong-random.test.mjs new file mode 100644 index 000000000..c6f98b0ee --- /dev/null +++ b/src/wasix/runtime/tools/strong-random.test.mjs @@ -0,0 +1,78 @@ +import assert from 'node:assert/strict'; +import { spawnSync } from 'node:child_process'; +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import path from 'node:path'; +import test from 'node:test'; + +test('actual patched entropy loop handles interruptions, short reads and failures', { + skip: process.platform === 'win32', +}, () => { + const patch = readFileSync( + new URL( + '../../../third-party/postgres/patches/wasix/0037-oliphaunt-wasix-use-checked-getrandom.patch', + import.meta.url, + ), + 'utf8', + ); + // Compile the exact added C implementation, not a second model of its loop. + const added = patch + .split('\n') + .filter((line) => line.startsWith('+') && !line.startsWith('+++')) + .map((line) => line.slice(1)) + .join('\n'); + const implementation = added.slice(added.indexOf('void\npg_strong_random_init')); + assert.ok(implementation.includes('bool\npg_strong_random(')); + const fixture = mkdtempSync(path.join(tmpdir(), 'oliphaunt-entropy-')); + try { + const source = path.join(fixture, 'entropy.c'); + const binary = path.join(fixture, 'entropy'); + writeFileSync( + source, + ` +#include +#include +#include +#include +#include +#include +static int step, mode; +static ssize_t getrandom(void *buffer, size_t length, unsigned int flags) { + assert(flags == 0); + ++step; + if (mode == 1) { errno = EIO; return -1; } + if (mode == 2) return 0; + if (step == 1) { errno = EINTR; return -1; } + size_t count = length > 2 ? 2 : length; + memset(buffer, 0xa5, count); + return (ssize_t) count; +} +${implementation} +int main(void) { + unsigned char output[5] = {0}; + pg_strong_random_init(); + assert(pg_strong_random(output, sizeof(output))); + assert(step == 4); + for (size_t i = 0; i < sizeof(output); ++i) assert(output[i] == 0xa5); + mode = 1; step = 0; + assert(!pg_strong_random(output, sizeof(output)) && step == 1); + mode = 2; step = 0; + assert(!pg_strong_random(output, sizeof(output)) && step == 1); + step = 0; + assert(pg_strong_random(NULL, 0) && step == 0); + return 0; +} +`, + ); + const compiled = spawnSync( + process.env.CC ?? 'cc', + ['-std=c11', '-Wall', '-Wextra', '-Werror', source, '-o', binary], + { encoding: 'utf8' }, + ); + assert.equal(compiled.status, 0, compiled.stderr); + const result = spawnSync(binary, [], { encoding: 'utf8' }); + assert.equal(result.status, 0, result.stderr); + } finally { + rmSync(fixture, { recursive: true, force: true }); + } +}); diff --git a/src/wasix/runtime/tools/wasix-aot-manifest.test.mts b/src/wasix/runtime/tools/wasix-aot-manifest.test.mts index 2303ec2d8..5705db1c8 100644 --- a/src/wasix/runtime/tools/wasix-aot-manifest.test.mts +++ b/src/wasix/runtime/tools/wasix-aot-manifest.test.mts @@ -29,6 +29,13 @@ test('accepts AOT metadata that exactly matches the canonical WASIX toolchain', ); }); +test('rejects artifacts compiled under a different memory-codegen profile', () => { + assert.throws( + () => assertCanonicalWasixAotManifest(manifest({ engine: 'llvm-opta-ro_ftable' })), + /engine must match canonical WASIX metadata/u, + ); +}); + test('rejects stale prerelease Wasmer metadata', () => { assert.throws( () => diff --git a/src/wasix/runtime/tools/xtask/src/aot_serializer.rs b/src/wasix/runtime/tools/xtask/src/aot_serializer.rs index a7eff7c42..647407b9d 100644 --- a/src/wasix/runtime/tools/xtask/src/aot_serializer.rs +++ b/src/wasix/runtime/tools/xtask/src/aot_serializer.rs @@ -16,7 +16,41 @@ use zstd::stream::write::Encoder as ZstdEncoder; #[cfg(feature = "aot-serializer")] use crate::value_after; +// Wasmer 7.2.1's deterministic_id omits these codegen choices. Keep our +// artifact identity explicit; never label strict and nonvolatile code alike. +pub(crate) const AOT_ENGINE_PROFILE: &str = "llvm-opta"; + +pub(crate) fn check_aot_codegen_environment() -> Result<()> { + for (name, expected) in [ + ("OLIPHAUNT_WASM_AOT_NON_VOLATILE_MEMOPS", true), + ("OLIPHAUNT_WASM_AOT_READONLY_FUNCREF_TABLE", true), + ] { + match std::env::var(name) { + Ok(value) => validate_profile_override(name, Some(&value), expected)?, + Err(std::env::VarError::NotPresent) => {} + Err(error) => bail!("invalid {name}: {error}"), + } + } + Ok(()) +} + +fn validate_profile_override(name: &str, value: Option<&str>, expected: bool) -> Result<()> { + let Some(value) = value else { return Ok(()) }; + let actual = match value.trim().to_ascii_lowercase().as_str() { + "1" | "true" | "yes" | "on" => true, + "" | "0" | "false" | "no" | "off" => false, + _ => bail!("invalid {name}={value:?}; remove this retired codegen override"), + }; + if actual != expected { + bail!( + "{name}={value:?} conflicts with fixed AOT profile {AOT_ENGINE_PROFILE}; remove this retired codegen override and rebuild AOT artifacts" + ); + } + Ok(()) +} + pub(crate) fn aot_serializer(args: Vec) -> Result<()> { + check_aot_codegen_environment()?; match args.first().map(String::as_str) { Some("serialize") => serialize_aot_cli(&args[1..]), Some("probe") => probe_aot_serializer_in_process(), @@ -107,12 +141,10 @@ fn llvm_aot_engine() -> wasmer::Engine { if env_flag("OLIPHAUNT_WASM_WASMER_PERFMAP") { llvm.enable_perfmap(); } - if env_flag_default_true("OLIPHAUNT_WASM_AOT_NON_VOLATILE_MEMOPS") { - llvm.enable_non_volatile_memops(); - } - if env_flag_default_true("OLIPHAUNT_WASM_AOT_READONLY_FUNCREF_TABLE") { - llvm.enable_readonly_funcref_table(); - } + // Retain main's codegen policy here; strict memory semantics are a separate + // correctness/performance change, not part of patch consolidation. + llvm.enable_non_volatile_memops(); + llvm.enable_readonly_funcref_table(); EngineBuilder::new(llvm) .set_target(Some(portable_aot_target())) .set_features(Some(features)) @@ -143,7 +175,8 @@ fn portable_aot_target() -> wasmer_types::target::Target { fn print_aot_engine_config(engine: &wasmer::Engine) { let target = portable_aot_target(); println!("wasmer-engine: llvm"); - println!("wasmer-engine-id: {}", engine.deterministic_id()); + println!("wasmer-engine-id: {AOT_ENGINE_PROFILE}"); + println!("wasmer-compiler-id: {}", engine.deterministic_id()); println!("wasmer-target-triple: {}", target.triple()); println!( "wasmer-target-cpu-features: {}", @@ -151,18 +184,8 @@ fn print_aot_engine_config(engine: &wasmer::Engine) { ); println!("wasmer-feature-exceptions: enabled"); println!("wasmer-llvm-target-cpu: generic"); - println!( - "wasmer-llvm-non-volatile-memops: {}", - enabled_label(env_flag_default_true( - "OLIPHAUNT_WASM_AOT_NON_VOLATILE_MEMOPS" - )) - ); - println!( - "wasmer-llvm-readonly-funcref-table: {}", - enabled_label(env_flag_default_true( - "OLIPHAUNT_WASM_AOT_READONLY_FUNCREF_TABLE" - )) - ); + println!("wasmer-llvm-non-volatile-memops: enabled"); + println!("wasmer-llvm-readonly-funcref-table: enabled"); } #[cfg(feature = "aot-serializer")] @@ -194,20 +217,22 @@ fn env_flag(name: &str) -> bool { .unwrap_or(false) } -#[cfg(feature = "aot-serializer")] -fn env_flag_default_true(name: &str) -> bool { - env::var(name) - .map(|value| { - let value = value.trim(); - !matches!( - value.to_ascii_lowercase().as_str(), - "" | "0" | "false" | "no" | "off" - ) - }) - .unwrap_or(true) -} - -#[cfg(feature = "aot-serializer")] -fn enabled_label(enabled: bool) -> &'static str { - if enabled { "enabled" } else { "disabled" } +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn fixed_codegen_profile_rejects_conflicting_and_ambiguous_overrides() { + assert_eq!(AOT_ENGINE_PROFILE, "llvm-opta"); + for expected in [false, true] { + validate_profile_override("test", None, expected).unwrap(); + validate_profile_override("test", Some(if expected { "1" } else { "0" }), expected) + .unwrap(); + assert!( + validate_profile_override("test", Some(if expected { "0" } else { "1" }), expected) + .is_err() + ); + assert!(validate_profile_override("test", Some("typo"), expected).is_err()); + } + } } diff --git a/src/wasix/runtime/tools/xtask/src/asset_pipeline.rs b/src/wasix/runtime/tools/xtask/src/asset_pipeline.rs index 3b1b105f8..e8a3ac51c 100644 --- a/src/wasix/runtime/tools/xtask/src/asset_pipeline.rs +++ b/src/wasix/runtime/tools/xtask/src/asset_pipeline.rs @@ -1279,14 +1279,25 @@ fn has_wasm_export(link: &WasmLinkMetadataOut, name: &str) -> bool { .any(|export| export.name == name || export.name == format!("_{name}")) } +fn aot_source_dir(product: AssetProduct, target: &str) -> PathBuf { + // Old producer outputs used a different memory-codegen policy. A fresh + // profile directory prevents packaging those bytes under the new identity. + product + .root() + .join("aot-source") + .join(target) + .join(crate::aot_serializer::AOT_ENGINE_PROFILE) +} + pub(crate) fn prepare_aot_artifacts( target: &str, source_lane: &str, product: AssetProduct, ) -> Result<()> { + crate::aot_serializer::check_aot_codegen_environment()?; ensure_supported_aot_target(target)?; let outputs = BuildOutputs::discover_product_for_aot(source_lane, product)?; - let source_dir = outputs.product.root().join("aot-source").join(target); + let source_dir = aot_source_dir(outputs.product, target); if source_dir.exists() { fs::remove_dir_all(&source_dir) .with_context(|| format!("remove {}", source_dir.display()))?; @@ -1920,7 +1931,8 @@ fn package_aot_artifacts( outputs: &BuildOutputs, sources: &SourcesManifest, ) -> Result<()> { - let source_dir = outputs.product.root().join("aot-source").join(target); + crate::aot_serializer::check_aot_codegen_environment()?; + let source_dir = aot_source_dir(outputs.product, target); if !source_dir.exists() { let source_lane_arg = if outputs.source_lane == DEFAULT_SOURCE_LANE { String::new() @@ -1988,7 +2000,7 @@ fn package_aot_artifacts( source_fingerprint: outputs.source_fingerprint.clone(), postgres_version: Some(outputs.postgres_version.clone()), target_triple: target.to_owned(), - engine: "llvm-opta".to_owned(), + engine: crate::aot_serializer::AOT_ENGINE_PROFILE.to_owned(), wasmer_version: sources.toolchain.wasmer.clone(), wasmer_wasix_version: sources.toolchain.wasmer_wasix.clone(), artifacts: manifest_artifacts, @@ -2008,8 +2020,9 @@ pub(crate) fn package_extension_aot_artifacts( target: &str, source_lane: &str, ) -> Result<()> { + crate::aot_serializer::check_aot_codegen_environment()?; let outputs = BuildOutputs::discover_product_for_aot(source_lane, AssetProduct::Extensions)?; - let source_dir = outputs.product.root().join("aot-source").join(target); + let source_dir = aot_source_dir(outputs.product, target); if !source_dir.exists() { let source_lane_arg = if outputs.source_lane == DEFAULT_SOURCE_LANE { String::new() @@ -2081,7 +2094,7 @@ pub(crate) fn package_extension_aot_artifacts( source_fingerprint: outputs.source_fingerprint.clone(), postgres_version: Some(outputs.postgres_version.clone()), target_triple: target.to_owned(), - engine: "llvm-opta".to_owned(), + engine: crate::aot_serializer::AOT_ENGINE_PROFILE.to_owned(), wasmer_version: sources.toolchain.wasmer.clone(), wasmer_wasix_version: sources.toolchain.wasmer_wasix.clone(), artifacts, @@ -2146,7 +2159,11 @@ pub(crate) fn check_aot_product_manifest( target, "AOT manifest target-triple", )?; - ensure_eq(&manifest.engine, "llvm-opta", "AOT manifest engine")?; + ensure_eq( + &manifest.engine, + crate::aot_serializer::AOT_ENGINE_PROFILE, + "AOT manifest engine", + )?; ensure_eq( &manifest.wasmer_version, &sources.toolchain.wasmer, @@ -2558,6 +2575,23 @@ fn extension_control_files_for_asset_manifest( mod tests { use super::*; + #[test] + fn strict_codegen_does_not_reuse_legacy_producer_outputs() { + let target = "x86_64-unknown-linux-gnu"; + for product in [ + AssetProduct::Runtime, + AssetProduct::Tools, + AssetProduct::Extensions, + ] { + let legacy = product.root().join("aot-source").join(target); + assert_eq!( + aot_source_dir(product, target), + legacy.join(crate::aot_serializer::AOT_ENGINE_PROFILE) + ); + assert_ne!(aot_source_dir(product, target), legacy); + } + } + fn manifest_extension_metadata( create_extension: bool, control_files: Vec<&str>, @@ -2688,7 +2722,7 @@ mod tests { source_fingerprint: source_fingerprint.map(str::to_owned), postgres_version: postgres_version.map(str::to_owned), target_triple: "aarch64-apple-darwin".to_owned(), - engine: "llvm-opta".to_owned(), + engine: crate::aot_serializer::AOT_ENGINE_PROFILE.to_owned(), wasmer_version: "7.2.1".to_owned(), wasmer_wasix_version: "0.702.1".to_owned(), artifacts: vec![AotManifestArtifact { diff --git a/src/wasix/runtime/tools/xtask/src/main.rs b/src/wasix/runtime/tools/xtask/src/main.rs index 7b1c1f830..72a97740e 100644 --- a/src/wasix/runtime/tools/xtask/src/main.rs +++ b/src/wasix/runtime/tools/xtask/src/main.rs @@ -49,6 +49,8 @@ const GENERATED_AOT_DIR: &str = "target/oliphaunt-wasix/aot"; const RUNTIME_MODULE_ARCHIVE_MEMBER: &str = "oliphaunt/bin/postgres"; const REQUIRED_RUNTIME_ABI_EXPORTS: &[&str] = &[ "_start", + "oliphaunt_wasix_prepare_trusted_embedded_session", + "oliphaunt_wasix_startup_outcome_v1", "oliphaunt_wasix_set_active", "oliphaunt_wasix_start", "oliphaunt_wasix_get_proc_port", @@ -58,8 +60,6 @@ const REQUIRED_RUNTIME_ABI_EXPORTS: &[&str] = &[ "pq_buffer_remaining_data", "PostgresMainLoopOnce", "PostgresSendReadyForQueryIfNecessary", - "PostgresMainLongJmp", - "oliphaunt_wasix_set_force_host_error_recovery", "oliphaunt_wasix_protocol_stream_active", "oliphaunt_wasix_input_reset", "oliphaunt_wasix_input_reserve", @@ -69,6 +69,7 @@ const REQUIRED_RUNTIME_ABI_EXPORTS: &[&str] = &[ "oliphaunt_wasix_output_len", "oliphaunt_wasix_output_data", "oliphaunt_wasix_output_contains_error", + "oliphaunt_wasix_output_status", "oliphaunt_wasix_set_protocol_transport", ]; fn main() -> Result<()> { diff --git a/src/wasix/runtime/tools/xtask/src/postgres_guard.rs b/src/wasix/runtime/tools/xtask/src/postgres_guard.rs index b34eab2a9..36c6fc622 100644 --- a/src/wasix/runtime/tools/xtask/src/postgres_guard.rs +++ b/src/wasix/runtime/tools/xtask/src/postgres_guard.rs @@ -71,7 +71,13 @@ pub(crate) fn postgres_expected_source_fingerprint( fn postgres_patch_series_hash(patches: &[String]) -> Result { let mut hasher = Sha256::new(); - let inputs = std::iter::once(repo_relative_path(POSTGRES_PATCH_SERIES_PATH)) + let input_manifest = repo_relative_path("src/wasix/runtime/postgres/source-inputs"); + let input_names = fs::read_to_string(&input_manifest) + .with_context(|| format!("read {}", input_manifest.display()))?; + let inputs = input_names + .lines() + .filter(|line| !line.is_empty() && !line.starts_with('#')) + .map(repo_relative_path) .chain(patches.iter().map(repo_relative_path)); for path in inputs { let hash = sha256_text_file_lf(&path)?; diff --git a/src/wasix/sdks/rust/README.md b/src/wasix/sdks/rust/README.md index c557dec31..7922d9168 100644 --- a/src/wasix/sdks/rust/README.md +++ b/src/wasix/sdks/rust/README.md @@ -27,6 +27,10 @@ fn main() -> Result<(), Box> { Default storage is a memory filesystem and is discarded on close. Use the quickstart's persistent-storage example for application data; run it as an alternative to this disposable example. Always close database handles explicitly. +SQL `statement_timeout` is not a reliable execution deadline for CPU-bound WASIX queries. Timing out a caller's future does not stop guest work. + +The host-selected username is the real PostgreSQL session principal: LOGIN/CONNECT restrictions, role defaults and login triggers apply. `RESET ROLE` and `DISCARD ALL` do not restore the bootstrap superuser. The host authenticates the caller; initialize new storage as `postgres` before selecting an existing application role. + ## Build your integration - [Guide](https://oliphaunt.dev/docs/sdk/wasix-rust/guide): parameters, transactions, extensions, backups, and shutdown. diff --git a/src/wasix/sdks/rust/moon.yml b/src/wasix/sdks/rust/moon.yml index 95803bc5e..5cd5a3370 100644 --- a/src/wasix/sdks/rust/moon.yml +++ b/src/wasix/sdks/rust/moon.yml @@ -128,7 +128,7 @@ tasks: cargo test -p oliphaunt-wasix --doc --locked cargo test -p oliphaunt-wasix --doc --locked --features tools cargo nextest run -p oliphaunt-wasix --locked --profile ci --no-default-features --features extensions,tools,extension-vector --test public_api --no-tests=fail --test-threads=1 - cargo nextest run -p oliphaunt-wasix --locked --profile ci --no-default-features --lib --no-tests=fail --test-threads=1 + cargo nextest run -p oliphaunt-wasix --locked --profile ci --no-default-features --features tools-execution --lib --no-tests=fail --test-threads=1 env: # Cargo releases its build lock before rustdoc consumes the compiled libraries. CARGO_TARGET_DIR: "target/moon/oliphaunt-wasix-rust/unit" diff --git a/src/wasix/sdks/rust/src/oliphaunt/backend.rs b/src/wasix/sdks/rust/src/oliphaunt/backend.rs index 449818c4e..dc6a61695 100644 --- a/src/wasix/sdks/rust/src/oliphaunt/backend.rs +++ b/src/wasix/sdks/rust/src/oliphaunt/backend.rs @@ -160,6 +160,7 @@ impl WasixBackendSession { self.pg.attach_protocol_stream(stream) } + #[cfg(feature = "tools-execution")] pub(crate) fn send_with_protocol_pump( &mut self, message: &[u8], @@ -172,6 +173,18 @@ impl WasixBackendSession { .send_protocol_pump(message, Vec::new, ProtocolPumpScope::Copy) } + pub(crate) fn send_with_output_stream( + &mut self, + message: &[u8], + ) -> Result { + ensure!( + self.supports_protocol_pump(), + "WASIX runtime is missing backend-owned protocol pump exports" + ); + self.pg + .send_protocol_pump(message, Vec::new, ProtocolPumpScope::OutputStream) + } + pub(crate) fn send_with_connection_protocol_pump( &mut self, message: &[u8], @@ -259,6 +272,7 @@ impl BackendSession { self.0.attach_protocol_stream(stream) } + #[cfg(feature = "tools-execution")] pub(crate) fn send_with_protocol_pump( &mut self, message: &[u8], @@ -266,6 +280,13 @@ impl BackendSession { self.0.send_with_protocol_pump(message) } + pub(crate) fn send_with_output_stream( + &mut self, + message: &[u8], + ) -> Result { + self.0.send_with_output_stream(message) + } + pub(crate) fn send_with_connection_protocol_pump( &mut self, message: &[u8], diff --git a/src/wasix/sdks/rust/src/oliphaunt/client.rs b/src/wasix/sdks/rust/src/oliphaunt/client.rs index fc7af8911..c036dd15b 100644 --- a/src/wasix/sdks/rust/src/oliphaunt/client.rs +++ b/src/wasix/sdks/rust/src/oliphaunt/client.rs @@ -47,10 +47,18 @@ use crate::oliphaunt::tools::{ #[cfg(feature = "tools-execution")] use oliphaunt_query::wire::{FrontendFrameKind, FrontendFrameReader, classify_frontend_message}; -const PROTOCOL_CALLBACK_CHUNK_BYTES: usize = 64 * 1024; +// Host callback maximum from runtime/protocol-contract/contract.json. +// A callback borrows its slice synchronously; this is not a response-size limit. +use super::protocol_limits_generated::PROTOCOL_CALLBACK_CHUNK_BYTES; +// Tool socket read batching, independent of the callback ABI above. A large +// frontend frame spans reads; increasing this also increases local stack use. #[cfg(feature = "tools-execution")] const DIRECT_TOOL_READ_BUFFER_BYTES: usize = 64 * 1024; +fn protocol_callback_chunks(bytes: &[u8]) -> std::slice::Chunks<'_, u8> { + bytes.chunks(PROTOCOL_CALLBACK_CHUNK_BYTES) +} + /// Direct, single-session Oliphaunt WASIX database. pub struct Oliphaunt { backend: TeardownOwnership, @@ -423,7 +431,7 @@ impl Write for CallbackProtocolStream { if state.error.is_some() || state.panic.is_some() { return Ok(buffer.len()); } - for chunk in buffer.chunks(PROTOCOL_CALLBACK_CHUNK_BYTES) { + for chunk in protocol_callback_chunks(buffer) { let result = { let callback = state.callback.as_mut().ok_or_else(|| { io::Error::new( @@ -508,16 +516,16 @@ impl Oliphaunt { extensions: &[Extension], ) -> Result { let backend = if extensions.is_empty() { - BackendSession::open(outcome, postgres_config, startup_config.clone())? + BackendSession::open(outcome, postgres_config, startup_config)? } else { BackendSession::open_with_extension_preload( outcome, postgres_config, - startup_config.clone(), + startup_config, extensions, )? }; - Self::finish_open(backend, startup_config) + Ok(Self::finish_open(backend)) } #[cfg(not(feature = "extensions"))] @@ -526,12 +534,12 @@ impl Oliphaunt { postgres_config: PostgresConfig, startup_config: StartupConfig, ) -> Result { - let backend = BackendSession::open(outcome, postgres_config, startup_config.clone())?; - Self::finish_open(backend, startup_config) + let backend = BackendSession::open(outcome, postgres_config, startup_config)?; + Ok(Self::finish_open(backend)) } - fn finish_open(backend: BackendSession, startup_config: StartupConfig) -> Result { - let mut instance = Self { + fn finish_open(backend: BackendSession) -> Self { + Self { backend: TeardownOwnership::new(backend), _workspace: TeardownOwnership::new(None), _directory_lock: TeardownOwnership::new(None), @@ -543,15 +551,7 @@ impl Oliphaunt { close_result: None, protocol_stream: Arc::new(Mutex::new(CallbackProtocolState::default())), protocol_stream_attached: false, - }; - if startup_config.username != "postgres" { - let sql = format!( - "SET ROLE {}", - crate::oliphaunt::sql::quote_identifier(&startup_config.username) - ); - instance.execute_inner(&sql)?; } - Ok(instance) } /// Restore a validated physical backup into an absent or empty managed directory root. @@ -566,12 +566,12 @@ impl Oliphaunt { Sql::database(self, sql) } - /// Whether this database is permanently retired after a close attempt. + /// Whether this database can no longer accept work. /// - /// This becomes `true` when shutdown begins, even when cleanup reports an - /// error. + /// This becomes `true` after a close attempt or a terminal runtime failure. + /// After a failure, still call [`Self::close`] to release storage ownership. pub fn is_closed(&self) -> bool { - self.closed + self.closed || self.transaction_outcome_unknown || self.backup_mode_exit_unconfirmed } /// Execute a PostgreSQL command. Row-producing SQL must use [`Self::query`]. @@ -964,7 +964,7 @@ impl Oliphaunt { state.error = None; state.panic = None; } - let outcome = self.backend.send_with_protocol_pump(request); + let outcome = self.backend.send_with_output_stream(request); let (callback_error, callback_panic, callback, outcome) = match (self.protocol_stream.lock(), outcome) { (Ok(mut state), outcome) => ( @@ -994,7 +994,7 @@ impl Oliphaunt { })?; match outcome { ProtocolPumpOutcome::Buffered(response) => { - for chunk in response.chunks(PROTOCOL_CALLBACK_CHUNK_BYTES) { + for chunk in protocol_callback_chunks(&response) { match invoke_protocol_callback(&mut callback, chunk) { Ok(()) => {} Err(ProtocolCallbackFailure::Error(error)) => { @@ -1108,14 +1108,6 @@ impl Oliphaunt { .context("roll back embedded session")?; self.execute_inner("DISCARD ALL") .context("discard embedded session state")?; - let username = self.backend.startup_config().username.clone(); - if username != "postgres" { - self.execute_inner(&format!( - "SET ROLE {}", - crate::oliphaunt::sql::quote_identifier(&username) - )) - .context("restore embedded session role")?; - } Ok(()) } @@ -1498,9 +1490,10 @@ impl Oliphaunt { /// Validation before shutdown, such as an active callback transaction, /// leaves the database open and may be retried. Once shutdown begins, the /// database is permanently retired; repeated calls replay the same success - /// or failure. Successful teardown releases the backend and storage root; - /// failed teardown retains that ownership until process exit rather than - /// attempting an unproven second destructive cleanup. + /// or failure. A previously failed guest is discarded without running more + /// guest code, releasing the storage root for recovery on reopen. This does + /// not establish whether an interrupted write committed. A failure during + /// teardown itself retains ownership instead of retrying destructive cleanup. pub fn close(&mut self) -> crate::Result<()> { self.close_inner() } @@ -1553,10 +1546,14 @@ impl Oliphaunt { return Err(crate::error::lifecycle("Oliphaunt is closed")); } if self.transaction_outcome_unknown { - bail!("Oliphaunt PostgreSQL session state is unknown; close and reopen it"); + return Err(crate::error::lifecycle( + "Oliphaunt PostgreSQL session state is unknown; close and reopen it", + )); } if self.backup_mode_exit_unconfirmed { - bail!("Oliphaunt backup-mode exit is unconfirmed; close and reopen it"); + return Err(crate::error::lifecycle( + "Oliphaunt backup-mode exit is unconfirmed; close and reopen it", + )); } Ok(()) } @@ -2122,7 +2119,7 @@ mod transaction_state_tests { } #[cfg(test)] -mod protocol_callback_tests { +mod protocol_callback_outcome_tests { use std::sync::atomic::{AtomicUsize, Ordering}; use super::*; @@ -2240,6 +2237,40 @@ fn materialize_storage(storage: &PgDataStorage) -> Result { } } +#[cfg(test)] +mod protocol_callback_tests { + use super::*; + + #[test] + fn callback_chunks_are_bounded_and_preserve_exact_bytes() { + for (length, expected_lengths) in [ + (0, vec![]), + (PROTOCOL_CALLBACK_CHUNK_BYTES, vec![65_536]), + (PROTOCOL_CALLBACK_CHUNK_BYTES + 1, vec![65_536, 1]), + (PROTOCOL_CALLBACK_CHUNK_BYTES * 2, vec![65_536, 65_536]), + ( + PROTOCOL_CALLBACK_CHUNK_BYTES * 2 + 1, + vec![65_536, 65_536, 1], + ), + ] { + let input = (0..length) + .map(|offset| (offset % 251) as u8) + .collect::>(); + let chunks = protocol_callback_chunks(&input).collect::>(); + assert_eq!( + chunks.iter().map(|chunk| chunk.len()).collect::>(), + expected_lengths + ); + assert!( + chunks + .iter() + .all(|chunk| chunk.len() <= PROTOCOL_CALLBACK_CHUNK_BYTES) + ); + assert_eq!(chunks.concat(), input); + } + } +} + #[cfg(all(test, feature = "tools-execution"))] mod direct_tool_protocol_tests { use super::*; diff --git a/src/wasix/sdks/rust/src/oliphaunt/mod.rs b/src/wasix/sdks/rust/src/oliphaunt/mod.rs index bd1f00330..8baad0d33 100644 --- a/src/wasix/sdks/rust/src/oliphaunt/mod.rs +++ b/src/wasix/sdks/rust/src/oliphaunt/mod.rs @@ -11,9 +11,9 @@ pub(crate) mod database_root_descriptor; pub(crate) mod extensions; pub(crate) mod lifecycle; pub(crate) mod postgres_mod; +pub(crate) mod protocol_limits_generated; pub(crate) mod query; pub(crate) use oliphaunt_query as query_core; -pub(crate) mod sql; pub(crate) mod storage; pub(crate) mod sync_host_fs; #[cfg(test)] diff --git a/src/wasix/sdks/rust/src/oliphaunt/postgres_mod.rs b/src/wasix/sdks/rust/src/oliphaunt/postgres_mod.rs index bc7123e27..dbfdb8376 100644 --- a/src/wasix/sdks/rust/src/oliphaunt/postgres_mod.rs +++ b/src/wasix/sdks/rust/src/oliphaunt/postgres_mod.rs @@ -8,7 +8,7 @@ use std::sync::{Arc, Mutex, OnceLock}; use anyhow::{Context, Result, ensure}; use sha2::{Digest, Sha256}; use tokio::runtime::Runtime as TokioRuntime; -use tracing::{debug, warn}; +use tracing::debug; use wasmer::{Engine, Instance, Module, Store, TypedFunction, WasmTypeList}; use wasmer_config::package::{PackageHash, PackageId}; use wasmer_types::ModuleHash; @@ -36,6 +36,7 @@ mod stdio; mod task_policy; mod wasix_fs; +use super::protocol_limits_generated::BUFFERED_PROTOCOL_OUTPUT_LIMIT_BYTES; pub use stdio::ProtocolStream; use stdio::{ProtocolStdioAttachment, ProtocolStdioFile, TailCaptureFile, TailCaptureHandle}; use task_policy::{GuestWasmTasks, constrain_single_backend_tasks}; @@ -52,8 +53,18 @@ const RUNTIME_SIDE_MODULES: &[(&str, &str)] = &[ ("plpgsql.so", "runtime-support:plpgsql"), ("dict_snowball.so", "runtime-support:dict_snowball"), ]; +const OLIPHAUNT_EXIT_STARTUP_REJECTED: i32 = 98; const OLIPHAUNT_EXIT_ALIVE: i32 = 99; -const POSTGRES_MAIN_LONGJMP: i32 = 100; +const STARTUP_OUTCOME_ABI_VERSION: u32 = 1; +const STARTUP_OUTCOME_DESCRIPTOR_SIZE: usize = 32; +// Reject corrupt startup descriptors before reading guest memory. This bounds +// diagnostic output only, not query results; match the startup ABI in bridge.c. +const STARTUP_OUTCOME_MAX_PROTOCOL_BYTES: u64 = 1024 * 1024; +const STARTUP_OUTCOME_PENDING: u32 = 0; +const STARTUP_OUTCOME_REJECTED: u32 = 1; +// Retain only the latest startup/tool diagnostics per stream. Larger tails help +// debugging but increase per-instance retention; they never limit SQL output. +const DIAGNOSTIC_TAIL_BYTES: usize = 8 * 1024; static WASIX_PROCESS_RUNTIME: OnceLock, String>> = OnceLock::new(); @@ -91,6 +102,7 @@ pub struct PostgresMod { cluster_ready: bool, backend_started: bool, started: bool, + terminal_failure: Option, } pub struct StartupProtocolResponse { @@ -113,6 +125,25 @@ impl StartupErrorResponse { pub(crate) fn output(&self) -> &[u8] { &self.output } + + fn into_error(self) -> anyhow::Error { + let diagnostic = (|| { + let mut input = self.output.as_slice(); + while !input.is_empty() { + let (tag, body, rest) = crate::oliphaunt::query::read_backend_message(input)?; + if tag == b'E' { + return crate::oliphaunt::query::parse_postgres_error(body); + } + input = rest; + } + anyhow::bail!("startup response has no PostgreSQL ErrorResponse") + })(); + // Keep both typed identities: SDK callers get SQLSTATE/details, while + // the proxy can still forward the original startup response bytes. + diagnostic + .map_or_else(|error| error, anyhow::Error::new) + .context(self) + } } impl fmt::Display for StartupErrorResponse { @@ -140,33 +171,70 @@ pub enum ProtocolPumpOutcome { #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(crate) enum ProtocolPumpScope { + /// Stream output for a finite buffered request without waiting for COPY. + OutputStream, /// Return after PostgreSQL completes the COPY command that activated the stream. + #[cfg_attr(not(feature = "tools-execution"), allow(dead_code))] Copy, /// Keep pumping until the frontend connection ends. Connection, } #[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum MainLoopOutcome { + Processed, + Recovered, + InputEnded, +} + +impl MainLoopOutcome { + fn from_i32(value: i32) -> Option { + match value { + 0 => Some(Self::Processed), + 1 => Some(Self::Recovered), + 2 => Some(Self::InputEnded), + _ => None, + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[repr(i32)] enum ProtocolTransportMode { - Buffered = 0, - Stream = 1, - Hybrid = 2, + Buffered = super::protocol_limits_generated::PROTOCOL_BUFFERED, + Stream = super::protocol_limits_generated::PROTOCOL_STREAM, + Hybrid = super::protocol_limits_generated::PROTOCOL_HYBRID, + BufferedInputStreamedOutput = + super::protocol_limits_generated::PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT, } impl ProtocolTransportMode { fn from_i32(value: i32) -> Result { match value { - 0 => Ok(Self::Buffered), - 1 => Ok(Self::Stream), - 2 => Ok(Self::Hybrid), + super::protocol_limits_generated::PROTOCOL_BUFFERED => Ok(Self::Buffered), + super::protocol_limits_generated::PROTOCOL_STREAM => Ok(Self::Stream), + super::protocol_limits_generated::PROTOCOL_HYBRID => Ok(Self::Hybrid), + super::protocol_limits_generated::PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT => { + Ok(Self::BufferedInputStreamedOutput) + } other => anyhow::bail!("invalid WASIX protocol transport mode {other}"), } } } +impl ProtocolPumpScope { + fn transport_mode(self) -> ProtocolTransportMode { + match self { + Self::OutputStream => ProtocolTransportMode::BufferedInputStreamedOutput, + Self::Copy | Self::Connection => ProtocolTransportMode::Hybrid, + } + } +} + struct OliphauntLifecycleExports { + prepare_trusted_embedded_session: TypedFunction<(), i32>, + startup_outcome_ptr: i32, wasi_start: TypedFunction<(), ()>, - set_force_host_error_recovery: Option>, set_active: TypedFunction, start_oliphaunt: TypedFunction<(), ()>, #[cfg_attr(not(feature = "extensions"), allow(dead_code))] @@ -177,11 +245,10 @@ struct WasixProtocolExports { get_port: TypedFunction<(), i32>, process_startup: TypedFunction<(i32, i32, i32), i32>, send_conn_data: TypedFunction<(), ()>, - pq_flush: TypedFunction<(), ()>, + pq_flush: TypedFunction<(), i32>, pq_buffer_remaining_data: TypedFunction<(), i32>, - main_loop: TypedFunction<(), ()>, + main_loop: TypedFunction<(), i32>, send_ready: TypedFunction<(), ()>, - recover_error: TypedFunction<(), ()>, } #[derive(Clone)] @@ -198,7 +265,6 @@ struct WasixOliphauntIo { output_reset: TypedFunction<(), i32>, output_len: TypedFunction<(), i32>, output_data: TypedFunction<(), i32>, - output_contains_error: TypedFunction<(), i32>, } impl PostgresMod { @@ -259,6 +325,7 @@ impl PostgresMod { let protocol_stdio = WasixProtocolStdioExports::load(&mut store, &instance)?; (io, lifecycle, protocol, protocol_stdio) }; + validate_pending_startup_outcome(&store, &env, lifecycle.startup_outcome_ptr)?; let pg = Self { engine, @@ -283,16 +350,19 @@ impl PostgresMod { cluster_ready: false, backend_started: false, started: false, + terminal_failure: None, }; Ok(pg) } pub(crate) fn ensure_cluster(&mut self) -> Result<()> { + self.ensure_not_terminal()?; self.initialize_cluster()?; self.start_backend() } pub fn initialize_cluster(&mut self) -> Result<()> { + self.ensure_not_terminal()?; if self.cluster_ready { return Ok(()); } @@ -313,20 +383,43 @@ impl PostgresMod { } fn start_backend(&mut self) -> Result<()> { + self.ensure_not_terminal()?; if self.backend_started { return Ok(()); } - self.configure_host_error_recovery()?; { self.lifecycle .set_active .call(&mut self.store, 1) .context("oliphaunt_wasix_set_active(1)")?; } + { + let prepare_status = self + .lifecycle + .prepare_trusted_embedded_session + .call(&mut self.store) + .context("oliphaunt_wasix_prepare_trusted_embedded_session")?; + ensure!( + prepare_status == 0, + "oliphaunt_wasix_prepare_trusted_embedded_session rejected after PostgreSQL startup began" + ); + } { match self.lifecycle.wasi_start.call(&mut self.store) { - Ok(()) => {} + Ok(()) => { + let failure = format!( + "_start returned without an Oliphaunt lifecycle exit{}", + self.startup_failure_detail() + ); + self.poison_main_loop(failure.clone()); + return Err(anyhow::anyhow!("{failure}; the backend is closed")); + } Err(err) if runtime_error_exit_code(&err) == Some(OLIPHAUNT_EXIT_ALIVE) => {} + Err(err) + if runtime_error_exit_code(&err) == Some(OLIPHAUNT_EXIT_STARTUP_REJECTED) => + { + return self.atomic_startup_rejection(err); + } Err(err) => { return self.startup_failure(err, "_start Oliphaunt single-user backend"); } @@ -339,64 +432,58 @@ impl PostgresMod { Ok(()) } - fn configure_host_error_recovery(&mut self) -> Result<()> { - let force = host_requires_process_exit_error_recovery(); - let Some(set_force) = &self.lifecycle.set_force_host_error_recovery else { - if force { - anyhow::bail!( - "WASIX runtime does not export oliphaunt_wasix_set_force_host_error_recovery required by this host" - ); - } - return Ok(()); - }; - - set_force - .call(&mut self.store, i32::from(force)) - .context("oliphaunt_wasix_set_force_host_error_recovery")?; - Ok(()) - } - fn startup_failure(&mut self, err: wasmer::RuntimeError, context: &str) -> Result<()> { - if let Some(output) = self.take_startup_output_after_failure() { - if protocol_response_contains_error(&output) { - return Err(StartupErrorResponse::new(output).into()); + let failure = format!("{context}{}", self.startup_failure_detail()); + self.poison_main_loop(failure.clone()); + Err(anyhow::Error::from(err).context(format!("{failure}; the backend is closed"))) + } + + fn atomic_startup_rejection(&mut self, exit: wasmer::RuntimeError) -> Result<()> { + match read_rejected_startup_outcome( + &self.store, + &self.env, + self.lifecycle.startup_outcome_ptr, + ) { + Ok(output) => { + let rejection = StartupErrorResponse::new(output); + self.poison_main_loop(rejection.to_string()); + Err(rejection.into_error()) } - return Err(err).context(format!( - "{context}{}", - self.startup_failure_detail(Some(&output)) - )); - } - Err(err).context(format!("{context}{}", self.startup_failure_detail(None))) - } - - fn take_startup_output_after_failure(&mut self) -> Option> { - let _ = self.protocol.pq_flush.call(&mut self.store); - match self.io.take_output(&mut self.store, &self.env) { - Ok(output) if !output.is_empty() => Some(output), - Ok(_) => None, - Err(err) => { - warn!("failed to read startup output after backend failure: {err}"); - None + Err(error) => { + let failure = format!( + "_start returned controlled startup-rejection exit {OLIPHAUNT_EXIT_STARTUP_REJECTED}, but its atomic outcome was invalid: {error}{}", + self.startup_failure_detail() + ); + self.poison_main_loop(failure.clone()); + Err(error.context(format!( + "{failure}; original runtime error: {exit}; the backend is closed" + ))) } } } - fn startup_failure_detail(&self, output: Option<&[u8]>) -> String { + fn startup_failure_detail(&self) -> String { let mut detail = String::new(); let stderr = self.wasi_stderr.text(); if !stderr.trim().is_empty() { detail.push_str("\nWASIX stderr tail:\n"); detail.push_str(stderr.trim_end()); } - if let Some(output) = output { - detail.push_str("\nWASIX startup output tail:\n"); - detail.push_str(&format_output_tail(output)); - } detail } #[cfg_attr(not(feature = "extensions"), allow(dead_code))] pub(crate) fn shutdown_backend(&mut self) -> Result<()> { + if self.terminal_failure.is_some() { + // No guest is running: entry is synchronous and this runtime denies + // additional guest threads/processes. Retire host-owned descriptors + // without re-entering the failed guest or running its TLS/atexit + // destructors. Dropping this instance then permits crash recovery + // in a fresh instance; it does not confirm the failed write's outcome. + // Use WasiEnv, not WasiFunctionEnv::on_exit (which can call guest code). + self.env.data(&self.store).blocking_on_exit(Some(1.into())); + return Ok(()); + } self.lifecycle .set_active .call(&mut self.store, 0) @@ -456,6 +543,8 @@ impl PostgresMod { } pub fn send_protocol(&mut self, payload: &[u8]) -> Result> { + self.ensure_not_terminal()?; + validate_protocol_input_length(payload.len())?; { self.start_protocol()?; } @@ -469,6 +558,7 @@ impl PostgresMod { where S: ProtocolStream + 'static, { + self.ensure_not_terminal()?; ensure!( self.protocol_stdio.is_some(), "WASIX runtime does not export protocol stream transport" @@ -494,6 +584,8 @@ impl PostgresMod { continuation_prefix: impl FnOnce() -> Vec, scope: ProtocolPumpScope, ) -> Result { + self.ensure_not_terminal()?; + validate_protocol_input_length(payload.len())?; { self.start_protocol()?; } @@ -504,15 +596,50 @@ impl PostgresMod { self.protocol_stdio_attachment.is_some(), "WASIX protocol pump requires an attached stream" ); - let previous_mode = self.set_protocol_transport(ProtocolTransportMode::Hybrid)?; - ensure!( - previous_mode == ProtocolTransportMode::Buffered, - "WASIX protocol transport was not buffered before protocol pump" - ); + let transport_mode = scope.transport_mode(); + let previous_mode = self.set_protocol_transport(transport_mode)?; + if previous_mode != ProtocolTransportMode::Buffered { + return Err(self.terminal_guest_failure( + "WASIX protocol transport was not buffered before protocol pump", + )); + } let result = self.send_protocol_inner(payload); - let active = self.protocol_stream_active().unwrap_or(false); + if let Some(failure) = self.terminal_failure.as_ref() { + // The main loop failed before a streaming continuation could be + // trusted. Propagate its terminal result without another guest call. + return match result { + Err(error) => Err(error), + Ok(_) => Err(anyhow::anyhow!( + "WASIX protocol pump entered terminal state without an error result: {failure}; the backend is closed" + )), + }; + } + if scope == ProtocolPumpScope::OutputStream { + let execution = result.and_then(|output| { + if output.is_empty() { + Ok(()) + } else { + Err(self.terminal_guest_failure(format!( + "buffered-input/streamed-output transport retained {} buffered response bytes", + output.len() + ))) + } + }); + let restore_result = if self.terminal_failure.is_some() { + // A transport contract violation makes guest state untrustworthy. + Ok(()) + } else { + self.restore_protocol_transport(previous_mode, transport_mode) + }; + execution.and(restore_result)?; + return Ok(ProtocolPumpOutcome::Streamed); + } + let active = self.protocol_stream_active()?; if active { let stream_result = match scope { + ProtocolPumpScope::OutputStream => { + unreachable!("one-way output stream handled before COPY activation") + } // The triggering PostgresMainLoopOnce call synchronously completes // COPY. Starting another iteration would consume the next frontend // frame (normally Terminate) as part of the borrowed session. @@ -522,13 +649,19 @@ impl PostgresMod { self.serve_protocol_stream_inner() }), }; - let restore_result = self.restore_protocol_transport(previous_mode); + let restore_result = if self.terminal_failure.is_some() { + // A terminal streaming result can leave arbitrary guest state. + // Clear host-owned state below, but do not call a guest export. + Ok(()) + } else { + self.restore_protocol_transport(previous_mode, transport_mode) + }; let clear_result = self.clear_protocol_stream_prefix(); stream_result.and(restore_result).and(clear_result)?; Ok(ProtocolPumpOutcome::Streamed) } else { let output = result; - let restore_result = self.restore_protocol_transport(previous_mode); + let restore_result = self.restore_protocol_transport(previous_mode, transport_mode); restore_result?; let output = output?; Ok(ProtocolPumpOutcome::Buffered(output)) @@ -536,6 +669,12 @@ impl PostgresMod { } fn send_protocol_inner(&mut self, payload: &[u8]) -> Result> { + self.run_guest_phase("buffered protocol dispatch", |pg| { + pg.dispatch_buffered_protocol(payload) + }) + } + + fn dispatch_buffered_protocol(&mut self, payload: &[u8]) -> Result> { { self.io.reset(&mut self.store)?; } @@ -546,55 +685,36 @@ impl PostgresMod { { let max_attempts = (payload.len() / 5).saturating_add(2).max(1); let mut attempts = 0usize; - let mut recovered_protocol_error = false; while self.protocol_input_remaining()? > 0 { attempts += 1; ensure!( attempts <= max_attempts, "Postgres protocol dispatch did not drain buffered input after {attempts} attempts" ); - if let Err(err) = self.protocol.main_loop.call(&mut self.store) { - if runtime_error_exit_code(&err) == Some(POSTGRES_MAIN_LONGJMP) { - debug!( - "PostgresMainLoopOnce used host longjmp fallback; recovering protocol error" - ); - self.recover_protocol_error(payload.len())?; - recovered_protocol_error = true; - } else if is_wasm_uncaught_exception(&err) { + let status = match self.protocol.main_loop.call(&mut self.store) { + Ok(status) => status, + Err(err) => return Err(self.terminal_main_loop_error(err)), + }; + match self.decode_main_loop_outcome(status)? { + MainLoopOutcome::Processed => {} + MainLoopOutcome::Recovered => { + // The guest already ran PostgreSQL's top-level cleanup. + // Keep pumping so an extended-protocol Sync already in + // this buffer is consumed before ReadyForQuery is sent. debug!( - "PostgresMainLoopOnce trapped for PostgreSQL error; recovering protocol state: {err}" + "PostgresMainLoopOnce recovered a PostgreSQL error inside the guest" ); - self.recover_protocol_error(payload.len())?; - recovered_protocol_error = true; - } else { - warn!("PostgresMainLoopOnce trapped; attempting protocol recovery: {err}"); - self.recover_protocol_error(payload.len())?; - recovered_protocol_error = true; + } + MainLoopOutcome::InputEnded => { + return Err(self.terminal_main_loop_outcome( + "PostgresMainLoopOnce reported input end while dispatching buffered protocol input", + )); } } } - { - self.protocol - .send_ready - .call(&mut self.store) - .context("PostgresSendReadyForQueryIfNecessary")?; - } - { - self.protocol - .pq_flush - .call(&mut self.store) - .context("oliphaunt_wasix_pq_flush after protocol buffer")?; - } - let contains_error = self.io.output_contains_error(&mut self.store)?; - let output = { - self.io - .take_output(&mut self.store, &self.env) - .context("take backend output after protocol buffer")? - }; - if !recovered_protocol_error && contains_error { - self.recover_non_trapping_protocol_error()?; - } + self.finish_main_loop_output("after buffered protocol dispatch")?; + let output = self.take_buffered_protocol_output("after protocol dispatch")?; Ok(output) } } @@ -605,36 +725,22 @@ impl PostgresMod { fn serve_protocol_stream_inner(&mut self) -> Result<()> { loop { - if let Err(err) = self.protocol.main_loop.call(&mut self.store) { - if runtime_error_exit_code(&err) == Some(OLIPHAUNT_EXIT_ALIVE) { - break; - } - if runtime_error_exit_code(&err) == Some(POSTGRES_MAIN_LONGJMP) { - debug!( - "PostgresMainLoopOnce used host longjmp fallback while serving streaming protocol" - ); - self.protocol.recover_error.call(&mut self.store).context( - "recover Postgres main-loop error while serving streaming protocol", - )?; - } else if is_wasm_uncaught_exception(&err) { + let status = match self.protocol.main_loop.call(&mut self.store) { + Ok(status) => status, + Err(err) => return Err(self.terminal_main_loop_error(err)), + }; + match self.decode_main_loop_outcome(status)? { + MainLoopOutcome::Processed => {} + MainLoopOutcome::Recovered => { + // Recovery and ErrorResponse production completed before + // the typed return; only ReadyForQuery and flushing remain. debug!( - "PostgresMainLoopOnce trapped for PostgreSQL error while serving streaming protocol: {err}" + "PostgresMainLoopOnce recovered a PostgreSQL error while serving streaming protocol" ); - self.protocol.recover_error.call(&mut self.store).context( - "recover Postgres main-loop error while serving streaming protocol", - )?; - } else { - return Err(err).context("PostgresMainLoopOnce streaming protocol"); } + MainLoopOutcome::InputEnded => break, } - self.protocol - .send_ready - .call(&mut self.store) - .context("PostgresSendReadyForQueryIfNecessary streaming protocol")?; - self.protocol - .pq_flush - .call(&mut self.store) - .context("oliphaunt_wasix_pq_flush streaming protocol")?; + self.finish_main_loop_output("while serving streaming protocol")?; } Ok(()) } @@ -643,39 +749,55 @@ impl PostgresMod { &mut self, mode: ProtocolTransportMode, ) -> Result { - let stdio = self - .protocol_stdio - .as_ref() - .context("WASIX runtime does not export protocol stdio switching")?; - let previous = stdio - .set_protocol_transport - .call(&mut self.store, mode as i32) - .context("oliphaunt_wasix_set_protocol_transport")?; - ProtocolTransportMode::from_i32(previous) - } - - fn restore_protocol_transport(&mut self, previous_mode: ProtocolTransportMode) -> Result<()> { - let current = self.set_protocol_transport(previous_mode)?; ensure!( - current != previous_mode, - "oliphaunt_wasix_set_protocol_transport restore observed unchanged current mode" + self.protocol_stdio.is_some(), + "WASIX runtime does not export protocol stdio switching" ); - Ok(()) + self.run_guest_phase("set protocol transport", |pg| { + let stdio = pg.protocol_stdio.as_ref().expect("checked protocol stdio"); + let previous = stdio + .set_protocol_transport + .call(&mut pg.store, mode as i32) + .context("oliphaunt_wasix_set_protocol_transport")?; + ProtocolTransportMode::from_i32(previous) + }) + } + + fn restore_protocol_transport( + &mut self, + previous_mode: ProtocolTransportMode, + expected_current: ProtocolTransportMode, + ) -> Result<()> { + self.run_guest_phase("restore protocol transport", |pg| { + let current = pg.set_protocol_transport(previous_mode)?; + ensure!( + current == expected_current, + "oliphaunt_wasix_set_protocol_transport restore observed unexpected current mode {current:?}, expected {expected_current:?}" + ); + Ok(()) + }) } fn protocol_stream_active(&mut self) -> Result { - let stdio = self - .protocol_stdio - .as_ref() - .context("WASIX runtime does not export protocol stream state")?; - Ok(stdio - .protocol_stream_active - .call(&mut self.store) - .context("oliphaunt_wasix_protocol_stream_active")? - != 0) + self.run_guest_phase("read protocol stream state", |pg| { + let stdio = pg + .protocol_stdio + .as_ref() + .context("WASIX runtime does not export protocol stream state")?; + let active = stdio + .protocol_stream_active + .call(&mut pg.store) + .context("oliphaunt_wasix_protocol_stream_active")?; + match active { + 0 => Ok(false), + 1 => Ok(true), + other => anyhow::bail!("invalid WASIX protocol stream state {other}"), + } + }) } fn start_protocol(&mut self) -> Result<()> { + self.ensure_not_terminal()?; if self.started { return Ok(()); } @@ -698,6 +820,7 @@ impl PostgresMod { &mut self, startup: &[u8], ) -> Result { + self.ensure_not_terminal()?; self.ensure_cluster()?; ensure!( !self.started, @@ -729,8 +852,8 @@ impl PostgresMod { .context("ProcessStartupPacket")? }; if status != 0 { - let _ = self.protocol.pq_flush.call(&mut self.store); - let output = self.io.take_output(&mut self.store, &self.env)?; + self.flush_protocol_output("after rejected startup")?; + let output = self.take_buffered_protocol_output("after rejected protocol startup")?; return Ok(StartupProtocolResponse { output, accepted: false, @@ -743,13 +866,8 @@ impl PostgresMod { .call(&mut self.store) .context("oliphaunt_wasix_send_conn_data")?; } - { - self.protocol - .pq_flush - .call(&mut self.store) - .context("oliphaunt_wasix_pq_flush after startup")?; - } - self.io.take_output(&mut self.store, &self.env)? + self.flush_protocol_output("after accepted startup")?; + self.take_buffered_protocol_output("after accepted protocol startup")? }; self.started = true; self.startup_response = Some(output.clone()); @@ -769,73 +887,159 @@ impl PostgresMod { &self.startup_config } - fn recover_protocol_error(&mut self, payload_len: usize) -> Result<()> { - self.protocol - .recover_error - .call(&mut self.store) - .context("PostgresMainLongJmp after protocol trap")?; - - // PostgreSQL extended-query errors skip messages until Sync. If Sync was - // already in this host buffer, re-enter the loop to drain it and produce - // ReadyForQuery from PostgreSQL rather than inventing one in Rust. - let max_drain_attempts = (payload_len / 5).saturating_add(2).max(1); - let mut drain_attempts = 0usize; - while self.protocol_input_remaining()? > 0 { - drain_attempts += 1; - ensure!( - drain_attempts <= max_drain_attempts, - "Postgres protocol recovery did not drain buffered input after {drain_attempts} attempts" - ); - if let Err(drain_err) = self.protocol.main_loop.call(&mut self.store) { - if runtime_error_exit_code(&drain_err) == Some(POSTGRES_MAIN_LONGJMP) - || is_wasm_uncaught_exception(&drain_err) - { - debug!( - "PostgresMainLoopOnce trapped while draining after PostgreSQL error recovery: {drain_err}" - ); - } else { - warn!( - "PostgresMainLoopOnce trapped while draining after recovery: {drain_err}" - ); - } - self.protocol - .recover_error - .call(&mut self.store) - .context("PostgresMainLongJmp while draining after protocol trap")?; + fn ensure_not_terminal(&self) -> Result<()> { + ensure_guest_phase_live(self.terminal_failure.as_deref()) + } + + fn run_guest_phase( + &mut self, + phase: &str, + operation: impl FnOnce(&mut Self) -> Result, + ) -> Result { + self.ensure_not_terminal()?; + let result = operation(self); + finish_guest_phase(result, phase, |failure| { + if self.terminal_failure.is_none() { + self.poison_main_loop(failure); + } + }) + } + + fn decode_main_loop_outcome(&mut self, status: i32) -> Result { + MainLoopOutcome::from_i32(status).ok_or_else(|| { + self.terminal_main_loop_outcome(format!( + "PostgresMainLoopOnce returned invalid typed outcome {status}" + )) + }) + } + + fn terminal_main_loop_outcome(&mut self, failure: impl Into) -> anyhow::Error { + self.terminal_guest_failure(failure) + } + + fn terminal_guest_failure(&mut self, failure: impl Into) -> anyhow::Error { + let failure = failure.into(); + self.poison_main_loop(failure.clone()); + anyhow::anyhow!("{failure}; the backend is closed") + } + + fn take_buffered_protocol_output(&mut self, phase: &str) -> Result> { + match self.io.take_output(&mut self.store, &self.env) { + Ok(output) => Ok(output), + Err(error) => { + let failure = + format!("failed to take bounded WASIX protocol output {phase}: {error:#}"); + self.poison_main_loop(failure.clone()); + Err(error.context(format!("{failure}; the backend is closed"))) } } - Ok(()) } - fn recover_non_trapping_protocol_error(&mut self) -> Result<()> { - self.protocol - .recover_error - .call(&mut self.store) - .context("PostgresMainLongJmp after backend ErrorResponse")?; - self.protocol - .send_ready - .call(&mut self.store) - .context("PostgresSendReadyForQueryIfNecessary after backend ErrorResponse")?; - self.protocol - .pq_flush - .call(&mut self.store) - .context("oliphaunt_wasix_pq_flush after backend ErrorResponse recovery")?; - let _ = self.io.take_output(&mut self.store, &self.env)?; + fn terminal_main_loop_error(&mut self, err: wasmer::RuntimeError) -> anyhow::Error { + let failure = + "PostgresMainLoopOnce trapped instead of returning a typed outcome".to_owned(); + self.poison_main_loop(failure.clone()); + anyhow::Error::from(err).context(format!("{failure}; the backend is closed")) + } + + fn finish_main_loop_output(&mut self, phase: &str) -> Result<()> { + if let Err(err) = self.protocol.send_ready.call(&mut self.store) { + return Err(self.terminal_post_step_export_error( + "PostgresSendReadyForQueryIfNecessary", + phase, + err, + )); + } + self.flush_protocol_output(phase) + } + + fn flush_protocol_output(&mut self, phase: &str) -> Result<()> { + let status = match self.protocol.pq_flush.call(&mut self.store) { + Ok(status) => status, + Err(err) => { + return Err(self.terminal_post_step_export_error( + "oliphaunt_wasix_pq_flush", + phase, + err, + )); + } + }; + if status != 0 { + return Err(self.terminal_guest_failure(format!( + "oliphaunt_wasix_pq_flush returned failure status {status} {phase}" + ))); + } Ok(()) } + fn terminal_post_step_export_error( + &mut self, + export: &str, + phase: &str, + err: wasmer::RuntimeError, + ) -> anyhow::Error { + let failure = match runtime_error_exit_code(&err) { + Some(code) => format!("{export} failed {phase} with WASI exit code {code}"), + None => format!( + "{export} trapped {phase} outside the live PostgreSQL main-loop recovery boundary" + ), + }; + self.poison_main_loop(failure.clone()); + anyhow::Error::from(err).context(format!("{failure}; the backend is closed")) + } + + fn poison_main_loop(&mut self, failure: String) { + // A trap or invalid outcome can leave arbitrary guest state behind. Mark + // the host terminal without invoking another guest export, including + // shutdown or a separate recovery entry point. + self.terminal_failure = Some(failure); + self.backend_started = false; + self.started = false; + self.startup_response = None; + } + fn protocol_input_remaining(&mut self) -> Result { let host_remaining = self.io.available(&mut self.store)?; if host_remaining > 0 { return Ok(host_remaining); } - self.protocol + let buffered = self + .protocol .pq_buffer_remaining_data .call(&mut self.store) - .context("pq_buffer_remaining_data") + .context("pq_buffer_remaining_data")?; + ensure!( + buffered >= 0, + "pq_buffer_remaining_data returned negative length {buffered}" + ); + Ok(buffered) } } +fn validate_protocol_input_length(length: usize) -> Result<()> { + i32::try_from(length).context("protocol input exceeds i32")?; + Ok(()) +} + +fn ensure_guest_phase_live(terminal_failure: Option<&str>) -> Result<()> { + if let Some(failure) = terminal_failure { + anyhow::bail!( + "Oliphaunt WASIX PostgreSQL backend cannot be reused after a terminal guest failure: {failure}" + ); + } + Ok(()) +} + +fn finish_guest_phase(result: Result, phase: &str, poison: impl FnOnce(String)) -> Result { + result.map_err(|error| { + poison(format!("WASIX guest phase {phase} failed: {error:#}")); + // Keep the cause in the error chain, not duplicated inside its context. + error.context(format!( + "WASIX guest phase {phase} failed; the backend is closed" + )) + }) +} + fn process_wasix_runtime(engine: &Engine) -> Result> { WASIX_PROCESS_RUNTIME .get_or_init(|| { @@ -1063,8 +1267,8 @@ fn run_split_initdb(runtime_layout: &RuntimeLayout, pgdata_storage: &PgDataStora .read_dir(Path::new(PGDATA_DIR)) .with_context(|| format!("verify split initdb {PGDATA_DIR} mount"))?; - let (stdout_file, stdout_capture) = TailCaptureFile::new(8 * 1024); - let (stderr_file, stderr_capture) = TailCaptureFile::new(8 * 1024); + let (stdout_file, stdout_capture) = TailCaptureFile::new(DIAGNOSTIC_TAIL_BYTES); + let (stderr_file, stderr_capture) = TailCaptureFile::new(DIAGNOSTIC_TAIL_BYTES); let mut runner = WasiRunner::new(); runner @@ -1492,20 +1696,30 @@ where impl OliphauntLifecycleExports { fn load(store: &mut Store, instance: &Instance) -> Result { - let wasi_start = typed_export(store, instance, "_start")?; - let set_force_host_error_recovery = optional_typed_export( + let prepare_trusted_embedded_session = typed_export( store, instance, - "oliphaunt_wasix_set_force_host_error_recovery", + "oliphaunt_wasix_prepare_trusted_embedded_session", )?; + let startup_outcome = + typed_export::<(), i32>(store, instance, "oliphaunt_wasix_startup_outcome_v1")?; + let startup_outcome_ptr = startup_outcome + .call(&mut *store) + .context("oliphaunt_wasix_startup_outcome_v1")?; + ensure!( + startup_outcome_ptr != 0, + "oliphaunt_wasix_startup_outcome_v1 returned null" + ); + let wasi_start = typed_export(store, instance, "_start")?; let set_active = typed_export(store, instance, "oliphaunt_wasix_set_active")?; let start_oliphaunt = typed_export(store, instance, "oliphaunt_wasix_start")?; let run_atexit_funcs = optional_typed_export(store, instance, "oliphaunt_wasix_run_atexit_funcs")?; Ok(Self { + prepare_trusted_embedded_session, + startup_outcome_ptr, wasi_start, - set_force_host_error_recovery, set_active, start_oliphaunt, run_atexit_funcs, @@ -1522,7 +1736,6 @@ impl WasixProtocolExports { let pq_buffer_remaining_data = typed_export(store, instance, "pq_buffer_remaining_data")?; let main_loop = typed_export(store, instance, "PostgresMainLoopOnce")?; let send_ready = typed_export(store, instance, "PostgresSendReadyForQueryIfNecessary")?; - let recover_error = typed_export(store, instance, "PostgresMainLongJmp")?; Ok(Self { get_port, @@ -1532,7 +1745,6 @@ impl WasixProtocolExports { pq_buffer_remaining_data, main_loop, send_ready, - recover_error, }) } } @@ -1558,9 +1770,10 @@ impl WasixProtocolStdioExports { fn ensure_integrated_oliphaunt_contract(instance: &Instance) -> Result<()> { for name in [ + "oliphaunt_wasix_prepare_trusted_embedded_session", + "oliphaunt_wasix_startup_outcome_v1", "oliphaunt_wasix_start", "oliphaunt_wasix_set_active", - "PostgresMainLongJmp", ] { ensure!( instance.exports.get_function(name).is_ok() @@ -1581,11 +1794,6 @@ impl WasixOliphauntIo { output_reset: typed_export(store, instance, "oliphaunt_wasix_output_reset")?, output_len: typed_export(store, instance, "oliphaunt_wasix_output_len")?, output_data: typed_export(store, instance, "oliphaunt_wasix_output_data")?, - output_contains_error: typed_export( - store, - instance, - "oliphaunt_wasix_output_contains_error", - )?, }; io.reset(store)?; Ok(io) @@ -1650,14 +1858,11 @@ impl WasixOliphauntIo { } fn take_output(&self, store: &mut Store, env: &WasiFunctionEnv) -> Result> { - let len = self + let guest_len = self .output_len .call(&mut *store) .context("oliphaunt_wasix_output_len")?; - ensure!( - len >= 0, - "oliphaunt_wasix_output_len returned negative length {len}" - ); + let len = checked_buffered_protocol_output_len(guest_len)?; if len == 0 { return Ok(Vec::new()); } @@ -1669,7 +1874,7 @@ impl WasixOliphauntIo { ptr > 0, "oliphaunt_wasix_output_data returned null for non-empty output" ); - let mut bytes = vec![0u8; len as usize]; + let mut bytes = allocate_zeroed_buffered_protocol_output(len)?; let view = env .data(&*store) .try_memory_view(&*store) @@ -1685,14 +1890,27 @@ impl WasixOliphauntIo { ); Ok(bytes) } +} - fn output_contains_error(&self, store: &mut Store) -> Result { - Ok(self - .output_contains_error - .call(store) - .context("oliphaunt_wasix_output_contains_error")? - != 0) - } +fn checked_buffered_protocol_output_len(guest_len: i32) -> Result { + let len = usize::try_from(guest_len).with_context(|| { + format!("oliphaunt_wasix_output_len returned invalid length {guest_len}") + })?; + ensure!( + len <= BUFFERED_PROTOCOL_OUTPUT_LIMIT_BYTES, + "buffered WASIX protocol output is {len} bytes, exceeding the inclusive {}-byte host limit", + BUFFERED_PROTOCOL_OUTPUT_LIMIT_BYTES + ); + Ok(len) +} + +fn allocate_zeroed_buffered_protocol_output(len: usize) -> Result> { + let mut output = Vec::new(); + output + .try_reserve_exact(len) + .context("reserve bounded WASIX protocol output")?; + output.resize(len, 0); + Ok(output) } fn typed_export( @@ -1740,20 +1958,6 @@ fn runtime_error_exit_code(err: &wasmer::RuntimeError) -> Option { }) } -fn is_wasm_uncaught_exception(err: &wasmer::RuntimeError) -> bool { - // Wasmer reports an uncaught WebAssembly exception when PostgreSQL ERROR - // unwinds across the exported loop boundary. The C recovery export then - // performs the normal Postgres error cleanup and emits ErrorResponse. - err.message().contains("uncaught exception") -} - -fn host_requires_process_exit_error_recovery() -> bool { - // Wasmer 7.2.1 disables its WebAssembly exception-handling tests on - // Windows. Keep PostgreSQL's proven top-level process-exit recovery there; - // other hosts retain nested PG_TRY/PG_CATCH unwinding. - cfg!(target_env = "msvc") -} - fn wasix_icu_data_is_available(runtime_layout: &RuntimeLayout) -> bool { runtime_layout.mutable_root.is_dir(Path::new("/share/icu")) || runtime_layout.module_root.join("share/icu").is_dir() @@ -1881,6 +2085,144 @@ fn startup_packet(user: &str, database: &str) -> Vec { packet } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct StartupOutcomeDescriptor { + kind: u32, + protocol_ptr: u64, + protocol_len: u64, +} + +fn decode_startup_outcome_descriptor(bytes: &[u8]) -> Result { + ensure!( + bytes.len() == STARTUP_OUTCOME_DESCRIPTOR_SIZE, + "startup outcome descriptor has {} bytes, expected {STARTUP_OUTCOME_DESCRIPTOR_SIZE}", + bytes.len() + ); + let field = |offset: usize| { + u32::from_le_bytes( + bytes[offset..offset + 4] + .try_into() + .expect("fixed startup outcome u32 field"), + ) + }; + let wide_field = |offset: usize| { + u64::from_le_bytes( + bytes[offset..offset + 8] + .try_into() + .expect("fixed startup outcome u64 field"), + ) + }; + let version = field(0); + let byte_size = field(4); + let kind = field(8); + let reserved = field(12); + ensure!( + version == STARTUP_OUTCOME_ABI_VERSION, + "startup outcome ABI version {version}, expected {STARTUP_OUTCOME_ABI_VERSION}" + ); + ensure!( + byte_size as usize == STARTUP_OUTCOME_DESCRIPTOR_SIZE, + "startup outcome descriptor size {byte_size}, expected {STARTUP_OUTCOME_DESCRIPTOR_SIZE}" + ); + ensure!(reserved == 0, "startup outcome reserved field is nonzero"); + Ok(StartupOutcomeDescriptor { + kind, + protocol_ptr: wide_field(16), + protocol_len: wide_field(24), + }) +} + +fn read_startup_outcome_descriptor( + store: &Store, + env: &WasiFunctionEnv, + descriptor_ptr: i32, +) -> Result { + ensure!( + descriptor_ptr != 0, + "startup outcome descriptor pointer is null" + ); + let descriptor_ptr = descriptor_ptr as u32 as u64; + let view = env + .data(store) + .try_memory_view(store) + .context("get WASIX memory view for startup outcome")?; + let descriptor_end = descriptor_ptr + .checked_add(STARTUP_OUTCOME_DESCRIPTOR_SIZE as u64) + .context("startup outcome descriptor address overflow")?; + ensure!( + descriptor_end <= view.data_size(), + "startup outcome descriptor is outside guest memory" + ); + let mut bytes = [0u8; STARTUP_OUTCOME_DESCRIPTOR_SIZE]; + view.read(descriptor_ptr, &mut bytes) + .context("read startup outcome descriptor")?; + decode_startup_outcome_descriptor(&bytes) +} + +fn validate_pending_startup_outcome( + store: &Store, + env: &WasiFunctionEnv, + descriptor_ptr: i32, +) -> Result<()> { + let descriptor = read_startup_outcome_descriptor(store, env, descriptor_ptr)?; + ensure!( + descriptor.kind == STARTUP_OUTCOME_PENDING, + "startup outcome was not pending before _start" + ); + ensure!( + descriptor.protocol_ptr == 0 && descriptor.protocol_len == 0, + "pending startup outcome exposed protocol bytes" + ); + Ok(()) +} + +fn read_rejected_startup_outcome( + store: &Store, + env: &WasiFunctionEnv, + descriptor_ptr: i32, +) -> Result> { + let descriptor = read_startup_outcome_descriptor(store, env, descriptor_ptr)?; + ensure!( + descriptor.kind == STARTUP_OUTCOME_REJECTED, + "startup outcome kind {} is not rejected", + descriptor.kind + ); + ensure!( + descriptor.protocol_ptr != 0 && descriptor.protocol_len != 0, + "rejected startup outcome has no protocol bytes" + ); + ensure!( + descriptor.protocol_len <= STARTUP_OUTCOME_MAX_PROTOCOL_BYTES, + "startup rejection protocol response exceeds the {}-byte host limit", + STARTUP_OUTCOME_MAX_PROTOCOL_BYTES + ); + let view = env + .data(store) + .try_memory_view(store) + .context("get WASIX memory view for startup rejection")?; + let protocol_end = descriptor + .protocol_ptr + .checked_add(descriptor.protocol_len) + .context("startup rejection protocol address overflow")?; + ensure!( + protocol_end <= view.data_size(), + "startup rejection protocol response is outside guest memory" + ); + let protocol_len = descriptor.protocol_len as usize; + let mut output = Vec::new(); + output + .try_reserve_exact(protocol_len) + .context("reserve startup rejection protocol response")?; + output.resize(protocol_len, 0); + view.read(descriptor.protocol_ptr, &mut output) + .context("read atomic startup rejection protocol response")?; + ensure!( + complete_protocol_response_contains_error(&output), + "atomic startup rejection is not a complete PostgreSQL response containing ErrorResponse" + ); + Ok(output) +} + fn protocol_response_contains_error(response: &[u8]) -> bool { let mut cursor = 0usize; while cursor + 5 <= response.len() { @@ -1901,23 +2243,66 @@ fn protocol_response_contains_error(response: &[u8]) -> bool { false } -fn format_output_tail(bytes: &[u8]) -> String { - const LIMIT: usize = 512; - let skipped = bytes.len().saturating_sub(LIMIT); - let tail = &bytes[skipped..]; - let mut hex = String::new(); - for (index, byte) in tail.iter().enumerate() { - if index > 0 { - hex.push(' '); +fn complete_protocol_response_contains_error(response: &[u8]) -> bool { + let mut cursor = 0usize; + let mut contains_error = false; + while cursor < response.len() { + let Some(header_end) = cursor.checked_add(5) else { + return false; + }; + if header_end > response.len() { + return false; } - hex.push_str(&format!("{byte:02x}")); + let tag = response[cursor]; + let len = i32::from_be_bytes( + response[cursor + 1..cursor + 5] + .try_into() + .expect("checked protocol length field"), + ); + if len < 4 { + return false; + } + let Some(next) = cursor.checked_add(1 + len as usize) else { + return false; + }; + if next > response.len() { + return false; + } + if tag == b'E' { + if !error_response_has_valid_sqlstate(&response[cursor + 5..next]) { + return false; + } + contains_error = true; + } + cursor = next; } - let text = String::from_utf8_lossy(tail); - format!( - "{} bytes total, showing last {} bytes\nhex: {hex}\nutf8-lossy:\n{text}", - bytes.len(), - tail.len() - ) + contains_error +} + +fn error_response_has_valid_sqlstate(fields: &[u8]) -> bool { + if fields.last() != Some(&0) { + return false; + } + let mut cursor = 0usize; + let mut contains_sqlstate = false; + while cursor < fields.len() - 1 { + let field_type = fields[cursor]; + if field_type == 0 { + return false; + } + cursor += 1; + let Some(terminator) = fields[cursor..].iter().position(|byte| *byte == 0) else { + return false; + }; + if field_type == b'C' { + if terminator != 5 { + return false; + } + contains_sqlstate = true; + } + cursor += terminator + 1; + } + cursor == fields.len() - 1 && contains_sqlstate } fn seed_exported_c_string_value( @@ -2037,6 +2422,198 @@ mod tests { use std::io; use std::pin::Pin; + #[test] + fn startup_rejection_preserves_public_diagnostic_and_proxy_bytes() { + let output = oliphaunt_query::wire::error_response("FATAL", "28000", "login denied"); + let error = StartupErrorResponse::new(output.clone()) + .into_error() + .context("open database"); + assert_eq!( + startup_error_response_output(&error), + Some(output.as_slice()) + ); + let error = crate::Error::from_anyhow(error); + assert_eq!(error.kind(), crate::ErrorKind::Postgres); + let diagnostic = error.postgres_error().expect("typed startup error"); + assert_eq!(diagnostic.sqlstate.as_deref(), Some("28000")); + assert_eq!(diagnostic.message, "login denied"); + } + #[test] + fn main_loop_outcome_classification_is_exact() { + assert_eq!( + MainLoopOutcome::from_i32(0), + Some(MainLoopOutcome::Processed) + ); + assert_eq!( + MainLoopOutcome::from_i32(1), + Some(MainLoopOutcome::Recovered) + ); + assert_eq!( + MainLoopOutcome::from_i32(2), + Some(MainLoopOutcome::InputEnded) + ); + assert_eq!(MainLoopOutcome::from_i32(-1), None); + assert_eq!(MainLoopOutcome::from_i32(3), None); + assert_eq!(MainLoopOutcome::from_i32(99), None); + assert_eq!(MainLoopOutcome::from_i32(100), None); + } + + #[test] + fn protocol_transport_modes_and_pump_intent_are_exact() { + assert_eq!( + ProtocolTransportMode::from_i32(0).unwrap(), + ProtocolTransportMode::Buffered + ); + assert_eq!( + ProtocolTransportMode::from_i32(1).unwrap(), + ProtocolTransportMode::Stream + ); + assert_eq!( + ProtocolTransportMode::from_i32(2).unwrap(), + ProtocolTransportMode::Hybrid + ); + assert_eq!( + ProtocolTransportMode::from_i32(3).unwrap(), + ProtocolTransportMode::BufferedInputStreamedOutput + ); + for invalid in [-1, 4, 99] { + assert!(ProtocolTransportMode::from_i32(invalid).is_err()); + } + + assert_eq!( + ProtocolPumpScope::OutputStream.transport_mode(), + ProtocolTransportMode::BufferedInputStreamedOutput + ); + assert_eq!( + ProtocolPumpScope::Copy.transport_mode(), + ProtocolTransportMode::Hybrid + ); + assert_eq!( + ProtocolPumpScope::Connection.transport_mode(), + ProtocolTransportMode::Hybrid + ); + } + + #[test] + fn buffered_protocol_output_length_is_inclusively_bounded() { + assert_eq!(checked_buffered_protocol_output_len(0).unwrap(), 0); + assert_eq!( + checked_buffered_protocol_output_len( + i32::try_from(BUFFERED_PROTOCOL_OUTPUT_LIMIT_BYTES).unwrap() + ) + .unwrap(), + BUFFERED_PROTOCOL_OUTPUT_LIMIT_BYTES + ); + + // Ordinary responses above the old 64 MiB quota are valid. Lengths + // beyond the signed bridge range arrive as negative i32 values. + assert_eq!( + checked_buffered_protocol_output_len(80 * 1024 * 1024).unwrap(), + 80 * 1024 * 1024 + ); + assert!(checked_buffered_protocol_output_len(i32::MIN).is_err()); + + let negative = checked_buffered_protocol_output_len(-1).unwrap_err(); + assert!(negative.to_string().contains("invalid length -1")); + } + + #[test] + fn buffered_protocol_output_allocation_is_fallible_and_zeroed() { + let output = allocate_zeroed_buffered_protocol_output(4).unwrap(); + assert_eq!(output, [0, 0, 0, 0]); + + let allocation_failure = allocate_zeroed_buffered_protocol_output(usize::MAX).unwrap_err(); + assert!( + allocation_failure + .to_string() + .contains("reserve bounded WASIX protocol output") + ); + } + + #[test] + fn startup_outcome_descriptor_is_versioned_and_exact() -> Result<()> { + let mut bytes = [0u8; STARTUP_OUTCOME_DESCRIPTOR_SIZE]; + bytes[0..4].copy_from_slice(&STARTUP_OUTCOME_ABI_VERSION.to_le_bytes()); + bytes[4..8].copy_from_slice(&(STARTUP_OUTCOME_DESCRIPTOR_SIZE as u32).to_le_bytes()); + bytes[8..12].copy_from_slice(&STARTUP_OUTCOME_REJECTED.to_le_bytes()); + bytes[16..24].copy_from_slice(&0x1234_u64.to_le_bytes()); + bytes[24..32].copy_from_slice(&0x5678_u64.to_le_bytes()); + + assert_eq!( + decode_startup_outcome_descriptor(&bytes)?, + StartupOutcomeDescriptor { + kind: STARTUP_OUTCOME_REJECTED, + protocol_ptr: 0x1234, + protocol_len: 0x5678, + } + ); + + let mut wrong_version = bytes; + wrong_version[0..4].copy_from_slice(&2_u32.to_le_bytes()); + assert!( + decode_startup_outcome_descriptor(&wrong_version) + .unwrap_err() + .to_string() + .contains("ABI version 2") + ); + let mut wrong_size = bytes; + wrong_size[4..8].copy_from_slice(&24_u32.to_le_bytes()); + assert!( + decode_startup_outcome_descriptor(&wrong_size) + .unwrap_err() + .to_string() + .contains("descriptor size 24") + ); + let mut reserved = bytes; + reserved[12..16].copy_from_slice(&1_u32.to_le_bytes()); + assert!( + decode_startup_outcome_descriptor(&reserved) + .unwrap_err() + .to_string() + .contains("reserved field") + ); + Ok(()) + } + + #[test] + fn atomic_startup_response_requires_complete_protocol_frames_and_error() { + let error = + oliphaunt_query::wire::error_response("FATAL", "3D000", "database does not exist"); + assert!(complete_protocol_response_contains_error(&error)); + + let mut notice_then_error = vec![b'N', 0, 0, 0, 5, 0]; + notice_then_error.extend_from_slice(&error); + assert!(complete_protocol_response_contains_error( + ¬ice_then_error + )); + + let mut trailing = error.clone(); + trailing.push(0xff); + assert!(!complete_protocol_response_contains_error(&trailing)); + assert!(!complete_protocol_response_contains_error( + &error[..error.len() - 1] + )); + assert!(!complete_protocol_response_contains_error(&[ + b'N', 0, 0, 0, 5, 0 + ])); + assert!(!complete_protocol_response_contains_error(&[ + b'E', 0, 0, 0, 4 + ])); + assert!(!complete_protocol_response_contains_error(&[ + b'E', 0, 0, 0, 5, 0 + ])); + assert!(!complete_protocol_response_contains_error(&[ + b'E', 0, 0, 0, 10, b'C', b'3', b'D', b'0', b'0', 0 + ])); + assert!(!complete_protocol_response_contains_error(&[ + b'E', 0, 0, 0, 13, b'C', b'3', b'D', b'0', b'0', b'0', b'0', 0, 0 + ])); + assert!(!complete_protocol_response_contains_error(&[ + b'E', 0, 0, 0, 10, b'C', b'3', b'D', b'0', b'0', b'0' + ])); + assert!(!complete_protocol_response_contains_error(&[])); + } + #[test] fn postgres_argv_delimits_an_option_like_database_name() -> Result<()> { let startup = StartupConfig { @@ -2055,6 +2632,12 @@ mod tests { #[test] fn split_initdb_selects_exact_collation_profile_environment() { + // Main's initdb patch uses environment policy, not a seed-profile CLI. + assert!( + split_initdb_args() + .iter() + .all(|arg| arg != "--oliphaunt-seed-profile") + ); assert_eq!( split_initdb_profile_environment(false), vec![(SKIP_ICU_COLLATION_DISCOVERY_ENV, "1")] @@ -2068,6 +2651,60 @@ mod tests { ); } + #[test] + fn guest_phase_failure_blocks_restore_and_reuse() { + for phase in [ + "input reset", + "input reservation", + "input availability", + "dispatch progress", + "set protocol transport", + "read protocol stream state", + "restore protocol transport", + ] { + let mut terminal = None; + let mut subsequent_guest_calls = 0; + let error = finish_guest_phase::<()>( + Err(anyhow::anyhow!("injected trap or invalid result")), + phase, + |failure| terminal = Some(failure), + ) + .unwrap_err(); + assert!(error.to_string().contains("the backend is closed")); + assert_eq!( + format!("{error:#}") + .matches("injected trap or invalid result") + .count(), + 1 + ); + assert!(terminal.as_deref().unwrap().contains(phase)); + // Both cleanup restores and future dispatch use this admission gate. + for _ in 0..2 { + let attempt = ensure_guest_phase_live(terminal.as_deref()).map(|()| { + subsequent_guest_calls += 1; + }); + assert!(attempt.is_err()); + } + assert_eq!(subsequent_guest_calls, 0); + } + } + + #[test] + fn successful_guest_phase_and_predispatch_validation_do_not_poison() { + let mut terminal = None; + assert_eq!( + finish_guest_phase(Ok(7), "dispatch", |failure| { + terminal = Some(failure); + }) + .unwrap(), + 7 + ); + assert!(terminal.is_none()); + assert!(validate_protocol_input_length(i32::MAX as usize).is_ok()); + assert!(validate_protocol_input_length(i32::MAX as usize + 1).is_err()); + assert!(ensure_guest_phase_live(terminal.as_deref()).is_ok()); + } + #[test] fn private_initdb_settings_do_not_escape_bootstrap() -> Result<()> { let args = split_initdb_args(); diff --git a/src/wasix/sdks/rust/src/oliphaunt/protocol_limits_generated.rs b/src/wasix/sdks/rust/src/oliphaunt/protocol_limits_generated.rs new file mode 100644 index 000000000..6cdb96f97 --- /dev/null +++ b/src/wasix/sdks/rust/src/oliphaunt/protocol_limits_generated.rs @@ -0,0 +1,10 @@ +// Generated by src/wasix/runtime/protocol-contract/generate.mjs. +// Edit contract.json, not these ABI limits. See src/docs/maintainers/runtime-resource-budgets.md. +// Signed i32 bridge length ceiling, not an application quota or eager allocation. +pub(crate) const BUFFERED_PROTOCOL_OUTPUT_LIMIT_BYTES: usize = 2147483647; +// Maximum callback slice; independent of socket read and COPY frame sizes. +pub(crate) const PROTOCOL_CALLBACK_CHUNK_BYTES: usize = 65536; +pub(crate) const PROTOCOL_BUFFERED: i32 = 0; +pub(crate) const PROTOCOL_STREAM: i32 = 1; +pub(crate) const PROTOCOL_HYBRID: i32 = 2; +pub(crate) const PROTOCOL_BUFFERED_INPUT_STREAMED_OUTPUT: i32 = 3; diff --git a/src/wasix/sdks/rust/src/oliphaunt/query.rs b/src/wasix/sdks/rust/src/oliphaunt/query.rs index 53814c7f0..57365fc29 100644 --- a/src/wasix/sdks/rust/src/oliphaunt/query.rs +++ b/src/wasix/sdks/rust/src/oliphaunt/query.rs @@ -13,6 +13,18 @@ pub(crate) fn simple_query(sql: &str) -> Result> { query_core_result(query_core::simple_query(sql)) } +pub(crate) fn read_backend_message(bytes: &[u8]) -> Result<(u8, &[u8], &[u8])> { + query_core_result(query_core::read_backend_message(bytes)) +} + +pub(crate) fn parse_postgres_error(body: &[u8]) -> Result { + let fields = query_core_result(query_core::parse_diagnostic_fields(body, "ErrorResponse"))?; + Ok(PostgresError::from_core(query_core::diagnostic( + fields, + "PostgreSQL ErrorResponse", + ))) +} + fn query_core_result(result: query_core::Result) -> Result { result.map_err(query_core_error) } diff --git a/src/wasix/sdks/rust/src/oliphaunt/sql.rs b/src/wasix/sdks/rust/src/oliphaunt/sql.rs deleted file mode 100644 index be9766c20..000000000 --- a/src/wasix/sdks/rust/src/oliphaunt/sql.rs +++ /dev/null @@ -1,3 +0,0 @@ -pub(crate) fn quote_identifier(identifier: &str) -> String { - format!("\"{}\"", identifier.replace('"', "\"\"")) -} diff --git a/src/wasix/sdks/rust/src/oliphaunt/tools.rs b/src/wasix/sdks/rust/src/oliphaunt/tools.rs index f30c7895c..d509373af 100644 --- a/src/wasix/sdks/rust/src/oliphaunt/tools.rs +++ b/src/wasix/sdks/rust/src/oliphaunt/tools.rs @@ -62,7 +62,10 @@ const PSQL_VALUE_OPTIONS: &[&str] = &[ "--variable", ]; -/// Options for the bundled WASIX `pg_dump` runner. +/// Options for the bundled, in-memory WASIX `pg_dump` runner. +/// +/// Captures complete output in memory, subject to available memory and host +/// string-size limits. A streaming or sink API is not currently available. #[derive(Debug, Clone, Default, PartialEq, Eq)] pub struct PgDumpOptions { args: Vec, @@ -102,6 +105,9 @@ impl PgDumpOptions { } /// Structured failure from a packaged PostgreSQL frontend program. +/// +/// On output-capture failure, [`Self::stdout`] and [`Self::stderr`] are empty +/// so neither stream exposes an incomplete prefix. #[derive(Debug)] pub struct PostgresToolError { tool: &'static str, @@ -113,6 +119,55 @@ pub struct PostgresToolError { cause: anyhow::Error, } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum ToolOutputCaptureFailure { + AllocationFailed, + LockPoisoned, +} + +impl ToolOutputCaptureFailure { + fn io_error(self) -> std::io::Error { + let kind = match self { + Self::AllocationFailed => std::io::ErrorKind::OutOfMemory, + Self::LockPoisoned => std::io::ErrorKind::Other, + }; + std::io::Error::new(kind, self) + } +} + +impl fmt::Display for ToolOutputCaptureFailure { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::AllocationFailed => { + formatter.write_str("could not reserve memory for tool output") + } + Self::LockPoisoned => { + formatter.write_str("stdout and stderr capture lock was poisoned") + } + } + } +} + +impl StdError for ToolOutputCaptureFailure {} + +#[derive(Debug)] +struct ToolOutputCaptureError { + failure: ToolOutputCaptureFailure, + runner_failure: Option, +} + +impl fmt::Display for ToolOutputCaptureError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(formatter, "{}", self.failure)?; + if let Some(runner_failure) = &self.runner_failure { + write!(formatter, "; WASIX tool also failed: {runner_failure}")?; + } + Ok(()) + } +} + +impl StdError for ToolOutputCaptureError {} + /// Internal marker for a broken virtual connection after a tool may have sent work. #[derive(Debug)] pub(crate) struct DirectToolOutcomeUnknown; @@ -152,14 +207,21 @@ impl PostgresToolError { self.tool } + /// Return the frontend program's exit code when one is available. pub fn exit_code(&self) -> Option { self.exit_code } + /// Return captured standard output. + /// + /// This is empty after output-capture overflow. pub fn stdout(&self) -> &str { &self.stdout } + /// Return captured standard error. + /// + /// This is empty after output-capture overflow. pub fn stderr(&self) -> &str { &self.stderr } @@ -177,6 +239,9 @@ impl PostgresToolError { impl fmt::Display for PostgresToolError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + if let Some(cause) = self.cause.downcast_ref::() { + return write!(formatter, "{} output capture failed: {cause}", self.tool); + } if self .cause .downcast_ref::() @@ -233,7 +298,10 @@ impl PgDumpOptions { } } -/// Options for the bundled WASIX `psql` runner. +/// Options for the bundled, in-memory WASIX `psql` runner. +/// +/// Captures complete output in memory, subject to available memory and host +/// string-size limits. A streaming or sink API is not currently available. #[derive(Debug, Clone, Default, PartialEq, Eq)] pub struct PsqlOptions { args: Vec, @@ -507,6 +575,8 @@ impl PostgresToolOutput { } } +type SharedToolOutputCapture = Arc>; + struct ToolInvocation<'a, N> { name: &'static str, wasm: &'a [u8], @@ -574,8 +644,7 @@ where (host_fs, wasix_runtime) }; - let stdout = Arc::new(Mutex::new(Vec::new())); - let stderr = Arc::new(Mutex::new(Vec::new())); + let capture = Arc::new(Mutex::new(ToolOutputCapture::default())); let mut runner = WasiRunner::new(); runner .with_mount("/".to_owned(), Arc::clone(&host_fs)) @@ -587,8 +656,14 @@ where ("PGSSLMODE", "disable"), ("PGCLIENTENCODING", "UTF8"), ]) - .with_stdout(Box::new(CaptureFile::new(Arc::clone(&stdout)))) - .with_stderr(Box::new(CaptureFile::new(Arc::clone(&stderr)))); + .with_stdout(Box::new(CaptureFile::new( + Arc::clone(&capture), + ToolOutputStream::Stdout, + ))) + .with_stderr(Box::new(CaptureFile::new( + Arc::clone(&capture), + ToolOutputStream::Stderr, + ))); if name == "psql" { runner.with_envs([("OLIPHAUNT_PSQL_NONINTERACTIVE", "1")]); } @@ -596,27 +671,66 @@ where Some(input) => runner.with_stdin(Box::new(virtual_fs::StaticFile::new(input))), None => runner.with_stdin(Box::::default()), }; - if let Err(cause) = runner.run_wasm( + let run_result = runner.run_wasm( RuntimeOrEngine::Runtime(Arc::new(wasix_runtime)), name, module, ModuleHash::sha256(wasm), - ) { - let exit_code = cause + ); + finish_wasix_client_tool_run(name, run_result, capture) +} + +fn finish_wasix_client_tool_run( + tool: &'static str, + run_result: Result<()>, + capture: SharedToolOutputCapture, +) -> Result { + let exit_code = match &run_result { + Ok(()) => Some(0), + Err(cause) => cause .chain() .find_map(|error| error.downcast_ref::()) .and_then(WasiRuntimeError::as_exit_code) - .map(|code| code.raw()); - let stdout = stdout.lock().expect("stdout capture poisoned").clone(); - let stderr = stderr.lock().expect("stderr capture poisoned").clone(); - return Err(anyhow::Error::new(PostgresToolError::from_output( - name, exit_code, stdout, stderr, cause, - ))); + .map(|code| code.raw()), + }; + + let output = take_tool_output(&capture); + match (run_result, output) { + (Ok(()), Ok(output)) => Ok(output), + (Err(cause), Ok(PostgresToolOutput { stdout, stderr })) => Err(anyhow::Error::new( + PostgresToolError::from_output(tool, exit_code, stdout, stderr, cause), + )), + (run_result, Err(failure)) => { + let runner_failure = run_result.err().map(|cause| format!("{cause:#}")); + Err(anyhow::Error::new(PostgresToolError::from_output( + tool, + exit_code, + Vec::new(), + Vec::new(), + anyhow::Error::new(ToolOutputCaptureError { + failure, + runner_failure, + }), + ))) + } } +} - let stdout = std::mem::take(&mut *stdout.lock().expect("stdout capture poisoned")); - let stderr = std::mem::take(&mut *stderr.lock().expect("stderr capture poisoned")); - Ok(PostgresToolOutput { stdout, stderr }) +fn take_tool_output( + capture: &SharedToolOutputCapture, +) -> std::result::Result { + let mut capture = capture + .lock() + .map_err(|_| ToolOutputCaptureFailure::LockPoisoned)?; + if let Some(failure) = capture.failure { + capture.clear_buffers(); + return Err(failure); + } + let output = PostgresToolOutput { + stdout: std::mem::take(&mut capture.stdout), + stderr: std::mem::take(&mut capture.stderr), + }; + Ok(output) } fn pg_dump_with_networking( @@ -1022,14 +1136,96 @@ fn receive_direct_tool_socket( } } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum ToolOutputStream { + Stdout, + Stderr, +} + +#[derive(Debug, Default)] +struct ToolOutputCapture { + stdout: Vec, + stderr: Vec, + failure: Option, +} + +impl ToolOutputCapture { + fn stream_len(&self, stream: ToolOutputStream) -> usize { + match stream { + ToolOutputStream::Stdout => self.stdout.len(), + ToolOutputStream::Stderr => self.stderr.len(), + } + } + + fn stream_mut(&mut self, stream: ToolOutputStream) -> &mut Vec { + match stream { + ToolOutputStream::Stdout => &mut self.stdout, + ToolOutputStream::Stderr => &mut self.stderr, + } + } + + fn clear_buffers(&mut self) { + self.stdout = Vec::new(); + self.stderr = Vec::new(); + } + + fn fail(&mut self, failure: ToolOutputCaptureFailure) -> std::io::Error { + self.clear_buffers(); + let failure = *self.failure.get_or_insert(failure); + failure.io_error() + } + + fn reserve_for(&mut self, stream: ToolOutputStream, additional: usize) -> std::io::Result<()> { + if let Some(failure) = self.failure { + return Err(failure.io_error()); + } + if self.stream_mut(stream).try_reserve(additional).is_err() { + return Err(self.fail(ToolOutputCaptureFailure::AllocationFailed)); + } + Ok(()) + } + + fn write(&mut self, stream: ToolOutputStream, bytes: &[u8]) -> std::io::Result { + self.reserve_for(stream, bytes.len())?; + self.stream_mut(stream).extend_from_slice(bytes); + Ok(bytes.len()) + } + + fn write_vectored( + &mut self, + stream: ToolOutputStream, + buffers: &[std::io::IoSlice<'_>], + ) -> std::io::Result { + let Some(additional) = buffers + .iter() + .try_fold(0usize, |total, buffer| total.checked_add(buffer.len())) + else { + return Err(self.fail(ToolOutputCaptureFailure::AllocationFailed)); + }; + self.reserve_for(stream, additional)?; + let output = self.stream_mut(stream); + for buffer in buffers { + output.extend_from_slice(buffer); + } + Ok(additional) + } +} + #[derive(Debug)] struct CaptureFile { - buffer: Arc>>, + capture: SharedToolOutputCapture, + stream: ToolOutputStream, } impl CaptureFile { - fn new(buffer: Arc>>) -> Self { - Self { buffer } + fn new(capture: SharedToolOutputCapture, stream: ToolOutputStream) -> Self { + Self { capture, stream } + } + + fn capture_lock(&self) -> std::io::Result> { + self.capture + .lock() + .map_err(|_| ToolOutputCaptureFailure::LockPoisoned.io_error()) } } @@ -1047,7 +1243,10 @@ impl VirtualFile for CaptureFile { } fn size(&self) -> u64 { - self.buffer.lock().expect("capture lock poisoned").len() as u64 + self.capture + .lock() + .map(|capture| capture.stream_len(self.stream) as u64) + .unwrap_or(0) } fn set_len(&mut self, _new_size: u64) -> Result<(), wasmer_wasix::FsError> { @@ -1069,7 +1268,13 @@ impl VirtualFile for CaptureFile { self: Pin<&mut Self>, _cx: &mut TaskContext<'_>, ) -> Poll> { - Poll::Ready(Ok(8192)) + let ready = self + .capture_lock() + .and_then(|capture| match capture.failure { + Some(failure) => Err(failure.io_error()), + None => Ok(8192), + }); + Poll::Ready(ready) } } @@ -1089,7 +1294,7 @@ impl AsyncWrite for CaptureFile { _cx: &mut TaskContext<'_>, buf: &[u8], ) -> Poll> { - Poll::Ready(self.write(buf)) + Poll::Ready(Write::write(self.as_mut().get_mut(), buf)) } fn poll_flush(self: Pin<&mut Self>, _cx: &mut TaskContext<'_>) -> Poll> { @@ -1099,6 +1304,18 @@ impl AsyncWrite for CaptureFile { fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut TaskContext<'_>) -> Poll> { Poll::Ready(Ok(())) } + + fn poll_write_vectored( + mut self: Pin<&mut Self>, + _cx: &mut TaskContext<'_>, + buffers: &[std::io::IoSlice<'_>], + ) -> Poll> { + Poll::Ready(Write::write_vectored(self.as_mut().get_mut(), buffers)) + } + + fn is_write_vectored(&self) -> bool { + true + } } impl AsyncSeek for CaptureFile { @@ -1122,11 +1339,13 @@ impl Read for CaptureFile { impl Write for CaptureFile { fn write(&mut self, buf: &[u8]) -> std::io::Result { - self.buffer - .lock() - .expect("capture lock poisoned") - .extend_from_slice(buf); - Ok(buf.len()) + let stream = self.stream; + self.capture_lock()?.write(stream, buf) + } + + fn write_vectored(&mut self, buffers: &[std::io::IoSlice<'_>]) -> std::io::Result { + let stream = self.stream; + self.capture_lock()?.write_vectored(stream, buffers) } fn flush(&mut self) -> std::io::Result<()> { @@ -1144,6 +1363,125 @@ impl Seek for CaptureFile { mod tests { use super::*; + fn capture_files() -> (SharedToolOutputCapture, CaptureFile, CaptureFile) { + let capture = Arc::new(Mutex::new(ToolOutputCapture::default())); + let stdout = CaptureFile::new(Arc::clone(&capture), ToolOutputStream::Stdout); + let stderr = CaptureFile::new(Arc::clone(&capture), ToolOutputStream::Stderr); + (capture, stdout, stderr) + } + + #[test] + fn tool_output_capture_preserves_exact_bytes() { + let (capture, mut stdout, mut stderr) = capture_files(); + Write::write_all(&mut stdout, b"dump\0").unwrap(); + Write::write_all(&mut stderr, b"warn\n").unwrap(); + + let output = finish_wasix_client_tool_run("pg_dump", Ok(()), capture).unwrap(); + assert_eq!(output.stdout, b"dump\0"); + assert_eq!(output.stderr, b"warn\n"); + } + + #[test] + fn tool_output_capture_accepts_more_than_64_mib() { + let (capture, mut stdout, mut stderr) = capture_files(); + let chunk = vec![b'x'; 1024 * 1024]; + let buffers = [std::io::IoSlice::new(&chunk), std::io::IoSlice::new(b"!")]; + for _ in 0..65 { + assert_eq!( + Write::write_vectored(&mut stdout, &buffers).unwrap(), + chunk.len() + 1 + ); + } + Write::write_all(&mut stderr, b"diagnostic").unwrap(); + let output = finish_wasix_client_tool_run("pg_dump", Ok(()), capture).unwrap(); + assert_eq!(output.stdout.len(), 65 * (chunk.len() + 1)); + for block in output.stdout.chunks_exact(chunk.len() + 1) { + assert_eq!(&block[..chunk.len()], chunk); + assert_eq!(block[chunk.len()], b'!'); + } + assert_eq!(output.stderr, b"diagnostic"); + } + + #[test] + fn tool_output_failure_is_sticky_and_fail_closed() { + let (capture, mut stdout, mut stderr) = capture_files(); + Write::write_all(&mut stdout, b"12345").unwrap(); + Write::write_all(&mut stderr, b"67890").unwrap(); + + let failure = capture + .lock() + .unwrap() + .reserve_for(ToolOutputStream::Stdout, usize::MAX) + .unwrap_err(); + assert!(failure.to_string().contains("could not reserve memory")); + assert_eq!(capture.lock().unwrap().stdout.capacity(), 0); + assert_eq!(capture.lock().unwrap().stderr.capacity(), 0); + let sticky = Write::write(&mut stderr, b"still rejected").unwrap_err(); + assert!(sticky.to_string().contains("could not reserve memory")); + + let error = finish_wasix_client_tool_run("pg_dump", Ok(()), capture) + .expect_err("a sticky capture failure must override guest exit 0"); + let structured = error + .downcast_ref::() + .expect("capture failure must remain a structured tool failure"); + assert_eq!(structured.exit_code(), Some(0)); + assert_eq!(structured.stdout(), ""); + assert_eq!(structured.stderr(), ""); + assert!( + structured + .to_string() + .contains("output capture failed: could not reserve memory") + ); + + let direct_error = finish_direct_tool::<()>(true, Ok(()), Err(error)) + .expect_err("capture failure must remain an ordinary completed tool failure"); + assert!(!is_direct_tool_outcome_unknown(&direct_error)); + } + + #[test] + fn tool_output_capture_rejects_set_len_without_changing_output() { + let (capture, mut stdout, _stderr) = capture_files(); + Write::write_all(&mut stdout, b"data").unwrap(); + assert!(matches!( + VirtualFile::set_len(&mut stdout, u64::MAX), + Err(wasmer_wasix::FsError::PermissionDenied) + )); + assert_eq!(VirtualFile::size(&stdout), 4); + + let output = finish_wasix_client_tool_run("pg_dump", Ok(()), capture).unwrap(); + assert_eq!(output.stdout, b"data"); + assert!(output.stderr.is_empty()); + } + + #[test] + fn poisoned_tool_output_lock_returns_errors_instead_of_panicking() { + let (capture, mut stdout, _stderr) = capture_files(); + let poisoned = Arc::clone(&capture); + assert!( + thread::spawn(move || { + let _guard = poisoned.lock().unwrap(); + panic!("intentionally poison tool output capture"); + }) + .join() + .is_err() + ); + + assert_eq!(VirtualFile::size(&stdout), 0); + assert!( + Write::write(&mut stdout, b"data") + .unwrap_err() + .to_string() + .contains("lock was poisoned") + ); + let error = finish_wasix_client_tool_run("psql", Ok(()), capture) + .expect_err("a poisoned capture lock must fail the invocation"); + let structured = error.downcast_ref::().unwrap(); + assert_eq!(structured.exit_code(), Some(0)); + assert_eq!(structured.stdout(), ""); + assert_eq!(structured.stderr(), ""); + assert!(structured.to_string().contains("capture lock was poisoned")); + } + #[test] fn shared_option_fixture_matches_wasix_validation() { let fixture: serde_json::Value = serde_json::from_str( diff --git a/src/wasix/sdks/rust/tests/runtime_smoke.rs b/src/wasix/sdks/rust/tests/runtime_smoke.rs index d3ee6be00..5ac5b2684 100644 --- a/src/wasix/sdks/rust/tests/runtime_smoke.rs +++ b/src/wasix/sdks/rust/tests/runtime_smoke.rs @@ -1,5 +1,3 @@ -#![cfg(feature = "extensions")] - use anyhow::Result; use oliphaunt_wasix::{ AsyncOliphaunt, AsyncTransaction, DatabaseStorage, Error, Oliphaunt, TransactionResult, @@ -25,6 +23,185 @@ fn synthetic_sdk_error() -> oliphaunt_wasix::Error { .expect_err("invalid archive creates a public SDK error") } +#[test] +fn terminal_raw_exchange_reports_closed_without_hiding_cleanup() -> Result<()> { + let workspace = tempfile::TempDir::new()?; + let storage = DatabaseStorage::Directory(workspace.path().join("terminal")); + let mut database = Oliphaunt::builder().storage(storage.clone()).open()?; + database.execute("CREATE TABLE kept(value int)")?; + database.execute("INSERT INTO kept VALUES (1)")?; + database.exec_protocol_raw(oliphaunt_query::simple_query( + "BEGIN; INSERT INTO kept VALUES (2)", + )?)?; + // A frontend Terminate ends the backend; there is no response to collect. + database + .exec_protocol_raw([b'X', 0, 0, 0, 4]) + .expect_err("terminated backend"); + assert!(database.is_closed()); + assert_eq!( + database.query("SELECT 1").unwrap_err().kind(), + oliphaunt_wasix::ErrorKind::Lifecycle + ); + // Discard the failed guest and release the root lock without re-entering it. + database.close()?; + database.close()?; + let mut reopened = Oliphaunt::builder().storage(storage).open()?; + assert_eq!( + reopened + .query("SELECT sum(value)::text AS value FROM kept")? + .get_text(0, "value")?, + Some("1") + ); + reopened.close()?; + Ok(()) +} + +#[test] +fn buffered_query_above_64_mib_keeps_session_usable() -> Result<()> { + let workspace = tempfile::TempDir::new()?; + for storage in [ + DatabaseStorage::Memory, + DatabaseStorage::Directory(workspace.path().join("large")), + ] { + let mut database = Oliphaunt::builder().storage(storage).open()?; + // Many ordinary rows: callers should not need a transport flag just + // because their complete result exceeds an internal buffer budget. + let result = + database.query("SELECT repeat('x', 8192) AS payload FROM generate_series(1, 10240)")?; + assert_eq!(result.rows().len(), 10240); + for row in [0, 10239] { + assert_eq!(result.get_text(row, "payload")?.unwrap().len(), 8192); + } + drop(result); + assert_eq!( + database.query("SELECT 42 AS value")?.get_text(0, "value")?, + Some("42") + ); + database.close()?; + } + Ok(()) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn terminal_async_exchange_can_close_and_reopen_the_same_directory() -> Result<()> { + let workspace = tempfile::TempDir::new()?; + let storage = DatabaseStorage::Directory(workspace.path().join("terminal-async")); + let database = AsyncOliphaunt::builder() + .storage(storage.clone()) + .open() + .await?; + database + .exec_protocol_raw([b'X', 0, 0, 0, 4]) + .await + .expect_err("terminated backend"); + assert_eq!( + database.query("SELECT 1").await.unwrap_err().kind(), + oliphaunt_wasix::ErrorKind::Lifecycle + ); + database.close().await?; + assert!(database.is_closed()); + let reopened = AsyncOliphaunt::builder().storage(storage).open().await?; + assert_eq!( + reopened + .query("SELECT 42 AS value") + .await? + .get_text(0, "value")?, + Some("42") + ); + reopened.close().await?; + Ok(()) +} + +#[test] +fn configured_startup_identity_survives_reset_role() -> Result<()> { + let workspace = tempfile::TempDir::new()?; + let storage = DatabaseStorage::Directory(workspace.path().join("identity")); + let mut setup = Oliphaunt::builder().storage(storage.clone()).open()?; + setup.execute("CREATE ROLE patch_app LOGIN")?; + setup.execute("CREATE ROLE patch_member NOLOGIN")?; + setup.execute("GRANT patch_member TO patch_app")?; + setup.execute("ALTER ROLE patch_app SET work_mem = '9MB'")?; + setup.execute("CREATE ROLE patch_blocked LOGIN")?; + setup.execute("CREATE ROLE patch_limited LOGIN CONNECTION LIMIT 0")?; + setup.execute("CREATE ROLE patch_no_connect LOGIN")?; + setup.execute("REVOKE CONNECT ON DATABASE postgres FROM PUBLIC")?; + setup.execute("GRANT CONNECT ON DATABASE postgres TO patch_app, patch_blocked")?; + setup.execute( + "CREATE FUNCTION patch_login() RETURNS event_trigger LANGUAGE plpgsql AS $$ + BEGIN + IF session_user = 'patch_blocked' THEN + RAISE EXCEPTION 'patch login denied' USING ERRCODE = '28000'; + END IF; + PERFORM set_config('patch.login', 'fired', false); + END $$", + )?; + setup.execute("CREATE EVENT TRIGGER patch_login ON login EXECUTE FUNCTION patch_login()")?; + setup.close()?; + + let mut database = Oliphaunt::builder() + .storage(storage.clone()) + .username("patch_app") + .open()?; + let identity = database.query( + "SELECT current_user::text AS current_role, session_user::text AS session_role, current_setting('work_mem') AS work_mem", + )?; + assert_eq!(identity.get_text(0, "current_role")?, Some("patch_app")); + assert_eq!(identity.get_text(0, "session_role")?, Some("patch_app")); + assert_eq!(identity.get_text(0, "work_mem")?, Some("9MB")); + let login = database.query("SELECT current_setting('patch.login') AS login")?; + assert_eq!(login.get_text(0, "login")?, Some("fired")); + database.execute("SET ROLE patch_member")?; + database.execute("RESET ROLE")?; + let reset = database.query("SELECT current_user::text AS current_role")?; + assert_eq!(reset.get_text(0, "current_role")?, Some("patch_app")); + database.execute("SET work_mem = '12MB'")?; + database.execute("DISCARD ALL")?; + let discarded = database.query( + "SELECT current_user::text AS current_role, current_setting('work_mem') AS work_mem", + )?; + assert_eq!(discarded.get_text(0, "current_role")?, Some("patch_app")); + assert_eq!(discarded.get_text(0, "work_mem")?, Some("9MB")); + let denied = database + .execute("SET ROLE postgres") + .expect_err("no bootstrap-superuser escape"); + assert_eq!( + denied + .postgres_error() + .and_then(|error| error.sqlstate.as_deref()), + Some("42501") + ); + database.close()?; + for (username, expected, sqlstate) in [ + ("patch_member", "not permitted to log in", "28000"), + ("patch_missing", "does not exist", "28000"), + ("patch_blocked", "patch login denied", "28000"), + ("patch_limited", "too many connections", "53300"), + ( + "patch_no_connect", + "permission denied for database", + "42501", + ), + ] { + let error = Oliphaunt::builder() + .storage(storage.clone()) + .username(username) + .open() + .err() + .expect("startup must enforce the configured role's admission policy"); + assert!(error.to_string().contains(expected), "{username}: {error}"); + assert_eq!( + error + .postgres_error() + .and_then(|error| error.sqlstate.as_deref()), + Some(sqlstate), + "{username}: admission must preserve PostgreSQL's structured error", + ); + } + // Rejected startup must leave the directory usable by a permitted session. + Oliphaunt::builder().storage(storage).open()?.close()?; + Ok(()) +} + #[test] fn direct_api_query_transaction_persistence_and_backup() -> Result<()> { let workspace = tempfile::TempDir::new()?; @@ -209,6 +386,74 @@ fn direct_protocol_callback_error_and_panic_recover_before_returning() -> Result Ok(()) } +#[test] +fn copy_output_streams_exact_bytes_and_recovers_after_callback_failure() -> Result<()> { + let workspace = tempfile::TempDir::new()?; + let expected: String = (1..=10_000) + .map(|row| format!("{row}\trow-{row}\n")) + .collect(); + let request = oliphaunt_query::simple_query( + "COPY (SELECT i, 'row-' || i FROM generate_series(1, 10000) AS i) TO STDOUT", + )?; + for storage in [ + DatabaseStorage::Memory, + DatabaseStorage::Directory(workspace.path().join("copy")), + ] { + let mut database = Oliphaunt::builder().storage(storage).open()?; + let output = Arc::new(Mutex::new((Vec::new(), 0))); + let capture = Arc::clone(&output); + database.exec_protocol_raw_stream(&request, move |chunk| { + assert!( + chunk.len() <= 64 * 1024, + "callback exceeds transport contract" + ); + let mut capture = capture.lock().expect("COPY capture"); + capture.1 += 1; + capture.0.extend_from_slice(chunk); + })?; + let captured = output.lock().expect("COPY capture"); + let (output, callbacks) = &*captured; + assert!(*callbacks > 1, "COPY must cross a callback boundary"); + let mut frames = output.as_slice(); + let mut data = Vec::new(); + let mut complete = false; + let mut ready = false; + while !frames.is_empty() { + let (tag, body, rest) = oliphaunt_query::read_backend_message(frames)?; + match tag { + b'd' => data.extend_from_slice(body), + b'C' => { + assert_eq!(body, b"COPY 10000\0"); + complete = true; + } + b'Z' => { + assert_eq!(body, b"I"); + ready = true; + } + b'E' => panic!("unexpected COPY ErrorResponse: {body:?}"), + _ => {} + } + frames = rest; + } + assert_eq!(data, expected.as_bytes()); + assert!(complete && ready, "COPY did not finish at ReadyForQuery"); + let mut failure = Some(synthetic_sdk_error()); + database + .exec_protocol_raw_stream(&request, move |_| { + Err(failure.take().expect("one callback")) + }) + .expect_err("callback failure must propagate after draining COPY"); + assert_eq!( + database + .query("SELECT 42::int4 AS answer")? + .get_text(0, "answer")?, + Some("42") + ); + database.close()?; + } + Ok(()) +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn async_api_owns_the_runtime_and_serializes_clones() -> Result<()> { let caller_thread = std::thread::current().id(); diff --git a/src/wasix/sdks/ts/ARCHITECTURE.md b/src/wasix/sdks/ts/ARCHITECTURE.md index 8b037b8fe..fd1aec4bf 100644 --- a/src/wasix/sdks/ts/ARCHITECTURE.md +++ b/src/wasix/sdks/ts/ARCHITECTURE.md @@ -127,9 +127,12 @@ smaller qualified side modules remain supported in a direct Window. 5. The direct export driver completes the exported startup transition before exposing the session. Selected carriers contribute verified artifacts and required startup/preload configuration only; database-local extension SQL is - application/ORM-owned. A requested non-default user is selected from existing - roles with `SET ROLE`; standalone bootstrap remains the fixed `postgres` - identity. + application/ORM-owned. PostgreSQL selects the trusted configured `PGUSER` + as the actual session principal during startup, applying catalog login, + database access, settings and login-trigger policy. The host does not emulate + that identity with later `SET ROLE` SQL; `RESET ROLE` and `DISCARD ALL` restore + the configured session principal. Fresh cluster seeds retain `postgres` as + their bootstrap owner. 6. The binding frames later responses through `ReadyForQuery` and exposes serialized `query`, `execute`, buffered `execProtocolRaw`, callback `execProtocolRawStream`, and callback-scoped `transaction` calls @@ -142,11 +145,11 @@ smaller qualified side modules remain supported in a direct Window. final boundary. A new persistent synchronous-OPFS root uses a separate internal full-publication boundary after initialization; it is not a public database operation. PostgreSQL `CHECKPOINT` remains available through ordinary - `execute`. If a - PostgreSQL `ERROR` crosses the host boundary, the direct host - invokes `PostgresMainLongJmp`, sends and flushes readiness, and continues - through `PostgresMainLoopOnce`. Normal ErrorResponse returns receive the same - top-level cleanup as trapping errors. + `execute`. PostgreSQL `ERROR` recovery stays inside the live guest invocation: + `PostgresMainLoopOnce` returns a typed outcome (processed, recovered, or input + ended). The host never resumes a guest `longjmp` after unwinding into the host. + Unexpected traps, invalid outcomes, and failed guest phases close the backend; + ordinary SQL errors remain pgwire ErrorResponses with their original SQLSTATE. 7. `close` establishes a terminal admission cutoff and lets already accepted database work drain. The direct owner sends PostgreSQL Terminate through the same direct bridge, deactivates the @@ -485,16 +488,19 @@ shortcut. Realtime uses the JavaScript epoch clock, while monotonic reads calibrate the host's monotonic clock against the canonical Rust fallback epoch, so fast and fallback reads cannot jump between domains. Process and thread CPU clocks remain on the canonical fallback because wall time is not an equivalent -clock. Synthetic clock offsets remain honored by declining the direct import -for guests that import `clock_time_set`, and pending WASIX operations are -checked on a real-time bound. Invalid clock IDs, pointers, or host values use +clock. Calling `clock_time_set` switches every reader sharing the database's +memory to the canonical WASIX clock, including readers linked later. Importing +a setter alone leaves the fast path enabled. Pending WASIX operations are +checked after 16 observed milliseconds or 1,024 direct reads per clock domain. +The read bound also covers coarsened or stalled clocks. Invalid clock IDs, pointers, or host values use the complete Rust syscall. Other WASIX programs retain the complete upstream per-call path. -The exact pairing is qualified for the single-process direct Oliphaunt export -path in both execution surfaces, including repeated PostgreSQL `ERROR` recovery. The -direct driver treats every `PostgresMainLoopOnce` trap as the guest's exported -top-level recovery boundary and also cleans up non-trapping ErrorResponses. +The pairing requires single-process direct Oliphaunt integration checks in both +execution surfaces, including repeated PostgreSQL `ERROR` recovery. A recovered +typed outcome confirms that the live guest boundary performed top-level cleanup. +The direct driver treats every `PostgresMainLoopOnce` trap as terminal, not as +permission to invoke another guest recovery export. Its JavaScript memory bridge is limited to the direct Oliphaunt driver: generic WASIX streams keep their normal ownership and scheduling semantics. Copy failures are caught before guest buffers are released, and protocol responses are copied @@ -511,7 +517,7 @@ stock `@wasmer/sdk`; the published binding owns the source-pinned host. A larger current-Wasmer JS port is outside this host's compatibility contract. The version skew is upstream-owned rather than a loose Oliphaunt dependency. -The commit referenced by the latest npm `@wasmer/sdk` 0.10.0 release identifies +The commit referenced by the historical npm `@wasmer/sdk` 0.10.0 release identifies its checked-in source as 0.8.0 and embeds Wasmer 6.1 with the 0.601 Wasmer support family. `wasmer-wasix` 0.702.1 embeds Wasmer 7.2.1 and matching 0.702.1 virtual filesystem/network, package, configuration, backend, @@ -525,16 +531,18 @@ that port exists. ## PGlite reference, not product inheritance -PGlite independently validates the recovery shape used here. Its Emscripten +PGlite is useful prior art for error reporting, not proof of our runtime's +recovery safety. In the pinned reference below, its Emscripten guest turns the active PostgreSQL top-level `longjmp` into a known exit status; the TypeScript host then calls `PostgresMainLongJmp`, sends readiness, flushes, and resumes `PostgresMainLoopOnce`. Its public database error is separately decoded from pgwire. See PGlite's [runtime loop](https://github.com/electric-sql/pglite/blob/67872123b637ba132cceb8dbb3f739a09685ee87/packages/pglite/src/pglite.ts#L932-L965) and [guest shim](https://github.com/electric-sql/postgres-pglite/blob/7b4ee5086055dc5e54ae1e13e487888249438e68/pglite/src/pglitec/pglitec.c#L52-L84). -Oliphaunt deliberately uses an environment-gated Wasmer exception discriminator -instead of Emscripten's numeric sentinel, but preserves the same separation -between control-flow recovery and the pgwire `PostgresError` seen by callers. +Oliphaunt does not copy that host-side reentry: its guest catches PostgreSQL +errors before returning a typed outcome to Wasmer. There is no environment-gated +exception discriminator or process-exit sentinel for query recovery. Control-flow +recovery remains separate from the pgwire `PostgresError` seen by callers. Lifecycle SQL for a selectively imported extension runs in the owning realm. Isolated-host errors are serialized by PostgreSQL field and rebuilt in the caller; direct errors retain the same `PostgresError` identity in place. Generic diff --git a/src/wasix/sdks/ts/src/__tests__/direct-client-common.test.ts b/src/wasix/sdks/ts/src/__tests__/direct-client-common.test.ts index 25a5c277c..375c7ef31 100644 --- a/src/wasix/sdks/ts/src/__tests__/direct-client-common.test.ts +++ b/src/wasix/sdks/ts/src/__tests__/direct-client-common.test.ts @@ -270,7 +270,7 @@ describe('direct WASIX session lifecycle', () => { await session.close(); }); - it('keeps an existing configured role across a tool session without loading a seed', async () => { + it('keeps startup identity across a tool session without post-startup role or extension SQL', async () => { const queries: string[] = []; const options = openOptions(); options.username = 'app"role'; @@ -305,15 +305,7 @@ describe('direct WASIX session lifecycle', () => { await session.close(); expect(seedLoads).toBe(0); - expect(queries).toEqual([ - 'SET ROLE "app""role"', - 'ROLLBACK', - 'DISCARD ALL', - 'SET ROLE "app""role"', - 'ROLLBACK', - 'DISCARD ALL', - 'SET ROLE "app""role"', - ]); + expect(queries).toEqual(['ROLLBACK', 'DISCARD ALL', 'ROLLBACK', 'DISCARD ALL']); }); it('reuses prepared pg_dump but creates fresh processes and publishes once per run', async () => { @@ -609,7 +601,7 @@ describe('direct WASIX session lifecycle', () => { expect(events).toEqual(['startup', 'exec', 'close', 'storage:failed', 'free']); }); - it('keeps a direct session usable after the host confirms stream callback recovery', async () => { + it('drains after a callback abort and keeps the successfully completed session usable', async () => { const events: string[] = []; const storage = fakeLease(async (_directory, outcome) => { events.push(`storage:${outcome}`); @@ -622,23 +614,23 @@ describe('direct WASIX session lifecycle', () => { fakeHost({ events, execProtocolStream(onChunk) { - try { - onChunk(Uint8Array.of(1)); - return 0; - } catch { - return 1; - } + onChunk(Uint8Array.of(1)); + onChunk(Uint8Array.of(2)); + return 0; }, }), fakeDependencies(storage), ); const callbackFailure = new Error('consumer stopped'); + let callbackCount = 0; await expect( session.execStream(Uint8Array.of(1), () => { + callbackCount += 1; throw callbackFailure; }), ).resolves.toBe('callbackAborted'); + expect(callbackCount).toBe(1); await expect(session.exec(Uint8Array.of(2))).resolves.toEqual(querySuccess()); await session.close(); @@ -654,18 +646,307 @@ describe('direct WASIX session lifecycle', () => { ]); }); + it('keeps callback reentry blocked while the guest allocation is freed', async () => { + let session!: DirectWasixSession; + let nestedClose: Promise | undefined; + let nestedServe: Promise | undefined; + session = await DirectWasixSession.open( + openOptions(), + fakeHost({ + free() { + nestedClose = session.close(); + nestedServe = session.serve( + { frontend: createWasixByteChannel(), backend: createWasixByteChannel() }, + 'server', + ); + void nestedClose.catch(() => undefined); + void nestedServe.catch(() => undefined); + }, + }), + fakeDependencies(fakeLease(async () => undefined)), + ); + + await expect(session.close()).resolves.toBeUndefined(); + await expect(nestedClose).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedServe).rejects.toThrow('cannot be reentered synchronously'); + }); + + it('guards replacement startup callbacks until backend ownership is published', async () => { + let session!: DirectWasixSession; + let instantiations = 0; + let initialCloseCalls = 0; + let initialFreeCalls = 0; + let replacementCloseCalls = 0; + let replacementFreeCalls = 0; + let storageCloseCalls = 0; + let nestedClose: Promise | undefined; + let nestedServe: Promise | undefined; + let callbackState: + | { + instantiations: number; + replacementCloseCalls: number; + replacementFreeCalls: number; + storageCloseCalls: number; + } + | undefined; + const initialHost = fakeHost({ + close() { + initialCloseCalls += 1; + }, + free() { + initialFreeCalls += 1; + }, + }); + const replacementHost = fakeHost({ + startup() { + nestedClose = session.close(); + nestedServe = session.serve( + { frontend: createWasixByteChannel(), backend: createWasixByteChannel() }, + 'server', + ); + void nestedClose.catch(() => undefined); + void nestedServe.catch(() => undefined); + callbackState = { + instantiations, + replacementCloseCalls, + replacementFreeCalls, + storageCloseCalls, + }; + return startupSuccess(); + }, + close() { + replacementCloseCalls += 1; + }, + free() { + replacementFreeCalls += 1; + }, + }); + const host: DirectWasixHost = { + ...initialHost, + async instantiateOliphauntDirect(module, moduleBytes, options) { + instantiations += 1; + const owner = instantiations === 1 ? initialHost : replacementHost; + return owner.instantiateOliphauntDirect(module, moduleBytes, options); + }, + }; + const storage = fakeLease(async () => { + storageCloseCalls += 1; + }); + session = await DirectWasixSession.open(openOptions(), host, fakeDependencies(storage)); + const frontend = createWasixByteChannel(); + closeWasixByteChannel(frontend); + + await expect( + session.serve({ frontend, backend: createWasixByteChannel() }, 'server'), + ).resolves.toBeUndefined(); + + await expect(nestedClose).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedServe).rejects.toThrow('cannot be reentered synchronously'); + expect(callbackState).toEqual({ + instantiations: 2, + replacementCloseCalls: 0, + replacementFreeCalls: 0, + storageCloseCalls: 0, + }); + expect(initialCloseCalls).toBe(1); + expect(initialFreeCalls).toBe(1); + expect(replacementCloseCalls).toBe(0); + expect(replacementFreeCalls).toBe(0); + expect(storageCloseCalls).toBe(0); + await expect(session.exec(Uint8Array.of(1), 'defer')).resolves.toEqual(querySuccess()); + await session.close(); + expect(replacementCloseCalls).toBe(1); + expect(replacementFreeCalls).toBe(1); + expect(storageCloseCalls).toBe(1); + }); + + it('blocks lifecycle reentry while replacement instantiation is pending', async () => { + let session!: DirectWasixSession; + let instantiations = 0; + let nestedClose: Promise | undefined; + const baseHost = fakeHost({}); + const host: DirectWasixHost = { + ...baseHost, + async instantiateOliphauntDirect(module, moduleBytes, options) { + instantiations += 1; + if (instantiations === 2) { + await Promise.resolve(); + nestedClose = session.close(); + void nestedClose.catch(() => undefined); + } + return baseHost.instantiateOliphauntDirect(module, moduleBytes, options); + }, + }; + let storageCloseCalls = 0; + const storage = fakeLease(async () => { + storageCloseCalls += 1; + }); + session = await DirectWasixSession.open(openOptions(), host, fakeDependencies(storage)); + const frontend = createWasixByteChannel(); + closeWasixByteChannel(frontend); + + await session.serve({ frontend, backend: createWasixByteChannel() }, 'server'); + await expect(nestedClose).rejects.toThrow('cannot be reentered synchronously'); + expect(storageCloseCalls).toBe(0); + await expect(session.exec(Uint8Array.of(1), 'defer')).resolves.toEqual(querySuccess()); + await session.close(); + expect(storageCloseCalls).toBe(1); + }); + + it('guards replacement cleanup callbacks after startup fails', async () => { + let session!: DirectWasixSession; + let instantiations = 0; + let replacementCloseCalls = 0; + let replacementFreeCalls = 0; + let storageCloseCalls = 0; + let nestedClose: Promise | undefined; + let nestedServe: Promise | undefined; + let cleanupState: + | { + instantiations: number; + replacementFreeCalls: number; + storageCloseCalls: number; + } + | undefined; + const initialHost = fakeHost({}); + const replacementHost = fakeHost({ + startup() { + throw new Error('injected replacement startup failure'); + }, + close() { + replacementCloseCalls += 1; + nestedClose = session.close(); + nestedServe = session.serve( + { frontend: createWasixByteChannel(), backend: createWasixByteChannel() }, + 'server', + ); + void nestedClose.catch(() => undefined); + void nestedServe.catch(() => undefined); + cleanupState = { instantiations, replacementFreeCalls, storageCloseCalls }; + }, + free() { + replacementFreeCalls += 1; + }, + }); + const host: DirectWasixHost = { + ...initialHost, + async instantiateOliphauntDirect(module, moduleBytes, options) { + instantiations += 1; + const owner = instantiations === 1 ? initialHost : replacementHost; + return owner.instantiateOliphauntDirect(module, moduleBytes, options); + }, + }; + const storage = fakeLease(async () => { + storageCloseCalls += 1; + }); + session = await DirectWasixSession.open(openOptions(), host, fakeDependencies(storage)); + const frontend = createWasixByteChannel(); + closeWasixByteChannel(frontend); + + await expect( + session.serve({ frontend, backend: createWasixByteChannel() }, 'server'), + ).rejects.toThrow('injected replacement startup failure'); + + await expect(nestedClose).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedServe).rejects.toThrow('cannot be reentered synchronously'); + expect(cleanupState).toEqual({ + instantiations: 2, + replacementFreeCalls: 0, + storageCloseCalls: 0, + }); + expect(replacementCloseCalls).toBe(1); + expect(replacementFreeCalls).toBe(1); + expect(storageCloseCalls).toBe(0); + await expect(session.exec(Uint8Array.of(1), 'defer')).rejects.toThrow('database failed'); + await session.close(); + expect(storageCloseCalls).toBe(1); + }); + + it('rejects every guest-owning callback reentry before lifecycle state changes', async () => { + let guestCloseCalls = 0; + let guestFreeCalls = 0; + let storageSyncCalls = 0; + const storage = fakeLease(async () => undefined); + storage.sync = async () => { + storageSyncCalls += 1; + }; + const session = await DirectWasixSession.open( + openOptions(), + fakeHost({ + execProtocolStream(onChunk) { + onChunk(querySuccess()); + }, + close() { + guestCloseCalls += 1; + }, + free() { + guestFreeCalls += 1; + }, + }), + fakeDependencies(storage), + ); + let nestedExec: Promise | undefined; + let nestedExecStream: Promise | undefined; + let nestedServe: Promise | undefined; + let nestedPgDump: Promise | undefined; + let nestedBackup: Promise | undefined; + let nestedSync: Promise | undefined; + let nestedClose: Promise | undefined; + + await session.execStream( + Uint8Array.of(1), + () => { + nestedExec = session.exec(Uint8Array.of(2), 'defer'); + nestedExecStream = session.execStream(Uint8Array.of(2), () => undefined, 'defer'); + nestedServe = session.serve( + { frontend: createWasixByteChannel(), backend: createWasixByteChannel() }, + 'server', + ); + nestedPgDump = session.runPgDump({ tool: pgDumpDescriptor, args: [] }); + nestedBackup = session.backup(); + nestedSync = session.sync('operation'); + nestedClose = session.close(); + for (const pending of [ + nestedExec, + nestedExecStream, + nestedServe, + nestedPgDump, + nestedBackup, + nestedSync, + nestedClose, + ]) { + void pending.catch(() => undefined); + } + }, + 'defer', + ); + + await expect(nestedExec).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedExecStream).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedServe).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedPgDump).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedBackup).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedSync).rejects.toThrow('cannot be reentered synchronously'); + await expect(nestedClose).rejects.toThrow('cannot be reentered synchronously'); + expect(guestCloseCalls).toBe(0); + expect(guestFreeCalls).toBe(0); + expect(storageSyncCalls).toBe(0); + await expect(session.exec(Uint8Array.of(3), 'defer')).resolves.toEqual(querySuccess()); + await session.close(); + expect(guestCloseCalls).toBe(1); + expect(guestFreeCalls).toBe(1); + }); + it('makes a failed stream recovery authoritative and poisons the direct session', async () => { const recoveryFailure = new Error('ReadyForQuery recovery failed'); const session = await DirectWasixSession.open( openOptions(), fakeHost({ execProtocolStream(onChunk) { - try { - onChunk(Uint8Array.of(1)); - } catch { - throw recoveryFailure; - } - return 0; + onChunk(Uint8Array.of(1)); + // Even after the consumer abort is contained, a later guest/flush + // failure must win over a seemingly recoverable callback error. + throw recoveryFailure; }, }), fakeDependencies(fakeLease(async () => undefined)), diff --git a/src/wasix/sdks/ts/src/__tests__/wasix-runtime.test.ts b/src/wasix/sdks/ts/src/__tests__/wasix-runtime.test.ts index a68a6b49e..ef5c76234 100644 --- a/src/wasix/sdks/ts/src/__tests__/wasix-runtime.test.ts +++ b/src/wasix/sdks/ts/src/__tests__/wasix-runtime.test.ts @@ -5,7 +5,6 @@ import { PostgresError } from '../protocol/query.js'; import { compileWasixModule, composeLifecycleFailure, - configureWasixDatabase, describeError, materializeWasixMounts, wasixPostgresEnvironment, @@ -127,18 +126,14 @@ describe('WASIX host runtime helpers', () => { ).toBe('open failed; cleanup failed: trap'); }); - it('does not install selected extensions and only applies the quoted caller role', async () => { + it('passes the exact caller identity to PostgreSQL startup without SQL quoting', () => { const options = workerOpenOptions(); options.username = 'app"role'; - const inputs: Uint8Array[] = []; - await configureWasixDatabase(options, async (input) => { - inputs.push(input); - return querySuccess(); + expect(wasixPostgresEnvironment(options)).toMatchObject({ + PGUSER: 'app"role', + USER: 'app"role', + LOGNAME: 'app"role', }); - expect(inputs).toHaveLength(1); - const sql = new TextDecoder().decode(inputs[0]); - expect(sql).toContain('SET ROLE "app""role"'); - expect(sql).not.toMatch(/CREATE EXTENSION|\bLOAD\b|CREATE SCHEMA/u); }); }); @@ -154,32 +149,3 @@ class RecordingDirectory { this.#created.push(path); } } - -function querySuccess(): Uint8Array { - return concatenate([ - backendMessage('C', new TextEncoder().encode('SELECT 1\0')), - backendMessage('Z', Uint8Array.of('I'.charCodeAt(0))), - ]); -} - -function backendMessage(tag: string, body: Uint8Array): Uint8Array { - const length = body.length + 4; - return Uint8Array.of( - tag.charCodeAt(0), - (length >>> 24) & 0xff, - (length >>> 16) & 0xff, - (length >>> 8) & 0xff, - length & 0xff, - ...body, - ); -} - -function concatenate(chunks: readonly Uint8Array[]): Uint8Array { - const bytes = new Uint8Array(chunks.reduce((total, chunk) => total + chunk.length, 0)); - let offset = 0; - for (const chunk of chunks) { - bytes.set(chunk, offset); - offset += chunk.length; - } - return bytes; -} diff --git a/src/wasix/sdks/ts/src/core/database.ts b/src/wasix/sdks/ts/src/core/database.ts index c0e02c5b6..c9a141caa 100644 --- a/src/wasix/sdks/ts/src/core/database.ts +++ b/src/wasix/sdks/ts/src/core/database.ts @@ -1,4 +1,5 @@ import { composeWasixStorageFailure, WasixStorageError } from './errors.js'; +import { PROTOCOL_CALLBACK_CHUNK_BYTES } from '../protocol-limits.generated.js'; import { assertNoTransactionChain, decodeQueryResult, @@ -44,8 +45,11 @@ import type { const transactionPinnedMessage = 'Oliphaunt WASIX database is pinned to an active transaction; use the callback transaction handle'; const CLOSE_DEADLINE_MS = 120_000; -/** @internal Buffered protocol fallbacks use the same observable callback granularity. */ -export const WASIX_PROTOCOL_CALLBACK_CHUNK_BYTES = 64 * 1024; +/** + * @internal Host callback maximum owned by postgres-protocol-transport-contract. + * Buffered fallbacks preserve this granularity; it does not cap total output. + */ +export const WASIX_PROTOCOL_CALLBACK_CHUNK_BYTES = PROTOCOL_CALLBACK_CHUNK_BYTES; const pgDumpTargets = new WeakSet(); const nativeToolTargets = new WeakSet(); const protocolConnectionTargets = new WeakSet(); diff --git a/src/wasix/sdks/ts/src/core/types.ts b/src/wasix/sdks/ts/src/core/types.ts index cf4c81fb9..f5c968d73 100644 --- a/src/wasix/sdks/ts/src/core/types.ts +++ b/src/wasix/sdks/ts/src/core/types.ts @@ -166,7 +166,7 @@ export type WasixAssetManifest = { }; export type OpenConfig = { - /** Existing PostgreSQL role selected after the fixed superuser bootstrap. */ + /** Existing login role. Honors role defaults and LOGIN/CONNECT restrictions; the host authenticates the caller. */ username?: string; database?: string; /** PostgreSQL `-c name=value` settings applied before the database opens. */ diff --git a/src/wasix/sdks/ts/src/host/index.d.mts b/src/wasix/sdks/ts/src/host/index.d.mts index 114cbecbc..c4a45ba4b 100644 --- a/src/wasix/sdks/ts/src/host/index.d.mts +++ b/src/wasix/sdks/ts/src/host/index.d.mts @@ -113,7 +113,7 @@ export function runOliphauntToolDirect( prepared: OliphauntPreparedTool, options: RunWasixOptions, protocolRead: (maximumBytes: number) => Uint8Array, - /** Borrowed bytes: synchronously copy; never mutate or retain this view. */ + /** Owned bytes: the receiver may retain or mutate this callback-local copy. */ protocolWrite: (chunk: Uint8Array) => void, ): Promise; export function instantiateOliphauntDirect( diff --git a/src/wasix/sdks/ts/src/hosts/browser/direct-client-common.ts b/src/wasix/sdks/ts/src/hosts/browser/direct-client-common.ts index 6e2c26386..f4ff23376 100644 --- a/src/wasix/sdks/ts/src/hosts/browser/direct-client-common.ts +++ b/src/wasix/sdks/ts/src/hosts/browser/direct-client-common.ts @@ -55,8 +55,6 @@ import { normalizeWasixStartupGUCs } from '../../core/startup-config.js'; import { compileWasixModule, composeLifecycleFailure, - configureWasixDatabase, - configureWasixRole, describeError, materializeWasixMounts, wasixPostgresArgs, @@ -78,7 +76,7 @@ export type DirectWasixHost = Readonly<{ prepared: OliphauntPreparedTool, options: RunWasixOptions, protocolRead: (maximumBytes: number) => Uint8Array, - /** Borrowed bytes: synchronously copy; never mutate or retain this view. */ + /** Owned bytes: the receiver may retain or mutate this callback-local copy. */ protocolWrite: (chunk: Uint8Array) => void, ): Promise; }>; @@ -96,6 +94,13 @@ export type DirectWasixDependencies = Readonly<{ export type DirectWasixEnvironment = 'browser-main' | 'browser-worker' | 'node'; +class DirectGuestReentryError extends Error { + constructor() { + super('Oliphaunt WASIX direct guest execution cannot be reentered synchronously'); + this.name = 'DirectGuestReentryError'; + } +} + type DirectInstanceFactory = () => Promise; type DirectInstanceInitializer = ( instance: OliphauntDirectInstance, @@ -152,7 +157,6 @@ export class DirectWasixSession implements WasixDatabaseSession { readonly #baseDirectory: Directory; readonly #instantiate: DirectInstanceFactory; readonly #initialize: DirectInstanceInitializer; - readonly #username: string; readonly #host: DirectWasixHost; #pgDump: Promise | undefined; #pgDumpIdentity: string | undefined; @@ -160,6 +164,7 @@ export class DirectWasixSession implements WasixDatabaseSession { #failed = false; #closeAttempt: Promise | undefined; #startupResponse: Uint8Array = new Uint8Array(); + #guestCallActive = false; private constructor( instance: OliphauntDirectInstance, @@ -176,7 +181,6 @@ export class DirectWasixSession implements WasixDatabaseSession { this.#baseDirectory = baseDirectory; this.#instantiate = instantiate; this.#initialize = initialize; - this.#username = username; this.#host = host; this.identity = normalizeWasixDatabaseIdentity(username, database); } @@ -240,7 +244,6 @@ export class DirectWasixSession implements WasixDatabaseSession { const initialize: DirectInstanceInitializer = async (candidate, _storageState) => { const response = candidate.startup(startupPacket(options.username, options.database)); assertSuccessfulStartupResponse(response); - await configureWasixDatabase(options, async (input) => candidate.execProtocolRaw(input)); return new Uint8Array(response); }; instance = await instantiate(); @@ -297,9 +300,10 @@ export class DirectWasixSession implements WasixDatabaseSession { } async exec(input: Uint8Array, persistence: WasixPersistenceMode = 'sync'): Promise { + this.#assertGuestEntryAllowed(); this.#assertHealthy(); try { - const response = this.#currentInstance().execProtocolRaw(input); + const response = this.#callGuest((instance) => instance.execProtocolRaw(input)); if (persistence === 'sync') { await this.#storage.sync(this.#baseDirectory, 'operation'); } @@ -321,16 +325,31 @@ export class DirectWasixSession implements WasixDatabaseSession { onChunk: (chunk: Uint8Array) => void, persistence: WasixPersistenceMode = 'sync', ): Promise { + this.#assertGuestEntryAllowed(); this.#assertHealthy(); try { - const status = this.#currentInstance().execProtocolStream(input, onChunk); + let callbackAborted = false; + const status = this.#callGuest((instance) => + instance.execProtocolStream(input, (chunk) => { + if (callbackAborted) return; + try { + onChunk(chunk); + } catch { + // Finish the guest exchange without invoking the failed consumer + // again. Only a successful host flush/restore may confirm this abort. + callbackAborted = true; + } + }), + ); if (status !== PROTOCOL_STREAM_COMPLETE && status !== PROTOCOL_STREAM_CALLBACK_ABORTED) { throw new Error(`WASIX host returned unknown protocol stream status ${status}`); } if (persistence === 'sync') { await this.#storage.sync(this.#baseDirectory, 'operation'); } - return status === PROTOCOL_STREAM_CALLBACK_ABORTED ? 'callbackAborted' : 'complete'; + return callbackAborted || status === PROTOCOL_STREAM_CALLBACK_ABORTED + ? 'callbackAborted' + : 'complete'; } catch (error) { this.#failed = true; if (error instanceof WasixStorageError) throw error; @@ -342,6 +361,7 @@ export class DirectWasixSession implements WasixDatabaseSession { } async runPgDump(options: WasixPgDumpProcessOptions): Promise { + this.#assertGuestEntryAllowed(); this.#assertHealthy(); if (options.tool.name !== 'pg_dump') { throw new TypeError('the same-realm tool path supports only pg_dump'); @@ -362,7 +382,7 @@ export class DirectWasixSession implements WasixDatabaseSession { startupResponse: this.#startupResponse, startupIdentity: this.identity, exec: (input, onChunk) => { - this.#currentInstance().execProtocolStream(input, onChunk); + this.#callGuest((instance) => instance.execProtocolStream(input, onChunk)); }, }); connection = activeConnection; @@ -394,6 +414,7 @@ export class DirectWasixSession implements WasixDatabaseSession { connection: WasixProtocolConnection, mode: WasixProtocolConnectionMode, ): Promise { + this.#assertGuestEntryAllowed(); this.#assertHealthy(); if (mode === 'tool') { await this.#runToolSession(() => @@ -401,11 +422,13 @@ export class DirectWasixSession implements WasixDatabaseSession { startupResponse: this.#startupResponse, startupIdentity: this.identity, execDuplex: (input, onRead, onWrite) => { - this.#currentInstance().execProtocolDuplex(input, onRead, onWrite); + this.#callGuest((instance) => instance.execProtocolDuplex(input, onRead, onWrite)); }, publishIdle: async () => undefined, rollback: async () => { - const response = this.#currentInstance().execProtocolRaw(simpleQuery('ROLLBACK')); + const response = this.#callGuest((instance) => + instance.execProtocolRaw(simpleQuery('ROLLBACK')), + ); assertSuccessfulQueryResponse(response); }, }), @@ -424,11 +447,13 @@ export class DirectWasixSession implements WasixDatabaseSession { startupResponse: this.#startupResponse, startupIdentity: this.identity, execDuplex: (input, onRead, onWrite) => { - this.#currentInstance().execProtocolDuplex(input, onRead, onWrite); + this.#callGuest((instance) => instance.execProtocolDuplex(input, onRead, onWrite)); }, publishIdle: () => this.#storage.sync(this.#baseDirectory, 'operation'), rollback: async () => { - const response = this.#currentInstance().execProtocolRaw(simpleQuery('ROLLBACK')); + const response = this.#callGuest((instance) => + instance.execProtocolRaw(simpleQuery('ROLLBACK')), + ); assertSuccessfulQueryResponse(response); await this.#storage.sync(this.#baseDirectory, 'operation'); }, @@ -527,20 +552,20 @@ export class DirectWasixSession implements WasixDatabaseSession { async #resetProtocolSession(): Promise { for (const statement of ['ROLLBACK', 'DISCARD ALL']) { - const response = this.#currentInstance().execProtocolRaw(simpleQuery(statement)); + const response = this.#callGuest((instance) => + instance.execProtocolRaw(simpleQuery(statement)), + ); assertSuccessfulQueryResponse(response); } - await configureWasixRole(this.#username, async (input) => - this.#currentInstance().execProtocolRaw(input), - ); } async #restartProtocolBackend(): Promise { + this.#assertGuestEntryAllowed(); const previous = this.#currentInstance(); this.#instance = undefined; let failure: Error | undefined; try { - previous.close(); + this.#withGuestCall(() => previous.close()); } catch (error) { failure = new Error( `WASIX PostgreSQL backend restart close failed: ${describeError(error)}`, @@ -550,7 +575,7 @@ export class DirectWasixSession implements WasixDatabaseSession { ); } try { - previous.free(); + this.#withGuestCall(() => previous.free()); } catch (error) { failure = failure === undefined @@ -563,15 +588,19 @@ export class DirectWasixSession implements WasixDatabaseSession { let replacement: OliphauntDirectInstance | undefined; try { - replacement = await this.#instantiate(); - const startupResponse = await this.#initialize(replacement, 'existing'); - this.#instance = replacement; - this.#startupResponse = startupResponse; + await this.#withGuestCallAsync(async () => { + const candidate = await this.#instantiate(); + replacement = candidate; + const startupResponse = await this.#initialize(candidate, 'existing'); + this.#instance = candidate; + this.#startupResponse = startupResponse; + }); } catch (error) { failure = directStartupFailure(error); - if (replacement !== undefined) { + const failedReplacement = replacement; + if (failedReplacement !== undefined) { try { - replacement.close(); + this.#withGuestCall(() => failedReplacement.close()); } catch (closeError) { failure = composeLifecycleFailure( failure, @@ -580,7 +609,7 @@ export class DirectWasixSession implements WasixDatabaseSession { ); } try { - replacement.free(); + this.#withGuestCall(() => failedReplacement.free()); } catch (freeError) { failure = composeLifecycleFailure( failure, @@ -594,6 +623,7 @@ export class DirectWasixSession implements WasixDatabaseSession { } async sync(boundary: WasixStorageSyncBoundary): Promise { + this.#assertGuestEntryAllowed(); this.#assertHealthy(); try { await this.#storage.sync(this.#baseDirectory, boundary); @@ -611,6 +641,7 @@ export class DirectWasixSession implements WasixDatabaseSession { } async backup(): Promise { + this.#assertGuestEntryAllowed(); this.#assertHealthy(); try { return await createPhysicalArchive( @@ -624,10 +655,15 @@ export class DirectWasixSession implements WasixDatabaseSession { } async #execBackupProtocol(input: Uint8Array): Promise { - return this.#currentInstance().execProtocolRaw(input); + return this.#callGuest((instance) => instance.execProtocolRaw(input)); } close(): Promise { + try { + this.#assertGuestEntryAllowed(); + } catch (error) { + return Promise.reject(error); + } if (this.#closeAttempt !== undefined) return this.#closeAttempt; // Establish the admission cutoff before any guest/provider teardown can // invoke userland code or yield. @@ -648,7 +684,7 @@ export class DirectWasixSession implements WasixDatabaseSession { this.#instance = undefined; if (instance !== undefined) { try { - instance.close(); + this.#withGuestCall(() => instance.close()); } catch (error) { failure = new Error(`WASIX PostgreSQL direct close failed: ${describeError(error)}`, { cause: error, @@ -670,7 +706,7 @@ export class DirectWasixSession implements WasixDatabaseSession { if (instance !== undefined) { try { - instance.free(); + this.#withGuestCall(() => instance.free()); } catch (error) { failure = failure === undefined @@ -684,7 +720,7 @@ export class DirectWasixSession implements WasixDatabaseSession { const pgDump = await pendingPgDump?.catch(() => undefined); if (pgDump !== undefined) { try { - pgDump.prepared.free(); + this.#withGuestCall(() => pgDump.prepared.free()); } catch (error) { failure = failure === undefined @@ -716,6 +752,36 @@ export class DirectWasixSession implements WasixDatabaseSession { return instance; } + #callGuest(operation: (instance: OliphauntDirectInstance) => Result): Result { + return this.#withGuestCall(() => operation(this.#currentInstance())); + } + + #withGuestCall(operation: () => Result): Result { + // Reject callback recursion before wasm-bindgen can borrow or free the + // same allocation. Destructors belong inside this boundary too. + this.#assertGuestEntryAllowed(); + this.#guestCallActive = true; + try { + return operation(); + } finally { + this.#guestCallActive = false; + } + } + + async #withGuestCallAsync(operation: () => Promise): Promise { + this.#assertGuestEntryAllowed(); + this.#guestCallActive = true; + try { + return await operation(); + } finally { + this.#guestCallActive = false; + } + } + + #assertGuestEntryAllowed(): void { + if (this.#guestCallActive) throw new DirectGuestReentryError(); + } + async #closeAfterOpenFailure(failure: Error): Promise { this.#closed = true; this.#failed = true; @@ -723,7 +789,7 @@ export class DirectWasixSession implements WasixDatabaseSession { this.#instance = undefined; if (instance !== undefined) { try { - instance.close(); + this.#withGuestCall(() => instance.close()); } catch (closeError) { failure = composeLifecycleFailure( failure, @@ -739,7 +805,7 @@ export class DirectWasixSession implements WasixDatabaseSession { } if (instance !== undefined) { try { - instance.free(); + this.#withGuestCall(() => instance.free()); } catch (freeError) { failure = composeLifecycleFailure( failure, diff --git a/src/wasix/sdks/ts/src/hosts/browser/wasix-runtime.ts b/src/wasix/sdks/ts/src/hosts/browser/wasix-runtime.ts index 464166dd1..3cc59b1d9 100644 --- a/src/wasix/sdks/ts/src/hosts/browser/wasix-runtime.ts +++ b/src/wasix/sdks/ts/src/hosts/browser/wasix-runtime.ts @@ -1,8 +1,7 @@ import type { WasixDirectoryMount, WasixRuntimeLayout } from '../../resources/archive.js'; import { WasixStorageError } from '../../core/errors.js'; import type { Directory } from '../../host/index.mjs'; -import { simpleQuery } from '../../protocol/protocol.js'; -import { assertSuccessfulQueryResponse, PostgresError } from '../../protocol/query.js'; +import { PostgresError } from '../../protocol/query.js'; import type { SerializedOpenOptions } from '../../workers/rpc.js'; import { normalizeWasixStartupGUCs } from '../../core/startup-config.js'; import { releaseWasixToolMounts } from '../../resources/tool-runtime.js'; @@ -228,23 +227,3 @@ export function composeLifecycleFailure(primary: Error, label: string, secondary } return new Error(message, { cause }); } - -/** @internal Apply caller role after the direct bridge reaches ReadyForQuery. */ -export async function configureWasixDatabase( - options: SerializedOpenOptions, - exec: (input: Uint8Array) => Promise, -): Promise { - // Extension selection owns files and startup configuration only. Database- - // local CREATE EXTENSION/LOAD/schema/migration SQL remains application-owned. - await configureWasixRole(options.username, exec); -} - -/** @internal Restore the configured application role after DISCARD ALL. */ -export async function configureWasixRole( - username: string, - exec: (input: Uint8Array) => Promise, -): Promise { - if (username === 'postgres') return; - const quoted = username.replaceAll('"', '""'); - assertSuccessfulQueryResponse(await exec(simpleQuery(`SET ROLE "${quoted}"`))); -} diff --git a/src/wasix/sdks/ts/src/protocol-limits.generated.ts b/src/wasix/sdks/ts/src/protocol-limits.generated.ts new file mode 100644 index 000000000..08ce1e17c --- /dev/null +++ b/src/wasix/sdks/ts/src/protocol-limits.generated.ts @@ -0,0 +1,3 @@ +// Generated by src/wasix/runtime/protocol-contract/generate.mjs. +// Edit contract.json, not this callback limit. +export const PROTOCOL_CALLBACK_CHUNK_BYTES = 65536; diff --git a/src/wasix/sdks/ts/src/protocol/byte-channel.ts b/src/wasix/sdks/ts/src/protocol/byte-channel.ts index 8715028a5..43c303df4 100644 --- a/src/wasix/sdks/ts/src/protocol/byte-channel.ts +++ b/src/wasix/sdks/ts/src/protocol/byte-channel.ts @@ -8,7 +8,10 @@ const PROTOCOL_IDLE = 0; const PROTOCOL_ACTIVE = 1; const PROTOCOL_COMPLETE = 2; +// Read batching only; independent of the public protocol callback maximum. const WASIX_BYTE_CHANNEL_CHUNK_BYTES = 64 * 1024; +// Fixed shared-memory allocation. One sentinel byte distinguishes full/empty, +// leaving 256 KiB usable; writers wait for space rather than buffering the stream. const WASIX_CHANNEL_BYTES = 256 * 1024 + 1; /** @internal One bounded single-producer/single-consumer byte channel. */ diff --git a/src/wasix/sdks/ts/src/protocol/pgwire-connection.ts b/src/wasix/sdks/ts/src/protocol/pgwire-connection.ts index 87a9484e2..62de41389 100644 --- a/src/wasix/sdks/ts/src/protocol/pgwire-connection.ts +++ b/src/wasix/sdks/ts/src/protocol/pgwire-connection.ts @@ -9,6 +9,8 @@ const SSL_REQUEST = 80_877_103; const GSSENC_REQUEST = 80_877_104; const CANCEL_REQUEST = 80_877_102; const PROTOCOL_3 = 196_608; +// One complete frontend frame including its header, not an eager allocation or +// stream-size limit. Keep the same admission policy as the Rust wire reader. const MAX_FRONTEND_MESSAGE_BYTES = 128 * 1024 * 1024; const POSTGRES_IDENTIFIER_BYTES = 63; diff --git a/tools/ci/workflow-moon-transfers.test.mts b/tools/ci/workflow-moon-transfers.test.mts index 2338df792..6214c3e74 100644 --- a/tools/ci/workflow-moon-transfers.test.mts +++ b/tools/ci/workflow-moon-transfers.test.mts @@ -141,6 +141,31 @@ if (!process.env.OLIPHAUNT_TRANSFER_FIXTURE_PHASE) } }); +if (!process.env.OLIPHAUNT_TRANSFER_FIXTURE_PHASE) + test('mandatory artifact consumers check direct producer success explicitly', () => { + for (const id of [ + 'extension-artifacts-native-android', + 'extension-artifacts-wasix', + 'mobile-extension-packages-android', + 'mobile-extension-packages-ios', + 'liboliphaunt-native-release-assets', + 'swift-sdk-package', + 'liboliphaunt-wasix-release-assets', + 'mobile-build-ios', + 'mobile-e2e-ios', + ]) { + const job = workflow.jobs[id]; + assert(job.if.startsWith('${{ !cancelled() && '), `${id} can inherit skipped ancestors`); + for (const dependency of [job.needs].flat()) { + assert( + job.if.includes(`needs.${dependency}.result == 'success'`), + `${id} must require successful ${dependency}`, + ); + } + assert(!job.if.includes('needs.*.result'), `${id} relies on wildcard status filtering`); + } + }); + if (!process.env.OLIPHAUNT_TRANSFER_FIXTURE_PHASE) test('WASIX aggregates require successful selected hosts before accepting their artifacts', () => { for (const id of ['wasix-napi', 'liboliphaunt-wasix-aot']) { diff --git a/tools/packaging/wasix-cargo-payload.test.mts b/tools/packaging/wasix-cargo-payload.test.mts index 3bf474269..aa079305c 100644 --- a/tools/packaging/wasix-cargo-payload.test.mts +++ b/tools/packaging/wasix-cargo-payload.test.mts @@ -7,6 +7,7 @@ import { import { createDeterministicTar } from './cargo-source-package.mts'; import { extractPortableArchiveTree, releaseZstdCompressSync } from './portable-archive.mts'; import { packageSpec } from './wasix-cargo-payload.mts'; +import { WASIX_AOT_ENGINE } from '../../src/wasix/runtime/tools/wasix-aot-manifest.mts'; const root = path.resolve(import.meta.dir, '../..'); const scratch = process.argv[2]; @@ -93,6 +94,7 @@ for (const [id, template, kind, payloadDirName, ancestorAssets, variable, expres JSON.stringify( aot ? { + engine: WASIX_AOT_ENGINE, artifacts: [ { name: id === 'tools-aot' ? 'tool:pg_dump' : 'runtime:oliphaunt', diff --git a/tools/packaging/wasix-cargo-payload.test.sh b/tools/packaging/wasix-cargo-payload.test.sh index 2240cff37..96fec9862 100644 --- a/tools/packaging/wasix-cargo-payload.test.sh +++ b/tools/packaging/wasix-cargo-payload.test.sh @@ -13,6 +13,15 @@ while IFS=$'\t' read -r crate payload variable external; do CARGO_TARGET_DIR="$scratch/cargo-target" cargo run --locked --offline --quiet \ --manifest-path "$crate/Cargo.toml" --example probe if [[ "$payload" == artifacts ]]; then + cp "$crate/$payload/manifest.json" "$crate/current-profile.json" + bun -e 'const p = process.argv[1]; const m = await Bun.file(p).json(); m.engine += "-incompatible"; await Bun.write(p, JSON.stringify(m));' "$crate/$payload/manifest.json" + if CARGO_TARGET_DIR="$scratch/cargo-target" cargo check --locked --offline --quiet \ + --manifest-path "$crate/Cargo.toml" --lib > "$scratch/stale-profile.log" 2>&1; then + echo "Published carrier accepted stale AOT codegen profile: $crate" >&2 + exit 1 + fi + rg -q 'stale WASIX AOT profile' "$scratch/stale-profile.log" + mv "$crate/current-profile.json" "$crate/$payload/manifest.json" mkdir "$crate/removed-aot" mv "$crate/$payload/"*.zst "$crate/removed-aot/" if CARGO_TARGET_DIR="$scratch/cargo-target" cargo check --locked --offline --quiet \