diff --git a/.fuse_hidden0002bd66000001f9 b/.fuse_hidden0002bd66000001f9 index 6ebf0d6..c47927c 100644 --- a/.fuse_hidden0002bd66000001f9 +++ b/.fuse_hidden0002bd66000001f9 @@ -1444,3 +1444,78 @@ Max iterations: unlimited [11:00:31] [INFO] [agent] DGR-031 is complete: I added `packages/node/meshnet_node/shard_engine.py` (the `ShardEngine` ABC with load/capabilities/prefill/decode/cancel/release/health/metrics, project-owned tensor/bundle/token-output types, reserved-but-disabled MTP and architecture aux-state hooks), a reusable `tests/shard_engine_contract.py` lifecycle contract for DGR-032/DGR-037 to reuse, `tests/test_shard_engine.py` proving it against a minimal reference engine, and the evidence README/prd.json/issue markdown updates. All targeted tests (12 new + 95 dependency-adjacent) pass, `compileall` and `git diff --check` are clean, and the pre-existing 27-module collection failure in `tests/` was confirmed unrelated via `git stash`. [11:00:31] [INFO] [agent] [ScheduleWakeup] [11:00:32] [INFO] [agent] COMPLETE +[11:00:33] [INFO] [progress] Iteration 2 finished. Task DGR-031: COMPLETED. Duration: 564s +[11:00:33] [INFO] [tracker] Task DGR-031 completed in iteration 2 +[11:00:34] [INFO] [progress] Iteration 3/∞: Working on DGR-032 - Implement deterministic fake `ShardEngine` +[11:00:38] [INFO] [agent] I'll start by reading the context files and the DGR-031 ShardEngine interface this task builds on. +[11:00:39] [INFO] [agent] [Bash] $ cat .scratch/distributed-gguf-runtime/RALPH-CONTEXT.md 2>/dev/null | head -100 +[11:00:40] [INFO] [agent] [Bash] $ find . -path ./node_modules -prune -o -iname "*shard_engine*" -print 2>/dev/null +[11:00:43] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/meshnet_node/shard_engine.py +[11:00:44] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/shard_engine_contract.py +[11:00:45] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_shard_engine.py +[11:00:49] [INFO] [agent] [Bash] $ ls .scratch/distributed-gguf-runtime/evidence/ 2>/dev/null +[11:00:51] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/meshnet_node/shard_lifecycle.py +[11:00:55] [INFO] [agent] Now let me check the DGR-031 evidence README and prd.json entry for DGR-032 to understand exact expectations, plus check existing test directory structure and conventions. +[11:00:56] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.scratch/distributed-gguf-runtime/evidence/DGR-031/README.md +[11:00:57] [INFO] [agent] [Bash] $ python3 -c " import json data = json.load(open('.scratch/distributed-gguf-runtime/prd.json')) storie... +[11:01:00] [INFO] [agent] [Bash] $ python3 -c " import json data = json.load(open('.scratch/distributed-gguf-runtime/prd.json')) print(... +[11:01:03] [INFO] [agent] [Bash] $ python3 -c " import json data = json.load(open('.scratch/distributed-gguf-runtime/prd.json')) for s ... +[11:01:10] [INFO] [agent] [Bash] $ ls tests/ | sort | grep -v __pycache__ +[11:01:14] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/meshnet_node/testing.py +[11:01:23] [INFO] [agent] [Bash] $ grep -n "ShardEngine\|shard_engine\|fake\|Fake" packages/node/meshnet_node/shard_runtime_server.py |... +[11:01:25] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.scratch/distributed-gguf-runtime/issues/033-build-a-standalone-fake-c-grpc-shard-worker.md +[11:01:38] [INFO] [agent] [Bash] $ grep -n "class \|def __init__\|Servicer\|import" packages/node/meshnet_node/shard_runtime_server.py ... +[11:02:45] [INFO] [agent] [Bash] $ grep -rn "dtype_size\|bfloat16\|itemsize\|byte_size\|dtype ==" packages/node/meshnet_node/protocol.p... +[11:05:05] [INFO] [agent] Now I have a clear picture. Let me write the fake `ShardEngine` module. +[11:05:45] [INFO] [agent] [Write] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/meshnet_node/fake_shard_engine.py """"Deterministic fake ``ShardEngine`` fixture (DGR-032). +[11:05:45] [INFO] [agent] ``FakeShardEngine`` is a pure-Python, allocation-cheap subclass of +[11:05:45] [INFO] [agent] :class:`~meshnet_node.shard_engine.ShardEngine`: no llama.cpp, no native +[11:05:45] [INFO] [agent] bu... (12102 chars)" +[11:05:48] [INFO] [agent] Now let me write the test file covering the DGR-032-specific behaviors. +[11:06:17] [INFO] [agent] [Write] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_fake_shard_engine.py """"DGR-032 ``FakeShardEngine`` tests. +[11:06:17] [INFO] [agent] ``FakeShardEngine`` obeys the exact same lifecycle contract every +[11:06:17] [INFO] [agent] ``ShardEngine`` implementation must (see ``shard_engine_contract.py``); the +[11:06:17] [INFO] [agent] tests here addition... (8385 chars)" +[11:06:20] [INFO] [agent] Now let's run the tests. +[11:06:21] [INFO] [agent] [Bash] $ .venv/bin/python3 -m pytest -q tests/test_fake_shard_engine.py tests/test_shard_engine.py 2>&1 | tai... +[11:06:28] [INFO] [agent] [Edit] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_fake_shard_engine.py edit: " engine = FakeShardEngine(FakeShardEngineConfig..." → " engine = FakeShardEngine(FakeShardEngineConfig..." +[11:06:31] [INFO] [agent] [Bash] $ .venv/bin/python3 -m pytest -q tests/test_fake_shard_engine.py tests/test_shard_engine.py 2>&1 | tai... +[11:06:35] [INFO] [agent] All 26 tests pass. Now let's run the full targeted suite plus compileall and git diff --check. +[11:06:36] [INFO] [agent] [Bash] $ \ PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q \ tests/test_fake_shard... +[11:06:44] [INFO] [agent] [Bash] $ echo "exit: $?" +[11:06:48] [INFO] [agent] [Bash] $ echo "rc=$?" +[11:06:52] [INFO] [agent] Also run the broader `python -m compileall packages tests` gate as required. +[11:06:52] [INFO] [agent] [Bash] $ echo "rc=$?" +[11:06:54] [INFO] [agent] Clean. Now let's update `prd.json`, regenerate the issue markdown, and write the evidence README. +[11:06:55] [INFO] [agent] [Bash] $ python3 scripts/ralph_prd_schema.py --help 2>&1 | head -30 +[11:06:58] [INFO] [agent] [Bash] $ python3 -c " import json data = json.load(open('.scratch/distributed-gguf-runtime/prd.json')) for s ... +[11:07:04] [INFO] [agent] Now let's write the evidence README before updating prd.json. +[11:08:00] [INFO] [agent] [Write] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md "# DGR-032 evidence — deterministic fake `ShardEngine` +[11:08:00] [INFO] [agent] **Completed:** 2026-07-23 +[11:08:00] [INFO] [agent] **Branch:** `ralph/distributed-gguf-runtime` +[11:08:00] [INFO] [agent] **Authority:** `.scratch/distributed-gguf-runtime/prd.json` +[11:08:00] [INFO] [agent] **Dependencies... (13965 chars)" +[11:08:03] [INFO] [agent] Now let's update `prd.json` to mark DGR-032 as passing, then regenerate the issue markdown. +[11:08:05] [INFO] [agent] [Bash] $ python3 -c " import json path = '.scratch/distributed-gguf-runtime/prd.json' with open(path) as f: ... +[11:08:07] [INFO] [agent] [Bash] $ python3 scripts/ralph_prd_schema.py validate .scratch/distributed-gguf-runtime/prd.json 2>&1 | tail ... +[11:08:10] [INFO] [agent] [Bash] $ python3 scripts/ralph_prd_schema.py render .scratch/distributed-gguf-runtime/prd.json DGR-032 > .scr... +[11:08:12] [INFO] [agent] Good, that matches the pattern. Let's do a final check of `git status` and `git diff --check`. +[11:08:13] [INFO] [agent] [Bash] $ echo "diffcheck_rc=$?" +[11:08:18] [INFO] [agent] Now let's append learnings to `.ralph-tui/progress.md` as instructed. +[11:08:19] [INFO] [agent] [Bash] $ ls -la .ralph-tui/progress.md 2>&1 +[11:08:25] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.ralph-tui/progress.md +[11:08:28] [INFO] [agent] Now let me append a new pattern and the DGR-032 completion entry. +[11:08:28] [INFO] [agent] [Bash] $ wc -l .ralph-tui/progress.md +[11:08:34] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.ralph-tui/progress.md +[11:08:47] [INFO] [agent] [Edit] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.ralph-tui/progress.md edit: "- **Proving "this module imports no forbidden depe..." → "- **Proving "this module imports no forbidden depe..." +[11:08:49] [INFO] [agent] Now append the DGR-032 entry at the end of the file. +[11:08:50] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.ralph-tui/progress.md +[11:08:53] [INFO] [agent] [Read] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.ralph-tui/progress.md +[11:09:06] [INFO] [agent] [Edit] /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/.ralph-tui/progress.md edit: "- `tests/` as a whole still fails to collect in th..." → "- `tests/` as a whole still fails to collect in th..." +[11:09:15] [INFO] [agent] All acceptance criteria are met and evidence is documented. Summary of the work: +[11:09:15] [INFO] [agent] **Implemented DGR-032** — `FakeShardEngine`, a deterministic fixture `ShardEngine` subclass: +[11:09:15] [INFO] [agent] - `packages/node/meshnet_node/fake_shard_engine.py` (new) — SHA-256-hash-derived deterministic output per `(range, inputs, idempotency_step)`; head/middle/tail boundary-point tagging with token-ID sideband pass-through; isolated per-session epoch/cancel state; four opt-in fault-injection knobs via `FakeShardEngineConfig` (delay, memory-pressure budget, malformed-but-valid output, crash-after-N-calls raising a raw uncaught exception); `EVIDENCE_CLASS = "fixture"` marker for DGR-036's future fixture-vs-real comparison. +[11:09:15] [INFO] [agent] - `tests/test_fake_shard_engine.py` (new) — 15 tests covering the shared contract, head/middle/tail shape, session isolation, and each fault-injection knob (26 tests total combined with DGR-031's suite, all passing). +[11:09:15] [INFO] [agent] - `.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md` (new), `prd.json` marked `passes: true`, issue markdown regenerated. +[11:09:15] [INFO] [agent] Quality gates: targeted pytest (109 passed, 3 skipped, pre-existing/unrelated), `python -m compileall packages tests` clean, `git diff --check` clean. No commit was made per instructions — the engine will commit automatically. +[11:09:15] [INFO] [agent] COMPLETE diff --git a/.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md b/.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md new file mode 100644 index 0000000..284806d --- /dev/null +++ b/.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md @@ -0,0 +1,259 @@ +# DGR-032 evidence — deterministic fake `ShardEngine` + +**Completed:** 2026-07-23 +**Branch:** `ralph/distributed-gguf-runtime` +**Authority:** `.scratch/distributed-gguf-runtime/prd.json` +**Dependencies:** DGR-031 (`evidence/DGR-031/README.md` — the project-owned +`ShardEngine` abstract contract, `tests/shard_engine_contract.py`'s +`assert_shard_engine_contract`, and its own dependency-handoff note that +DGR-032 should "subclass `ShardEngine`, add delay/memory-pressure/malformed- +output/crash injection, and pass the *same* `assert_shard_engine_contract` +... — no new contract vocabulary should be needed"). + +## Objective + +Provide an engine fixture that deterministically transforms typed boundary +bundles and session state: head/middle/tail, prefill/decode, cancellation, +release, isolated per-session epoch state, deterministic cache-miss/stale- +epoch failures, and configurable delay/memory-pressure/malformed-output/ +crash-injection fault surfaces — all without llama.cpp, a GPU, or any I/O. + +## What was found live before changing code + +- `packages/node/meshnet_node/shard_engine.py` (DGR-031): the abstract + `ShardEngine` with eight operations (`load`, `capabilities`, `prefill`, + `decode`, `cancel`, `release`, `health`, `metrics`) and its project-owned + dataclasses (`LoadRequest`, `EngineCapabilities`, `PrefillRequest`/ + `DecodeRequest`, `StepResult`, `BoundaryBundle`/`EngineTensor`, + `TokenOutput`, `HealthResult`, `MetricsResult`). +- `tests/shard_engine_contract.py` (DGR-031): the reusable + `assert_shard_engine_contract(make_engine)` helper — nine lifecycle checks + any implementation must pass, explicitly designed to be imported by + DGR-032 and DGR-037 against their own engines. +- `tests/test_shard_engine.py` (DGR-031): its `_ReferenceEngine` is + explicitly documented as *not* the DGR-032 fixture ("no delay/memory- + pressure/malformed/crash injection... that is a separate, larger story") — + confirming this story starts from nothing, not inherited credit. +- `grep -rn -i "fakeshardengine\|fake_shard_engine"` across `.py`/`.md` + returned no prior matches — no fake engine existed before this story. +- No file in `packages/node/meshnet_node/` wires a `ShardEngine` into + `shard_runtime_server.py` yet (confirmed by grep for `ShardEngine`/ + `shard_engine` in that file — no matches); that wiring is DGR-037's scope, + so this fixture is a standalone, importable engine only. + +Live verification of the pre-existing dependency contract before adding new +code: +```bash +PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q tests/test_shard_engine.py +``` +```text +12 passed in 0.13s +``` + +## What was added (this story's change) + +### `packages/node/meshnet_node/fake_shard_engine.py` (new) + +`FakeShardEngine(ShardEngine)` — a pure-Python, deterministic fixture: + +- **Determinism.** Every `prefill`/`decode` output is `SHA-256(seed_bytes + + idempotency_step)`, where `seed_bytes` is derived from `token_ids` (head) + or the input `BoundaryBundle`'s tensor bytes plus any `token_id_sideband` + (middle/tail-in). Replaying identical inputs on a brand-new session + produces byte-identical output — proven by + `assert_shard_engine_contract`'s own determinism check and reused directly. +- **Head/middle/tail.** Tail shards (`shard_end >= total_layers - 1`) return + a `TokenOutput` sampled into `[0, TOKEN_ID_VOCAB_SIZE)`; head/middle shards + return a `BoundaryBundle` tagged `boundary_point="post_head_residual"` or + `"post_middle_residual"` respectively, so the three cases are + distinguishable in fixture output, not just in the load request. A middle + shard's `token_id_sideband` passes through unchanged from its input bundle + to its output bundle (the V4 first-three-hash-routed-layers requirement + RALPH-CONTEXT documents), never invented or dropped. +- **Isolated session/epoch state.** `_sessions: dict[str, _SessionState]` + keyed by `session_id`; each session tracks its own `epoch`/`cancelled` + flag. A stale epoch, cancel, or release on one session never touches + another's state (`test_session_state_is_isolated_between_two_concurrent_sessions` + proves a stale-epoch rejection and a cancel on session `"a"` leave session + `"b"` fully serviceable). Decoding an unopened session is a deterministic + `NOT_FOUND`/`CacheResult.MISS`, not an exception. +- **Configurable delay.** `FakeShardEngineConfig.step_delay_seconds` + + injectable `sleep` hook (defaults to `time.sleep`, overridable in tests so + they don't block wall-clock time) — invoked once per `prefill`/`decode` + call before computing the deterministic output. +- **Configurable memory pressure.** `FakeShardEngineConfig.memory_budget_bytes` + — the engine accumulates `_bytes_used` across every step's seed bytes; + once a step would push cumulative usage past the budget, that step + deterministically returns `StatusCode.RESOURCE_EXHAUSTED` (`retryable=True`) + with no output, instead of computing one. +- **Configurable malformed output.** `FakeShardEngineConfig.malformed_output` + — when set, the engine still reports `StatusCode.OK` (the point is a + buggy-but-"successful"-looking response, not a status-coded failure) but + the payload is structurally valid, semantically wrong: a tail `TokenOutput` + is pushed past `MALFORMED_TOKEN_ID_FLOOR` (outside the fixture's own + advertised vocab), and a head/middle `BoundaryBundle` gets an + `architecture` field prefixed `"malformed:"` and its tensor `data` + truncated to one byte — both structurally valid per `EngineTensor`'s and + `BoundaryBundle`'s own `__post_init__` validation (which does not + cross-check `data` length against `shape`/`dtype`), so a consumer must + actually check shape/semantics, not just status codes, to catch it. +- **Configurable crash injection.** `FakeShardEngineConfig.crash_after_calls` + + `crash_exception_factory` — after the configured number of + `prefill`/`decode` calls, the engine raises an arbitrary exception (default + `RuntimeError`, injectable) directly out of the call instead of returning a + `StepResult`. This is deliberately *not* wrapped in `EngineError`/ + `StructuredStatus`: it simulates a whole-process failure (what a worker + supervisor — DGR-040 — must catch and restart around), which is a + different failure mode from a graceful status-coded rejection. +- **Fixture-vs-real marker.** `FakeShardEngine.EVIDENCE_CLASS = "fixture"` — + a structural constant (not just docstring prose) so DGR-036's fixture-vs- + real-model parity check can assert programmatically that it is comparing a + fixture engine against a real one, never two fixtures. +- Every fault-injection knob defaults to off (`0`/`None`/`False`), so a bare + `FakeShardEngine()` passes `assert_shard_engine_contract` unmodified — + fault injection is opt-in, never a baseline behavior change. + +### `tests/test_fake_shard_engine.py` (new) + +- `test_fake_shard_engine_obeys_the_shared_shard_engine_contract` — runs the + full DGR-031 contract against a bare `FakeShardEngine`. +- `test_fake_shard_engine_declares_fixture_evidence_class` — pins the + `EVIDENCE_CLASS` marker DGR-036 will rely on. +- Head/middle/tail output-shape tests (`boundary_point`, token-id-sideband + pass-through, tail vocab range). +- `test_session_state_is_isolated_between_two_concurrent_sessions` — a + stale-epoch rejection and a cancel on one session leave a second, + concurrently open session fully serviceable. +- One test per fault-injection knob (delay hook invocation, memory-budget + trip, malformed tail/boundary-bundle output, crash-after-N-calls, + configurable crash exception type) plus `FakeShardEngineConfig`'s own + `__post_init__` validation (negative delay, negative budget, non-positive + `crash_after_calls`). +- `test_load_result_and_capabilities_report_recipe_architecture` — the + fixture threads `LoadRequest.recipe["architecture"]` through to both + `LoadResult.architecture` and `EngineCapabilities.architecture` rather than + hardcoding `"dense"`/`"fake"` everywhere, so a future V4 recipe is visible + in fixture output too. + +### `.scratch/distributed-gguf-runtime/prd.json` / issue markdown + +Marked `DGR-032.passes = true` with `completionNotes`; regenerated +`issues/032-implement-deterministic-fake-shardengine.md` via +`scripts/ralph_prd_schema.py render` so it matches `prd.json` byte-for-byte. + +## Acceptance criteria → evidence + +1. **Head, middle, tail, prefill, decode, cancellation, release with + deterministic outputs** — `FakeShardEngine`'s `_transform`, boundary-point + tagging, and `assert_shard_engine_contract`'s own determinism/cancel/ + release checks. Verified by + `test_fake_shard_engine_obeys_the_shared_shard_engine_contract`, + `test_head_shard_returns_boundary_bundle_with_post_head_residual_point`, + `test_middle_shard_returns_boundary_bundle_and_passes_through_token_sideband`, + `test_tail_shard_returns_token_output_within_advertised_vocab`. +2. **Isolated session/epoch state and deterministic cache-miss/stale-epoch + failures** — `_sessions` dict keyed per session; + `test_session_state_is_isolated_between_two_concurrent_sessions` plus the + shared contract's own cache-miss/stale-epoch checks. +3. **Configurable delay, memory pressure, malformed output, crash + injection** — `FakeShardEngineConfig`; verified by + `test_step_delay_seconds_invokes_the_configured_sleep_hook`, + `test_memory_budget_bytes_trips_deterministic_resource_exhausted`, + `test_malformed_output_is_structurally_valid_but_semantically_wrong_for_tail`, + `test_malformed_output_is_structurally_valid_but_semantically_wrong_for_boundary_bundle`, + `test_crash_after_calls_raises_instead_of_returning_a_structured_status`, + `test_crash_exception_factory_is_configurable`, + `test_config_rejects_invalid_knob_values`. +4. **Contract tests distinguish fixture evidence from real-model + certification** — module docstring and this README are explicit that + this is FIXTURE evidence only (numeric parity is DGR-036 onward); the + `EVIDENCE_CLASS = "fixture"` constant makes that distinction structurally + checkable, not just prose, pinned by + `test_fake_shard_engine_declares_fixture_evidence_class`. +5. **Gates + this handoff** — below. + +## Commands and results + +```bash +PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q tests/test_fake_shard_engine.py tests/test_shard_engine.py +``` +```text +26 passed in 0.17s +``` + +```bash +PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q \ + tests/test_fake_shard_engine.py tests/test_shard_engine.py tests/test_shard_lifecycle.py \ + tests/test_architecture_boundary.py tests/test_activation_envelope.py \ + tests/test_native_shard_protocol.py tests/test_shard_runtime_harness.py +``` +```text +109 passed, 3 skipped in 3.78s +``` + +```bash +.venv/bin/python3 -m compileall -q packages tests +``` +```text +(no output — clean; exit 0) +``` + +```bash +git diff --check +``` +```text +(no output — clean) +``` + +## Limitations + +- `tests/` as a whole does not collect cleanly in this environment: the same + pre-existing collection errors DGR-031's evidence recorded (missing + optional dependencies such as `cryptography`) are still present and are + unrelated to this story. This story's own gates were run as the targeted, + scoped test set above per the shared quality gates' wording ("Targeted + deterministic tests pass"). +- This is FIXTURE evidence only. `FakeShardEngine` proves lifecycle, + session/epoch isolation, and fault-injection semantics; it proves nothing + about numerical parity with a real model. That is DGR-036's explicit job + once DGR-037's real engine exists, and DGR-053/054 for V4 alpha + certification. +- `FakeShardEngine` is not wired into `shard_runtime_server.py` or any gRPC + surface — it is a standalone, importable engine only. Wiring a + `ShardEngine` (fake or real) into the gRPC servicer is DGR-037's scope for + the real engine; DGR-033 covers a C++ worker surface, which is a separate + native executable, not a consumer of this Python module. +- No gRPC/CMake/native-build changes were needed or made — this story is + pure Python fixture code (`evidenceClass: fixture`, `hardware: none`), so + the native CMake/CTest and patch-stack gates in the shared quality-gate + list do not apply here, consistent with DGR-031's own README recording the + same non-applicability. + +## Dependency handoff + +- **DGR-033** (standalone fake C++ gRPC Shard worker): its own issue + describes a native C++ executable serving the lifecycle/stream RPC + contract "using the fake engine" — that is a native analogue, not a + consumer of this Python module; DGR-033 should still read this README for + the exact deterministic-output/session-isolation/fault-injection semantics + its C++ fake engine needs to reproduce so both fakes behave identically + from a client's point of view. +- **DGR-034/DGR-035** (range-aware GGUF ownership, boundary I/O): + `FakeShardEngine` already demonstrates range-driven head/middle/tail + behavior purely from `LoadRequest.shard_start`/`shard_end`/`total_layers`; + no new range vocabulary was needed. +- **DGR-036** (fixture vs real-model parity): compare a `FakeShardEngine` + instance's `EVIDENCE_CLASS` (`"fixture"`) against DGR-037's real engine's + equivalent marker (expected `"real"`) to assert the parity check is + actually comparing two different implementations; reuse + `assert_shard_engine_contract` against both to prove lifecycle parity + before attempting numeric parity. +- **DGR-037** (bind llama.cpp to the worker): `FakeShardEngine` is the + reference implementation to diff a real engine's lifecycle behavior + against — same request/result types, same session/epoch model, no new + contract vocabulary. +- **DGR-040** (worker supervision): the crash-injection knob + (`crash_after_calls`/`crash_exception_factory`) exists specifically so + supervision/restart logic has a deterministic way to trigger and test an + unhandled engine failure distinct from a graceful `StructuredStatus` + rejection. diff --git a/.scratch/distributed-gguf-runtime/issues/032-implement-deterministic-fake-shardengine.md b/.scratch/distributed-gguf-runtime/issues/032-implement-deterministic-fake-shardengine.md index b22d2e0..3ab5802 100644 --- a/.scratch/distributed-gguf-runtime/issues/032-implement-deterministic-fake-shardengine.md +++ b/.scratch/distributed-gguf-runtime/issues/032-implement-deterministic-fake-shardengine.md @@ -1,7 +1,7 @@ # DGR-032: Implement deterministic fake `ShardEngine` -- **Status / triage:** specification only; `ready-for-agent`; `passes: false` +- **Status / triage:** completed; `passes: true` - **Execution mode:** `AFK` - **Milestone:** `M1` - **Dependencies:** `DGR-031` @@ -18,11 +18,11 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`, ## Acceptance criteria -- [ ] Support head, middle, tail, prefill, decode, cancellation, and release with deterministic outputs. -- [ ] Model isolated session/epoch state and deterministic cache-miss/stale-epoch failures. -- [ ] Support configurable delay, memory pressure, malformed output, and crash injection. -- [ ] Contract tests distinguish fixture evidence from real-model certification. -- [ ] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff. +- [x] Support head, middle, tail, prefill, decode, cancellation, and release with deterministic outputs. +- [x] Model isolated session/epoch state and deterministic cache-miss/stale-epoch failures. +- [x] Support configurable delay, memory pressure, malformed output, and crash injection. +- [x] Contract tests distinguish fixture evidence from real-model certification. +- [x] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff. ## Shared quality gates @@ -30,10 +30,7 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`, - `git diff --check` passes. - Default tests are model-download-free, API-credit-free, and GPU-free. - Evidence README records exact changed files, commands/results, limitations, and dependency handoff; no fabricated evidence or inherited completion credit. -- Native changes pass focused out-of-tree CMake build and CTest; patch changes verify clean apply/check/reverse against the exact llama.cpp pin. -- Runs are opt-in and record exact artifact/split hashes, runtime/upstream pin, backend/driver, hardware, network, commands, and raw metrics. Model artifacts use configured mounted-drive storage and never `/home`. -- Preserve existing Transformers behavior and backend-agnostic Tracker routing/load balancing/billing/relay semantics unless an explicit versioned contract says otherwise. One scoped story commit is expected during execution, but this specification-materialization change is not committed. ## Evidence handoff -Write and verify `.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md`. Until every criterion and applicable gate has real evidence, this story remains `passes: false`. Legacy evidence is provenance only, not completion credit. +Verified evidence: `.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md`. Legacy evidence remains provenance only and grants no implementation completion credit. diff --git a/.scratch/distributed-gguf-runtime/prd.json b/.scratch/distributed-gguf-runtime/prd.json index ea02a4b..d065c3b 100644 --- a/.scratch/distributed-gguf-runtime/prd.json +++ b/.scratch/distributed-gguf-runtime/prd.json @@ -617,11 +617,12 @@ "Contract tests distinguish fixture evidence from real-model certification.", "Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff." ], - "passes": false, + "passes": true, "notes": "Generated source issue: .scratch/distributed-gguf-runtime/issues/032-implement-deterministic-fake-shardengine.md; prd.json is authoritative.", "blocks": [ "DGR-033" - ] + ], + "completionNotes": "Completed by agent" }, { "id": "DGR-033", @@ -2163,6 +2164,6 @@ } ], "metadata": { - "updatedAt": "2026-07-23T08:00:32.939Z" + "updatedAt": "2026-07-23T08:09:16.081Z" } } \ No newline at end of file diff --git a/packages/node/meshnet_node/fake_shard_engine.py b/packages/node/meshnet_node/fake_shard_engine.py new file mode 100644 index 0000000..e862a82 --- /dev/null +++ b/packages/node/meshnet_node/fake_shard_engine.py @@ -0,0 +1,302 @@ +"""Deterministic fake ``ShardEngine`` fixture (DGR-032). + +``FakeShardEngine`` is a pure-Python, allocation-cheap subclass of +:class:`~meshnet_node.shard_engine.ShardEngine`: no llama.cpp, no native +buffers, no GPU, no filesystem or network I/O. Every prefill/decode output is +a deterministic pure function of ``(loaded range, request inputs, +idempotency_step)`` — hashed with SHA-256 — so replaying identical inputs on +a fresh session always yields byte-identical output. It exists so worker +wiring, gRPC harnesses (DGR-033), and lifecycle/session logic can be +exercised end-to-end before a real llama.cpp-backed engine (DGR-037) exists. + +This is FIXTURE evidence only. ``EVIDENCE_CLASS`` is set to ``"fixture"`` (as +opposed to ``"real"``) precisely so a later story comparing engines +programmatically — DGR-036's fixture-vs-real-model parity check — can assert +it is actually comparing a fixture against a real engine rather than two +fixtures. This module proves lifecycle/session/epoch/fault-injection +semantics; it says nothing about numerical parity with a real model. Real- +model certification is DGR-036 onward (DGR-053/DGR-054 for V4 alpha). + +Fault injection (delay, memory pressure, malformed output, crash) is +deterministic and opt-in via :class:`FakeShardEngineConfig`. Every knob +defaults to off, so a bare ``FakeShardEngine()`` reproduces plain +deterministic fixture behavior and passes +:func:`tests.shard_engine_contract.assert_shard_engine_contract` unmodified. +""" + +from __future__ import annotations + +import hashlib +import time +from dataclasses import dataclass +from typing import Callable + +from .shard_engine import ( + BoundaryBundle, + DecodeRequest, + EngineCapabilities, + EngineTensor, + HealthResult, + LoadRequest, + LoadResult, + MetricsResult, + PrefillRequest, + ShardEngine, + StepResult, + TokenOutput, +) +from .shard_lifecycle import CacheResult, StatusCode, StructuredStatus + +__all__ = ["FakeShardEngineConfig", "FakeShardEngine", "TOKEN_ID_VOCAB_SIZE", "MALFORMED_TOKEN_ID_FLOOR"] + +TOKEN_ID_VOCAB_SIZE = 50_000 +# A malformed tail output is deterministically pushed past the fixture's own +# advertised vocabulary range, so a downstream consumer checking "is this +# token_id within the vocab this fixture promises" can detect it without any +# extra signalling from the engine. +MALFORMED_TOKEN_ID_FLOOR = 100_000_000 + + +def _default_crash_exception() -> BaseException: + return RuntimeError( + "FakeShardEngine: injected crash (simulated process failure, not a StructuredStatus)" + ) + + +@dataclass(frozen=True) +class FakeShardEngineConfig: + """Deterministic fault-injection knobs. + + Every knob is off (``0``/``None``/``False``) by default. ``sleep`` is + injectable so tests can assert a delay was requested without an actual + process sleep; ``crash_exception_factory`` is injectable so tests can + assert on a specific exception type/instance. + """ + + step_delay_seconds: float = 0.0 + sleep: Callable[[float], None] = time.sleep + memory_budget_bytes: int | None = None + malformed_output: bool = False + crash_after_calls: int | None = None + crash_exception_factory: Callable[[], BaseException] = _default_crash_exception + + def __post_init__(self) -> None: + if self.step_delay_seconds < 0: + raise ValueError("step_delay_seconds must be non-negative") + if self.memory_budget_bytes is not None and self.memory_budget_bytes < 0: + raise ValueError("memory_budget_bytes must be non-negative") + if self.crash_after_calls is not None and self.crash_after_calls <= 0: + raise ValueError("crash_after_calls must be positive when set") + + +@dataclass +class _SessionState: + epoch: int + cancelled: bool = False + + +class FakeShardEngine(ShardEngine): + """Deterministic fixture ``ShardEngine``. See module docstring.""" + + EVIDENCE_CLASS = "fixture" + + def __init__(self, config: FakeShardEngineConfig | None = None) -> None: + self._config = config or FakeShardEngineConfig() + self._loaded: LoadRequest | None = None + self._sessions: dict[str, _SessionState] = {} + self._cancelled_total = 0 + self._generated_tokens = 0 + self._call_count = 0 + self._bytes_used = 0 + + # -- lifecycle ----------------------------------------------------- + + def load(self, request: LoadRequest) -> LoadResult: + self._loaded = request + return LoadResult( + status=StructuredStatus(StatusCode.OK, "fake engine loaded"), + effective_start=request.shard_start, + architecture=str(request.recipe.get("architecture", "fake")), + ) + + def capabilities(self) -> EngineCapabilities: + if self._loaded is None: + return EngineCapabilities( + status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "engine not loaded") + ) + request = self._loaded + return EngineCapabilities( + status=StructuredStatus(StatusCode.OK, "ready"), + shard_start=request.shard_start, + shard_end=request.shard_end, + effective_start=request.shard_start, + total_layers=request.total_layers, + architecture=str(request.recipe.get("architecture", "fake")), + max_concurrent_sessions=64, + max_context_tokens=131072, + supports_mtp=False, + ) + + def prefill(self, request: PrefillRequest) -> StepResult: + return self._step( + session_id=request.session_id, + route_epoch=request.route_epoch, + idempotency_step=request.idempotency_step, + token_ids=request.token_ids, + input_bundle=request.input, + cache_result_on_success=CacheResult.STORED, + opens_session=True, + ) + + def decode(self, request: DecodeRequest) -> StepResult: + token_ids = (request.token_id,) if request.token_id is not None else None + return self._step( + session_id=request.session_id, + route_epoch=request.route_epoch, + idempotency_step=request.idempotency_step, + token_ids=token_ids, + input_bundle=request.input, + cache_result_on_success=CacheResult.HIT, + opens_session=False, + ) + + def cancel(self, session_id: str, *, work_id: str = "", reason: str = "") -> StructuredStatus: + session = self._sessions.get(session_id) + if session is None: + session = _SessionState(epoch=0) + self._sessions[session_id] = session + if not session.cancelled: + self._cancelled_total += 1 + session.cancelled = True + return StructuredStatus(StatusCode.CANCELLED, reason or "fake engine: session cancelled") + + def release(self, session_id: str) -> StructuredStatus: + self._sessions.pop(session_id, None) + return StructuredStatus(StatusCode.OK, "fake engine: session released") + + def health(self) -> HealthResult: + loaded = self._loaded is not None + return HealthResult( + status=StructuredStatus(StatusCode.OK, "ok"), + serving=loaded, + state="SERVING" if loaded else "NOT_LOADED", + active_sessions=len(self._sessions), + ) + + def metrics(self) -> MetricsResult: + return MetricsResult( + status=StructuredStatus(StatusCode.OK, "ok"), + active_sessions=len(self._sessions), + queued_frames=0, + inflight_bytes=0, + kv_entries=len(self._sessions), + generated_tokens=self._generated_tokens, + cancelled_sessions=self._cancelled_total, + ) + + # -- shared step machinery ------------------------------------------ + + def _step( + self, + *, + session_id: str, + route_epoch: int, + idempotency_step: int, + token_ids: tuple[int, ...] | None, + input_bundle: BoundaryBundle | None, + cache_result_on_success: CacheResult, + opens_session: bool, + ) -> StepResult: + if self._loaded is None: + return StepResult(status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "engine not loaded")) + + self._call_count += 1 + if self._config.crash_after_calls is not None and self._call_count == self._config.crash_after_calls: + raise self._config.crash_exception_factory() + + session = self._sessions.get(session_id) + if session is None: + if not opens_session: + return StepResult( + status=StructuredStatus(StatusCode.NOT_FOUND, "no cached session state for decode"), + cache_result=CacheResult.MISS, + ) + session = _SessionState(epoch=route_epoch) + self._sessions[session_id] = session + + if session.cancelled: + return StepResult(status=StructuredStatus(StatusCode.CANCELLED, "session cancelled")) + + if route_epoch < session.epoch: + return StepResult(status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "stale route epoch")) + session.epoch = route_epoch + + if self._config.step_delay_seconds: + self._config.sleep(self._config.step_delay_seconds) + + seed = self._seed_bytes(token_ids, input_bundle) + self._bytes_used += len(seed) + budget = self._config.memory_budget_bytes + if budget is not None and self._bytes_used > budget: + return StepResult( + status=StructuredStatus( + StatusCode.RESOURCE_EXHAUSTED, + "fake engine memory pressure budget exceeded", + retryable=True, + details={"memory_budget_bytes": str(budget), "bytes_used": str(self._bytes_used)}, + ) + ) + + output = self._transform(seed, idempotency_step, input_bundle) + if isinstance(output, TokenOutput): + self._generated_tokens += 1 + return StepResult(status=StructuredStatus(StatusCode.OK, "ok"), cache_result=cache_result_on_success, output=output) + + @staticmethod + def _seed_bytes(token_ids: tuple[int, ...] | None, bundle: BoundaryBundle | None) -> bytes: + if token_ids: + seed = b"".join(int(t).to_bytes(8, "big") for t in token_ids) + elif bundle is not None: + seed = b"".join(tensor.data for tensor in bundle.tensors) + if bundle.token_id_sideband: + seed += b"".join(int(t).to_bytes(8, "big") for t in bundle.token_id_sideband) + else: + seed = b"" + return seed + + def _transform( + self, seed: bytes, idempotency_step: int, input_bundle: BoundaryBundle | None + ) -> BoundaryBundle | TokenOutput: + assert self._loaded is not None + digest = hashlib.sha256(seed + idempotency_step.to_bytes(8, "big")).digest() + loaded = self._loaded + is_tail = loaded.shard_end >= loaded.total_layers - 1 + is_head = loaded.shard_start == 0 + + if is_tail: + token_id = int.from_bytes(digest[:4], "big") % TOKEN_ID_VOCAB_SIZE + if self._config.malformed_output: + token_id = MALFORMED_TOKEN_ID_FLOOR + token_id + return TokenOutput(token_id=token_id) + + boundary_point = "post_head_residual" if is_head else "post_middle_residual" + architecture = ( + input_bundle.architecture if input_bundle is not None else str(loaded.recipe.get("architecture", "fake")) + ) + data = digest + if self._config.malformed_output: + architecture = f"malformed:{architecture}" + data = digest[:1] + tensor = EngineTensor( + name="hidden_states", + shape=(1, max(len(seed) // 8, 1)), + dtype="bfloat16", + data=data, + ) + token_id_sideband = input_bundle.token_id_sideband if input_bundle is not None else None + return BoundaryBundle( + tensors=(tensor,), + architecture=architecture, + boundary_point=boundary_point, + token_id_sideband=token_id_sideband, + ) diff --git a/tests/test_fake_shard_engine.py b/tests/test_fake_shard_engine.py new file mode 100644 index 0000000..413d149 --- /dev/null +++ b/tests/test_fake_shard_engine.py @@ -0,0 +1,200 @@ +"""DGR-032 ``FakeShardEngine`` tests. + +``FakeShardEngine`` obeys the exact same lifecycle contract every +``ShardEngine`` implementation must (see ``shard_engine_contract.py``); the +tests here additionally cover this story's own scope: head/middle/tail +output shape, isolated multi-session state, and the delay/memory-pressure/ +malformed-output/crash fault-injection knobs. None of this is real-model +evidence — see the module docstring in ``fake_shard_engine.py`` and +``test_fake_shard_engine_declares_fixture_evidence_class`` below, which pins +the marker DGR-036 will rely on to tell a fixture engine apart from a real +one when it certifies fixture-vs-real parity. +""" + +from __future__ import annotations + +import pytest + +from meshnet_node.fake_shard_engine import ( + MALFORMED_TOKEN_ID_FLOOR, + TOKEN_ID_VOCAB_SIZE, + FakeShardEngine, + FakeShardEngineConfig, +) +from meshnet_node.shard_engine import ( + BoundaryBundle, + DecodeRequest, + EngineTensor, + LoadRequest, + PrefillRequest, + TokenOutput, +) +from meshnet_node.shard_lifecycle import StatusCode + +from shard_engine_contract import assert_shard_engine_contract + + +def _load(engine: FakeShardEngine, *, shard_start=0, shard_end=3, total_layers=4, recipe=None): + return engine.load( + LoadRequest( + artifact_path="fixture://fake-shard-engine", + shard_start=shard_start, + shard_end=shard_end, + total_layers=total_layers, + recipe=recipe or {}, + ) + ) + + +def test_fake_shard_engine_obeys_the_shared_shard_engine_contract(): + assert_shard_engine_contract(FakeShardEngine) + + +def test_fake_shard_engine_declares_fixture_evidence_class(): + # DGR-036's fixture-vs-real parity check needs a structural way to tell + # a fixture engine apart from a real one; this constant is that marker. + assert FakeShardEngine.EVIDENCE_CLASS == "fixture" + + +def test_head_shard_returns_boundary_bundle_with_post_head_residual_point(): + engine = FakeShardEngine() + _load(engine, shard_start=0, shard_end=1, total_layers=4) + result = engine.prefill( + PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2)) + ) + assert result.status.code is StatusCode.OK + assert isinstance(result.output, BoundaryBundle) + assert result.output.boundary_point == "post_head_residual" + + +def test_middle_shard_returns_boundary_bundle_and_passes_through_token_sideband(): + engine = FakeShardEngine() + _load(engine, shard_start=1, shard_end=2, total_layers=8) + bundle_in = BoundaryBundle( + tensors=(EngineTensor(name="hidden_states", shape=(1, 2), dtype="bfloat16", data=b"\x01\x02\x03\x04"),), + architecture="dense", + boundary_point="pre_tail_residual", + token_id_sideband=(7, 8, 9), + ) + result = engine.prefill( + PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, input=bundle_in) + ) + assert result.status.code is StatusCode.OK + assert isinstance(result.output, BoundaryBundle) + assert result.output.boundary_point == "post_middle_residual" + assert result.output.token_id_sideband == (7, 8, 9) + + +def test_tail_shard_returns_token_output_within_advertised_vocab(): + engine = FakeShardEngine() + _load(engine, shard_start=0, shard_end=3, total_layers=4) + result = engine.prefill( + PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2, 3)) + ) + assert isinstance(result.output, TokenOutput) + assert 0 <= result.output.token_id < TOKEN_ID_VOCAB_SIZE + + +def test_session_state_is_isolated_between_two_concurrent_sessions(): + engine = FakeShardEngine() + _load(engine) + engine.prefill(PrefillRequest(session_id="a", route_epoch=5, position=0, idempotency_step=0, token_ids=(1,))) + engine.prefill(PrefillRequest(session_id="b", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,))) + + # A stale epoch against session "a" must not affect session "b" at all. + stale = engine.decode(DecodeRequest(session_id="a", route_epoch=4, position=1, idempotency_step=1, token_id=2)) + assert stale.status.code is StatusCode.FAILED_PRECONDITION + + still_fine = engine.decode(DecodeRequest(session_id="b", route_epoch=1, position=1, idempotency_step=1, token_id=2)) + assert still_fine.status.code is StatusCode.OK + + engine.cancel("a") + after_cancel_b = engine.decode( + DecodeRequest(session_id="b", route_epoch=1, position=2, idempotency_step=2, token_id=3) + ) + assert after_cancel_b.status.code is StatusCode.OK, "cancelling session a must not cancel session b" + + +def test_step_delay_seconds_invokes_the_configured_sleep_hook(): + calls: list[float] = [] + engine = FakeShardEngine(FakeShardEngineConfig(step_delay_seconds=0.25, sleep=calls.append)) + _load(engine) + engine.prefill(PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,))) + engine.decode(DecodeRequest(session_id="s", route_epoch=1, position=1, idempotency_step=1, token_id=2)) + assert calls == [0.25, 0.25] + + +def test_memory_budget_bytes_trips_deterministic_resource_exhausted(): + engine = FakeShardEngine(FakeShardEngineConfig(memory_budget_bytes=8)) + _load(engine) + first = engine.prefill( + PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,)) + ) + assert first.status.code is StatusCode.OK # 8 bytes used, exactly at budget + + second = engine.decode(DecodeRequest(session_id="s", route_epoch=1, position=1, idempotency_step=1, token_id=2)) + assert second.status.code is StatusCode.RESOURCE_EXHAUSTED + assert second.status.retryable is True + assert second.output is None + + +def test_malformed_output_is_structurally_valid_but_semantically_wrong_for_tail(): + engine = FakeShardEngine(FakeShardEngineConfig(malformed_output=True)) + _load(engine, shard_start=0, shard_end=3, total_layers=4) + result = engine.prefill( + PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2, 3)) + ) + assert result.status.code is StatusCode.OK + assert isinstance(result.output, TokenOutput) + assert result.output.token_id >= MALFORMED_TOKEN_ID_FLOOR + + +def test_malformed_output_is_structurally_valid_but_semantically_wrong_for_boundary_bundle(): + engine = FakeShardEngine(FakeShardEngineConfig(malformed_output=True)) + _load(engine, shard_start=0, shard_end=1, total_layers=4) + result = engine.prefill( + PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2, 3)) + ) + assert result.status.code is StatusCode.OK + assert isinstance(result.output, BoundaryBundle) + assert result.output.architecture.startswith("malformed:") + assert len(result.output.tensors[0].data) == 1 + + +def test_crash_after_calls_raises_instead_of_returning_a_structured_status(): + engine = FakeShardEngine(FakeShardEngineConfig(crash_after_calls=2)) + _load(engine) + ok = engine.prefill(PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,))) + assert ok.status.code is StatusCode.OK + + with pytest.raises(RuntimeError): + engine.decode(DecodeRequest(session_id="s", route_epoch=1, position=1, idempotency_step=1, token_id=2)) + + +def test_crash_exception_factory_is_configurable(): + class _SimulatedSegfault(Exception): + pass + + engine = FakeShardEngine( + FakeShardEngineConfig(crash_after_calls=1, crash_exception_factory=_SimulatedSegfault) + ) + _load(engine) + with pytest.raises(_SimulatedSegfault): + engine.prefill(PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,))) + + +def test_config_rejects_invalid_knob_values(): + with pytest.raises(ValueError): + FakeShardEngineConfig(step_delay_seconds=-1.0) + with pytest.raises(ValueError): + FakeShardEngineConfig(memory_budget_bytes=-1) + with pytest.raises(ValueError): + FakeShardEngineConfig(crash_after_calls=0) + + +def test_load_result_and_capabilities_report_recipe_architecture(): + engine = FakeShardEngine() + load_result = _load(engine, recipe={"architecture": "deepseek-v4-flash"}) + assert load_result.architecture == "deepseek-v4-flash" + caps = engine.capabilities() + assert caps.architecture == "deepseek-v4-flash"