Compare commits
3 Commits
ralph/dist
...
25e53bfeab
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
25e53bfeab | ||
|
|
c34ab059cc | ||
|
|
fd742d35c0 |
1405
.fuse_hidden0002bd66000001f0
Normal file
1405
.fuse_hidden0002bd66000001f0
Normal file
File diff suppressed because it is too large
Load Diff
1521
.fuse_hidden0002bd66000001f9
Normal file
1521
.fuse_hidden0002bd66000001f9
Normal file
File diff suppressed because it is too large
Load Diff
5
.ralph-supervisor.log
Normal file
5
.ralph-supervisor.log
Normal file
@@ -0,0 +1,5 @@
|
||||
[2026-07-23 10:24:53] supervisor started, tailer pid=1460238
|
||||
[2026-07-23 10:24:53] cycle 1: running ralph-tui resume (log starts at line 978)
|
||||
[2026-07-23 10:25:59] ralph-tui exited without a recognized stop reason; retrying resume in 5 min
|
||||
[2026-07-23 10:33:51] supervisor started, tailer pid=1465293
|
||||
[2026-07-23 10:33:51] cycle 1: running ralph-tui run (log starts at line 1150)
|
||||
@@ -976,3 +976,329 @@ reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=DGR-030 dry_run=False
|
||||
reconciled DGR-017 #1 completed
|
||||
reconciled DGR-018 #2 completed
|
||||
reconciled DGR-019 #3 completed
|
||||
reconciled DGR-020 #4 completed
|
||||
reconciled DGR-021 #5 completed
|
||||
reconciled DGR-022 #6 completed
|
||||
reconciled DGR-023 #7 completed
|
||||
reconciled DGR-024 #8 completed
|
||||
reconciled DGR-025 #9 completed
|
||||
reconciled DGR-026 #10 completed
|
||||
reconciled DGR-027 #11 completed
|
||||
reconciled DGR-028 #12 completed
|
||||
reconciled DGR-029 #13 completed
|
||||
reconciled DGR-030 #14 in-progress
|
||||
reconciled DGR-031 #15 ready
|
||||
reconciled DGR-032 #16 blocked
|
||||
reconciled DGR-033 #17 blocked
|
||||
reconciled DGR-034 #18 blocked
|
||||
reconciled DGR-035 #19 blocked
|
||||
reconciled DGR-036 #20 blocked
|
||||
reconciled DGR-037 #21 blocked
|
||||
reconciled DGR-038 #22 blocked
|
||||
reconciled DGR-039 #23 blocked
|
||||
reconciled DGR-040 #24 blocked
|
||||
reconciled DGR-041 #25 blocked
|
||||
reconciled DGR-042 #26 blocked
|
||||
reconciled DGR-043 #27 blocked
|
||||
reconciled DGR-044 #28 ready
|
||||
reconciled DGR-045 #29 blocked
|
||||
reconciled DGR-046 #30 blocked
|
||||
reconciled DGR-047 #31 blocked
|
||||
reconciled DGR-048 #32 blocked
|
||||
reconciled DGR-049 #33 blocked
|
||||
reconciled DGR-050 #34 blocked
|
||||
reconciled DGR-051 #35 blocked
|
||||
reconciled DGR-052 #36 blocked
|
||||
reconciled DGR-053 #37 blocked
|
||||
reconciled DGR-054 #38 blocked
|
||||
reconciled DGR-055 #39 blocked
|
||||
reconciled DGR-056 #40 blocked
|
||||
reconciled DGR-057 #41 blocked
|
||||
reconciled DGR-058 #42 blocked
|
||||
reconciled DGR-059 #43 blocked
|
||||
reconciled DGR-060 #44 blocked
|
||||
reconciled DGR-061 #45 blocked
|
||||
reconciled DGR-062 #46 blocked
|
||||
reconciled DGR-063 #47 blocked
|
||||
reconciled DGR-064 #48 blocked
|
||||
reconciled DGR-065 #49 blocked
|
||||
reconciled DGR-066 #50 blocked
|
||||
reconciled DGR-067 #51 blocked
|
||||
reconciled DGR-068 #52 blocked
|
||||
reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=DGR-030 dry_run=False
|
||||
|
||||
📦 Upgrading ralph-tui configuration...
|
||||
Installing bundled skills for detected agents...
|
||||
Installing skills for Claude Code...
|
||||
✓ Skills installed for Claude Code (claude-code)
|
||||
Installing skills for OpenCode...
|
||||
✓ Skills installed for OpenCode (opencode)
|
||||
· Skipping Factory Droid (not installed)
|
||||
· Skipping Gemini CLI (not installed)
|
||||
Installing skills for Codex CLI...
|
||||
✓ Skills installed for Codex CLI (codex)
|
||||
· Skipping Kiro CLI (not installed)
|
||||
Installing skills for Cursor Agent...
|
||||
✓ Skills installed for Cursor Agent (cursor)
|
||||
· Skipping GitHub Copilot (not installed)
|
||||
Installing skills for Kimi CLI...
|
||||
✗ Failed for Kimi CLI
|
||||
· Skipping Pi Coding Agent (not installed)
|
||||
✓ Installed 3 template(s) to /home/popov/.config/ralph-tui/templates
|
||||
✓ Updated config version
|
||||
|
||||
✅ Upgraded to config version 2.1
|
||||
|
||||
⚠️ Warnings:
|
||||
• Failed to install skills for Kimi CLI:
|
||||
[33m[1mDEPRECATED:[0m[33m 'add-skill' has been renamed to 'skills'[0m
|
||||
|
||||
Please use: [1mnpx skills add <package>[0m
|
||||
|
||||
Example: npx skills add vercel-labs/agent-skills
|
||||
|
||||
[33mForwarding to 'npx skills add'...[0m
|
||||
|
||||
|
||||
[90m│[39m
|
||||
[34m●[39m [46m[30m[1m claude-code_2-1-216_agent [22m[39m[49m Agent detected — installing non-interactively
|
||||
[?25l[90m│[39m
|
||||
[32m◇[39m Source: https://github.com/subsy/ralph-tui.git
|
||||
[?25h[?25l[90m│[39m
|
||||
[35m◒[39m Cloning repository…[1G[J[35m◐[39m Cloning repository…[1G[J[35m◓[39m Cloning repository…[1G[J[35m◑[39m Cloning repository…[1G[J[35m◒[39m Cloning repository…[1G[J[35m◐[39m Cloning repository…[1G[J[35m◓[39m Cloning repository…[1G[J[35m◑[39m Cloning repository…[1G[J[35m◒[39m Cloning repository….[1G[J[35m◐[39m Cloning repository….[1G[J[35m◓[39m Cloning repository….[1G[J[35m◑[39m Cloning repository….[1G[J[35m◒[39m Cloning repository….[1G[J[35m◐[39m Cloning repository….[1G[J[35m◓[39m Cloning repository….[1G[J[35m◑[39m Cloning repository….[1G[J[35m◒[39m Cloning repository…..[1G[J[35m◐[39m Cloning repository…..[1G[J[35m◓[39m Cloning repository…..[1G[J[35m◑[39m Cloning repository…..[1G[J[35m◒[39m Cloning repository…..[1G[J[35m◐[39m Cloning repository…..[1G[J[32m◇[39m Repository cloned
|
||||
[?25h[?25l[90m│[39m
|
||||
[1G[J[32m◇[39m Found [32m4[39m skills
|
||||
[?25h[90m│[39m
|
||||
[34m●[39m Installing all 4 skills
|
||||
[90m│[39m
|
||||
[31m■[39m Invalid agents: kimi-cli
|
||||
[90m│[39m
|
||||
[34m●[39m Valid agents: aider-desk, amp, antigravity, antigravity-cli, astrbot, autohand-code, augment, bob, claude-code, openclaw, cline, codearts-agent, codebuddy, codemaker, codestudio, codex, command-code, continue, cortex, crush, cursor, deepagents, devin, dexto, droid, eve, firebender, forgecode, gemini-cli, github-copilot, goose, grok, hermes-agent, inference-sh, jazz, junie, iflow-cli, kilo, kimchi, kimi-code-cli, kiro-cli, kode, lingma, loaf, mcpjam, mistral-vibe, moxby, mux, opencode, openhands, ona, pi, qoder, qoder-cn, qwen-code, replit, reasonix, rovodev, roo, tabnine-cli, terramind, tinycloud, trae, trae-cn, warp, windsurf, zed, zcode, zencoder, zenflow, neovate, pochi, promptscript, adal, universal
|
||||
|
||||
|
||||
Initializing Ralph TUI...
|
||||
Env filter: no vars matched exclusion patterns (*_API_KEY, *_SECRET_KEY, *_SECRET)
|
||||
|
||||
|
||||
⚠️ Recovered stale session
|
||||
Cleared 5 stuck in-progress task(s)
|
||||
Session status set to "interrupted" (resumable)
|
||||
|
||||
Resuming previous session...
|
||||
[0m[31mFailed to resume session[0m
|
||||
reconciled DGR-017 #1 completed
|
||||
reconciled DGR-018 #2 completed
|
||||
reconciled DGR-019 #3 completed
|
||||
reconciled DGR-020 #4 completed
|
||||
reconciled DGR-021 #5 completed
|
||||
reconciled DGR-022 #6 completed
|
||||
reconciled DGR-023 #7 completed
|
||||
reconciled DGR-024 #8 completed
|
||||
reconciled DGR-025 #9 completed
|
||||
reconciled DGR-026 #10 completed
|
||||
reconciled DGR-027 #11 completed
|
||||
reconciled DGR-028 #12 completed
|
||||
reconciled DGR-029 #13 completed
|
||||
reconciled DGR-030 #14 ready
|
||||
reconciled DGR-031 #15 ready
|
||||
reconciled DGR-032 #16 blocked
|
||||
reconciled DGR-033 #17 blocked
|
||||
reconciled DGR-034 #18 blocked
|
||||
reconciled DGR-035 #19 blocked
|
||||
reconciled DGR-036 #20 blocked
|
||||
reconciled DGR-037 #21 blocked
|
||||
reconciled DGR-038 #22 blocked
|
||||
reconciled DGR-039 #23 blocked
|
||||
reconciled DGR-040 #24 blocked
|
||||
reconciled DGR-041 #25 blocked
|
||||
reconciled DGR-042 #26 blocked
|
||||
reconciled DGR-043 #27 blocked
|
||||
reconciled DGR-044 #28 ready
|
||||
reconciled DGR-045 #29 blocked
|
||||
reconciled DGR-046 #30 blocked
|
||||
reconciled DGR-047 #31 blocked
|
||||
reconciled DGR-048 #32 blocked
|
||||
reconciled DGR-049 #33 blocked
|
||||
reconciled DGR-050 #34 blocked
|
||||
reconciled DGR-051 #35 blocked
|
||||
reconciled DGR-052 #36 blocked
|
||||
reconciled DGR-053 #37 blocked
|
||||
reconciled DGR-054 #38 blocked
|
||||
reconciled DGR-055 #39 blocked
|
||||
reconciled DGR-056 #40 blocked
|
||||
reconciled DGR-057 #41 blocked
|
||||
reconciled DGR-058 #42 blocked
|
||||
reconciled DGR-059 #43 blocked
|
||||
reconciled DGR-060 #44 blocked
|
||||
reconciled DGR-061 #45 blocked
|
||||
reconciled DGR-062 #46 blocked
|
||||
reconciled DGR-063 #47 blocked
|
||||
reconciled DGR-064 #48 blocked
|
||||
reconciled DGR-065 #49 blocked
|
||||
reconciled DGR-066 #50 blocked
|
||||
reconciled DGR-067 #51 blocked
|
||||
reconciled DGR-068 #52 blocked
|
||||
reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=none dry_run=False
|
||||
reconciled DGR-017 #1 completed
|
||||
reconciled DGR-018 #2 completed
|
||||
reconciled DGR-019 #3 completed
|
||||
reconciled DGR-020 #4 completed
|
||||
reconciled DGR-021 #5 completed
|
||||
reconciled DGR-022 #6 completed
|
||||
reconciled DGR-023 #7 completed
|
||||
reconciled DGR-024 #8 completed
|
||||
reconciled DGR-025 #9 completed
|
||||
reconciled DGR-026 #10 completed
|
||||
reconciled DGR-027 #11 completed
|
||||
reconciled DGR-028 #12 completed
|
||||
reconciled DGR-029 #13 completed
|
||||
reconciled DGR-030 #14 in-progress
|
||||
reconciled DGR-031 #15 ready
|
||||
reconciled DGR-032 #16 blocked
|
||||
reconciled DGR-033 #17 blocked
|
||||
reconciled DGR-034 #18 blocked
|
||||
reconciled DGR-035 #19 blocked
|
||||
reconciled DGR-036 #20 blocked
|
||||
reconciled DGR-037 #21 blocked
|
||||
reconciled DGR-038 #22 blocked
|
||||
reconciled DGR-039 #23 blocked
|
||||
reconciled DGR-040 #24 blocked
|
||||
reconciled DGR-041 #25 blocked
|
||||
reconciled DGR-042 #26 blocked
|
||||
reconciled DGR-043 #27 blocked
|
||||
reconciled DGR-044 #28 ready
|
||||
reconciled DGR-045 #29 blocked
|
||||
reconciled DGR-046 #30 blocked
|
||||
reconciled DGR-047 #31 blocked
|
||||
reconciled DGR-048 #32 blocked
|
||||
reconciled DGR-049 #33 blocked
|
||||
reconciled DGR-050 #34 blocked
|
||||
reconciled DGR-051 #35 blocked
|
||||
reconciled DGR-052 #36 blocked
|
||||
reconciled DGR-053 #37 blocked
|
||||
reconciled DGR-054 #38 blocked
|
||||
reconciled DGR-055 #39 blocked
|
||||
reconciled DGR-056 #40 blocked
|
||||
reconciled DGR-057 #41 blocked
|
||||
reconciled DGR-058 #42 blocked
|
||||
reconciled DGR-059 #43 blocked
|
||||
reconciled DGR-060 #44 blocked
|
||||
reconciled DGR-061 #45 blocked
|
||||
reconciled DGR-062 #46 blocked
|
||||
reconciled DGR-063 #47 blocked
|
||||
reconciled DGR-064 #48 blocked
|
||||
reconciled DGR-065 #49 blocked
|
||||
reconciled DGR-066 #50 blocked
|
||||
reconciled DGR-067 #51 blocked
|
||||
reconciled DGR-068 #52 blocked
|
||||
reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=DGR-030 dry_run=False
|
||||
Initializing Ralph TUI...
|
||||
Env filter: no vars matched exclusion patterns (*_API_KEY, *_SECRET_KEY, *_SECRET)
|
||||
|
||||
Session: b92af6fa-2fb6-4e92-9bf2-d571be5c8561
|
||||
Agent: claude
|
||||
Tracker: json
|
||||
PRD: .scratch/distributed-gguf-runtime/prd.json
|
||||
Max iterations: unlimited
|
||||
|
||||
[10:34:45] [INFO] [session] Session b92af6fa-2fb6-4e92-9bf2-d571be5c8561 created. Agent: claude, Tracker: json
|
||||
[10:34:45] [INFO] [engine] Ralph started. Total tasks: 42
|
||||
[10:34:45] [INFO] [progress] Iteration 1/∞: Working on DGR-030 - Add accelerator build presets and native CI matrix
|
||||
[10:34:51] [INFO] [agent] I'll start by reading the context files for this task.
|
||||
[10:34:52] [INFO] [agent] [94m[Bash][0m $ echo "--- RALPH-CONTEXT ---" && cat .scratch/distributed-gguf-runtime/RALPH-CONTEXT.md 2>/dev/null |...
|
||||
[10:34:53] [INFO] [agent] [94m[Bash][0m $ echo "--- ISSUE ---" && cat .scratch/distributed-gguf-runtime/issues/030-add-accelerator-build-prese...
|
||||
[10:34:54] [INFO] [agent] [94m[Bash][0m $ echo "--- PROGRESS ---" && cat .ralph-tui/progress.md 2>/dev/null | tail -300
|
||||
[10:34:58] [INFO] [agent] [94m[Bash][0m $ echo "--- DGR-029 evidence ---" && cat .scratch/distributed-gguf-runtime/evidence/DGR-029/README.md ...
|
||||
[10:34:59] [INFO] [agent] [94m[Bash][0m $ echo "--- find native/cmake dirs ---" && find . -iname "*CMakeLists*" -not -path "*/node_modules/*" ...
|
||||
[10:35:14] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/native/llama/UPSTREAM_LOCK.json[0m
|
||||
[10:35:14] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m
|
||||
[10:35:21] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/native/CMakeLists.txt[0m
|
||||
[10:35:23] [INFO] [agent] [94m[Bash][0m $ find /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime -maxdept...
|
||||
[10:35:30] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime find . -ma...
|
||||
[10:35:30] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_llama_cpp_dependency.py[0m
|
||||
[10:35:44] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -rn "...
|
||||
[10:35:44] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -rln ...
|
||||
[10:35:48] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -n "r...
|
||||
[10:35:49] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/tracker/meshnet_tracker/capability.py[0m
|
||||
[10:36:21] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -n -B...
|
||||
[10:36:21] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -n -B...
|
||||
[10:37:10] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime ls build/l...
|
||||
[10:37:25] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime ls .venv/b...
|
||||
[10:37:29] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime timeout 8 ...
|
||||
[10:37:36] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime .venv/bin/...
|
||||
[10:37:41] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime timeout 90...
|
||||
[10:37:47] [INFO] [agent] Good, cmake is available now in `.venv/bin`. Now let's check for GPU SDK availability more thoroughly and check the c++ compiler.
|
||||
[10:37:48] [INFO] [agent] [94m[Bash][0m $ c++ --version 2>&1 | head -2 echo "---" which nvidia-smi rocminfo 2>&1 echo "--- CUDA_HOME / ROCM re...
|
||||
[10:37:55] [INFO] [agent] [94m[Bash][0m $ which glslc glslangValidator 2>&1 echo "---" find / -maxdepth 6 -iname "vulkan.h" 2>/dev/null | grep...
|
||||
[10:39:23] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime python3 -c...
|
||||
[10:39:27] [INFO] [agent] No separate shared-gates block; they're embedded per-story. Let's check the DGR-025 evidence README for the "isolated out-of-tree" and recipe-identity conventions, briefly, then start implementing.
|
||||
[10:39:28] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime cat packag...
|
||||
[10:39:31] [INFO] [agent] Now let's implement. First, update `UPSTREAM_LOCK.json` to add the accelerator presets.
|
||||
[10:39:35] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/native/llama/UPSTREAM_LOCK.json[0m edit: " "smoke_binary": "bin/llama-gguf-hash",
|
||||
[10:39:35] [INFO] [agent] "sm..." → " "smoke_binary": "bin/llama-gguf-hash",
|
||||
[10:39:35] [INFO] [agent] "sm..."
|
||||
[10:39:39] [INFO] [agent] Now let's update `scripts/llama_cpp_dependency.py` with the accelerator lane primitives.
|
||||
[10:39:47] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " if lock["expected_source"] != {"git_tree": loc..." → " if lock["expected_source"] != {"git_tree": loc..."
|
||||
[10:39:51] [INFO] [agent] Now let's add the accelerator lane functions after `ctest_lane`:
|
||||
[10:40:07] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: "def ctest_lane(build_dir: pathlib.Path) -> None:
|
||||
[10:40:07] [INFO] [agent] ..." → "def ctest_lane(build_dir: pathlib.Path) -> None:
|
||||
[10:40:07] [INFO] [agent] ..."
|
||||
[10:40:10] [INFO] [agent] Now add the CLI subcommands:
|
||||
[10:40:13] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " ctest_parser = subcommands.add_parser("ctest")..." → " ctest_parser = subcommands.add_parser("ctest")..."
|
||||
[10:40:22] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " elif args.command == "ctest":
|
||||
[10:40:22] [INFO] [agent] ..." → " elif args.command == "ctest":
|
||||
[10:40:22] [INFO] [agent] ..."
|
||||
[10:40:29] [INFO] [agent] I need to check that this doesn't break the existing test whose fixture lock has `"build": {}`. Let me guard against that.
|
||||
[10:40:30] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m
|
||||
[10:40:36] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " presets = lock.get("accelerator_presets", {})
|
||||
[10:40:36] [INFO] [agent] ..." → " presets = lock.get("accelerator_presets", {})
|
||||
[10:40:36] [INFO] [agent] ..."
|
||||
[10:40:40] [INFO] [agent] Now let's write the CI matrix script.
|
||||
[10:40:53] [INFO] [agent] [94m[Write][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/native_accelerator_matrix.py[0m "#!/usr/bin/env python3
|
||||
[10:40:53] [INFO] [agent] """DGR-030: native CI/build matrix over the CPU default plus accelerator lanes.
|
||||
[10:40:53] [INFO] [agent] Runs the exact deterministic CPU lane DGR-029 locked (unchanged), then probes
|
||||
[10:40:53] [INFO] [agent] each accelerator p... (4553 chars)"
|
||||
[10:40:59] [INFO] [agent] Now let's verify the Python compiles and existing tests still pass.
|
||||
[10:41:00] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime python3 -m...
|
||||
[10:41:03] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:41:10] [INFO] [agent] Existing tests pass. Now let's quickly test the CLI additions and the SDK-probe logic live.
|
||||
[10:41:11] [INFO] [agent] [94m[Bash][0m $ do echo "--- $lane ---" python3 scripts/llama_cpp_dependency.py accelerator-status --name "$lane...
|
||||
[10:41:14] [INFO] [agent] Now let's run the full matrix live end-to-end to verify the CPU lane still builds correctly and accelerator lanes report skipped, then verify the reverse-cleanup contract.
|
||||
[10:41:16] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:43:19] [INFO] [agent] It's compiling in the background (matches DGR-029's ~2m16s wall-clock for the CPU lane). I'll wait for it.
|
||||
[10:43:20] [INFO] [agent] [94m[Bash][0m $ sleep 90 && cat /tmp/claude-1000/-run-media-popov-d-DEV-repos-d-popov-com-AI--claude-worktrees-distr...
|
||||
[10:43:20] [INFO] [agent] [91m[Error: <tool_use_error>Blocked: sleep 90 followed by: cat /tmp/claude-1000/-run-media-popov-d-DEV-repos-d-popov-com-AI--claude-worktrees-distributed-gguf-runtime/972c183b-ce77-43c0-9fb1-d513602df13b/tasks/b1g4st9uy.output tail -100. To wait for a condition, use Monitor with an until-loop (e.g. `until <check>; do sleep 2; done`). To wait for a command you started, use run_in_background: true. Do not chain shorter sleeps to work around this block.</tool_use_error>][0m
|
||||
[10:43:25] [INFO] [agent] The build moved to background automatically; I'll get notified when it finishes. Meanwhile, let's write tests for the new accelerator functionality.
|
||||
[10:43:26] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_llama_cpp_dependency.py[0m
|
||||
[10:44:52] [INFO] [agent] [94m[Bash][0m $ tail -50 /tmp/claude-1000/-run-media-popov-d-DEV-repos-d-popov-com-AI--claude-worktrees-distributed-...
|
||||
[10:44:57] [INFO] [agent] The live matrix run succeeded: CPU lane built and CTest-passed exactly as before, and all four accelerator lanes correctly reported `skipped` with an explicit reason (no false success). Let's confirm the source checkout was properly reversed to pristine and the build dirs exist as expected.
|
||||
[10:44:58] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime git -C bui...
|
||||
[10:45:06] [INFO] [agent] Good — the checkout is pristine (matches the locked pin/tree), and only the CPU `build` dir was created (accelerator lanes never created their build dirs since they were skipped, as designed). Now let's write the unit tests.
|
||||
[10:45:26] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_llama_cpp_dependency.py[0m edit: " dependency._load_lock = lambda: _lock_with_reg..." → " dependency._load_lock = lambda: _lock_with_reg..."
|
||||
[10:45:30] [INFO] [agent] Now let's create the matrix orchestration test file.
|
||||
[10:45:51] [INFO] [agent] [94m[Write][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_native_accelerator_matrix.py[0m """"Offline behavior tests for DGR-030's native CI/build matrix orchestration.
|
||||
[10:45:51] [INFO] [agent] These tests never fetch or compile llama.cpp: `llama_cpp_dependency`'s fetch/
|
||||
[10:45:51] [INFO] [agent] apply/reverse/build/smoke/ctest_lane/accele... (7187 chars)"
|
||||
[10:46:02] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_native_accelerator_matrix.py[0m edit: " assert report["failed_lanes"] == []
|
||||
[10:46:02] [INFO] [agent] assert..." → " assert report["failed_lanes"] == []
|
||||
[10:46:02] [INFO] [agent] assert..."
|
||||
[10:46:17] [INFO] [agent] [94m[Bash][0m $ grep -n "workspace / \"build\"" tests/test_native_accelerator_matrix.py
|
||||
[10:46:21] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_native_accelerator_matrix.py[0m edit: " assert smoke_calls == [("smoke", (workspace / ..." → " assert smoke_calls == [("smoke", (workspace.re..."
|
||||
[10:46:28] [INFO] [agent] Now let's run the full test suite.
|
||||
[10:46:29] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:46:42] [INFO] [agent] All 19 tests pass. Now let's run the broader shared quality gates.
|
||||
[10:46:43] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:46:55] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:47:00] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime git stash ...
|
||||
|
||||
2
.ralph-tui/config.toml
Normal file
2
.ralph-tui/config.toml
Normal file
@@ -0,0 +1,2 @@
|
||||
autoCommit = true
|
||||
configVersion = "2.1"
|
||||
275
.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md
Normal file
275
.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md
Normal file
@@ -0,0 +1,275 @@
|
||||
# DGR-030 evidence — accelerator build presets and native CI/build matrix
|
||||
|
||||
**Status:** implementation complete, live-verified in this session (2026-07-23).
|
||||
**Authority:** local `prd.json` is authoritative; Gitea is a projection.
|
||||
**Upstream pin:** `e920c523e3b8a0163fe498af5bf90df35ff51d25` (`llama.cpp`, unchanged from DGR-027..029).
|
||||
|
||||
## What existed before this session
|
||||
|
||||
DGR-029 locked exactly one build lane — the deterministic CPU-only lane — in
|
||||
`UPSTREAM_LOCK.json`'s `build` section, plus `scripts/llama_cpp_dependency.py`'s
|
||||
`build()`/`smoke()`/`ctest_lane()`/`reproduce()`. There was no accelerator
|
||||
preset, no SDK-availability probing, and no matrix runner: only the one CPU
|
||||
lane existed, and there was no mechanism that could ever advertise a GPU
|
||||
backend as compiled or capable.
|
||||
|
||||
## What changed in this session
|
||||
|
||||
- `packages/node/native/llama/UPSTREAM_LOCK.json`: added a new top-level
|
||||
`accelerator_presets` object with one entry each for `cuda` (`GGML_CUDA`),
|
||||
`rocm` (`GGML_HIP`), `vulkan` (`GGML_VULKAN`), and `metal` (`GGML_METAL`).
|
||||
Each entry names only the one backend flag it flips and an `sdk_probe`
|
||||
(a binary to resolve on `PATH`, an optional env-var override, and — for
|
||||
Metal — a `platform_only: "darwin"` gate). **The existing `build` section
|
||||
— the deterministic CPU default DGR-029 locked — is untouched.**
|
||||
- `scripts/llama_cpp_dependency.py`:
|
||||
- `_load_lock()` now calls a new `_verify_accelerator_presets()`, which
|
||||
fail-closed-rejects any preset whose named backend flag is not `OFF` in
|
||||
the CPU default's `configure_flags` — structurally guaranteeing a preset
|
||||
can only ever *add* one backend on top of the untouched CPU baseline,
|
||||
never redefine it.
|
||||
- `accelerator_configure_flags(lock, name)` returns a **new** flag list —
|
||||
the CPU default's own `configure_flags` list is never mutated — with
|
||||
exactly the named preset's backend flag flipped `ON` and every other flag
|
||||
(including `GGML_CPU=ON`, the fallback ops backend GPU builds still need)
|
||||
left exactly as the CPU default declares it.
|
||||
- `_sdk_probe(probe)` / `accelerator_status(name, lock)` resolve a lane's
|
||||
SDK without ever raising: an absent SDK is returned as
|
||||
`{"available": false, "reason": "<binary> is unavailable on PATH"}` (or
|
||||
a platform-mismatch reason for Metal), so "unavailable" is data a caller
|
||||
reports, never an exception a caller has to remember to catch.
|
||||
- `accelerator_build(source, name, build_dir)` compiles one lane into its
|
||||
own out-of-tree `build_dir` (an isolated directory, never DGR-029's CPU
|
||||
`build_dir`), using the same patched-source verification and
|
||||
`native_targets` as the CPU lane, then writes a
|
||||
`meshnet-build-metadata.json` recording the exact `commit`/`commit_tree`,
|
||||
per-patch SHA-256 digests, the lane's overridden `configure_flags`, the
|
||||
resolved `cmake`/`cxx`/SDK-binary versions/paths, and explicit
|
||||
`model_downloads: false`, `hardware_execution: false`,
|
||||
`hardware_certified: false`, `semantic_certification: false` fields plus
|
||||
a `note` stating the lane is registered-dark until a real-hardware
|
||||
certification record exists. It **never** calls `smoke()`/`ctest_lane()`
|
||||
— running a binary linked against a real accelerator backend would touch
|
||||
real hardware, which this story deliberately keeps out of scope.
|
||||
- Added `accelerator-status --name <lane>` and
|
||||
`accelerator-build --name <lane> --source-dir --build-dir` CLI
|
||||
subcommands, mirroring the existing `ctest`/`build` subcommand pattern.
|
||||
- `scripts/native_accelerator_matrix.py` (new): the native CI/build matrix.
|
||||
`run_matrix(workspace)` fetches and applies the locked pin/patch stack once,
|
||||
runs the unchanged CPU lane (build → smoke → ctest, exactly DGR-029's
|
||||
contract), then for each `accelerator_presets` entry either reports
|
||||
`{"status": "skipped", "reason": ...}` (SDK absent) or compiles it via
|
||||
`accelerator_build` and reports `{"status": "built", ...}` — never silently
|
||||
treating a skip as a pass. Any `DependencyError` from a lane (CPU or
|
||||
accelerator) is caught per-lane and reported as `{"status": "failed", ...}`
|
||||
without aborting the remaining lanes or skipping cleanup. `reverse()` always
|
||||
runs in a `finally`, restoring the exact pristine pin/tree regardless of
|
||||
lane outcomes. The CLI prints a JSON report and exits non-zero only if any
|
||||
lane actually `failed` (a `skipped` lane never fails the run).
|
||||
- `tests/test_llama_cpp_dependency.py`: added 7 new tests —
|
||||
`test_accelerator_presets_isolate_one_backend_without_touching_the_cpu_default`
|
||||
(every preset flips exactly its own flag and the CPU default list is never
|
||||
mutated), `test_accelerator_configure_flags_rejects_an_unknown_lane`,
|
||||
`test_accelerator_status_reports_unavailable_sdks_without_raising` (asserts
|
||||
the exact reason string for cuda/rocm/vulkan/metal absence),
|
||||
`test_accelerator_status_honors_an_explicit_sdk_override`,
|
||||
`test_accelerator_status_rejects_an_unknown_lane`,
|
||||
`test_accelerator_build_refuses_to_compile_an_unavailable_lane` (asserts no
|
||||
build directory is created), and a `requires_cmake`-gated
|
||||
`test_accelerator_build_compiles_the_available_lane_with_isolated_evidence`,
|
||||
which builds a tiny synthetic CMake project (not the full llama.cpp tree) to
|
||||
prove `accelerator_build`'s "SDK present" path really configures with the
|
||||
overridden flag, compiles, and writes the registered-dark metadata — in
|
||||
about a second, without a real GPU SDK.
|
||||
- `tests/test_native_accelerator_matrix.py` (new): 3 offline tests exercising
|
||||
`run_matrix`'s orchestration with `llama_cpp_dependency`'s
|
||||
fetch/apply/reverse/build/smoke/ctest_lane/accelerator_status/
|
||||
accelerator_build stubbed out — proving unavailable SDKs are reported
|
||||
`skipped` (never a false pass), an available accelerator lane is compiled
|
||||
without ever calling `smoke`/`ctest_lane`, and a lane failure is reported
|
||||
per-lane without aborting sibling lanes or skipping the `reverse()` cleanup.
|
||||
|
||||
## Toolchain note
|
||||
|
||||
As in DGR-029, neither the ambient system Python nor `.venv-rocm` has `cmake`;
|
||||
this session's `.venv` also had no `cmake` (a prior session's install did not
|
||||
persist). This session ran `.venv/bin/python3 -m ensurepip --upgrade` (no
|
||||
`pip` was present in `.venv` either) and then
|
||||
`.venv/bin/python3 -m pip install cmake`, landing the same PyPI wheel
|
||||
(`cmake==4.4.0`) DGR-029 used, at `.venv/bin/cmake` / `.venv/bin/ctest`. All
|
||||
commands below were run with that `.venv/bin` prepended to `PATH`. No CUDA,
|
||||
ROCm, or Vulkan SDK (`nvcc`, `hipcc`, `glslc`) is installed in this
|
||||
environment, and the host platform is Linux, not `darwin` — so all four
|
||||
accelerator lanes are genuinely `skipped` in this environment's own live run
|
||||
below, which is real evidence for AC2 ("unavailable SDKs ... explicit
|
||||
unavailable/skipped lanes"), not a simulated one.
|
||||
|
||||
## Verification — live native CI/build matrix run
|
||||
|
||||
```text
|
||||
$ rm -rf build/llama.cpp/build build/llama.cpp/build-cuda build/llama.cpp/build-rocm build/llama.cpp/build-vulkan build/llama.cpp/build-metal
|
||||
$ python3 scripts/native_accelerator_matrix.py
|
||||
reused verified offline cache: .../build/llama.cpp/source
|
||||
usage: .../build/llama.cpp/build/bin/llama-gguf-hash [options] GGUF_IN
|
||||
...
|
||||
Test project .../build/llama.cpp/build
|
||||
Start 27: test-meshnet-range-ownership
|
||||
1/1 Test #27: test-meshnet-range-ownership ..... Passed 0.01 sec
|
||||
100% tests passed out of 1
|
||||
{
|
||||
"failed_lanes": [],
|
||||
"hardware_certified": false,
|
||||
"lanes": [
|
||||
{
|
||||
"build_dir": ".../build/llama.cpp/build",
|
||||
"lane": "cpu",
|
||||
"metadata": {
|
||||
"cmake": "cmake version 4.4.0",
|
||||
"commit": "e920c523e3b8a0163fe498af5bf90df35ff51d25",
|
||||
"commit_tree": "6c91a11407a3a3fb160f5dac705f9c59718f54f1",
|
||||
"configure_flags": [
|
||||
"-DCMAKE_BUILD_TYPE=Release", "-DLLAMA_BUILD_TESTS=ON",
|
||||
"-DLLAMA_BUILD_EXAMPLES=ON", "-DLLAMA_BUILD_SERVER=OFF",
|
||||
"-DLLAMA_BUILD_TOOLS=OFF", "-DLLAMA_BUILD_APP=OFF", "-DLLAMA_CURL=OFF",
|
||||
"-DGGML_CPU=ON", "-DGGML_BLAS=OFF", "-DGGML_CUDA=OFF",
|
||||
"-DGGML_HIP=OFF", "-DGGML_VULKAN=OFF", "-DGGML_METAL=OFF"
|
||||
],
|
||||
"cxx": "c++ (GCC) 15.2.1 20260123 (Red Hat 15.2.1-7)",
|
||||
"model_downloads": false,
|
||||
"patches": { "...": "... (5 entries, unchanged sha256 digests from DGR-029)" },
|
||||
"semantic_certification": false
|
||||
},
|
||||
"status": "built"
|
||||
},
|
||||
{"lane": "cuda", "reason": "nvcc is unavailable on PATH", "status": "skipped"},
|
||||
{"lane": "rocm", "reason": "hipcc is unavailable on PATH", "status": "skipped"},
|
||||
{"lane": "vulkan", "reason": "glslc is unavailable on PATH", "status": "skipped"},
|
||||
{"lane": "metal", "reason": "platform 'linux' is not 'darwin'", "status": "skipped"}
|
||||
],
|
||||
"note": "A `built` lane means it compiled with the exact recorded compiler/SDK/upstream-pin/patch-stack/build-option evidence — it never means an accelerator device was exercised. Every backend/model/recipe lane stays registered-dark until a separate real-hardware certification record exists."
|
||||
}
|
||||
$ echo $?
|
||||
0
|
||||
```
|
||||
|
||||
Wall-clock: `real 2m19.797s` — matches DGR-029's ~2m16s CPU-lane compile; no
|
||||
accelerator lane actually compiled in this environment (all four SDKs are
|
||||
genuinely absent), so this run's added cost over DGR-029's own CPU-only
|
||||
`reproduce()` is just the four fast SDK probes.
|
||||
|
||||
Post-run checks (source checkout left pristine by the matrix's `reverse()`):
|
||||
|
||||
```text
|
||||
$ git -C build/llama.cpp/source status --short --branch --untracked-files=all
|
||||
## HEAD (no branch)
|
||||
$ git -C build/llama.cpp/source rev-parse HEAD HEAD^{tree}
|
||||
e920c523e3b8a0163fe498af5bf90df35ff51d25
|
||||
6c91a11407a3a3fb160f5dac705f9c59718f54f1
|
||||
$ ls build/llama.cpp/ | grep build
|
||||
build
|
||||
```
|
||||
|
||||
Only the CPU lane's `build/` directory was created — no `build-cuda`,
|
||||
`build-rocm`, `build-vulkan`, or `build-metal` directory exists, because every
|
||||
accelerator lane was genuinely skipped rather than attempted.
|
||||
|
||||
## Verification — targeted test suites and shared gates
|
||||
|
||||
| Command | Result |
|
||||
| --- | --- |
|
||||
| `python3 -m pytest -q tests/test_llama_cpp_dependency.py tests/test_native_accelerator_matrix.py` | `19 passed` (9 pre-existing + 7 new accelerator-lane tests in `test_llama_cpp_dependency.py`, 3 new in `test_native_accelerator_matrix.py`; the `requires_cmake`-gated compile test ran for real, not skipped) |
|
||||
| `python3 -m compileall -q packages tests` | exit 0 |
|
||||
| `git diff --check -- packages/node/native/llama/UPSTREAM_LOCK.json scripts/llama_cpp_dependency.py tests/test_llama_cpp_dependency.py scripts/native_accelerator_matrix.py tests/test_native_accelerator_matrix.py` | exit 0 |
|
||||
| `python3 scripts/ralph_prd_schema.py validate .scratch/distributed-gguf-runtime/prd.json` | `OK: 55 stories validated.` |
|
||||
|
||||
`git diff --check` against the full working tree separately reports one
|
||||
pre-existing trailing-whitespace line in `.ralph-tui-run.log`, which was
|
||||
already modified before this session started (see the session's initial
|
||||
`git status`) and is unrelated to this story's scope; it is excluded above by
|
||||
naming this story's own changed files explicitly.
|
||||
|
||||
`python3 -m pytest -q tests/test_ralph_prd_schema.py` reports `55 failed, 53
|
||||
passed` in this session (all `test_render_issue_markdown_matches_committed_file`
|
||||
drift between `prd.json` and committed issue Markdown for other stories,
|
||||
e.g. `DGR-053`..`DGR-071`). `git stash`-ing this session's changes and rerunning
|
||||
reproduces `56 failed, 52 passed` identically — the same 56 failures minus the
|
||||
one this session's own `DGR-030` regeneration fixed, confirming the remaining
|
||||
55 predate this story and are out of scope to fix here. This session did
|
||||
regenerate `.scratch/distributed-gguf-runtime/issues/030-add-accelerator-
|
||||
build-presets-and-native-ci-matrix.md` via
|
||||
`python3 scripts/ralph_prd_schema.py render ... DGR-030` so DGR-030's own
|
||||
generated issue Markdown matches `prd.json` byte-for-byte (confirmed by the
|
||||
`test_render_issue_markdown_matches_committed_file[DGR-030]` case no longer
|
||||
appearing in the failure list).
|
||||
|
||||
## Ensuring build success does not advertise capability
|
||||
|
||||
- Every accelerator lane's `meshnet-build-metadata.json` explicitly records
|
||||
`hardware_execution: false`, `hardware_certified: false`, and
|
||||
`semantic_certification: false`, plus a `note` stating the lane is
|
||||
registered-dark until a separate real-hardware certification record exists
|
||||
— the same "artifact states this, not just prose" pattern DGR-029 used for
|
||||
the CPU lane's `model_downloads`/`semantic_certification` fields.
|
||||
- `accelerator_build` never runs `smoke()` or `ctest_lane()`: it only
|
||||
configures and compiles the exact `native_targets` DGR-029 already locked
|
||||
(`llama-gguf-hash`, `test-meshnet-range-ownership`) — no binary linked
|
||||
against a real accelerator backend is ever executed by this story's code.
|
||||
- `_verify_accelerator_presets()` structurally refuses any preset whose
|
||||
backend flag is not `OFF` in the locked CPU default, so a preset can never
|
||||
be defined in a way that redefines (rather than adds one backend on top of)
|
||||
DGR-029's deterministic CPU lane.
|
||||
- The matrix's top-level report always carries `"hardware_certified": false`
|
||||
regardless of how many lanes built, and its `note` field states this
|
||||
explicitly for any consumer reading only the report, not the per-lane
|
||||
metadata.
|
||||
|
||||
## Limitations
|
||||
|
||||
- This story proves accelerator lanes *compile* with correct, isolated
|
||||
flags and preserves exact evidence when a lane's SDK is present. It proves
|
||||
nothing about numerical correctness, performance, or any backend/model/
|
||||
recipe capability on real accelerator hardware — that is explicitly
|
||||
deferred to DGR-041 (capability registration), DGR-053 (real 2-4 stage
|
||||
certification), and DGR-067 (capability matrix certification), all of which
|
||||
remain unimplemented.
|
||||
- No CUDA, ROCm, or Vulkan SDK, and no macOS/Metal toolchain, is available in
|
||||
this session's environment, so the "compile an available accelerator lane"
|
||||
path is proven end-to-end only via the `requires_cmake`-gated synthetic-
|
||||
project unit test and the offline matrix-orchestration tests, not via a
|
||||
live compile of the real llama.cpp tree under `GGML_CUDA=ON` (etc.). A
|
||||
future session with a real SDK installed will exercise
|
||||
`accelerator_build`'s real-lane path against the genuine llama.cpp source
|
||||
for the first time; nothing in this story's design assumes that hasn't
|
||||
happened yet.
|
||||
- The accelerator lanes reuse the CPU lane's exact `native_targets`
|
||||
(`llama-gguf-hash`, `test-meshnet-range-ownership`), so a passing
|
||||
accelerator compile also proves the DGR-027/DGR-028 patch stack's
|
||||
range-ownership code compiles under that backend flag combination — but,
|
||||
per the point above, only structurally; it says nothing about GPU
|
||||
execution correctness.
|
||||
- `cmake`/`ctest` remain absent system-wide in this environment; this session
|
||||
reinstalled them into `.venv` exactly as DGR-029 did, and that install does
|
||||
not appear to persist across sessions (this session found `.venv` without
|
||||
`cmake` despite DGR-029's evidence recording its earlier install). A future
|
||||
session without a `cmake`-equipped `.venv` will see the same actionable
|
||||
"cmake is unavailable" failure DGR-029 demonstrated, not a silent pass, and
|
||||
the new `requires_cmake`-gated tests will be skipped rather than failing.
|
||||
- `git diff --check` and `tests/test_ralph_prd_schema.py` both carry
|
||||
pre-existing, out-of-scope failures unrelated to this story (see the gates
|
||||
table above); this story's own changed files pass both checks cleanly.
|
||||
|
||||
## Dependency handoff
|
||||
|
||||
DGR-053 (real 2-4 stage certification), DGR-067 (capability matrix
|
||||
certification), and DGR-068 (packaged releases) may rely on: four isolated,
|
||||
out-of-tree accelerator build presets (`cuda`/`rocm`/`vulkan`/`metal`) in
|
||||
`UPSTREAM_LOCK.json`'s `accelerator_presets`, each toggling exactly one
|
||||
backend flag on top of DGR-029's unchanged CPU default; a native CI/build
|
||||
matrix (`scripts/native_accelerator_matrix.py`) that compiles every
|
||||
SDK-available lane with full compiler/SDK/upstream-pin/patch-stack/build-
|
||||
option evidence and reports SDK-unavailable lanes as explicit `skipped`
|
||||
lanes, never a false pass; and a compile-only contract (no lane here ever
|
||||
runs a binary against real accelerator hardware). Real-hardware execution,
|
||||
numerical correctness, performance measurement, and backend/model/recipe
|
||||
certification for any accelerator remain entirely unimplemented and must not
|
||||
be assumed from any lane's green compile.
|
||||
237
.scratch/distributed-gguf-runtime/evidence/DGR-031/README.md
Normal file
237
.scratch/distributed-gguf-runtime/evidence/DGR-031/README.md
Normal file
@@ -0,0 +1,237 @@
|
||||
# DGR-031 evidence — the project-owned `ShardEngine` interface
|
||||
|
||||
**Completed:** 2026-07-23
|
||||
**Branch:** `ralph/distributed-gguf-runtime`
|
||||
**Authority:** `.scratch/distributed-gguf-runtime/prd.json`
|
||||
**Dependencies:** DGR-021 (`evidence/DGR-021/README.md` — versioned activation
|
||||
envelope, `NamedTensor`/`ActivationEnvelope` as the project-owned wire-envelope
|
||||
layer), DGR-025 (`evidence/DGR-025/README.md` — exact artifact/runtime recipe
|
||||
identity; both read before changing code).
|
||||
|
||||
## Objective
|
||||
|
||||
Isolate worker/protocol code from llama.cpp internals behind a stable
|
||||
project-owned engine contract, so a fake fixture engine (DGR-032) and a real
|
||||
llama.cpp-backed engine (DGR-037) are interchangeable subclasses of one
|
||||
interface.
|
||||
|
||||
## What was found live before changing code
|
||||
|
||||
Per RALPH-CONTEXT, legacy pass states were not trusted; the live surrounding
|
||||
contracts were read and exercised before designing this one:
|
||||
|
||||
- `packages/node/meshnet_node/shard_lifecycle.py` (DGR-022) already defines a
|
||||
versioned RPC/session lifecycle contract — `StructuredStatus`, `StatusCode`,
|
||||
`CacheExpectation`, `CacheResult`, `LifecycleState`, `SessionLifecycle` — but
|
||||
it is explicitly the *wire RPC* contract "consumed by a future generated
|
||||
gRPC binding," not an execution-engine boundary.
|
||||
- `packages/node/meshnet_node/native_backend.py` (DGR-025) is the identity
|
||||
boundary for the native GGUF artifact — it derives and attests a
|
||||
`ShardIdentity`, but does not define an execution contract either.
|
||||
- `packages/node/meshnet_node/protocol.py` (DGR-021) defines a project-owned
|
||||
`NamedTensor`/`ActivationEnvelope` for activation traffic *between shard
|
||||
hops over the network*, distinct from the generated-protobuf wire ABI in
|
||||
`native_protocol`.
|
||||
- `packages/node/meshnet_node/shard_runtime_server.py` (DGR-024) is today a
|
||||
real gRPC servicer that proves wire fidelity by checksumming and echoing
|
||||
bytes — it has no execution engine behind it yet; that seam is exactly
|
||||
where `ShardEngine` plugs in for DGR-037.
|
||||
- `packages/node/meshnet_node/architecture_boundary.py` established the
|
||||
precedent this story follows for tail output: `TailOutput.sampled_token()`
|
||||
never exposes raw logits, only a sampled token id.
|
||||
- No `ShardEngine` (or `shard_engine`) symbol existed anywhere in the
|
||||
repository prior to this story (confirmed by
|
||||
`grep -rn -i "shardengine\|shard_engine"` across `.py`/`.md`, which returned
|
||||
only planning-document prose naming it as future work).
|
||||
|
||||
Live verification of the pre-existing dependency contracts before adding new
|
||||
code: `PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q
|
||||
tests/test_shard_lifecycle.py tests/test_activation_envelope.py
|
||||
tests/test_architecture_boundary.py tests/test_native_shard_protocol.py
|
||||
tests/test_shard_runtime_harness.py` → `95 passed, 3 skipped`.
|
||||
|
||||
## What was added (this story's change)
|
||||
|
||||
### `packages/node/meshnet_node/shard_engine.py` (new)
|
||||
|
||||
The `ShardEngine` boundary: an `abc.ABC` with eight abstract operations —
|
||||
`load`, `capabilities`, `prefill`, `decode`, `cancel`, `release`, `health`,
|
||||
`metrics` — matching the acceptance criterion's list exactly (`prefill`/
|
||||
`decode` share one operation family; their shared result type is what the
|
||||
criterion calls the "boundary/logits result"). Every request/result type is a
|
||||
frozen dataclass built from plain `str`/`int`/`bytes`/`Mapping` values:
|
||||
|
||||
- `EngineTensor` / `BoundaryBundle` — the project-owned named-tensor
|
||||
activation crossing a shard boundary (head/middle/tail-in). Deliberately a
|
||||
*new*, minimal type distinct from both `native_protocol.pb.TensorBundle`
|
||||
(generated-protobuf ABI) and `protocol.NamedTensor`/`ActivationEnvelope`
|
||||
(wire-framing/fragmentation concerns irrelevant to model execution) — a
|
||||
fourth, execution-facing layer underneath the three that already existed.
|
||||
- `TokenOutput` — a tail shard's sampled result: a token id (+ optional
|
||||
decoded text), never a raw logits tensor.
|
||||
- `MtpHook` — reserved multi-token-prediction hook; its own `__post_init__`
|
||||
raises if constructed with `enabled=True`, so the type exists (fixing its
|
||||
field shape for DGR-051/DGR-066) without any code path being able to turn it
|
||||
on before DGR-066, matching RALPH-CONTEXT's "MTP is reserved and off for
|
||||
alpha."
|
||||
- `ArchitectureAuxStateHook` — reserved per-shard architecture auxiliary state
|
||||
(V4 CSA/HCA/SWA/indexer/compressor and similar); has no wire encoding and is
|
||||
never embedded in a `BoundaryBundle`, matching RALPH-CONTEXT's "remain local
|
||||
... never carried over the WAN seam."
|
||||
- `LoadRequest`/`LoadResult`, `EngineCapabilities`, `PrefillRequest`/
|
||||
`DecodeRequest` (exactly one of `token_ids`/`token_id` (head) or `input`
|
||||
(middle/tail) required — enforced in `__post_init__`), `StepResult` (a
|
||||
successful result must carry an output; `cache_result` reuses
|
||||
`shard_lifecycle.CacheResult`), `HealthResult`, `MetricsResult`.
|
||||
- Status vocabulary is reused, not reinvented: `StructuredStatus`/
|
||||
`StatusCode`/`CacheExpectation`/`CacheResult` are imported from
|
||||
`shard_lifecycle` (already project-owned and version-stable) rather than a
|
||||
parallel enum living alongside it.
|
||||
- The module imports nothing from `native_protocol`, `grpc`, or `ctypes` —
|
||||
verified structurally, not just by convention (see tests below).
|
||||
|
||||
### `tests/shard_engine_contract.py` (new)
|
||||
|
||||
A reusable, non-`test_`-prefixed helper: `assert_shard_engine_contract(make_engine)`
|
||||
takes a zero-arg engine factory and runs nine lifecycle checks — health before
|
||||
load, load→capabilities range/MTP-off, prefill→decode determinism (byte-identical
|
||||
output replayed on a fresh session), middle-shard boundary-bundle-in/out vs.
|
||||
head/tail token-output, deterministic cache-miss on an unopened session,
|
||||
stale-route-epoch rejection, cancel-then-decode rejection (+ cancel
|
||||
idempotency), release-then-decode rejection (+ release idempotency), and
|
||||
metrics reporting cancelled sessions. DGR-032's fixture and DGR-037's
|
||||
llama.cpp binding are both expected to import this and pass it against their
|
||||
own engine, proving identical lifecycle semantics without duplicating the
|
||||
checks.
|
||||
|
||||
### `tests/test_shard_engine.py` (new)
|
||||
|
||||
- `_ReferenceEngine`: a minimal in-memory `ShardEngine` used only to prove the
|
||||
shared contract is non-vacuous. It is explicitly *not* the DGR-032
|
||||
deterministic fixture (no delay/memory-pressure/malformed/crash injection —
|
||||
that is DGR-032's own, larger scope); the docstring says so to prevent this
|
||||
story's evidence from being read as inherited completion credit for DGR-032.
|
||||
- Dataclass validation tests: abstract-class instantiation refusal, tensor/
|
||||
bundle/token-output field validation, MTP-hook enable refusal, exactly-one-
|
||||
input-kind enforcement on `PrefillRequest`/`DecodeRequest`, `LoadRequest`
|
||||
shard-range-vs-total-layers validation, `StepResult` output-required-on-OK.
|
||||
- `test_shard_engine_module_imports_no_native_or_grpc_or_wire_abi_types`:
|
||||
walks `vars(shard_engine_module)` and asserts no bound name's `__name__` is
|
||||
`ctypes`, `grpc`, or `meshnet_node.native_protocol` — a structural check
|
||||
(not a docstring-text grep, which produced a false positive on first draft
|
||||
because the module's own docstring *names* `ggml_tensor` as an example of
|
||||
what must never appear) that the ABI-isolation acceptance criterion holds.
|
||||
|
||||
### `.scratch/distributed-gguf-runtime/prd.json` / issue markdown
|
||||
|
||||
Marked `DGR-031.passes = true` with `completionNotes`; regenerated
|
||||
`issues/031-introduce-the-project-owned-shardengine-interface.md` via
|
||||
`scripts/ralph_prd_schema.py render` so it matches `prd.json` byte-for-byte.
|
||||
|
||||
## Acceptance criteria → evidence
|
||||
|
||||
1. **load/capabilities/prefill/decode/boundary-logits-result/cancel/release/
|
||||
health/metrics** — `ShardEngine`'s eight abstract methods plus
|
||||
`StepResult.output: BoundaryBundle | TokenOutput | None`. Verified by
|
||||
`test_reference_engine_obeys_the_shared_shard_engine_contract` and the
|
||||
middle-shard-vs-tail-shard assertion inside
|
||||
`assert_shard_engine_contract`.
|
||||
2. **No `ggml_tensor`/llama context/scheduler/ABI-owned structure** — every
|
||||
type in `shard_engine.py` is a plain dataclass over `str`/`int`/`bytes`/
|
||||
`Mapping`; no import of `native_protocol`, `grpc`, or `ctypes`. Verified by
|
||||
`test_shard_engine_module_imports_no_native_or_grpc_or_wire_abi_types`.
|
||||
3. **Reserved typed MTP/architecture-aux-state hooks, not enabled** —
|
||||
`MtpHook.__post_init__` raises on `enabled=True`; `ArchitectureAuxStateHook`
|
||||
carries opaque shard-local state with no wire path. Verified by
|
||||
`test_mtp_hook_is_reserved_and_refuses_to_enable` and
|
||||
`test_architecture_aux_state_hook_carries_opaque_shard_local_state`, plus
|
||||
`assert_shard_engine_contract`'s `caps.supports_mtp is False` check.
|
||||
4. **Contract tests proving fake and future llama implementations obey
|
||||
identical lifecycle semantics** — `tests/shard_engine_contract.py` is
|
||||
written to be imported by DGR-032 and DGR-037 against their own engines;
|
||||
`test_shard_engine.py` proves it is real by running it against
|
||||
`_ReferenceEngine`.
|
||||
5. **Gates + this handoff** — below.
|
||||
|
||||
## Commands and results
|
||||
|
||||
```bash
|
||||
PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q tests/test_shard_engine.py
|
||||
```
|
||||
```text
|
||||
12 passed in 0.13s
|
||||
```
|
||||
|
||||
```bash
|
||||
PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q \
|
||||
tests/test_shard_engine.py tests/test_shard_lifecycle.py \
|
||||
tests/test_architecture_boundary.py tests/test_activation_envelope.py \
|
||||
tests/test_native_shard_protocol.py tests/test_shard_runtime_harness.py
|
||||
```
|
||||
```text
|
||||
95 passed, 3 skipped in 3.65s
|
||||
```
|
||||
|
||||
```bash
|
||||
.venv/bin/python3 -m compileall packages/node/meshnet_node/shard_engine.py tests/shard_engine_contract.py tests/test_shard_engine.py
|
||||
```
|
||||
```text
|
||||
Compiling 'packages/node/meshnet_node/shard_engine.py'...
|
||||
Compiling 'tests/shard_engine_contract.py'...
|
||||
Compiling 'tests/test_shard_engine.py'...
|
||||
```
|
||||
|
||||
```bash
|
||||
git diff --check
|
||||
```
|
||||
```text
|
||||
(no output — clean)
|
||||
```
|
||||
|
||||
## Limitations
|
||||
|
||||
- `tests/` as a whole does not collect cleanly in this environment: 27
|
||||
pre-existing test modules fail to import for missing optional dependencies
|
||||
(`cryptography`, etc.) unrelated to this story. Reproduced identically with
|
||||
`git stash` before this session's change (`27 errors during collection`),
|
||||
so this is pre-existing environment state, not a regression introduced
|
||||
here. This story's own gates were run as the targeted, scoped test set
|
||||
above per the shared quality gates' own wording ("Targeted deterministic
|
||||
tests pass").
|
||||
- The contract in `shard_engine_contract.py` proves *lifecycle* semantics
|
||||
(gating, cache-miss/stale-epoch/cancel/release, boundary-vs-token output
|
||||
shape) are identical across implementations. It does not — and cannot yet
|
||||
— prove numerical parity between a fake and a real engine; that is
|
||||
DGR-036's explicit job once DGR-032 and DGR-037 both exist.
|
||||
- `_ReferenceEngine` in `test_shard_engine.py` is intentionally minimal
|
||||
(no delay/memory-pressure/malformed-output/crash injection). DGR-032's
|
||||
acceptance criteria require those independently; nothing here should be
|
||||
read as satisfying them.
|
||||
- No gRPC/CMake/native-build changes were needed or made — this story is
|
||||
pure Python interface/type definition (`evidenceClass: model-free`,
|
||||
`hardware: none`), so the native CMake/CTest and patch-stack gates in the
|
||||
shared quality-gate list do not apply here (consistent with DGR-021/DGR-025,
|
||||
which record the same non-applicability for non-native stories).
|
||||
|
||||
## Dependency handoff
|
||||
|
||||
- **DGR-032** (fake `ShardEngine`): subclass `ShardEngine`, add delay/memory-
|
||||
pressure/malformed-output/crash injection, and pass the *same*
|
||||
`assert_shard_engine_contract` from `tests/shard_engine_contract.py`
|
||||
against it — no new contract vocabulary should be needed.
|
||||
- **DGR-034/DGR-035** (range-aware GGUF ownership, boundary I/O): `LoadRequest`
|
||||
already carries `shard_start`/`shard_end`/`total_layers`/`recipe`; `capabilities()`
|
||||
reports the authoritative range via `EngineCapabilities.is_head`/`is_tail`.
|
||||
`BoundaryBundle.token_id_sideband` is reserved for the first-three-hash-
|
||||
routed-layers V4 requirement RALPH-CONTEXT documents.
|
||||
- **DGR-037** (bind llama.cpp to the worker): implement `ShardEngine` as a
|
||||
thin wrapper around the native artifact from `native_backend.py`/
|
||||
`runtime_recipe.py`; `shard_runtime_server.py`'s `Session`/`GetCapability`/
|
||||
`Health`/`Cancel`/`Release` handlers become the translation layer between
|
||||
`pb.*` wire messages and this module's request/result types — this story
|
||||
intentionally does not touch `shard_runtime_server.py` itself, since that
|
||||
wiring is DGR-037's scope.
|
||||
- **DGR-051** (V4 `ShardEngine` adapter): `MtpHook`/`ArchitectureAuxStateHook`
|
||||
fix the field shape now so the V4 adapter does not need a breaking change
|
||||
to enable MTP after DGR-066 or to carry CSA/HCA/SWA/indexer/compressor
|
||||
state.
|
||||
259
.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md
Normal file
259
.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md
Normal file
@@ -0,0 +1,259 @@
|
||||
# DGR-032 evidence — deterministic fake `ShardEngine`
|
||||
|
||||
**Completed:** 2026-07-23
|
||||
**Branch:** `ralph/distributed-gguf-runtime`
|
||||
**Authority:** `.scratch/distributed-gguf-runtime/prd.json`
|
||||
**Dependencies:** DGR-031 (`evidence/DGR-031/README.md` — the project-owned
|
||||
`ShardEngine` abstract contract, `tests/shard_engine_contract.py`'s
|
||||
`assert_shard_engine_contract`, and its own dependency-handoff note that
|
||||
DGR-032 should "subclass `ShardEngine`, add delay/memory-pressure/malformed-
|
||||
output/crash injection, and pass the *same* `assert_shard_engine_contract`
|
||||
... — no new contract vocabulary should be needed").
|
||||
|
||||
## Objective
|
||||
|
||||
Provide an engine fixture that deterministically transforms typed boundary
|
||||
bundles and session state: head/middle/tail, prefill/decode, cancellation,
|
||||
release, isolated per-session epoch state, deterministic cache-miss/stale-
|
||||
epoch failures, and configurable delay/memory-pressure/malformed-output/
|
||||
crash-injection fault surfaces — all without llama.cpp, a GPU, or any I/O.
|
||||
|
||||
## What was found live before changing code
|
||||
|
||||
- `packages/node/meshnet_node/shard_engine.py` (DGR-031): the abstract
|
||||
`ShardEngine` with eight operations (`load`, `capabilities`, `prefill`,
|
||||
`decode`, `cancel`, `release`, `health`, `metrics`) and its project-owned
|
||||
dataclasses (`LoadRequest`, `EngineCapabilities`, `PrefillRequest`/
|
||||
`DecodeRequest`, `StepResult`, `BoundaryBundle`/`EngineTensor`,
|
||||
`TokenOutput`, `HealthResult`, `MetricsResult`).
|
||||
- `tests/shard_engine_contract.py` (DGR-031): the reusable
|
||||
`assert_shard_engine_contract(make_engine)` helper — nine lifecycle checks
|
||||
any implementation must pass, explicitly designed to be imported by
|
||||
DGR-032 and DGR-037 against their own engines.
|
||||
- `tests/test_shard_engine.py` (DGR-031): its `_ReferenceEngine` is
|
||||
explicitly documented as *not* the DGR-032 fixture ("no delay/memory-
|
||||
pressure/malformed/crash injection... that is a separate, larger story") —
|
||||
confirming this story starts from nothing, not inherited credit.
|
||||
- `grep -rn -i "fakeshardengine\|fake_shard_engine"` across `.py`/`.md`
|
||||
returned no prior matches — no fake engine existed before this story.
|
||||
- No file in `packages/node/meshnet_node/` wires a `ShardEngine` into
|
||||
`shard_runtime_server.py` yet (confirmed by grep for `ShardEngine`/
|
||||
`shard_engine` in that file — no matches); that wiring is DGR-037's scope,
|
||||
so this fixture is a standalone, importable engine only.
|
||||
|
||||
Live verification of the pre-existing dependency contract before adding new
|
||||
code:
|
||||
```bash
|
||||
PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q tests/test_shard_engine.py
|
||||
```
|
||||
```text
|
||||
12 passed in 0.13s
|
||||
```
|
||||
|
||||
## What was added (this story's change)
|
||||
|
||||
### `packages/node/meshnet_node/fake_shard_engine.py` (new)
|
||||
|
||||
`FakeShardEngine(ShardEngine)` — a pure-Python, deterministic fixture:
|
||||
|
||||
- **Determinism.** Every `prefill`/`decode` output is `SHA-256(seed_bytes +
|
||||
idempotency_step)`, where `seed_bytes` is derived from `token_ids` (head)
|
||||
or the input `BoundaryBundle`'s tensor bytes plus any `token_id_sideband`
|
||||
(middle/tail-in). Replaying identical inputs on a brand-new session
|
||||
produces byte-identical output — proven by
|
||||
`assert_shard_engine_contract`'s own determinism check and reused directly.
|
||||
- **Head/middle/tail.** Tail shards (`shard_end >= total_layers - 1`) return
|
||||
a `TokenOutput` sampled into `[0, TOKEN_ID_VOCAB_SIZE)`; head/middle shards
|
||||
return a `BoundaryBundle` tagged `boundary_point="post_head_residual"` or
|
||||
`"post_middle_residual"` respectively, so the three cases are
|
||||
distinguishable in fixture output, not just in the load request. A middle
|
||||
shard's `token_id_sideband` passes through unchanged from its input bundle
|
||||
to its output bundle (the V4 first-three-hash-routed-layers requirement
|
||||
RALPH-CONTEXT documents), never invented or dropped.
|
||||
- **Isolated session/epoch state.** `_sessions: dict[str, _SessionState]`
|
||||
keyed by `session_id`; each session tracks its own `epoch`/`cancelled`
|
||||
flag. A stale epoch, cancel, or release on one session never touches
|
||||
another's state (`test_session_state_is_isolated_between_two_concurrent_sessions`
|
||||
proves a stale-epoch rejection and a cancel on session `"a"` leave session
|
||||
`"b"` fully serviceable). Decoding an unopened session is a deterministic
|
||||
`NOT_FOUND`/`CacheResult.MISS`, not an exception.
|
||||
- **Configurable delay.** `FakeShardEngineConfig.step_delay_seconds` +
|
||||
injectable `sleep` hook (defaults to `time.sleep`, overridable in tests so
|
||||
they don't block wall-clock time) — invoked once per `prefill`/`decode`
|
||||
call before computing the deterministic output.
|
||||
- **Configurable memory pressure.** `FakeShardEngineConfig.memory_budget_bytes`
|
||||
— the engine accumulates `_bytes_used` across every step's seed bytes;
|
||||
once a step would push cumulative usage past the budget, that step
|
||||
deterministically returns `StatusCode.RESOURCE_EXHAUSTED` (`retryable=True`)
|
||||
with no output, instead of computing one.
|
||||
- **Configurable malformed output.** `FakeShardEngineConfig.malformed_output`
|
||||
— when set, the engine still reports `StatusCode.OK` (the point is a
|
||||
buggy-but-"successful"-looking response, not a status-coded failure) but
|
||||
the payload is structurally valid, semantically wrong: a tail `TokenOutput`
|
||||
is pushed past `MALFORMED_TOKEN_ID_FLOOR` (outside the fixture's own
|
||||
advertised vocab), and a head/middle `BoundaryBundle` gets an
|
||||
`architecture` field prefixed `"malformed:"` and its tensor `data`
|
||||
truncated to one byte — both structurally valid per `EngineTensor`'s and
|
||||
`BoundaryBundle`'s own `__post_init__` validation (which does not
|
||||
cross-check `data` length against `shape`/`dtype`), so a consumer must
|
||||
actually check shape/semantics, not just status codes, to catch it.
|
||||
- **Configurable crash injection.** `FakeShardEngineConfig.crash_after_calls`
|
||||
+ `crash_exception_factory` — after the configured number of
|
||||
`prefill`/`decode` calls, the engine raises an arbitrary exception (default
|
||||
`RuntimeError`, injectable) directly out of the call instead of returning a
|
||||
`StepResult`. This is deliberately *not* wrapped in `EngineError`/
|
||||
`StructuredStatus`: it simulates a whole-process failure (what a worker
|
||||
supervisor — DGR-040 — must catch and restart around), which is a
|
||||
different failure mode from a graceful status-coded rejection.
|
||||
- **Fixture-vs-real marker.** `FakeShardEngine.EVIDENCE_CLASS = "fixture"` —
|
||||
a structural constant (not just docstring prose) so DGR-036's fixture-vs-
|
||||
real-model parity check can assert programmatically that it is comparing a
|
||||
fixture engine against a real one, never two fixtures.
|
||||
- Every fault-injection knob defaults to off (`0`/`None`/`False`), so a bare
|
||||
`FakeShardEngine()` passes `assert_shard_engine_contract` unmodified —
|
||||
fault injection is opt-in, never a baseline behavior change.
|
||||
|
||||
### `tests/test_fake_shard_engine.py` (new)
|
||||
|
||||
- `test_fake_shard_engine_obeys_the_shared_shard_engine_contract` — runs the
|
||||
full DGR-031 contract against a bare `FakeShardEngine`.
|
||||
- `test_fake_shard_engine_declares_fixture_evidence_class` — pins the
|
||||
`EVIDENCE_CLASS` marker DGR-036 will rely on.
|
||||
- Head/middle/tail output-shape tests (`boundary_point`, token-id-sideband
|
||||
pass-through, tail vocab range).
|
||||
- `test_session_state_is_isolated_between_two_concurrent_sessions` — a
|
||||
stale-epoch rejection and a cancel on one session leave a second,
|
||||
concurrently open session fully serviceable.
|
||||
- One test per fault-injection knob (delay hook invocation, memory-budget
|
||||
trip, malformed tail/boundary-bundle output, crash-after-N-calls,
|
||||
configurable crash exception type) plus `FakeShardEngineConfig`'s own
|
||||
`__post_init__` validation (negative delay, negative budget, non-positive
|
||||
`crash_after_calls`).
|
||||
- `test_load_result_and_capabilities_report_recipe_architecture` — the
|
||||
fixture threads `LoadRequest.recipe["architecture"]` through to both
|
||||
`LoadResult.architecture` and `EngineCapabilities.architecture` rather than
|
||||
hardcoding `"dense"`/`"fake"` everywhere, so a future V4 recipe is visible
|
||||
in fixture output too.
|
||||
|
||||
### `.scratch/distributed-gguf-runtime/prd.json` / issue markdown
|
||||
|
||||
Marked `DGR-032.passes = true` with `completionNotes`; regenerated
|
||||
`issues/032-implement-deterministic-fake-shardengine.md` via
|
||||
`scripts/ralph_prd_schema.py render` so it matches `prd.json` byte-for-byte.
|
||||
|
||||
## Acceptance criteria → evidence
|
||||
|
||||
1. **Head, middle, tail, prefill, decode, cancellation, release with
|
||||
deterministic outputs** — `FakeShardEngine`'s `_transform`, boundary-point
|
||||
tagging, and `assert_shard_engine_contract`'s own determinism/cancel/
|
||||
release checks. Verified by
|
||||
`test_fake_shard_engine_obeys_the_shared_shard_engine_contract`,
|
||||
`test_head_shard_returns_boundary_bundle_with_post_head_residual_point`,
|
||||
`test_middle_shard_returns_boundary_bundle_and_passes_through_token_sideband`,
|
||||
`test_tail_shard_returns_token_output_within_advertised_vocab`.
|
||||
2. **Isolated session/epoch state and deterministic cache-miss/stale-epoch
|
||||
failures** — `_sessions` dict keyed per session;
|
||||
`test_session_state_is_isolated_between_two_concurrent_sessions` plus the
|
||||
shared contract's own cache-miss/stale-epoch checks.
|
||||
3. **Configurable delay, memory pressure, malformed output, crash
|
||||
injection** — `FakeShardEngineConfig`; verified by
|
||||
`test_step_delay_seconds_invokes_the_configured_sleep_hook`,
|
||||
`test_memory_budget_bytes_trips_deterministic_resource_exhausted`,
|
||||
`test_malformed_output_is_structurally_valid_but_semantically_wrong_for_tail`,
|
||||
`test_malformed_output_is_structurally_valid_but_semantically_wrong_for_boundary_bundle`,
|
||||
`test_crash_after_calls_raises_instead_of_returning_a_structured_status`,
|
||||
`test_crash_exception_factory_is_configurable`,
|
||||
`test_config_rejects_invalid_knob_values`.
|
||||
4. **Contract tests distinguish fixture evidence from real-model
|
||||
certification** — module docstring and this README are explicit that
|
||||
this is FIXTURE evidence only (numeric parity is DGR-036 onward); the
|
||||
`EVIDENCE_CLASS = "fixture"` constant makes that distinction structurally
|
||||
checkable, not just prose, pinned by
|
||||
`test_fake_shard_engine_declares_fixture_evidence_class`.
|
||||
5. **Gates + this handoff** — below.
|
||||
|
||||
## Commands and results
|
||||
|
||||
```bash
|
||||
PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q tests/test_fake_shard_engine.py tests/test_shard_engine.py
|
||||
```
|
||||
```text
|
||||
26 passed in 0.17s
|
||||
```
|
||||
|
||||
```bash
|
||||
PYTHONPATH=packages/node:packages/tracker .venv/bin/python3 -m pytest -q \
|
||||
tests/test_fake_shard_engine.py tests/test_shard_engine.py tests/test_shard_lifecycle.py \
|
||||
tests/test_architecture_boundary.py tests/test_activation_envelope.py \
|
||||
tests/test_native_shard_protocol.py tests/test_shard_runtime_harness.py
|
||||
```
|
||||
```text
|
||||
109 passed, 3 skipped in 3.78s
|
||||
```
|
||||
|
||||
```bash
|
||||
.venv/bin/python3 -m compileall -q packages tests
|
||||
```
|
||||
```text
|
||||
(no output — clean; exit 0)
|
||||
```
|
||||
|
||||
```bash
|
||||
git diff --check
|
||||
```
|
||||
```text
|
||||
(no output — clean)
|
||||
```
|
||||
|
||||
## Limitations
|
||||
|
||||
- `tests/` as a whole does not collect cleanly in this environment: the same
|
||||
pre-existing collection errors DGR-031's evidence recorded (missing
|
||||
optional dependencies such as `cryptography`) are still present and are
|
||||
unrelated to this story. This story's own gates were run as the targeted,
|
||||
scoped test set above per the shared quality gates' wording ("Targeted
|
||||
deterministic tests pass").
|
||||
- This is FIXTURE evidence only. `FakeShardEngine` proves lifecycle,
|
||||
session/epoch isolation, and fault-injection semantics; it proves nothing
|
||||
about numerical parity with a real model. That is DGR-036's explicit job
|
||||
once DGR-037's real engine exists, and DGR-053/054 for V4 alpha
|
||||
certification.
|
||||
- `FakeShardEngine` is not wired into `shard_runtime_server.py` or any gRPC
|
||||
surface — it is a standalone, importable engine only. Wiring a
|
||||
`ShardEngine` (fake or real) into the gRPC servicer is DGR-037's scope for
|
||||
the real engine; DGR-033 covers a C++ worker surface, which is a separate
|
||||
native executable, not a consumer of this Python module.
|
||||
- No gRPC/CMake/native-build changes were needed or made — this story is
|
||||
pure Python fixture code (`evidenceClass: fixture`, `hardware: none`), so
|
||||
the native CMake/CTest and patch-stack gates in the shared quality-gate
|
||||
list do not apply here, consistent with DGR-031's own README recording the
|
||||
same non-applicability.
|
||||
|
||||
## Dependency handoff
|
||||
|
||||
- **DGR-033** (standalone fake C++ gRPC Shard worker): its own issue
|
||||
describes a native C++ executable serving the lifecycle/stream RPC
|
||||
contract "using the fake engine" — that is a native analogue, not a
|
||||
consumer of this Python module; DGR-033 should still read this README for
|
||||
the exact deterministic-output/session-isolation/fault-injection semantics
|
||||
its C++ fake engine needs to reproduce so both fakes behave identically
|
||||
from a client's point of view.
|
||||
- **DGR-034/DGR-035** (range-aware GGUF ownership, boundary I/O):
|
||||
`FakeShardEngine` already demonstrates range-driven head/middle/tail
|
||||
behavior purely from `LoadRequest.shard_start`/`shard_end`/`total_layers`;
|
||||
no new range vocabulary was needed.
|
||||
- **DGR-036** (fixture vs real-model parity): compare a `FakeShardEngine`
|
||||
instance's `EVIDENCE_CLASS` (`"fixture"`) against DGR-037's real engine's
|
||||
equivalent marker (expected `"real"`) to assert the parity check is
|
||||
actually comparing two different implementations; reuse
|
||||
`assert_shard_engine_contract` against both to prove lifecycle parity
|
||||
before attempting numeric parity.
|
||||
- **DGR-037** (bind llama.cpp to the worker): `FakeShardEngine` is the
|
||||
reference implementation to diff a real engine's lifecycle behavior
|
||||
against — same request/result types, same session/epoch model, no new
|
||||
contract vocabulary.
|
||||
- **DGR-040** (worker supervision): the crash-injection knob
|
||||
(`crash_after_calls`/`crash_exception_factory`) exists specifically so
|
||||
supervision/restart logic has a deterministic way to trigger and test an
|
||||
unhandled engine failure distinct from a graceful `StructuredStatus`
|
||||
rejection.
|
||||
@@ -1,7 +1,7 @@
|
||||
<!-- GENERATED FROM prd.json — DO NOT EDIT AS AN INDEPENDENT SOURCE. prd.json IS AUTHORITATIVE. -->
|
||||
# DGR-030: Add accelerator build presets and native CI matrix
|
||||
|
||||
- **Status / triage:** specification only; `ready-for-agent`; `passes: false`
|
||||
- **Status / triage:** completed; `passes: true`
|
||||
- **Execution mode:** `AFK`
|
||||
- **Milestone:** `M1`
|
||||
- **Dependencies:** `DGR-029`
|
||||
@@ -18,11 +18,11 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Add isolated out-of-tree presets for CUDA, ROCm, Vulkan, and Metal without changing the deterministic CPU default.
|
||||
- [ ] Add a native CI/build matrix that reports unavailable SDKs as explicit unavailable/skipped lanes rather than false success.
|
||||
- [ ] Compile each available lane and preserve exact compiler, SDK, upstream pin, patch-stack, and build-option evidence.
|
||||
- [ ] Keep every backend/model/recipe lane registered-dark until a separate real-hardware certification record exists.
|
||||
- [ ] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
- [x] Add isolated out-of-tree presets for CUDA, ROCm, Vulkan, and Metal without changing the deterministic CPU default.
|
||||
- [x] Add a native CI/build matrix that reports unavailable SDKs as explicit unavailable/skipped lanes rather than false success.
|
||||
- [x] Compile each available lane and preserve exact compiler, SDK, upstream pin, patch-stack, and build-option evidence.
|
||||
- [x] Keep every backend/model/recipe lane registered-dark until a separate real-hardware certification record exists.
|
||||
- [x] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
|
||||
## Shared quality gates
|
||||
|
||||
@@ -30,10 +30,7 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
- `git diff --check` passes.
|
||||
- Default tests are model-download-free, API-credit-free, and GPU-free.
|
||||
- Evidence README records exact changed files, commands/results, limitations, and dependency handoff; no fabricated evidence or inherited completion credit.
|
||||
- Native changes pass focused out-of-tree CMake build and CTest; patch changes verify clean apply/check/reverse against the exact llama.cpp pin.
|
||||
- Runs are opt-in and record exact artifact/split hashes, runtime/upstream pin, backend/driver, hardware, network, commands, and raw metrics. Model artifacts use configured mounted-drive storage and never `/home`.
|
||||
- Preserve existing Transformers behavior and backend-agnostic Tracker routing/load balancing/billing/relay semantics unless an explicit versioned contract says otherwise. One scoped story commit is expected during execution, but this specification-materialization change is not committed.
|
||||
|
||||
## Evidence handoff
|
||||
|
||||
Write and verify `.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md`. Until every criterion and applicable gate has real evidence, this story remains `passes: false`. Legacy evidence is provenance only, not completion credit.
|
||||
Verified evidence: `.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md`. Legacy evidence remains provenance only and grants no implementation completion credit.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
<!-- GENERATED FROM prd.json — DO NOT EDIT AS AN INDEPENDENT SOURCE. prd.json IS AUTHORITATIVE. -->
|
||||
# DGR-031: Introduce the project-owned `ShardEngine` interface
|
||||
|
||||
- **Status / triage:** specification only; `ready-for-agent`; `passes: false`
|
||||
- **Status / triage:** completed; `passes: true`
|
||||
- **Execution mode:** `AFK`
|
||||
- **Milestone:** `M1`
|
||||
- **Dependencies:** `DGR-021`, `DGR-025`
|
||||
@@ -18,11 +18,11 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Define load, capabilities, prefill/decode, boundary/logits result, cancel, release, health, and metrics operations.
|
||||
- [ ] Use project-owned request/result/state types; expose no `ggml_tensor`, llama context, scheduler, or ABI-owned structure.
|
||||
- [ ] Reserve typed MTP and architecture auxiliary-state hooks without enabling them.
|
||||
- [ ] Add contract tests proving fake and future llama implementations obey identical lifecycle semantics.
|
||||
- [ ] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
- [x] Define load, capabilities, prefill/decode, boundary/logits result, cancel, release, health, and metrics operations.
|
||||
- [x] Use project-owned request/result/state types; expose no `ggml_tensor`, llama context, scheduler, or ABI-owned structure.
|
||||
- [x] Reserve typed MTP and architecture auxiliary-state hooks without enabling them.
|
||||
- [x] Add contract tests proving fake and future llama implementations obey identical lifecycle semantics.
|
||||
- [x] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
|
||||
## Shared quality gates
|
||||
|
||||
@@ -30,10 +30,7 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
- `git diff --check` passes.
|
||||
- Default tests are model-download-free, API-credit-free, and GPU-free.
|
||||
- Evidence README records exact changed files, commands/results, limitations, and dependency handoff; no fabricated evidence or inherited completion credit.
|
||||
- Native changes pass focused out-of-tree CMake build and CTest; patch changes verify clean apply/check/reverse against the exact llama.cpp pin.
|
||||
- Runs are opt-in and record exact artifact/split hashes, runtime/upstream pin, backend/driver, hardware, network, commands, and raw metrics. Model artifacts use configured mounted-drive storage and never `/home`.
|
||||
- Preserve existing Transformers behavior and backend-agnostic Tracker routing/load balancing/billing/relay semantics unless an explicit versioned contract says otherwise. One scoped story commit is expected during execution, but this specification-materialization change is not committed.
|
||||
|
||||
## Evidence handoff
|
||||
|
||||
Write and verify `.scratch/distributed-gguf-runtime/evidence/DGR-031/README.md`. Until every criterion and applicable gate has real evidence, this story remains `passes: false`. Legacy evidence is provenance only, not completion credit.
|
||||
Verified evidence: `.scratch/distributed-gguf-runtime/evidence/DGR-031/README.md`. Legacy evidence remains provenance only and grants no implementation completion credit.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
<!-- GENERATED FROM prd.json — DO NOT EDIT AS AN INDEPENDENT SOURCE. prd.json IS AUTHORITATIVE. -->
|
||||
# DGR-032: Implement deterministic fake `ShardEngine`
|
||||
|
||||
- **Status / triage:** specification only; `ready-for-agent`; `passes: false`
|
||||
- **Status / triage:** completed; `passes: true`
|
||||
- **Execution mode:** `AFK`
|
||||
- **Milestone:** `M1`
|
||||
- **Dependencies:** `DGR-031`
|
||||
@@ -18,11 +18,11 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Support head, middle, tail, prefill, decode, cancellation, and release with deterministic outputs.
|
||||
- [ ] Model isolated session/epoch state and deterministic cache-miss/stale-epoch failures.
|
||||
- [ ] Support configurable delay, memory pressure, malformed output, and crash injection.
|
||||
- [ ] Contract tests distinguish fixture evidence from real-model certification.
|
||||
- [ ] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
- [x] Support head, middle, tail, prefill, decode, cancellation, and release with deterministic outputs.
|
||||
- [x] Model isolated session/epoch state and deterministic cache-miss/stale-epoch failures.
|
||||
- [x] Support configurable delay, memory pressure, malformed output, and crash injection.
|
||||
- [x] Contract tests distinguish fixture evidence from real-model certification.
|
||||
- [x] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
|
||||
## Shared quality gates
|
||||
|
||||
@@ -30,10 +30,7 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
- `git diff --check` passes.
|
||||
- Default tests are model-download-free, API-credit-free, and GPU-free.
|
||||
- Evidence README records exact changed files, commands/results, limitations, and dependency handoff; no fabricated evidence or inherited completion credit.
|
||||
- Native changes pass focused out-of-tree CMake build and CTest; patch changes verify clean apply/check/reverse against the exact llama.cpp pin.
|
||||
- Runs are opt-in and record exact artifact/split hashes, runtime/upstream pin, backend/driver, hardware, network, commands, and raw metrics. Model artifacts use configured mounted-drive storage and never `/home`.
|
||||
- Preserve existing Transformers behavior and backend-agnostic Tracker routing/load balancing/billing/relay semantics unless an explicit versioned contract says otherwise. One scoped story commit is expected during execution, but this specification-materialization change is not committed.
|
||||
|
||||
## Evidence handoff
|
||||
|
||||
Write and verify `.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md`. Until every criterion and applicable gate has real evidence, this story remains `passes: false`. Legacy evidence is provenance only, not completion credit.
|
||||
Verified evidence: `.scratch/distributed-gguf-runtime/evidence/DGR-032/README.md`. Legacy evidence remains provenance only and grants no implementation completion credit.
|
||||
|
||||
@@ -538,13 +538,14 @@
|
||||
"Keep every backend/model/recipe lane registered-dark until a separate real-hardware certification record exists.",
|
||||
"Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff."
|
||||
],
|
||||
"passes": false,
|
||||
"passes": true,
|
||||
"notes": "Generated source issue: .scratch/distributed-gguf-runtime/issues/030-add-accelerator-build-presets-and-native-ci-matrix.md; prd.json is authoritative.",
|
||||
"blocks": [
|
||||
"DGR-053",
|
||||
"DGR-067",
|
||||
"DGR-068"
|
||||
]
|
||||
],
|
||||
"completionNotes": "Completed by agent"
|
||||
},
|
||||
{
|
||||
"id": "DGR-031",
|
||||
@@ -576,14 +577,15 @@
|
||||
"Add contract tests proving fake and future llama implementations obey identical lifecycle semantics.",
|
||||
"Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff."
|
||||
],
|
||||
"passes": false,
|
||||
"passes": true,
|
||||
"notes": "Generated source issue: .scratch/distributed-gguf-runtime/issues/031-introduce-the-project-owned-shardengine-interface.md; prd.json is authoritative.",
|
||||
"blocks": [
|
||||
"DGR-032",
|
||||
"DGR-034",
|
||||
"DGR-035",
|
||||
"DGR-037"
|
||||
]
|
||||
],
|
||||
"completionNotes": "Completed by agent"
|
||||
},
|
||||
{
|
||||
"id": "DGR-032",
|
||||
@@ -615,11 +617,12 @@
|
||||
"Contract tests distinguish fixture evidence from real-model certification.",
|
||||
"Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff."
|
||||
],
|
||||
"passes": false,
|
||||
"passes": true,
|
||||
"notes": "Generated source issue: .scratch/distributed-gguf-runtime/issues/032-implement-deterministic-fake-shardengine.md; prd.json is authoritative.",
|
||||
"blocks": [
|
||||
"DGR-033"
|
||||
]
|
||||
],
|
||||
"completionNotes": "Completed by agent"
|
||||
},
|
||||
{
|
||||
"id": "DGR-033",
|
||||
@@ -2161,6 +2164,6 @@
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"updatedAt": "2026-07-22T06:44:18.107Z"
|
||||
"updatedAt": "2026-07-23T08:09:16.081Z"
|
||||
}
|
||||
}
|
||||
@@ -120,5 +120,5 @@ M1: Build system + protocol (DGR-021..033)
|
||||
|
||||
- Ralph runs headless: reads backlog, spawns fresh Claude Code per ticket, verifies, reports
|
||||
- DGR-019/020 marked `ready-for-human` — needs review before certifying
|
||||
- Changes left uncommitted for review per Ralph policy (unless explicitly pushed)
|
||||
- As of July 23, 2026: `autoCommit = true` in `.ralph-tui/config.toml` — the engine now commits after every completed task, and a supervisor process pushes each commit to `origin/ralph/distributed-gguf-runtime` immediately.
|
||||
- `ralph-tui resume` picks up where it left off
|
||||
302
packages/node/meshnet_node/fake_shard_engine.py
Normal file
302
packages/node/meshnet_node/fake_shard_engine.py
Normal file
@@ -0,0 +1,302 @@
|
||||
"""Deterministic fake ``ShardEngine`` fixture (DGR-032).
|
||||
|
||||
``FakeShardEngine`` is a pure-Python, allocation-cheap subclass of
|
||||
:class:`~meshnet_node.shard_engine.ShardEngine`: no llama.cpp, no native
|
||||
buffers, no GPU, no filesystem or network I/O. Every prefill/decode output is
|
||||
a deterministic pure function of ``(loaded range, request inputs,
|
||||
idempotency_step)`` — hashed with SHA-256 — so replaying identical inputs on
|
||||
a fresh session always yields byte-identical output. It exists so worker
|
||||
wiring, gRPC harnesses (DGR-033), and lifecycle/session logic can be
|
||||
exercised end-to-end before a real llama.cpp-backed engine (DGR-037) exists.
|
||||
|
||||
This is FIXTURE evidence only. ``EVIDENCE_CLASS`` is set to ``"fixture"`` (as
|
||||
opposed to ``"real"``) precisely so a later story comparing engines
|
||||
programmatically — DGR-036's fixture-vs-real-model parity check — can assert
|
||||
it is actually comparing a fixture against a real engine rather than two
|
||||
fixtures. This module proves lifecycle/session/epoch/fault-injection
|
||||
semantics; it says nothing about numerical parity with a real model. Real-
|
||||
model certification is DGR-036 onward (DGR-053/DGR-054 for V4 alpha).
|
||||
|
||||
Fault injection (delay, memory pressure, malformed output, crash) is
|
||||
deterministic and opt-in via :class:`FakeShardEngineConfig`. Every knob
|
||||
defaults to off, so a bare ``FakeShardEngine()`` reproduces plain
|
||||
deterministic fixture behavior and passes
|
||||
:func:`tests.shard_engine_contract.assert_shard_engine_contract` unmodified.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable
|
||||
|
||||
from .shard_engine import (
|
||||
BoundaryBundle,
|
||||
DecodeRequest,
|
||||
EngineCapabilities,
|
||||
EngineTensor,
|
||||
HealthResult,
|
||||
LoadRequest,
|
||||
LoadResult,
|
||||
MetricsResult,
|
||||
PrefillRequest,
|
||||
ShardEngine,
|
||||
StepResult,
|
||||
TokenOutput,
|
||||
)
|
||||
from .shard_lifecycle import CacheResult, StatusCode, StructuredStatus
|
||||
|
||||
__all__ = ["FakeShardEngineConfig", "FakeShardEngine", "TOKEN_ID_VOCAB_SIZE", "MALFORMED_TOKEN_ID_FLOOR"]
|
||||
|
||||
TOKEN_ID_VOCAB_SIZE = 50_000
|
||||
# A malformed tail output is deterministically pushed past the fixture's own
|
||||
# advertised vocabulary range, so a downstream consumer checking "is this
|
||||
# token_id within the vocab this fixture promises" can detect it without any
|
||||
# extra signalling from the engine.
|
||||
MALFORMED_TOKEN_ID_FLOOR = 100_000_000
|
||||
|
||||
|
||||
def _default_crash_exception() -> BaseException:
|
||||
return RuntimeError(
|
||||
"FakeShardEngine: injected crash (simulated process failure, not a StructuredStatus)"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FakeShardEngineConfig:
|
||||
"""Deterministic fault-injection knobs.
|
||||
|
||||
Every knob is off (``0``/``None``/``False``) by default. ``sleep`` is
|
||||
injectable so tests can assert a delay was requested without an actual
|
||||
process sleep; ``crash_exception_factory`` is injectable so tests can
|
||||
assert on a specific exception type/instance.
|
||||
"""
|
||||
|
||||
step_delay_seconds: float = 0.0
|
||||
sleep: Callable[[float], None] = time.sleep
|
||||
memory_budget_bytes: int | None = None
|
||||
malformed_output: bool = False
|
||||
crash_after_calls: int | None = None
|
||||
crash_exception_factory: Callable[[], BaseException] = _default_crash_exception
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.step_delay_seconds < 0:
|
||||
raise ValueError("step_delay_seconds must be non-negative")
|
||||
if self.memory_budget_bytes is not None and self.memory_budget_bytes < 0:
|
||||
raise ValueError("memory_budget_bytes must be non-negative")
|
||||
if self.crash_after_calls is not None and self.crash_after_calls <= 0:
|
||||
raise ValueError("crash_after_calls must be positive when set")
|
||||
|
||||
|
||||
@dataclass
|
||||
class _SessionState:
|
||||
epoch: int
|
||||
cancelled: bool = False
|
||||
|
||||
|
||||
class FakeShardEngine(ShardEngine):
|
||||
"""Deterministic fixture ``ShardEngine``. See module docstring."""
|
||||
|
||||
EVIDENCE_CLASS = "fixture"
|
||||
|
||||
def __init__(self, config: FakeShardEngineConfig | None = None) -> None:
|
||||
self._config = config or FakeShardEngineConfig()
|
||||
self._loaded: LoadRequest | None = None
|
||||
self._sessions: dict[str, _SessionState] = {}
|
||||
self._cancelled_total = 0
|
||||
self._generated_tokens = 0
|
||||
self._call_count = 0
|
||||
self._bytes_used = 0
|
||||
|
||||
# -- lifecycle -----------------------------------------------------
|
||||
|
||||
def load(self, request: LoadRequest) -> LoadResult:
|
||||
self._loaded = request
|
||||
return LoadResult(
|
||||
status=StructuredStatus(StatusCode.OK, "fake engine loaded"),
|
||||
effective_start=request.shard_start,
|
||||
architecture=str(request.recipe.get("architecture", "fake")),
|
||||
)
|
||||
|
||||
def capabilities(self) -> EngineCapabilities:
|
||||
if self._loaded is None:
|
||||
return EngineCapabilities(
|
||||
status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "engine not loaded")
|
||||
)
|
||||
request = self._loaded
|
||||
return EngineCapabilities(
|
||||
status=StructuredStatus(StatusCode.OK, "ready"),
|
||||
shard_start=request.shard_start,
|
||||
shard_end=request.shard_end,
|
||||
effective_start=request.shard_start,
|
||||
total_layers=request.total_layers,
|
||||
architecture=str(request.recipe.get("architecture", "fake")),
|
||||
max_concurrent_sessions=64,
|
||||
max_context_tokens=131072,
|
||||
supports_mtp=False,
|
||||
)
|
||||
|
||||
def prefill(self, request: PrefillRequest) -> StepResult:
|
||||
return self._step(
|
||||
session_id=request.session_id,
|
||||
route_epoch=request.route_epoch,
|
||||
idempotency_step=request.idempotency_step,
|
||||
token_ids=request.token_ids,
|
||||
input_bundle=request.input,
|
||||
cache_result_on_success=CacheResult.STORED,
|
||||
opens_session=True,
|
||||
)
|
||||
|
||||
def decode(self, request: DecodeRequest) -> StepResult:
|
||||
token_ids = (request.token_id,) if request.token_id is not None else None
|
||||
return self._step(
|
||||
session_id=request.session_id,
|
||||
route_epoch=request.route_epoch,
|
||||
idempotency_step=request.idempotency_step,
|
||||
token_ids=token_ids,
|
||||
input_bundle=request.input,
|
||||
cache_result_on_success=CacheResult.HIT,
|
||||
opens_session=False,
|
||||
)
|
||||
|
||||
def cancel(self, session_id: str, *, work_id: str = "", reason: str = "") -> StructuredStatus:
|
||||
session = self._sessions.get(session_id)
|
||||
if session is None:
|
||||
session = _SessionState(epoch=0)
|
||||
self._sessions[session_id] = session
|
||||
if not session.cancelled:
|
||||
self._cancelled_total += 1
|
||||
session.cancelled = True
|
||||
return StructuredStatus(StatusCode.CANCELLED, reason or "fake engine: session cancelled")
|
||||
|
||||
def release(self, session_id: str) -> StructuredStatus:
|
||||
self._sessions.pop(session_id, None)
|
||||
return StructuredStatus(StatusCode.OK, "fake engine: session released")
|
||||
|
||||
def health(self) -> HealthResult:
|
||||
loaded = self._loaded is not None
|
||||
return HealthResult(
|
||||
status=StructuredStatus(StatusCode.OK, "ok"),
|
||||
serving=loaded,
|
||||
state="SERVING" if loaded else "NOT_LOADED",
|
||||
active_sessions=len(self._sessions),
|
||||
)
|
||||
|
||||
def metrics(self) -> MetricsResult:
|
||||
return MetricsResult(
|
||||
status=StructuredStatus(StatusCode.OK, "ok"),
|
||||
active_sessions=len(self._sessions),
|
||||
queued_frames=0,
|
||||
inflight_bytes=0,
|
||||
kv_entries=len(self._sessions),
|
||||
generated_tokens=self._generated_tokens,
|
||||
cancelled_sessions=self._cancelled_total,
|
||||
)
|
||||
|
||||
# -- shared step machinery ------------------------------------------
|
||||
|
||||
def _step(
|
||||
self,
|
||||
*,
|
||||
session_id: str,
|
||||
route_epoch: int,
|
||||
idempotency_step: int,
|
||||
token_ids: tuple[int, ...] | None,
|
||||
input_bundle: BoundaryBundle | None,
|
||||
cache_result_on_success: CacheResult,
|
||||
opens_session: bool,
|
||||
) -> StepResult:
|
||||
if self._loaded is None:
|
||||
return StepResult(status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "engine not loaded"))
|
||||
|
||||
self._call_count += 1
|
||||
if self._config.crash_after_calls is not None and self._call_count == self._config.crash_after_calls:
|
||||
raise self._config.crash_exception_factory()
|
||||
|
||||
session = self._sessions.get(session_id)
|
||||
if session is None:
|
||||
if not opens_session:
|
||||
return StepResult(
|
||||
status=StructuredStatus(StatusCode.NOT_FOUND, "no cached session state for decode"),
|
||||
cache_result=CacheResult.MISS,
|
||||
)
|
||||
session = _SessionState(epoch=route_epoch)
|
||||
self._sessions[session_id] = session
|
||||
|
||||
if session.cancelled:
|
||||
return StepResult(status=StructuredStatus(StatusCode.CANCELLED, "session cancelled"))
|
||||
|
||||
if route_epoch < session.epoch:
|
||||
return StepResult(status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "stale route epoch"))
|
||||
session.epoch = route_epoch
|
||||
|
||||
if self._config.step_delay_seconds:
|
||||
self._config.sleep(self._config.step_delay_seconds)
|
||||
|
||||
seed = self._seed_bytes(token_ids, input_bundle)
|
||||
self._bytes_used += len(seed)
|
||||
budget = self._config.memory_budget_bytes
|
||||
if budget is not None and self._bytes_used > budget:
|
||||
return StepResult(
|
||||
status=StructuredStatus(
|
||||
StatusCode.RESOURCE_EXHAUSTED,
|
||||
"fake engine memory pressure budget exceeded",
|
||||
retryable=True,
|
||||
details={"memory_budget_bytes": str(budget), "bytes_used": str(self._bytes_used)},
|
||||
)
|
||||
)
|
||||
|
||||
output = self._transform(seed, idempotency_step, input_bundle)
|
||||
if isinstance(output, TokenOutput):
|
||||
self._generated_tokens += 1
|
||||
return StepResult(status=StructuredStatus(StatusCode.OK, "ok"), cache_result=cache_result_on_success, output=output)
|
||||
|
||||
@staticmethod
|
||||
def _seed_bytes(token_ids: tuple[int, ...] | None, bundle: BoundaryBundle | None) -> bytes:
|
||||
if token_ids:
|
||||
seed = b"".join(int(t).to_bytes(8, "big") for t in token_ids)
|
||||
elif bundle is not None:
|
||||
seed = b"".join(tensor.data for tensor in bundle.tensors)
|
||||
if bundle.token_id_sideband:
|
||||
seed += b"".join(int(t).to_bytes(8, "big") for t in bundle.token_id_sideband)
|
||||
else:
|
||||
seed = b""
|
||||
return seed
|
||||
|
||||
def _transform(
|
||||
self, seed: bytes, idempotency_step: int, input_bundle: BoundaryBundle | None
|
||||
) -> BoundaryBundle | TokenOutput:
|
||||
assert self._loaded is not None
|
||||
digest = hashlib.sha256(seed + idempotency_step.to_bytes(8, "big")).digest()
|
||||
loaded = self._loaded
|
||||
is_tail = loaded.shard_end >= loaded.total_layers - 1
|
||||
is_head = loaded.shard_start == 0
|
||||
|
||||
if is_tail:
|
||||
token_id = int.from_bytes(digest[:4], "big") % TOKEN_ID_VOCAB_SIZE
|
||||
if self._config.malformed_output:
|
||||
token_id = MALFORMED_TOKEN_ID_FLOOR + token_id
|
||||
return TokenOutput(token_id=token_id)
|
||||
|
||||
boundary_point = "post_head_residual" if is_head else "post_middle_residual"
|
||||
architecture = (
|
||||
input_bundle.architecture if input_bundle is not None else str(loaded.recipe.get("architecture", "fake"))
|
||||
)
|
||||
data = digest
|
||||
if self._config.malformed_output:
|
||||
architecture = f"malformed:{architecture}"
|
||||
data = digest[:1]
|
||||
tensor = EngineTensor(
|
||||
name="hidden_states",
|
||||
shape=(1, max(len(seed) // 8, 1)),
|
||||
dtype="bfloat16",
|
||||
data=data,
|
||||
)
|
||||
token_id_sideband = input_bundle.token_id_sideband if input_bundle is not None else None
|
||||
return BoundaryBundle(
|
||||
tensors=(tensor,),
|
||||
architecture=architecture,
|
||||
boundary_point=boundary_point,
|
||||
token_id_sideband=token_id_sideband,
|
||||
)
|
||||
372
packages/node/meshnet_node/shard_engine.py
Normal file
372
packages/node/meshnet_node/shard_engine.py
Normal file
@@ -0,0 +1,372 @@
|
||||
"""The project-owned ``ShardEngine`` contract (DGR-031).
|
||||
|
||||
A worker process (the gRPC surface in ``shard_runtime_server.py``, or any
|
||||
future transport) never talks to llama.cpp directly. It talks to a
|
||||
``ShardEngine``. This module is the *only* place that boundary is defined, and
|
||||
every operation on it is built from project-owned dataclasses and plain
|
||||
Python values (``str``, ``int``, ``bytes``, ``Mapping``) — never a
|
||||
``ggml_tensor``, a llama context/scheduler handle, or a generated-protobuf
|
||||
(ABI) message. A fake fixture engine (DGR-032) and a real llama.cpp-backed
|
||||
engine (DGR-037) are both, structurally, nothing more than subclasses of
|
||||
:class:`ShardEngine`; the worker code that calls them does not change when one
|
||||
replaces the other.
|
||||
|
||||
This is deliberately a fourth, distinct layer from the three that already
|
||||
exist:
|
||||
|
||||
- ``native_protocol`` — the generated gRPC/Protobuf wire ABI (DGR-021/024).
|
||||
- ``protocol.ActivationEnvelope`` — the versioned wire envelope for activation
|
||||
traffic between shard *hops* over the network (DGR-021).
|
||||
- ``shard_lifecycle`` — the versioned RPC/session lifecycle contract a
|
||||
generated gRPC binding consumes (DGR-022).
|
||||
|
||||
``ShardEngine`` sits *inside* one worker process, below all three: it is the
|
||||
seam between "the code that speaks Meshnet's wire protocol" and "the code
|
||||
that actually runs model layers." It reuses :class:`~meshnet_node.shard_lifecycle.StructuredStatus`,
|
||||
:class:`~meshnet_node.shard_lifecycle.StatusCode`, :class:`~meshnet_node.shard_lifecycle.CacheExpectation`,
|
||||
and :class:`~meshnet_node.shard_lifecycle.CacheResult` rather than inventing a
|
||||
parallel status vocabulary, since those are already project-owned and
|
||||
version-stable.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import abc
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Mapping
|
||||
|
||||
from .shard_lifecycle import (
|
||||
CacheExpectation,
|
||||
CacheResult,
|
||||
StatusCode,
|
||||
StructuredStatus,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"EngineError",
|
||||
"EngineTensor",
|
||||
"BoundaryBundle",
|
||||
"TokenOutput",
|
||||
"MtpHook",
|
||||
"ArchitectureAuxStateHook",
|
||||
"LoadRequest",
|
||||
"LoadResult",
|
||||
"EngineCapabilities",
|
||||
"PrefillRequest",
|
||||
"DecodeRequest",
|
||||
"StepResult",
|
||||
"HealthResult",
|
||||
"MetricsResult",
|
||||
"ShardEngine",
|
||||
]
|
||||
|
||||
|
||||
class EngineError(RuntimeError):
|
||||
"""An engine-boundary failure represented by a structured status.
|
||||
|
||||
Mirrors :class:`~meshnet_node.shard_lifecycle.LifecycleContractError`:
|
||||
callers pattern-match on ``error.status.code`` rather than on exception
|
||||
subclasses, so a fake and a real engine can fail the exact same way for
|
||||
the exact same reason.
|
||||
"""
|
||||
|
||||
def __init__(self, status: StructuredStatus) -> None:
|
||||
self.status = status
|
||||
super().__init__(status.message)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EngineTensor:
|
||||
"""One named tensor crossing the engine boundary.
|
||||
|
||||
Intentionally not a ``ggml_tensor`` or a framework tensor object: ``data``
|
||||
is plain owned bytes, ``shape``/``dtype`` are plain metadata. An
|
||||
implementation constructs this from whatever internal representation it
|
||||
uses (a ``torch.Tensor``, a llama.cpp buffer, a synthetic fixture array)
|
||||
without leaking that representation across the boundary.
|
||||
"""
|
||||
|
||||
name: str
|
||||
shape: tuple[int, ...]
|
||||
dtype: str
|
||||
data: bytes
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not self.name:
|
||||
raise ValueError("engine tensor requires a name")
|
||||
if not self.shape or any(dim <= 0 for dim in self.shape):
|
||||
raise ValueError("engine tensor shape must be a non-empty tuple of positive ints")
|
||||
if not self.dtype:
|
||||
raise ValueError("engine tensor requires a dtype")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BoundaryBundle:
|
||||
"""A named-tensor activation crossing a shard boundary (head/middle/tail-in).
|
||||
|
||||
``token_id_sideband`` carries token IDs alongside the activation only
|
||||
where the architecture boundary requires them (V4's first three
|
||||
hash-routed MoE layers); it is ``None`` everywhere else. Per-shard hot
|
||||
KV/recurrent/CSA/HCA/SWA/indexer/compressor state never appears here — it
|
||||
stays local to a shard via :class:`ArchitectureAuxStateHook` and is never
|
||||
part of what crosses the wire.
|
||||
"""
|
||||
|
||||
tensors: tuple[EngineTensor, ...]
|
||||
architecture: str
|
||||
boundary_point: str
|
||||
token_id_sideband: tuple[int, ...] | None = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not self.tensors:
|
||||
raise ValueError("boundary bundle requires at least one tensor")
|
||||
if not self.architecture:
|
||||
raise ValueError("boundary bundle requires an architecture name")
|
||||
if not self.boundary_point:
|
||||
raise ValueError("boundary bundle requires a boundary point name")
|
||||
|
||||
def tensor(self, name: str) -> EngineTensor:
|
||||
for tensor in self.tensors:
|
||||
if tensor.name == name:
|
||||
return tensor
|
||||
raise KeyError(name)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TokenOutput:
|
||||
"""A tail shard's sampled decode result.
|
||||
|
||||
Never a raw logits tensor: the engine boundary only ever hands back the
|
||||
already-sampled token (mirroring
|
||||
:meth:`meshnet_node.architecture_boundary.TailOutput.sampled_token`, which
|
||||
likewise refuses anything but a sampled token id).
|
||||
"""
|
||||
|
||||
token_id: int
|
||||
text: str | None = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.token_id < 0:
|
||||
raise ValueError("sampled token id must be non-negative")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MtpHook:
|
||||
"""Reserved multi-token-prediction hook — typed, but refused when enabled.
|
||||
|
||||
RALPH-CONTEXT is explicit that "MTP is reserved and off for alpha; its
|
||||
ownership contract, implementation, and benchmark are required before
|
||||
beta" (DGR-065/DGR-066). Reserving the shape now means DGR-037's real
|
||||
engine and DGR-051's V4 adapter do not have to change this dataclass's
|
||||
field layout later; they only flip ``enabled`` once DGR-066 lands.
|
||||
"""
|
||||
|
||||
enabled: bool = False
|
||||
draft_token_count: int = 0
|
||||
aux_state: Mapping[str, Any] | None = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.enabled:
|
||||
raise ValueError(
|
||||
"MTP is reserved and must remain disabled before DGR-066; "
|
||||
"this hook exists to fix its shape, not to enable it"
|
||||
)
|
||||
if self.draft_token_count < 0:
|
||||
raise ValueError("draft_token_count must be non-negative")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ArchitectureAuxStateHook:
|
||||
"""Reserved per-shard architecture auxiliary-state hook.
|
||||
|
||||
Covers V4's CSA/HCA/SWA/indexer/compressor state and any other
|
||||
architecture-local state a future adapter needs. RALPH-CONTEXT locks this
|
||||
as shard-local, keyed by route session/epoch, and explicitly never carried
|
||||
over the WAN seam — so this hook has no wire encoding of its own and must
|
||||
never be embedded inside a :class:`BoundaryBundle`.
|
||||
"""
|
||||
|
||||
kind: str = ""
|
||||
state: Mapping[str, Any] | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LoadRequest:
|
||||
"""One exact artifact/recipe/range identity for a worker to load."""
|
||||
|
||||
artifact_path: str
|
||||
shard_start: int
|
||||
shard_end: int
|
||||
total_layers: int
|
||||
recipe: Mapping[str, Any] = field(default_factory=dict)
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not self.artifact_path:
|
||||
raise ValueError("load request requires an artifact path")
|
||||
if self.shard_start < 0 or self.shard_end < self.shard_start:
|
||||
raise ValueError("shard_start must be <= shard_end and non-negative")
|
||||
if self.total_layers <= self.shard_end:
|
||||
raise ValueError("total_layers must exceed shard_end (shard_end is inclusive)")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LoadResult:
|
||||
status: StructuredStatus
|
||||
effective_start: int = 0
|
||||
architecture: str = ""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EngineCapabilities:
|
||||
status: StructuredStatus
|
||||
shard_start: int = 0
|
||||
shard_end: int = 0
|
||||
effective_start: int = 0
|
||||
total_layers: int = 0
|
||||
architecture: str = ""
|
||||
max_concurrent_sessions: int = 0
|
||||
max_context_tokens: int = 0
|
||||
supports_mtp: bool = False
|
||||
|
||||
@property
|
||||
def is_head(self) -> bool:
|
||||
return self.shard_start == 0
|
||||
|
||||
@property
|
||||
def is_tail(self) -> bool:
|
||||
return self.shard_end >= self.total_layers - 1
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PrefillRequest:
|
||||
"""A prefill step. Exactly one of ``token_ids`` (head) or ``input`` (middle/tail) is set."""
|
||||
|
||||
session_id: str
|
||||
route_epoch: int
|
||||
position: int
|
||||
idempotency_step: int
|
||||
token_ids: tuple[int, ...] | None = None
|
||||
input: BoundaryBundle | None = None
|
||||
cache_expectation: CacheExpectation = CacheExpectation.NONE
|
||||
mtp: MtpHook = field(default_factory=MtpHook)
|
||||
architecture_aux_state: ArchitectureAuxStateHook | None = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_require_exactly_one_input(self.token_ids, self.input)
|
||||
if not self.session_id:
|
||||
raise ValueError("prefill request requires a session id")
|
||||
if self.route_epoch < 0 or self.position < 0 or self.idempotency_step < 0:
|
||||
raise ValueError("route_epoch, position, and idempotency_step must be non-negative")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DecodeRequest:
|
||||
"""A decode step. Exactly one of ``token_id`` (head) or ``input`` (middle/tail) is set."""
|
||||
|
||||
session_id: str
|
||||
route_epoch: int
|
||||
position: int
|
||||
idempotency_step: int
|
||||
token_id: int | None = None
|
||||
input: BoundaryBundle | None = None
|
||||
mtp: MtpHook = field(default_factory=MtpHook)
|
||||
architecture_aux_state: ArchitectureAuxStateHook | None = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_require_exactly_one_input(
|
||||
None if self.token_id is None else (self.token_id,), self.input
|
||||
)
|
||||
if not self.session_id:
|
||||
raise ValueError("decode request requires a session id")
|
||||
if self.route_epoch < 0 or self.position < 0 or self.idempotency_step < 0:
|
||||
raise ValueError("route_epoch, position, and idempotency_step must be non-negative")
|
||||
|
||||
|
||||
def _require_exactly_one_input(
|
||||
token_ids: tuple[int, ...] | None, bundle: BoundaryBundle | None
|
||||
) -> None:
|
||||
if (token_ids is None) == (bundle is None):
|
||||
raise ValueError("exactly one of token ids or a boundary bundle must be set")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StepResult:
|
||||
"""The result of a prefill or decode step.
|
||||
|
||||
``output`` is a :class:`BoundaryBundle` for a head/middle shard handing an
|
||||
activation to the next hop, or a :class:`TokenOutput` for a tail shard
|
||||
that sampled a token. It is ``None`` only when ``status.code`` is not
|
||||
``OK``.
|
||||
"""
|
||||
|
||||
status: StructuredStatus
|
||||
cache_result: CacheResult = CacheResult.NOT_REQUESTED
|
||||
output: BoundaryBundle | TokenOutput | None = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.status.code is StatusCode.OK and self.output is None:
|
||||
raise ValueError("a successful step result must carry an output")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HealthResult:
|
||||
status: StructuredStatus
|
||||
serving: bool = False
|
||||
state: str = "UNKNOWN"
|
||||
active_sessions: int = 0
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MetricsResult:
|
||||
status: StructuredStatus
|
||||
active_sessions: int = 0
|
||||
queued_frames: int = 0
|
||||
inflight_bytes: int = 0
|
||||
kv_entries: int = 0
|
||||
generated_tokens: int = 0
|
||||
cancelled_sessions: int = 0
|
||||
|
||||
|
||||
class ShardEngine(abc.ABC):
|
||||
"""The contract every shard execution engine (fake or real) must implement.
|
||||
|
||||
Every method returns a project-owned result carrying a
|
||||
:class:`~meshnet_node.shard_lifecycle.StructuredStatus` rather than
|
||||
raising for expected, protocol-visible outcomes (a cache miss, a stale
|
||||
epoch, an unknown session); an :class:`EngineError` is reserved for
|
||||
genuine programming errors at the call site (malformed request objects),
|
||||
which the request dataclasses' own ``__post_init__`` validation already
|
||||
catches before an implementation ever sees them.
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def load(self, request: LoadRequest) -> LoadResult:
|
||||
"""Load one exact artifact/recipe/range identity. Idempotent per engine instance."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def capabilities(self) -> EngineCapabilities:
|
||||
"""Report this engine's authoritative range and limits after ``load``."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def prefill(self, request: PrefillRequest) -> StepResult:
|
||||
"""Run one prefill step for a session."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def decode(self, request: DecodeRequest) -> StepResult:
|
||||
"""Run one decode step for a session."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def cancel(self, session_id: str, *, work_id: str = "", reason: str = "") -> StructuredStatus:
|
||||
"""Cancel a session (or one work item within it) in flight."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def release(self, session_id: str) -> StructuredStatus:
|
||||
"""Release a session's held state. Idempotent."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def health(self) -> HealthResult:
|
||||
"""Report liveness/serving state. Must never raise."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def metrics(self) -> MetricsResult:
|
||||
"""Report point-in-time operational counters. Must never raise."""
|
||||
@@ -52,6 +52,24 @@
|
||||
"smoke_output_token": "usage",
|
||||
"ctest_regex": "^test-meshnet-range-ownership$"
|
||||
},
|
||||
"accelerator_presets": {
|
||||
"cuda": {
|
||||
"backend_flag": "GGML_CUDA",
|
||||
"sdk_probe": {"binary": "nvcc", "env_var": "CUDACXX"}
|
||||
},
|
||||
"rocm": {
|
||||
"backend_flag": "GGML_HIP",
|
||||
"sdk_probe": {"binary": "hipcc", "env_var": "HIPCXX"}
|
||||
},
|
||||
"vulkan": {
|
||||
"backend_flag": "GGML_VULKAN",
|
||||
"sdk_probe": {"binary": "glslc", "env_var": "VULKAN_SDK_GLSLC"}
|
||||
},
|
||||
"metal": {
|
||||
"backend_flag": "GGML_METAL",
|
||||
"sdk_probe": {"binary": "xcrun", "platform_only": "darwin"}
|
||||
}
|
||||
},
|
||||
"required_upstream_blobs": {
|
||||
"CMakeLists.txt": "81f23d7e70b7378511af5d01be680c03aebc2b15"
|
||||
},
|
||||
|
||||
@@ -109,9 +109,41 @@ def _load_lock() -> dict[str, Any]:
|
||||
"workspace": "build/llama.cpp",
|
||||
}:
|
||||
raise DependencyError("retrieval must use the locked detached-commit build workspace")
|
||||
_verify_accelerator_presets(lock)
|
||||
return lock
|
||||
|
||||
|
||||
def _verify_accelerator_presets(lock: dict[str, Any]) -> None:
|
||||
"""Each preset must isolate one backend that the CPU default leaves OFF.
|
||||
|
||||
This is what keeps DGR-030's presets from ever being able to change the
|
||||
deterministic CPU default recorded in ``build.configure_flags``: a preset
|
||||
can only exist for a flag this lock already pins OFF, and
|
||||
``accelerator_configure_flags`` only ever returns a fresh list, never
|
||||
mutates ``build.configure_flags`` in place.
|
||||
"""
|
||||
presets = lock.get("accelerator_presets", {})
|
||||
if not isinstance(presets, dict):
|
||||
raise DependencyError("accelerator_presets must be a JSON object")
|
||||
if not presets:
|
||||
return
|
||||
base_flags = dict(flag[len("-D"):].split("=", 1) for flag in lock["build"]["configure_flags"])
|
||||
for name, preset in presets.items():
|
||||
if not isinstance(preset, dict):
|
||||
raise DependencyError(f"accelerator_presets.{name} must be a JSON object")
|
||||
backend_flag = preset.get("backend_flag")
|
||||
if not isinstance(backend_flag, str) or not backend_flag:
|
||||
raise DependencyError(f"accelerator_presets.{name} is missing backend_flag")
|
||||
if base_flags.get(backend_flag) != "OFF":
|
||||
raise DependencyError(
|
||||
f"accelerator_presets.{name} backend flag {backend_flag} must be OFF in "
|
||||
"the deterministic CPU default build.configure_flags"
|
||||
)
|
||||
probe = preset.get("sdk_probe")
|
||||
if not isinstance(probe, dict) or not isinstance(probe.get("binary"), str) or not probe["binary"]:
|
||||
raise DependencyError(f"accelerator_presets.{name} is missing an sdk_probe.binary")
|
||||
|
||||
|
||||
def _patches(lock: dict[str, Any]) -> list[pathlib.Path]:
|
||||
series = [line for line in (PATCH_DIR / "series").read_text().splitlines() if line]
|
||||
if series != lock["patch_series"] or series != sorted(series) or not series:
|
||||
@@ -507,6 +539,116 @@ def ctest_lane(build_dir: pathlib.Path) -> None:
|
||||
print(_run(_ctest(), "--test-dir", str(build_dir), "-R", regex, "--output-on-failure"))
|
||||
|
||||
|
||||
def _sdk_probe(probe: dict[str, Any]) -> str | None:
|
||||
"""Resolve one accelerator lane's SDK binary, or None if it is unavailable."""
|
||||
platform_only = probe.get("platform_only")
|
||||
if platform_only and sys.platform != platform_only:
|
||||
return None
|
||||
env_var = probe.get("env_var")
|
||||
if env_var:
|
||||
override = os.environ.get(env_var)
|
||||
if override:
|
||||
return override
|
||||
return shutil.which(probe["binary"])
|
||||
|
||||
|
||||
def accelerator_status(name: str, lock: dict[str, Any] | None = None) -> dict[str, Any]:
|
||||
"""Report whether lane `name`'s SDK is present, never raising for absence.
|
||||
|
||||
This is the single source of truth for DGR-030's "unavailable/skipped, not
|
||||
false success" contract: absence is reported as data, not swallowed and
|
||||
not escalated into a build attempt.
|
||||
"""
|
||||
lock = lock if lock is not None else _load_lock()
|
||||
presets = lock.get("accelerator_presets", {})
|
||||
if name not in presets:
|
||||
raise DependencyError(f"unknown accelerator lane: {name}")
|
||||
probe = presets[name]["sdk_probe"]
|
||||
resolved = _sdk_probe(probe)
|
||||
if resolved is None:
|
||||
platform_only = probe.get("platform_only")
|
||||
if platform_only and sys.platform != platform_only:
|
||||
reason = f"platform {sys.platform!r} is not {platform_only!r}"
|
||||
else:
|
||||
reason = f"{probe['binary']} is unavailable on PATH"
|
||||
return {"lane": name, "available": False, "reason": reason}
|
||||
return {"lane": name, "available": True, "sdk_binary": resolved}
|
||||
|
||||
|
||||
def accelerator_configure_flags(lock: dict[str, Any], name: str) -> list[str]:
|
||||
"""The CPU default's configure flags with exactly one backend flag flipped ON.
|
||||
|
||||
Returns a new list; `lock["build"]["configure_flags"]` (the deterministic
|
||||
CPU default DGR-029 locked) is never mutated.
|
||||
"""
|
||||
presets = lock.get("accelerator_presets", {})
|
||||
if name not in presets:
|
||||
raise DependencyError(f"unknown accelerator lane: {name}")
|
||||
backend_flag = presets[name]["backend_flag"]
|
||||
target = f"-D{backend_flag}="
|
||||
flags: list[str] = []
|
||||
replaced = False
|
||||
for flag in lock["build"]["configure_flags"]:
|
||||
if flag.startswith(target):
|
||||
flags.append(f"-D{backend_flag}=ON")
|
||||
replaced = True
|
||||
else:
|
||||
flags.append(flag)
|
||||
if not replaced:
|
||||
raise DependencyError(f"accelerator lane {name} backend flag {backend_flag} is not a locked base flag")
|
||||
return flags
|
||||
|
||||
|
||||
def accelerator_build(source: pathlib.Path, name: str, build_dir: pathlib.Path) -> pathlib.Path:
|
||||
"""Compile lane `name` into its own out-of-tree directory. Compile-only.
|
||||
|
||||
This never runs `smoke`/`ctest_lane`: exercising a binary linked against an
|
||||
accelerator backend would touch real hardware, and DGR-030 keeps every
|
||||
backend/model/recipe lane registered-dark (compiled, never certified)
|
||||
until a separate real-hardware certification record exists.
|
||||
"""
|
||||
lock = _load_lock()
|
||||
_patches(lock)
|
||||
_verify_source(source, lock, require_clean=False)
|
||||
_verify_patched_source(source, lock)
|
||||
expected_marker = source / "cmake/meshnet-patch-stack.cmake"
|
||||
if not expected_marker.is_file():
|
||||
raise DependencyError("patch stack is not applied: Meshnet CMake marker is absent")
|
||||
if build_dir.exists():
|
||||
raise DependencyError(f"accelerator build directory already exists; use a clean build dir: {build_dir}")
|
||||
status = accelerator_status(name, lock)
|
||||
if not status["available"]:
|
||||
raise DependencyError(f"accelerator lane {name} SDK is unavailable: {status['reason']}")
|
||||
flags = accelerator_configure_flags(lock, name)
|
||||
cmake = _cmake()
|
||||
_run(cmake, "-G", lock["build"]["generator"], "-S", str(source), "-B", str(build_dir), *flags)
|
||||
for target in lock["build"]["native_targets"]:
|
||||
_run(cmake, "--build", str(build_dir), "--target", target, "-j2")
|
||||
metadata = {
|
||||
"lane": name,
|
||||
"backend_flag": lock["accelerator_presets"][name]["backend_flag"],
|
||||
"commit": lock["commit"],
|
||||
"commit_tree": lock["commit_tree"],
|
||||
"patches": {patch.name: hashlib.sha256(patch.read_bytes()).hexdigest() for patch in _patches(lock)},
|
||||
"configure_flags": flags,
|
||||
"cmake": _run(cmake, "--version").splitlines()[0],
|
||||
"cxx": _run("c++", "--version").splitlines()[0],
|
||||
"sdk_binary": status["sdk_binary"],
|
||||
"model_downloads": False,
|
||||
"hardware_execution": False,
|
||||
"hardware_certified": False,
|
||||
"semantic_certification": False,
|
||||
"note": (
|
||||
"compiled only; no accelerator device was exercised or driven. "
|
||||
"Backend/model/recipe capability remains registered-dark until a "
|
||||
"separate real-hardware certification record exists (see "
|
||||
"DGR-041/053/067)."
|
||||
),
|
||||
}
|
||||
(build_dir / "meshnet-build-metadata.json").write_text(json.dumps(metadata, indent=2, sort_keys=True) + "\n")
|
||||
return build_dir
|
||||
|
||||
|
||||
def verify(workspace: pathlib.Path) -> None:
|
||||
"""Apply, verify, reverse, and leave the exact cached pin pristine."""
|
||||
source = fetch(workspace)
|
||||
@@ -562,6 +704,12 @@ def main() -> int:
|
||||
smoke_parser.add_argument("--binary", type=pathlib.Path, required=True)
|
||||
ctest_parser = subcommands.add_parser("ctest")
|
||||
ctest_parser.add_argument("--build-dir", type=pathlib.Path, required=True)
|
||||
accel_status_parser = subcommands.add_parser("accelerator-status")
|
||||
accel_status_parser.add_argument("--name", required=True)
|
||||
accel_build_parser = subcommands.add_parser("accelerator-build")
|
||||
accel_build_parser.add_argument("--name", required=True)
|
||||
accel_build_parser.add_argument("--source-dir", type=pathlib.Path, required=True)
|
||||
accel_build_parser.add_argument("--build-dir", type=pathlib.Path, required=True)
|
||||
reproduce_parser = subcommands.add_parser("reproduce")
|
||||
reproduce_parser.add_argument("--workspace", type=pathlib.Path, default=ROOT / "build/llama.cpp")
|
||||
args = parser.parse_args()
|
||||
@@ -582,6 +730,10 @@ def main() -> int:
|
||||
smoke(args.binary)
|
||||
elif args.command == "ctest":
|
||||
ctest_lane(args.build_dir)
|
||||
elif args.command == "accelerator-status":
|
||||
print(json.dumps(accelerator_status(args.name), indent=2, sort_keys=True))
|
||||
elif args.command == "accelerator-build":
|
||||
accelerator_build(args.source_dir, args.name, args.build_dir)
|
||||
else:
|
||||
reproduce(args.workspace)
|
||||
except DependencyError as error:
|
||||
|
||||
112
scripts/native_accelerator_matrix.py
Normal file
112
scripts/native_accelerator_matrix.py
Normal file
@@ -0,0 +1,112 @@
|
||||
#!/usr/bin/env python3
|
||||
"""DGR-030: native CI/build matrix over the CPU default plus accelerator lanes.
|
||||
|
||||
Runs the exact deterministic CPU lane DGR-029 locked (unchanged), then probes
|
||||
each accelerator preset (CUDA, ROCm, Vulkan, Metal) from `UPSTREAM_LOCK.json`
|
||||
and compiles the ones whose SDK is present on this machine into their own
|
||||
out-of-tree build directory.
|
||||
|
||||
A lane whose SDK is absent is reported as `skipped` with the exact probe
|
||||
reason, never treated as a false pass. A lane that compiles is reported as
|
||||
`built`, carrying exact compiler/SDK/upstream-pin/patch-stack/build-option
|
||||
evidence — never as a certified capability. This script never runs an
|
||||
accelerator binary and never certifies a backend/model/recipe: real-hardware
|
||||
certification is separate future work (DGR-041/053/067).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(ROOT / "scripts"))
|
||||
import llama_cpp_dependency as dep # noqa: E402
|
||||
|
||||
|
||||
def _cpu_lane(source: pathlib.Path, workspace: pathlib.Path) -> dict[str, Any]:
|
||||
build_dir = workspace.resolve() / "build"
|
||||
if build_dir.exists():
|
||||
return {
|
||||
"lane": "cpu",
|
||||
"status": "skipped",
|
||||
"reason": f"build directory already exists; remove for a clean rebuild: {build_dir}",
|
||||
}
|
||||
binary = dep.build(source, build_dir)
|
||||
dep.smoke(binary)
|
||||
dep.ctest_lane(build_dir)
|
||||
metadata = json.loads((build_dir / "meshnet-build-metadata.json").read_text())
|
||||
return {"lane": "cpu", "status": "built", "build_dir": str(build_dir), "metadata": metadata}
|
||||
|
||||
|
||||
def _accelerator_lane(source: pathlib.Path, workspace: pathlib.Path, name: str, lock: dict[str, Any]) -> dict[str, Any]:
|
||||
status = dep.accelerator_status(name, lock)
|
||||
if not status["available"]:
|
||||
return {"lane": name, "status": "skipped", "reason": status["reason"]}
|
||||
build_dir = workspace.resolve() / f"build-{name}"
|
||||
if build_dir.exists():
|
||||
return {
|
||||
"lane": name,
|
||||
"status": "skipped",
|
||||
"reason": f"build directory already exists; remove for a clean rebuild: {build_dir}",
|
||||
}
|
||||
dep.accelerator_build(source, name, build_dir)
|
||||
metadata = json.loads((build_dir / "meshnet-build-metadata.json").read_text())
|
||||
return {"lane": name, "status": "built", "build_dir": str(build_dir), "metadata": metadata}
|
||||
|
||||
|
||||
def run_matrix(workspace: pathlib.Path) -> dict[str, Any]:
|
||||
"""Fetch/apply once, run every lane, then always reverse the checkout."""
|
||||
source = dep.fetch(workspace)
|
||||
dep.apply(source)
|
||||
lanes: list[dict[str, Any]] = []
|
||||
try:
|
||||
lock = dep._load_lock()
|
||||
try:
|
||||
lanes.append(_cpu_lane(source, workspace))
|
||||
except dep.DependencyError as error:
|
||||
lanes.append({"lane": "cpu", "status": "failed", "reason": str(error)})
|
||||
for name in lock.get("accelerator_presets", {}):
|
||||
try:
|
||||
lanes.append(_accelerator_lane(source, workspace, name, lock))
|
||||
except dep.DependencyError as error:
|
||||
lanes.append({"lane": name, "status": "failed", "reason": str(error)})
|
||||
finally:
|
||||
dep.reverse(source)
|
||||
failed_lanes = [lane["lane"] for lane in lanes if lane["status"] == "failed"]
|
||||
return {
|
||||
"lanes": lanes,
|
||||
"hardware_certified": False,
|
||||
"note": (
|
||||
"A `built` lane means it compiled with the exact recorded compiler/SDK/"
|
||||
"upstream-pin/patch-stack/build-option evidence — it never means an "
|
||||
"accelerator device was exercised. Every backend/model/recipe lane "
|
||||
"stays registered-dark until a separate real-hardware certification "
|
||||
"record exists."
|
||||
),
|
||||
"failed_lanes": failed_lanes,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--workspace", type=pathlib.Path, default=ROOT / "build/llama.cpp")
|
||||
parser.add_argument("--out", type=pathlib.Path, default=None, help="also write the JSON report here")
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
report = run_matrix(args.workspace)
|
||||
except dep.DependencyError as error:
|
||||
print(f"DGR-030 dependency error: {error}", file=sys.stderr)
|
||||
return 2
|
||||
text = json.dumps(report, indent=2, sort_keys=True)
|
||||
print(text)
|
||||
if args.out:
|
||||
args.out.write_text(text + "\n")
|
||||
return 1 if report["failed_lanes"] else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
273
tests/shard_engine_contract.py
Normal file
273
tests/shard_engine_contract.py
Normal file
@@ -0,0 +1,273 @@
|
||||
"""Reusable ``ShardEngine`` lifecycle contract (DGR-031).
|
||||
|
||||
Any :class:`~meshnet_node.shard_engine.ShardEngine` implementation — the
|
||||
DGR-032 deterministic fixture, the DGR-037 llama.cpp binding, or a throwaway
|
||||
test double — can be checked against this contract by calling
|
||||
:func:`assert_shard_engine_contract` with a zero-argument factory that
|
||||
returns a fresh, unloaded engine instance. It proves the *lifecycle
|
||||
semantics* (load/capabilities gating, cache-miss/stale-epoch/cancel/release
|
||||
behavior, head vs. middle boundary-vs-token output) are identical across
|
||||
implementations. It says nothing about whether the numbers an implementation
|
||||
produces are numerically correct — that is DGR-036's job.
|
||||
|
||||
This module is not itself collected as a test file (it does not match
|
||||
``test_*.py``); import ``assert_shard_engine_contract`` from a real test file
|
||||
that supplies the engine factory, as ``test_shard_engine.py`` does here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Callable
|
||||
|
||||
from meshnet_node.shard_engine import (
|
||||
BoundaryBundle,
|
||||
DecodeRequest,
|
||||
EngineTensor,
|
||||
LoadRequest,
|
||||
PrefillRequest,
|
||||
ShardEngine,
|
||||
TokenOutput,
|
||||
)
|
||||
from meshnet_node.shard_lifecycle import CacheResult, StatusCode
|
||||
|
||||
|
||||
def assert_shard_engine_contract(make_engine: Callable[[], ShardEngine]) -> None:
|
||||
"""Run every lifecycle check against a fresh engine instance per check.
|
||||
|
||||
Each check gets its own ``make_engine()`` instance so one check's session
|
||||
state can never leak into another's.
|
||||
"""
|
||||
_assert_health_before_load_is_not_serving(make_engine())
|
||||
_assert_load_then_capabilities_matches_range(make_engine())
|
||||
_assert_prefill_then_decode_succeeds_and_is_deterministic(make_engine())
|
||||
_assert_middle_shard_accepts_boundary_bundle_not_token_ids(make_engine())
|
||||
_assert_decode_without_prefill_is_a_deterministic_cache_miss(make_engine())
|
||||
_assert_stale_epoch_is_rejected(make_engine())
|
||||
_assert_cancel_then_decode_is_rejected_and_cancel_is_idempotent(make_engine())
|
||||
_assert_release_then_decode_is_rejected_and_release_is_idempotent(make_engine())
|
||||
_assert_metrics_reports_cancelled_sessions(make_engine())
|
||||
|
||||
|
||||
def _load(
|
||||
engine: ShardEngine, *, shard_start: int = 0, shard_end: int = 3, total_layers: int = 4
|
||||
):
|
||||
result = engine.load(
|
||||
LoadRequest(
|
||||
artifact_path="fixture://contract-test",
|
||||
shard_start=shard_start,
|
||||
shard_end=shard_end,
|
||||
total_layers=total_layers,
|
||||
)
|
||||
)
|
||||
assert result.status.code is StatusCode.OK, result.status
|
||||
return result
|
||||
|
||||
|
||||
def _output_bytes(output: BoundaryBundle | TokenOutput | None) -> bytes:
|
||||
assert output is not None
|
||||
if isinstance(output, TokenOutput):
|
||||
return output.token_id.to_bytes(8, "big")
|
||||
return b"".join(tensor.data for tensor in output.tensors)
|
||||
|
||||
|
||||
def _assert_health_before_load_is_not_serving(engine: ShardEngine) -> None:
|
||||
health = engine.health()
|
||||
assert health.status.code is StatusCode.OK
|
||||
assert health.serving is False
|
||||
|
||||
|
||||
def _assert_load_then_capabilities_matches_range(engine: ShardEngine) -> None:
|
||||
_load(engine, shard_start=0, shard_end=3, total_layers=4)
|
||||
caps = engine.capabilities()
|
||||
assert caps.status.code is StatusCode.OK
|
||||
assert caps.shard_start == 0
|
||||
assert caps.shard_end == 3
|
||||
assert caps.total_layers == 4
|
||||
assert caps.is_head is True
|
||||
assert caps.is_tail is True
|
||||
assert caps.supports_mtp is False, "MTP must stay reserved-off until DGR-066"
|
||||
assert engine.health().serving is True
|
||||
|
||||
|
||||
def _assert_prefill_then_decode_succeeds_and_is_deterministic(engine: ShardEngine) -> None:
|
||||
_load(engine)
|
||||
prefill = engine.prefill(
|
||||
PrefillRequest(
|
||||
session_id="session-a",
|
||||
route_epoch=1,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_ids=(1, 2, 3),
|
||||
)
|
||||
)
|
||||
assert prefill.status.code is StatusCode.OK
|
||||
assert isinstance(prefill.output, (BoundaryBundle, TokenOutput))
|
||||
|
||||
decode = engine.decode(
|
||||
DecodeRequest(
|
||||
session_id="session-a",
|
||||
route_epoch=1,
|
||||
position=3,
|
||||
idempotency_step=1,
|
||||
token_id=4,
|
||||
)
|
||||
)
|
||||
assert decode.status.code is StatusCode.OK
|
||||
assert decode.cache_result is CacheResult.HIT
|
||||
assert isinstance(decode.output, (BoundaryBundle, TokenOutput))
|
||||
|
||||
# Determinism: the identical prefill replayed on a brand-new session
|
||||
# produces byte-identical output. The transform is a pure function of
|
||||
# its inputs, not of hidden randomness or cross-session state.
|
||||
replay = engine.prefill(
|
||||
PrefillRequest(
|
||||
session_id="session-b",
|
||||
route_epoch=1,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_ids=(1, 2, 3),
|
||||
)
|
||||
)
|
||||
assert _output_bytes(replay.output) == _output_bytes(prefill.output)
|
||||
|
||||
|
||||
def _assert_middle_shard_accepts_boundary_bundle_not_token_ids(engine: ShardEngine) -> None:
|
||||
_load(engine, shard_start=1, shard_end=2, total_layers=8)
|
||||
caps = engine.capabilities()
|
||||
assert caps.is_head is False
|
||||
assert caps.is_tail is False
|
||||
|
||||
input_bundle = BoundaryBundle(
|
||||
tensors=(
|
||||
EngineTensor(name="hidden_states", shape=(1, 3), dtype="bfloat16", data=b"\x00" * 8),
|
||||
),
|
||||
architecture="dense",
|
||||
boundary_point="pre_tail_residual",
|
||||
)
|
||||
result = engine.prefill(
|
||||
PrefillRequest(
|
||||
session_id="session-middle",
|
||||
route_epoch=1,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
input=input_bundle,
|
||||
)
|
||||
)
|
||||
assert result.status.code is StatusCode.OK
|
||||
assert isinstance(result.output, BoundaryBundle), "a non-tail shard must hand off a boundary bundle, never a sampled token"
|
||||
|
||||
|
||||
def _assert_decode_without_prefill_is_a_deterministic_cache_miss(engine: ShardEngine) -> None:
|
||||
_load(engine)
|
||||
result = engine.decode(
|
||||
DecodeRequest(
|
||||
session_id="never-opened",
|
||||
route_epoch=1,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_id=9,
|
||||
)
|
||||
)
|
||||
assert result.status.code is not StatusCode.OK
|
||||
assert result.cache_result is CacheResult.MISS
|
||||
assert result.output is None
|
||||
|
||||
|
||||
def _assert_stale_epoch_is_rejected(engine: ShardEngine) -> None:
|
||||
_load(engine)
|
||||
engine.prefill(
|
||||
PrefillRequest(
|
||||
session_id="session-epoch",
|
||||
route_epoch=5,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_ids=(1,),
|
||||
)
|
||||
)
|
||||
stale = engine.decode(
|
||||
DecodeRequest(
|
||||
session_id="session-epoch",
|
||||
route_epoch=4,
|
||||
position=1,
|
||||
idempotency_step=1,
|
||||
token_id=2,
|
||||
)
|
||||
)
|
||||
assert stale.status.code is not StatusCode.OK
|
||||
assert stale.output is None
|
||||
|
||||
|
||||
def _assert_cancel_then_decode_is_rejected_and_cancel_is_idempotent(engine: ShardEngine) -> None:
|
||||
_load(engine)
|
||||
engine.prefill(
|
||||
PrefillRequest(
|
||||
session_id="session-cancel",
|
||||
route_epoch=1,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_ids=(1,),
|
||||
)
|
||||
)
|
||||
cancelled = engine.cancel("session-cancel")
|
||||
assert cancelled.code is StatusCode.CANCELLED
|
||||
|
||||
after = engine.decode(
|
||||
DecodeRequest(
|
||||
session_id="session-cancel",
|
||||
route_epoch=1,
|
||||
position=1,
|
||||
idempotency_step=1,
|
||||
token_id=2,
|
||||
)
|
||||
)
|
||||
assert after.status.code is StatusCode.CANCELLED
|
||||
assert after.output is None
|
||||
|
||||
again = engine.cancel("session-cancel")
|
||||
assert again.code is StatusCode.CANCELLED
|
||||
|
||||
|
||||
def _assert_release_then_decode_is_rejected_and_release_is_idempotent(engine: ShardEngine) -> None:
|
||||
_load(engine)
|
||||
engine.prefill(
|
||||
PrefillRequest(
|
||||
session_id="session-release",
|
||||
route_epoch=1,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_ids=(1,),
|
||||
)
|
||||
)
|
||||
released = engine.release("session-release")
|
||||
assert released.code is StatusCode.OK
|
||||
|
||||
after = engine.decode(
|
||||
DecodeRequest(
|
||||
session_id="session-release",
|
||||
route_epoch=1,
|
||||
position=1,
|
||||
idempotency_step=1,
|
||||
token_id=2,
|
||||
)
|
||||
)
|
||||
assert after.status.code is not StatusCode.OK
|
||||
|
||||
again = engine.release("session-release")
|
||||
assert again.code is StatusCode.OK
|
||||
|
||||
|
||||
def _assert_metrics_reports_cancelled_sessions(engine: ShardEngine) -> None:
|
||||
_load(engine)
|
||||
engine.prefill(
|
||||
PrefillRequest(
|
||||
session_id="session-metrics",
|
||||
route_epoch=1,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_ids=(1,),
|
||||
)
|
||||
)
|
||||
engine.cancel("session-metrics")
|
||||
metrics = engine.metrics()
|
||||
assert metrics.status.code is StatusCode.OK
|
||||
assert metrics.cancelled_sessions >= 1
|
||||
200
tests/test_fake_shard_engine.py
Normal file
200
tests/test_fake_shard_engine.py
Normal file
@@ -0,0 +1,200 @@
|
||||
"""DGR-032 ``FakeShardEngine`` tests.
|
||||
|
||||
``FakeShardEngine`` obeys the exact same lifecycle contract every
|
||||
``ShardEngine`` implementation must (see ``shard_engine_contract.py``); the
|
||||
tests here additionally cover this story's own scope: head/middle/tail
|
||||
output shape, isolated multi-session state, and the delay/memory-pressure/
|
||||
malformed-output/crash fault-injection knobs. None of this is real-model
|
||||
evidence — see the module docstring in ``fake_shard_engine.py`` and
|
||||
``test_fake_shard_engine_declares_fixture_evidence_class`` below, which pins
|
||||
the marker DGR-036 will rely on to tell a fixture engine apart from a real
|
||||
one when it certifies fixture-vs-real parity.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from meshnet_node.fake_shard_engine import (
|
||||
MALFORMED_TOKEN_ID_FLOOR,
|
||||
TOKEN_ID_VOCAB_SIZE,
|
||||
FakeShardEngine,
|
||||
FakeShardEngineConfig,
|
||||
)
|
||||
from meshnet_node.shard_engine import (
|
||||
BoundaryBundle,
|
||||
DecodeRequest,
|
||||
EngineTensor,
|
||||
LoadRequest,
|
||||
PrefillRequest,
|
||||
TokenOutput,
|
||||
)
|
||||
from meshnet_node.shard_lifecycle import StatusCode
|
||||
|
||||
from shard_engine_contract import assert_shard_engine_contract
|
||||
|
||||
|
||||
def _load(engine: FakeShardEngine, *, shard_start=0, shard_end=3, total_layers=4, recipe=None):
|
||||
return engine.load(
|
||||
LoadRequest(
|
||||
artifact_path="fixture://fake-shard-engine",
|
||||
shard_start=shard_start,
|
||||
shard_end=shard_end,
|
||||
total_layers=total_layers,
|
||||
recipe=recipe or {},
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def test_fake_shard_engine_obeys_the_shared_shard_engine_contract():
|
||||
assert_shard_engine_contract(FakeShardEngine)
|
||||
|
||||
|
||||
def test_fake_shard_engine_declares_fixture_evidence_class():
|
||||
# DGR-036's fixture-vs-real parity check needs a structural way to tell
|
||||
# a fixture engine apart from a real one; this constant is that marker.
|
||||
assert FakeShardEngine.EVIDENCE_CLASS == "fixture"
|
||||
|
||||
|
||||
def test_head_shard_returns_boundary_bundle_with_post_head_residual_point():
|
||||
engine = FakeShardEngine()
|
||||
_load(engine, shard_start=0, shard_end=1, total_layers=4)
|
||||
result = engine.prefill(
|
||||
PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2))
|
||||
)
|
||||
assert result.status.code is StatusCode.OK
|
||||
assert isinstance(result.output, BoundaryBundle)
|
||||
assert result.output.boundary_point == "post_head_residual"
|
||||
|
||||
|
||||
def test_middle_shard_returns_boundary_bundle_and_passes_through_token_sideband():
|
||||
engine = FakeShardEngine()
|
||||
_load(engine, shard_start=1, shard_end=2, total_layers=8)
|
||||
bundle_in = BoundaryBundle(
|
||||
tensors=(EngineTensor(name="hidden_states", shape=(1, 2), dtype="bfloat16", data=b"\x01\x02\x03\x04"),),
|
||||
architecture="dense",
|
||||
boundary_point="pre_tail_residual",
|
||||
token_id_sideband=(7, 8, 9),
|
||||
)
|
||||
result = engine.prefill(
|
||||
PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, input=bundle_in)
|
||||
)
|
||||
assert result.status.code is StatusCode.OK
|
||||
assert isinstance(result.output, BoundaryBundle)
|
||||
assert result.output.boundary_point == "post_middle_residual"
|
||||
assert result.output.token_id_sideband == (7, 8, 9)
|
||||
|
||||
|
||||
def test_tail_shard_returns_token_output_within_advertised_vocab():
|
||||
engine = FakeShardEngine()
|
||||
_load(engine, shard_start=0, shard_end=3, total_layers=4)
|
||||
result = engine.prefill(
|
||||
PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2, 3))
|
||||
)
|
||||
assert isinstance(result.output, TokenOutput)
|
||||
assert 0 <= result.output.token_id < TOKEN_ID_VOCAB_SIZE
|
||||
|
||||
|
||||
def test_session_state_is_isolated_between_two_concurrent_sessions():
|
||||
engine = FakeShardEngine()
|
||||
_load(engine)
|
||||
engine.prefill(PrefillRequest(session_id="a", route_epoch=5, position=0, idempotency_step=0, token_ids=(1,)))
|
||||
engine.prefill(PrefillRequest(session_id="b", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,)))
|
||||
|
||||
# A stale epoch against session "a" must not affect session "b" at all.
|
||||
stale = engine.decode(DecodeRequest(session_id="a", route_epoch=4, position=1, idempotency_step=1, token_id=2))
|
||||
assert stale.status.code is StatusCode.FAILED_PRECONDITION
|
||||
|
||||
still_fine = engine.decode(DecodeRequest(session_id="b", route_epoch=1, position=1, idempotency_step=1, token_id=2))
|
||||
assert still_fine.status.code is StatusCode.OK
|
||||
|
||||
engine.cancel("a")
|
||||
after_cancel_b = engine.decode(
|
||||
DecodeRequest(session_id="b", route_epoch=1, position=2, idempotency_step=2, token_id=3)
|
||||
)
|
||||
assert after_cancel_b.status.code is StatusCode.OK, "cancelling session a must not cancel session b"
|
||||
|
||||
|
||||
def test_step_delay_seconds_invokes_the_configured_sleep_hook():
|
||||
calls: list[float] = []
|
||||
engine = FakeShardEngine(FakeShardEngineConfig(step_delay_seconds=0.25, sleep=calls.append))
|
||||
_load(engine)
|
||||
engine.prefill(PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,)))
|
||||
engine.decode(DecodeRequest(session_id="s", route_epoch=1, position=1, idempotency_step=1, token_id=2))
|
||||
assert calls == [0.25, 0.25]
|
||||
|
||||
|
||||
def test_memory_budget_bytes_trips_deterministic_resource_exhausted():
|
||||
engine = FakeShardEngine(FakeShardEngineConfig(memory_budget_bytes=8))
|
||||
_load(engine)
|
||||
first = engine.prefill(
|
||||
PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,))
|
||||
)
|
||||
assert first.status.code is StatusCode.OK # 8 bytes used, exactly at budget
|
||||
|
||||
second = engine.decode(DecodeRequest(session_id="s", route_epoch=1, position=1, idempotency_step=1, token_id=2))
|
||||
assert second.status.code is StatusCode.RESOURCE_EXHAUSTED
|
||||
assert second.status.retryable is True
|
||||
assert second.output is None
|
||||
|
||||
|
||||
def test_malformed_output_is_structurally_valid_but_semantically_wrong_for_tail():
|
||||
engine = FakeShardEngine(FakeShardEngineConfig(malformed_output=True))
|
||||
_load(engine, shard_start=0, shard_end=3, total_layers=4)
|
||||
result = engine.prefill(
|
||||
PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2, 3))
|
||||
)
|
||||
assert result.status.code is StatusCode.OK
|
||||
assert isinstance(result.output, TokenOutput)
|
||||
assert result.output.token_id >= MALFORMED_TOKEN_ID_FLOOR
|
||||
|
||||
|
||||
def test_malformed_output_is_structurally_valid_but_semantically_wrong_for_boundary_bundle():
|
||||
engine = FakeShardEngine(FakeShardEngineConfig(malformed_output=True))
|
||||
_load(engine, shard_start=0, shard_end=1, total_layers=4)
|
||||
result = engine.prefill(
|
||||
PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1, 2, 3))
|
||||
)
|
||||
assert result.status.code is StatusCode.OK
|
||||
assert isinstance(result.output, BoundaryBundle)
|
||||
assert result.output.architecture.startswith("malformed:")
|
||||
assert len(result.output.tensors[0].data) == 1
|
||||
|
||||
|
||||
def test_crash_after_calls_raises_instead_of_returning_a_structured_status():
|
||||
engine = FakeShardEngine(FakeShardEngineConfig(crash_after_calls=2))
|
||||
_load(engine)
|
||||
ok = engine.prefill(PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,)))
|
||||
assert ok.status.code is StatusCode.OK
|
||||
|
||||
with pytest.raises(RuntimeError):
|
||||
engine.decode(DecodeRequest(session_id="s", route_epoch=1, position=1, idempotency_step=1, token_id=2))
|
||||
|
||||
|
||||
def test_crash_exception_factory_is_configurable():
|
||||
class _SimulatedSegfault(Exception):
|
||||
pass
|
||||
|
||||
engine = FakeShardEngine(
|
||||
FakeShardEngineConfig(crash_after_calls=1, crash_exception_factory=_SimulatedSegfault)
|
||||
)
|
||||
_load(engine)
|
||||
with pytest.raises(_SimulatedSegfault):
|
||||
engine.prefill(PrefillRequest(session_id="s", route_epoch=1, position=0, idempotency_step=0, token_ids=(1,)))
|
||||
|
||||
|
||||
def test_config_rejects_invalid_knob_values():
|
||||
with pytest.raises(ValueError):
|
||||
FakeShardEngineConfig(step_delay_seconds=-1.0)
|
||||
with pytest.raises(ValueError):
|
||||
FakeShardEngineConfig(memory_budget_bytes=-1)
|
||||
with pytest.raises(ValueError):
|
||||
FakeShardEngineConfig(crash_after_calls=0)
|
||||
|
||||
|
||||
def test_load_result_and_capabilities_report_recipe_architecture():
|
||||
engine = FakeShardEngine()
|
||||
load_result = _load(engine, recipe={"architecture": "deepseek-v4-flash"})
|
||||
assert load_result.architecture == "deepseek-v4-flash"
|
||||
caps = engine.capabilities()
|
||||
assert caps.architecture == "deepseek-v4-flash"
|
||||
@@ -334,3 +334,157 @@ def test_ctest_lane_raises_an_actionable_error_for_a_failing_named_test(tmp_path
|
||||
assert "meshnet-fixture-fail" in str(error)
|
||||
else:
|
||||
raise AssertionError("a failing named CTest lane must raise DependencyError")
|
||||
|
||||
|
||||
def test_accelerator_presets_isolate_one_backend_without_touching_the_cpu_default() -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
presets = lock["accelerator_presets"]
|
||||
|
||||
assert set(presets) == {"cuda", "rocm", "vulkan", "metal"}
|
||||
base_flags = list(lock["build"]["configure_flags"])
|
||||
base_values = dict(flag[len("-D"):].split("=", 1) for flag in base_flags)
|
||||
|
||||
for name, preset in presets.items():
|
||||
flags = dependency.accelerator_configure_flags(lock, name)
|
||||
|
||||
# The CPU default's own flag list is never mutated by building a preset.
|
||||
assert lock["build"]["configure_flags"] == base_flags
|
||||
|
||||
new_values = dict(flag[len("-D"):].split("=", 1) for flag in flags)
|
||||
backend_flag = preset["backend_flag"]
|
||||
assert base_values[backend_flag] == "OFF"
|
||||
assert new_values[backend_flag] == "ON"
|
||||
for other_flag, value in base_values.items():
|
||||
if other_flag != backend_flag:
|
||||
assert new_values[other_flag] == value, f"{name}: {other_flag} drifted from the CPU default"
|
||||
|
||||
|
||||
def test_accelerator_configure_flags_rejects_an_unknown_lane() -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
try:
|
||||
dependency.accelerator_configure_flags(lock, "bogus")
|
||||
except dependency.DependencyError as error:
|
||||
assert "unknown accelerator lane" in str(error)
|
||||
else:
|
||||
raise AssertionError("an unknown accelerator lane must be refused")
|
||||
|
||||
|
||||
def test_accelerator_status_reports_unavailable_sdks_without_raising(monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
monkeypatch.setattr(dependency.shutil, "which", lambda name: None)
|
||||
|
||||
for name in ("cuda", "rocm", "vulkan"):
|
||||
env_var = lock["accelerator_presets"][name]["sdk_probe"]["env_var"]
|
||||
monkeypatch.delenv(env_var, raising=False)
|
||||
binary = lock["accelerator_presets"][name]["sdk_probe"]["binary"]
|
||||
assert dependency.accelerator_status(name, lock) == {
|
||||
"lane": name,
|
||||
"available": False,
|
||||
"reason": f"{binary} is unavailable on PATH",
|
||||
}
|
||||
|
||||
monkeypatch.setattr(dependency.sys, "platform", "linux")
|
||||
assert dependency.accelerator_status("metal", lock) == {
|
||||
"lane": "metal",
|
||||
"available": False,
|
||||
"reason": "platform 'linux' is not 'darwin'",
|
||||
}
|
||||
|
||||
|
||||
def test_accelerator_status_honors_an_explicit_sdk_override(tmp_path, monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
fake_nvcc = tmp_path / "nvcc"
|
||||
fake_nvcc.write_text("#!/bin/sh\nexit 0\n")
|
||||
fake_nvcc.chmod(0o755)
|
||||
monkeypatch.setenv("CUDACXX", str(fake_nvcc))
|
||||
|
||||
assert dependency.accelerator_status("cuda", lock) == {
|
||||
"lane": "cuda",
|
||||
"available": True,
|
||||
"sdk_binary": str(fake_nvcc),
|
||||
}
|
||||
|
||||
|
||||
def test_accelerator_status_rejects_an_unknown_lane() -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
try:
|
||||
dependency.accelerator_status("bogus", lock)
|
||||
except dependency.DependencyError as error:
|
||||
assert "unknown accelerator lane" in str(error)
|
||||
else:
|
||||
raise AssertionError("an unknown accelerator lane must be refused")
|
||||
|
||||
|
||||
def test_accelerator_build_refuses_to_compile_an_unavailable_lane(tmp_path, monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
source = tmp_path / "source"
|
||||
(source / "cmake").mkdir(parents=True)
|
||||
(source / "cmake" / "meshnet-patch-stack.cmake").write_text("# marker\n")
|
||||
monkeypatch.setattr(dependency, "_verify_source", lambda *a, **k: None)
|
||||
monkeypatch.setattr(dependency, "_verify_patched_source", lambda *a, **k: None)
|
||||
monkeypatch.setattr(dependency.shutil, "which", lambda name: None)
|
||||
monkeypatch.delenv("CUDACXX", raising=False)
|
||||
|
||||
build_dir = tmp_path / "build-cuda"
|
||||
try:
|
||||
dependency.accelerator_build(source, "cuda", build_dir)
|
||||
except dependency.DependencyError as error:
|
||||
assert "SDK is unavailable" in str(error)
|
||||
else:
|
||||
raise AssertionError("accelerator_build must refuse to compile an unavailable lane")
|
||||
assert not build_dir.exists()
|
||||
|
||||
|
||||
@requires_cmake
|
||||
def test_accelerator_build_compiles_the_available_lane_with_isolated_evidence(tmp_path, monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
|
||||
# A tiny synthetic project stands in for the patched llama.cpp checkout —
|
||||
# it only needs the meshnet patch-stack marker and one target, proving
|
||||
# accelerator_build's configure/build/evidence wiring without a multi-minute
|
||||
# llama.cpp compile or a real GPU SDK.
|
||||
source = tmp_path / "source"
|
||||
(source / "cmake").mkdir(parents=True)
|
||||
(source / "cmake" / "meshnet-patch-stack.cmake").write_text("# marker\n")
|
||||
(source / "CMakeLists.txt").write_text(
|
||||
"cmake_minimum_required(VERSION 3.14)\n"
|
||||
"project(accelerator_lane_fixture NONE)\n"
|
||||
"option(GGML_CUDA \"\" OFF)\n"
|
||||
"if(GGML_CUDA)\n"
|
||||
" file(WRITE ${CMAKE_BINARY_DIR}/lane-flag-on.txt \"on\")\n"
|
||||
"endif()\n"
|
||||
"add_custom_target(fixture-target ALL COMMAND ${CMAKE_COMMAND} -E true)\n"
|
||||
)
|
||||
|
||||
base_lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
fake_lock = dict(base_lock)
|
||||
fake_lock["build"] = {
|
||||
**base_lock["build"],
|
||||
"generator": "Unix Makefiles",
|
||||
"configure_flags": ["-DGGML_CUDA=OFF"],
|
||||
"native_targets": ["fixture-target"],
|
||||
}
|
||||
monkeypatch.setattr(dependency, "_load_lock", lambda: fake_lock)
|
||||
monkeypatch.setattr(dependency, "_patches", lambda lock: [])
|
||||
monkeypatch.setattr(dependency, "_verify_source", lambda *a, **k: None)
|
||||
monkeypatch.setattr(dependency, "_verify_patched_source", lambda *a, **k: None)
|
||||
monkeypatch.setenv("CUDACXX", str(dependency._cmake()))
|
||||
|
||||
build_dir = tmp_path / "build-cuda"
|
||||
result = dependency.accelerator_build(source, "cuda", build_dir)
|
||||
|
||||
assert result == build_dir
|
||||
assert (build_dir / "lane-flag-on.txt").is_file()
|
||||
metadata = json.loads((build_dir / "meshnet-build-metadata.json").read_text())
|
||||
assert metadata["lane"] == "cuda"
|
||||
assert metadata["backend_flag"] == "GGML_CUDA"
|
||||
assert metadata["configure_flags"] == ["-DGGML_CUDA=ON"]
|
||||
assert metadata["hardware_execution"] is False
|
||||
assert metadata["hardware_certified"] is False
|
||||
assert metadata["semantic_certification"] is False
|
||||
assert "registered-dark" in metadata["note"]
|
||||
|
||||
171
tests/test_native_accelerator_matrix.py
Normal file
171
tests/test_native_accelerator_matrix.py
Normal file
@@ -0,0 +1,171 @@
|
||||
"""Offline behavior tests for DGR-030's native CI/build matrix orchestration.
|
||||
|
||||
These tests never fetch or compile llama.cpp: `llama_cpp_dependency`'s fetch/
|
||||
apply/reverse/build/smoke/ctest_lane/accelerator_status/accelerator_build are
|
||||
stubbed so the matrix's own lane-reporting and cleanup contract is exercised
|
||||
in isolation. The real compile path is covered separately by
|
||||
`tests/test_llama_cpp_dependency.py`'s `accelerator_build`/CPU-lane tests and
|
||||
by a live run recorded in the DGR-030 evidence README.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||
MATRIX_SCRIPT = ROOT / "scripts/native_accelerator_matrix.py"
|
||||
DEP_SCRIPT = ROOT / "scripts/llama_cpp_dependency.py"
|
||||
|
||||
|
||||
def _load_matrix_module(monkeypatch):
|
||||
"""Load private copies of both modules with a controllable `dep`.
|
||||
|
||||
`native_accelerator_matrix.py` does `import llama_cpp_dependency as dep`
|
||||
after inserting `scripts/` onto `sys.path`; pre-registering our own module
|
||||
instance under that name in `sys.modules` (undone by monkeypatch at
|
||||
teardown) makes the matrix module bind to the stub instead of importing a
|
||||
fresh copy of the real dependency module.
|
||||
"""
|
||||
dep_spec = importlib.util.spec_from_file_location("llama_cpp_dependency_matrix_dep", DEP_SCRIPT)
|
||||
dep = importlib.util.module_from_spec(dep_spec)
|
||||
dep_spec.loader.exec_module(dep)
|
||||
monkeypatch.setitem(sys.modules, "llama_cpp_dependency", dep)
|
||||
|
||||
matrix_spec = importlib.util.spec_from_file_location("native_accelerator_matrix", MATRIX_SCRIPT)
|
||||
matrix = importlib.util.module_from_spec(matrix_spec)
|
||||
matrix_spec.loader.exec_module(matrix)
|
||||
return matrix, dep
|
||||
|
||||
|
||||
def test_matrix_reports_unavailable_accelerator_sdks_as_skipped_not_false_success(tmp_path, monkeypatch) -> None:
|
||||
matrix, dep = _load_matrix_module(monkeypatch)
|
||||
|
||||
workspace = tmp_path / "llama.cpp"
|
||||
source = workspace / "source"
|
||||
source.mkdir(parents=True)
|
||||
calls: list = []
|
||||
|
||||
monkeypatch.setattr(dep, "fetch", lambda ws: source)
|
||||
monkeypatch.setattr(dep, "apply", lambda src: calls.append(("apply", src)))
|
||||
monkeypatch.setattr(dep, "reverse", lambda src: calls.append(("reverse", src)))
|
||||
monkeypatch.setattr(
|
||||
dep,
|
||||
"_load_lock",
|
||||
lambda: {"accelerator_presets": {"cuda": {}, "rocm": {}, "vulkan": {}, "metal": {}}},
|
||||
)
|
||||
|
||||
def _cpu_build(src, build_dir):
|
||||
build_dir.mkdir(parents=True)
|
||||
(build_dir / "meshnet-build-metadata.json").write_text(json.dumps({"lane": "cpu"}))
|
||||
return build_dir / "bin/llama-gguf-hash"
|
||||
|
||||
monkeypatch.setattr(dep, "build", _cpu_build)
|
||||
monkeypatch.setattr(dep, "smoke", lambda binary: calls.append(("smoke", binary)))
|
||||
monkeypatch.setattr(dep, "ctest_lane", lambda build_dir: calls.append(("ctest", build_dir)))
|
||||
monkeypatch.setattr(
|
||||
dep,
|
||||
"accelerator_status",
|
||||
lambda name, lock: {"lane": name, "available": False, "reason": f"{name} SDK is unavailable on PATH"},
|
||||
)
|
||||
|
||||
report = matrix.run_matrix(workspace)
|
||||
|
||||
assert report["lanes"][0] == {
|
||||
"lane": "cpu",
|
||||
"status": "built",
|
||||
"build_dir": str((workspace / "build").resolve()),
|
||||
"metadata": {"lane": "cpu"},
|
||||
}
|
||||
accelerator_lanes = {lane["lane"]: lane for lane in report["lanes"][1:]}
|
||||
assert set(accelerator_lanes) == {"cuda", "rocm", "vulkan", "metal"}
|
||||
for name, lane in accelerator_lanes.items():
|
||||
assert lane["status"] == "skipped"
|
||||
assert "unavailable" in lane["reason"]
|
||||
|
||||
assert report["failed_lanes"] == []
|
||||
assert report["hardware_certified"] is False
|
||||
assert ("reverse", source) in calls # cleanup always runs
|
||||
# Only the CPU lane is ever smoke-tested/ctested; skipped accelerator lanes are not.
|
||||
smoke_calls = [call for call in calls if call[0] == "smoke"]
|
||||
ctest_calls = [call for call in calls if call[0] == "ctest"]
|
||||
assert smoke_calls == [("smoke", (workspace.resolve() / "build" / "bin/llama-gguf-hash"))]
|
||||
assert ctest_calls == [("ctest", (workspace.resolve() / "build"))]
|
||||
|
||||
|
||||
def test_matrix_compiles_an_available_accelerator_lane_without_smoke_or_ctest(tmp_path, monkeypatch) -> None:
|
||||
matrix, dep = _load_matrix_module(monkeypatch)
|
||||
|
||||
workspace = tmp_path / "llama.cpp"
|
||||
source = workspace / "source"
|
||||
source.mkdir(parents=True)
|
||||
calls: list = []
|
||||
|
||||
monkeypatch.setattr(dep, "fetch", lambda ws: source)
|
||||
monkeypatch.setattr(dep, "apply", lambda src: None)
|
||||
monkeypatch.setattr(dep, "reverse", lambda src: calls.append("reverse"))
|
||||
monkeypatch.setattr(dep, "_load_lock", lambda: {"accelerator_presets": {"cuda": {}}})
|
||||
monkeypatch.setattr(
|
||||
matrix,
|
||||
"_cpu_lane",
|
||||
lambda src, ws: {"lane": "cpu", "status": "skipped", "reason": "pre-existing build dir"},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
dep, "accelerator_status", lambda name, lock: {"lane": name, "available": True, "sdk_binary": "/fake/nvcc"}
|
||||
)
|
||||
|
||||
def _accelerator_build(src, name, build_dir):
|
||||
calls.append(("accelerator_build", name))
|
||||
build_dir.mkdir(parents=True)
|
||||
(build_dir / "meshnet-build-metadata.json").write_text(
|
||||
json.dumps({"lane": name, "hardware_certified": False})
|
||||
)
|
||||
return build_dir
|
||||
|
||||
monkeypatch.setattr(dep, "accelerator_build", _accelerator_build)
|
||||
monkeypatch.setattr(dep, "smoke", lambda binary: calls.append(("smoke", binary)))
|
||||
monkeypatch.setattr(dep, "ctest_lane", lambda build_dir: calls.append(("ctest", build_dir)))
|
||||
|
||||
report = matrix.run_matrix(workspace)
|
||||
|
||||
assert report["lanes"][1]["lane"] == "cuda"
|
||||
assert report["lanes"][1]["status"] == "built"
|
||||
assert report["lanes"][1]["metadata"]["hardware_certified"] is False
|
||||
assert ("accelerator_build", "cuda") in calls
|
||||
assert not any(call[0] in ("smoke", "ctest") for call in calls if isinstance(call, tuple))
|
||||
assert "reverse" in calls
|
||||
|
||||
|
||||
def test_matrix_reports_a_lane_failure_without_aborting_the_others_or_skipping_reverse(tmp_path, monkeypatch) -> None:
|
||||
matrix, dep = _load_matrix_module(monkeypatch)
|
||||
|
||||
workspace = tmp_path / "llama.cpp"
|
||||
source = workspace / "source"
|
||||
source.mkdir(parents=True)
|
||||
calls: list = []
|
||||
|
||||
monkeypatch.setattr(dep, "fetch", lambda ws: source)
|
||||
monkeypatch.setattr(dep, "apply", lambda src: None)
|
||||
monkeypatch.setattr(dep, "reverse", lambda src: calls.append("reverse"))
|
||||
monkeypatch.setattr(dep, "_load_lock", lambda: {"accelerator_presets": {"cuda": {}, "vulkan": {}}})
|
||||
|
||||
def _cpu_lane_raises(src, ws):
|
||||
raise dep.DependencyError("simulated cpu compile failure")
|
||||
|
||||
monkeypatch.setattr(matrix, "_cpu_lane", _cpu_lane_raises)
|
||||
monkeypatch.setattr(
|
||||
dep,
|
||||
"accelerator_status",
|
||||
lambda name, lock: {"lane": name, "available": False, "reason": f"{name} SDK is unavailable on PATH"},
|
||||
)
|
||||
|
||||
report = matrix.run_matrix(workspace)
|
||||
|
||||
assert report["lanes"][0] == {"lane": "cpu", "status": "failed", "reason": "simulated cpu compile failure"}
|
||||
assert report["failed_lanes"] == ["cpu"]
|
||||
accelerator_statuses = {lane["lane"]: lane["status"] for lane in report["lanes"][1:]}
|
||||
assert accelerator_statuses == {"cuda": "skipped", "vulkan": "skipped"}
|
||||
assert "reverse" in calls
|
||||
241
tests/test_shard_engine.py
Normal file
241
tests/test_shard_engine.py
Normal file
@@ -0,0 +1,241 @@
|
||||
"""DGR-031 ``ShardEngine`` contract tests.
|
||||
|
||||
``_ReferenceEngine`` below is a minimal, in-memory ``ShardEngine`` that exists
|
||||
only to prove :func:`assert_shard_engine_contract` is non-vacuous and to pin
|
||||
the abstract contract's own validation rules. It is deliberately not the
|
||||
DGR-032 deterministic fixture (delay/memory-pressure/malformed/crash
|
||||
injection, full session/epoch modeling for the fake worker) — that is a
|
||||
separate, larger story. DGR-032 and DGR-037 are expected to import
|
||||
``assert_shard_engine_contract`` from ``tests/shard_engine_contract.py``
|
||||
against their own engines.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
|
||||
import pytest
|
||||
|
||||
from meshnet_node.shard_engine import (
|
||||
ArchitectureAuxStateHook,
|
||||
BoundaryBundle,
|
||||
DecodeRequest,
|
||||
EngineCapabilities,
|
||||
EngineTensor,
|
||||
HealthResult,
|
||||
LoadRequest,
|
||||
LoadResult,
|
||||
MetricsResult,
|
||||
MtpHook,
|
||||
PrefillRequest,
|
||||
ShardEngine,
|
||||
StepResult,
|
||||
TokenOutput,
|
||||
)
|
||||
from meshnet_node.shard_lifecycle import CacheResult, StatusCode, StructuredStatus
|
||||
|
||||
from shard_engine_contract import assert_shard_engine_contract
|
||||
|
||||
|
||||
class _ReferenceEngine(ShardEngine):
|
||||
"""Minimal in-memory engine used only to exercise the shared contract."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._loaded: LoadRequest | None = None
|
||||
self._sessions: dict[str, dict] = {}
|
||||
self._cancelled_total = 0
|
||||
|
||||
def load(self, request: LoadRequest) -> LoadResult:
|
||||
self._loaded = request
|
||||
return LoadResult(
|
||||
status=StructuredStatus(StatusCode.OK, "loaded"),
|
||||
effective_start=request.shard_start,
|
||||
architecture="dense",
|
||||
)
|
||||
|
||||
def capabilities(self) -> EngineCapabilities:
|
||||
if self._loaded is None:
|
||||
return EngineCapabilities(status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "not loaded"))
|
||||
request = self._loaded
|
||||
return EngineCapabilities(
|
||||
status=StructuredStatus(StatusCode.OK, "ready"),
|
||||
shard_start=request.shard_start,
|
||||
shard_end=request.shard_end,
|
||||
effective_start=request.shard_start,
|
||||
total_layers=request.total_layers,
|
||||
architecture="dense",
|
||||
max_concurrent_sessions=8,
|
||||
max_context_tokens=131072,
|
||||
supports_mtp=False,
|
||||
)
|
||||
|
||||
def prefill(self, request: PrefillRequest) -> StepResult:
|
||||
if self._loaded is None:
|
||||
return StepResult(status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "engine not loaded"))
|
||||
self._sessions[request.session_id] = {"epoch": request.route_epoch, "cancelled": False}
|
||||
output = self._transform(self._seed_bytes(request.token_ids, request.input), request.idempotency_step)
|
||||
return StepResult(status=StructuredStatus(StatusCode.OK, "prefilled"), cache_result=CacheResult.STORED, output=output)
|
||||
|
||||
def decode(self, request: DecodeRequest) -> StepResult:
|
||||
session = self._sessions.get(request.session_id)
|
||||
if session is None:
|
||||
return StepResult(
|
||||
status=StructuredStatus(StatusCode.NOT_FOUND, "no cached session state"),
|
||||
cache_result=CacheResult.MISS,
|
||||
)
|
||||
if session["cancelled"]:
|
||||
return StepResult(status=StructuredStatus(StatusCode.CANCELLED, "session cancelled"))
|
||||
if request.route_epoch < session["epoch"]:
|
||||
return StepResult(status=StructuredStatus(StatusCode.FAILED_PRECONDITION, "stale route epoch"))
|
||||
session["epoch"] = request.route_epoch
|
||||
token_ids = (request.token_id,) if request.token_id is not None else None
|
||||
output = self._transform(self._seed_bytes(token_ids, request.input), request.idempotency_step)
|
||||
return StepResult(status=StructuredStatus(StatusCode.OK, "decoded"), cache_result=CacheResult.HIT, output=output)
|
||||
|
||||
def cancel(self, session_id: str, *, work_id: str = "", reason: str = "") -> StructuredStatus:
|
||||
session = self._sessions.setdefault(session_id, {"epoch": 0, "cancelled": False})
|
||||
if not session["cancelled"]:
|
||||
self._cancelled_total += 1
|
||||
session["cancelled"] = True
|
||||
return StructuredStatus(StatusCode.CANCELLED, reason or "cancelled")
|
||||
|
||||
def release(self, session_id: str) -> StructuredStatus:
|
||||
self._sessions.pop(session_id, None)
|
||||
return StructuredStatus(StatusCode.OK, "released")
|
||||
|
||||
def health(self) -> HealthResult:
|
||||
return HealthResult(
|
||||
status=StructuredStatus(StatusCode.OK, "ok"),
|
||||
serving=self._loaded is not None,
|
||||
state="SERVING" if self._loaded is not None else "NOT_LOADED",
|
||||
active_sessions=len(self._sessions),
|
||||
)
|
||||
|
||||
def metrics(self) -> MetricsResult:
|
||||
return MetricsResult(
|
||||
status=StructuredStatus(StatusCode.OK, "ok"),
|
||||
active_sessions=len(self._sessions),
|
||||
cancelled_sessions=self._cancelled_total,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _seed_bytes(token_ids, bundle: BoundaryBundle | None) -> bytes:
|
||||
if token_ids:
|
||||
return b"".join(int(t).to_bytes(4, "big") for t in token_ids)
|
||||
if bundle is not None:
|
||||
return b"".join(tensor.data for tensor in bundle.tensors)
|
||||
return b""
|
||||
|
||||
def _transform(self, seed: bytes, idempotency_step: int) -> BoundaryBundle | TokenOutput:
|
||||
digest = hashlib.sha256(seed + idempotency_step.to_bytes(4, "big")).digest()
|
||||
assert self._loaded is not None
|
||||
if self._loaded.shard_end >= self._loaded.total_layers - 1:
|
||||
token_id = int.from_bytes(digest[:4], "big") % 50_000
|
||||
return TokenOutput(token_id=token_id)
|
||||
tensor = EngineTensor(name="hidden_states", shape=(1, max(len(seed) // 4, 1)), dtype="bfloat16", data=digest)
|
||||
return BoundaryBundle(tensors=(tensor,), architecture="dense", boundary_point="pre_tail_residual")
|
||||
|
||||
|
||||
def test_reference_engine_obeys_the_shared_shard_engine_contract():
|
||||
assert_shard_engine_contract(_ReferenceEngine)
|
||||
|
||||
|
||||
def test_shard_engine_is_abstract_and_cannot_be_instantiated_directly():
|
||||
with pytest.raises(TypeError):
|
||||
ShardEngine() # type: ignore[abstract]
|
||||
|
||||
|
||||
def test_engine_tensor_rejects_empty_name_shape_or_dtype():
|
||||
with pytest.raises(ValueError):
|
||||
EngineTensor(name="", shape=(1,), dtype="bfloat16", data=b"x")
|
||||
with pytest.raises(ValueError):
|
||||
EngineTensor(name="t", shape=(), dtype="bfloat16", data=b"x")
|
||||
with pytest.raises(ValueError):
|
||||
EngineTensor(name="t", shape=(0,), dtype="bfloat16", data=b"x")
|
||||
with pytest.raises(ValueError):
|
||||
EngineTensor(name="t", shape=(1,), dtype="", data=b"x")
|
||||
|
||||
|
||||
def test_boundary_bundle_requires_at_least_one_tensor():
|
||||
with pytest.raises(ValueError):
|
||||
BoundaryBundle(tensors=(), architecture="dense", boundary_point="pre_tail_residual")
|
||||
|
||||
|
||||
def test_boundary_bundle_tensor_lookup_by_name():
|
||||
tensor = EngineTensor(name="hidden_states", shape=(1, 1), dtype="bfloat16", data=b"\x00\x00")
|
||||
bundle = BoundaryBundle(tensors=(tensor,), architecture="dense", boundary_point="pre_tail_residual")
|
||||
assert bundle.tensor("hidden_states") is tensor
|
||||
with pytest.raises(KeyError):
|
||||
bundle.tensor("router_logits")
|
||||
|
||||
|
||||
def test_token_output_rejects_negative_token_id():
|
||||
with pytest.raises(ValueError):
|
||||
TokenOutput(token_id=-1)
|
||||
|
||||
|
||||
def test_mtp_hook_is_reserved_and_refuses_to_enable():
|
||||
MtpHook() # disabled is fine
|
||||
with pytest.raises(ValueError):
|
||||
MtpHook(enabled=True)
|
||||
with pytest.raises(ValueError):
|
||||
MtpHook(draft_token_count=-1)
|
||||
|
||||
|
||||
def test_architecture_aux_state_hook_carries_opaque_shard_local_state():
|
||||
hook = ArchitectureAuxStateHook(kind="csa", state={"window": 128})
|
||||
assert hook.kind == "csa"
|
||||
assert hook.state == {"window": 128}
|
||||
|
||||
|
||||
def test_prefill_and_decode_requests_require_exactly_one_input_kind():
|
||||
with pytest.raises(ValueError):
|
||||
PrefillRequest(session_id="s", route_epoch=0, position=0, idempotency_step=0)
|
||||
with pytest.raises(ValueError):
|
||||
PrefillRequest(
|
||||
session_id="s",
|
||||
route_epoch=0,
|
||||
position=0,
|
||||
idempotency_step=0,
|
||||
token_ids=(1,),
|
||||
input=BoundaryBundle(
|
||||
tensors=(EngineTensor(name="hidden_states", shape=(1,), dtype="bfloat16", data=b"x"),),
|
||||
architecture="dense",
|
||||
boundary_point="pre_tail_residual",
|
||||
),
|
||||
)
|
||||
with pytest.raises(ValueError):
|
||||
DecodeRequest(session_id="s", route_epoch=0, position=0, idempotency_step=0)
|
||||
|
||||
|
||||
def test_load_request_validates_shard_range_against_total_layers():
|
||||
LoadRequest(artifact_path="a", shard_start=0, shard_end=3, total_layers=4)
|
||||
with pytest.raises(ValueError):
|
||||
LoadRequest(artifact_path="a", shard_start=0, shard_end=4, total_layers=4)
|
||||
with pytest.raises(ValueError):
|
||||
LoadRequest(artifact_path="", shard_start=0, shard_end=0, total_layers=1)
|
||||
with pytest.raises(ValueError):
|
||||
LoadRequest(artifact_path="a", shard_start=3, shard_end=1, total_layers=4)
|
||||
|
||||
|
||||
def test_step_result_requires_an_output_when_status_is_ok():
|
||||
with pytest.raises(ValueError):
|
||||
StepResult(status=StructuredStatus(StatusCode.OK, "ok"), output=None)
|
||||
# A non-OK status is allowed to carry no output.
|
||||
StepResult(status=StructuredStatus(StatusCode.NOT_FOUND, "missing"), output=None)
|
||||
|
||||
|
||||
def test_shard_engine_module_imports_no_native_or_grpc_or_wire_abi_types():
|
||||
import meshnet_node.shard_engine as shard_engine_module
|
||||
|
||||
# The boundary module must not *import* anything that would let a
|
||||
# ggml_tensor, llama context/scheduler handle, ctypes native handle, or a
|
||||
# generated-protobuf (ABI) message leak into a project-owned dataclass
|
||||
# field. Checking bound globals (not docstring prose) proves this
|
||||
# structurally rather than by convention.
|
||||
forbidden_modules = {"ctypes", "grpc", "meshnet_node.native_protocol"}
|
||||
for name, value in vars(shard_engine_module).items():
|
||||
module_name = getattr(value, "__name__", None)
|
||||
assert module_name not in forbidden_modules, (
|
||||
f"shard_engine.{name} binds forbidden module {module_name!r}"
|
||||
)
|
||||
Reference in New Issue
Block a user