story: DGR-030 Add accelerator build presets and native CI matrix
This commit is contained in:
1349
.fuse_hidden0002bd66000001f0
Normal file
1349
.fuse_hidden0002bd66000001f0
Normal file
File diff suppressed because it is too large
Load Diff
5
.ralph-supervisor.log
Normal file
5
.ralph-supervisor.log
Normal file
@@ -0,0 +1,5 @@
|
||||
[2026-07-23 10:24:53] supervisor started, tailer pid=1460238
|
||||
[2026-07-23 10:24:53] cycle 1: running ralph-tui resume (log starts at line 978)
|
||||
[2026-07-23 10:25:59] ralph-tui exited without a recognized stop reason; retrying resume in 5 min
|
||||
[2026-07-23 10:33:51] supervisor started, tailer pid=1465293
|
||||
[2026-07-23 10:33:51] cycle 1: running ralph-tui run (log starts at line 1150)
|
||||
@@ -976,3 +976,329 @@ reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=DGR-030 dry_run=False
|
||||
reconciled DGR-017 #1 completed
|
||||
reconciled DGR-018 #2 completed
|
||||
reconciled DGR-019 #3 completed
|
||||
reconciled DGR-020 #4 completed
|
||||
reconciled DGR-021 #5 completed
|
||||
reconciled DGR-022 #6 completed
|
||||
reconciled DGR-023 #7 completed
|
||||
reconciled DGR-024 #8 completed
|
||||
reconciled DGR-025 #9 completed
|
||||
reconciled DGR-026 #10 completed
|
||||
reconciled DGR-027 #11 completed
|
||||
reconciled DGR-028 #12 completed
|
||||
reconciled DGR-029 #13 completed
|
||||
reconciled DGR-030 #14 in-progress
|
||||
reconciled DGR-031 #15 ready
|
||||
reconciled DGR-032 #16 blocked
|
||||
reconciled DGR-033 #17 blocked
|
||||
reconciled DGR-034 #18 blocked
|
||||
reconciled DGR-035 #19 blocked
|
||||
reconciled DGR-036 #20 blocked
|
||||
reconciled DGR-037 #21 blocked
|
||||
reconciled DGR-038 #22 blocked
|
||||
reconciled DGR-039 #23 blocked
|
||||
reconciled DGR-040 #24 blocked
|
||||
reconciled DGR-041 #25 blocked
|
||||
reconciled DGR-042 #26 blocked
|
||||
reconciled DGR-043 #27 blocked
|
||||
reconciled DGR-044 #28 ready
|
||||
reconciled DGR-045 #29 blocked
|
||||
reconciled DGR-046 #30 blocked
|
||||
reconciled DGR-047 #31 blocked
|
||||
reconciled DGR-048 #32 blocked
|
||||
reconciled DGR-049 #33 blocked
|
||||
reconciled DGR-050 #34 blocked
|
||||
reconciled DGR-051 #35 blocked
|
||||
reconciled DGR-052 #36 blocked
|
||||
reconciled DGR-053 #37 blocked
|
||||
reconciled DGR-054 #38 blocked
|
||||
reconciled DGR-055 #39 blocked
|
||||
reconciled DGR-056 #40 blocked
|
||||
reconciled DGR-057 #41 blocked
|
||||
reconciled DGR-058 #42 blocked
|
||||
reconciled DGR-059 #43 blocked
|
||||
reconciled DGR-060 #44 blocked
|
||||
reconciled DGR-061 #45 blocked
|
||||
reconciled DGR-062 #46 blocked
|
||||
reconciled DGR-063 #47 blocked
|
||||
reconciled DGR-064 #48 blocked
|
||||
reconciled DGR-065 #49 blocked
|
||||
reconciled DGR-066 #50 blocked
|
||||
reconciled DGR-067 #51 blocked
|
||||
reconciled DGR-068 #52 blocked
|
||||
reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=DGR-030 dry_run=False
|
||||
|
||||
📦 Upgrading ralph-tui configuration...
|
||||
Installing bundled skills for detected agents...
|
||||
Installing skills for Claude Code...
|
||||
✓ Skills installed for Claude Code (claude-code)
|
||||
Installing skills for OpenCode...
|
||||
✓ Skills installed for OpenCode (opencode)
|
||||
· Skipping Factory Droid (not installed)
|
||||
· Skipping Gemini CLI (not installed)
|
||||
Installing skills for Codex CLI...
|
||||
✓ Skills installed for Codex CLI (codex)
|
||||
· Skipping Kiro CLI (not installed)
|
||||
Installing skills for Cursor Agent...
|
||||
✓ Skills installed for Cursor Agent (cursor)
|
||||
· Skipping GitHub Copilot (not installed)
|
||||
Installing skills for Kimi CLI...
|
||||
✗ Failed for Kimi CLI
|
||||
· Skipping Pi Coding Agent (not installed)
|
||||
✓ Installed 3 template(s) to /home/popov/.config/ralph-tui/templates
|
||||
✓ Updated config version
|
||||
|
||||
✅ Upgraded to config version 2.1
|
||||
|
||||
⚠️ Warnings:
|
||||
• Failed to install skills for Kimi CLI:
|
||||
[33m[1mDEPRECATED:[0m[33m 'add-skill' has been renamed to 'skills'[0m
|
||||
|
||||
Please use: [1mnpx skills add <package>[0m
|
||||
|
||||
Example: npx skills add vercel-labs/agent-skills
|
||||
|
||||
[33mForwarding to 'npx skills add'...[0m
|
||||
|
||||
|
||||
[90m│[39m
|
||||
[34m●[39m [46m[30m[1m claude-code_2-1-216_agent [22m[39m[49m Agent detected — installing non-interactively
|
||||
[?25l[90m│[39m
|
||||
[32m◇[39m Source: https://github.com/subsy/ralph-tui.git
|
||||
[?25h[?25l[90m│[39m
|
||||
[35m◒[39m Cloning repository…[1G[J[35m◐[39m Cloning repository…[1G[J[35m◓[39m Cloning repository…[1G[J[35m◑[39m Cloning repository…[1G[J[35m◒[39m Cloning repository…[1G[J[35m◐[39m Cloning repository…[1G[J[35m◓[39m Cloning repository…[1G[J[35m◑[39m Cloning repository…[1G[J[35m◒[39m Cloning repository….[1G[J[35m◐[39m Cloning repository….[1G[J[35m◓[39m Cloning repository….[1G[J[35m◑[39m Cloning repository….[1G[J[35m◒[39m Cloning repository….[1G[J[35m◐[39m Cloning repository….[1G[J[35m◓[39m Cloning repository….[1G[J[35m◑[39m Cloning repository….[1G[J[35m◒[39m Cloning repository…..[1G[J[35m◐[39m Cloning repository…..[1G[J[35m◓[39m Cloning repository…..[1G[J[35m◑[39m Cloning repository…..[1G[J[35m◒[39m Cloning repository…..[1G[J[35m◐[39m Cloning repository…..[1G[J[32m◇[39m Repository cloned
|
||||
[?25h[?25l[90m│[39m
|
||||
[1G[J[32m◇[39m Found [32m4[39m skills
|
||||
[?25h[90m│[39m
|
||||
[34m●[39m Installing all 4 skills
|
||||
[90m│[39m
|
||||
[31m■[39m Invalid agents: kimi-cli
|
||||
[90m│[39m
|
||||
[34m●[39m Valid agents: aider-desk, amp, antigravity, antigravity-cli, astrbot, autohand-code, augment, bob, claude-code, openclaw, cline, codearts-agent, codebuddy, codemaker, codestudio, codex, command-code, continue, cortex, crush, cursor, deepagents, devin, dexto, droid, eve, firebender, forgecode, gemini-cli, github-copilot, goose, grok, hermes-agent, inference-sh, jazz, junie, iflow-cli, kilo, kimchi, kimi-code-cli, kiro-cli, kode, lingma, loaf, mcpjam, mistral-vibe, moxby, mux, opencode, openhands, ona, pi, qoder, qoder-cn, qwen-code, replit, reasonix, rovodev, roo, tabnine-cli, terramind, tinycloud, trae, trae-cn, warp, windsurf, zed, zcode, zencoder, zenflow, neovate, pochi, promptscript, adal, universal
|
||||
|
||||
|
||||
Initializing Ralph TUI...
|
||||
Env filter: no vars matched exclusion patterns (*_API_KEY, *_SECRET_KEY, *_SECRET)
|
||||
|
||||
|
||||
⚠️ Recovered stale session
|
||||
Cleared 5 stuck in-progress task(s)
|
||||
Session status set to "interrupted" (resumable)
|
||||
|
||||
Resuming previous session...
|
||||
[0m[31mFailed to resume session[0m
|
||||
reconciled DGR-017 #1 completed
|
||||
reconciled DGR-018 #2 completed
|
||||
reconciled DGR-019 #3 completed
|
||||
reconciled DGR-020 #4 completed
|
||||
reconciled DGR-021 #5 completed
|
||||
reconciled DGR-022 #6 completed
|
||||
reconciled DGR-023 #7 completed
|
||||
reconciled DGR-024 #8 completed
|
||||
reconciled DGR-025 #9 completed
|
||||
reconciled DGR-026 #10 completed
|
||||
reconciled DGR-027 #11 completed
|
||||
reconciled DGR-028 #12 completed
|
||||
reconciled DGR-029 #13 completed
|
||||
reconciled DGR-030 #14 ready
|
||||
reconciled DGR-031 #15 ready
|
||||
reconciled DGR-032 #16 blocked
|
||||
reconciled DGR-033 #17 blocked
|
||||
reconciled DGR-034 #18 blocked
|
||||
reconciled DGR-035 #19 blocked
|
||||
reconciled DGR-036 #20 blocked
|
||||
reconciled DGR-037 #21 blocked
|
||||
reconciled DGR-038 #22 blocked
|
||||
reconciled DGR-039 #23 blocked
|
||||
reconciled DGR-040 #24 blocked
|
||||
reconciled DGR-041 #25 blocked
|
||||
reconciled DGR-042 #26 blocked
|
||||
reconciled DGR-043 #27 blocked
|
||||
reconciled DGR-044 #28 ready
|
||||
reconciled DGR-045 #29 blocked
|
||||
reconciled DGR-046 #30 blocked
|
||||
reconciled DGR-047 #31 blocked
|
||||
reconciled DGR-048 #32 blocked
|
||||
reconciled DGR-049 #33 blocked
|
||||
reconciled DGR-050 #34 blocked
|
||||
reconciled DGR-051 #35 blocked
|
||||
reconciled DGR-052 #36 blocked
|
||||
reconciled DGR-053 #37 blocked
|
||||
reconciled DGR-054 #38 blocked
|
||||
reconciled DGR-055 #39 blocked
|
||||
reconciled DGR-056 #40 blocked
|
||||
reconciled DGR-057 #41 blocked
|
||||
reconciled DGR-058 #42 blocked
|
||||
reconciled DGR-059 #43 blocked
|
||||
reconciled DGR-060 #44 blocked
|
||||
reconciled DGR-061 #45 blocked
|
||||
reconciled DGR-062 #46 blocked
|
||||
reconciled DGR-063 #47 blocked
|
||||
reconciled DGR-064 #48 blocked
|
||||
reconciled DGR-065 #49 blocked
|
||||
reconciled DGR-066 #50 blocked
|
||||
reconciled DGR-067 #51 blocked
|
||||
reconciled DGR-068 #52 blocked
|
||||
reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=none dry_run=False
|
||||
reconciled DGR-017 #1 completed
|
||||
reconciled DGR-018 #2 completed
|
||||
reconciled DGR-019 #3 completed
|
||||
reconciled DGR-020 #4 completed
|
||||
reconciled DGR-021 #5 completed
|
||||
reconciled DGR-022 #6 completed
|
||||
reconciled DGR-023 #7 completed
|
||||
reconciled DGR-024 #8 completed
|
||||
reconciled DGR-025 #9 completed
|
||||
reconciled DGR-026 #10 completed
|
||||
reconciled DGR-027 #11 completed
|
||||
reconciled DGR-028 #12 completed
|
||||
reconciled DGR-029 #13 completed
|
||||
reconciled DGR-030 #14 in-progress
|
||||
reconciled DGR-031 #15 ready
|
||||
reconciled DGR-032 #16 blocked
|
||||
reconciled DGR-033 #17 blocked
|
||||
reconciled DGR-034 #18 blocked
|
||||
reconciled DGR-035 #19 blocked
|
||||
reconciled DGR-036 #20 blocked
|
||||
reconciled DGR-037 #21 blocked
|
||||
reconciled DGR-038 #22 blocked
|
||||
reconciled DGR-039 #23 blocked
|
||||
reconciled DGR-040 #24 blocked
|
||||
reconciled DGR-041 #25 blocked
|
||||
reconciled DGR-042 #26 blocked
|
||||
reconciled DGR-043 #27 blocked
|
||||
reconciled DGR-044 #28 ready
|
||||
reconciled DGR-045 #29 blocked
|
||||
reconciled DGR-046 #30 blocked
|
||||
reconciled DGR-047 #31 blocked
|
||||
reconciled DGR-048 #32 blocked
|
||||
reconciled DGR-049 #33 blocked
|
||||
reconciled DGR-050 #34 blocked
|
||||
reconciled DGR-051 #35 blocked
|
||||
reconciled DGR-052 #36 blocked
|
||||
reconciled DGR-053 #37 blocked
|
||||
reconciled DGR-054 #38 blocked
|
||||
reconciled DGR-055 #39 blocked
|
||||
reconciled DGR-056 #40 blocked
|
||||
reconciled DGR-057 #41 blocked
|
||||
reconciled DGR-058 #42 blocked
|
||||
reconciled DGR-059 #43 blocked
|
||||
reconciled DGR-060 #44 blocked
|
||||
reconciled DGR-061 #45 blocked
|
||||
reconciled DGR-062 #46 blocked
|
||||
reconciled DGR-063 #47 blocked
|
||||
reconciled DGR-064 #48 blocked
|
||||
reconciled DGR-065 #49 blocked
|
||||
reconciled DGR-066 #50 blocked
|
||||
reconciled DGR-067 #51 blocked
|
||||
reconciled DGR-068 #52 blocked
|
||||
reconciled DGR-069 #53 blocked
|
||||
reconciled DGR-070 #54 blocked
|
||||
reconciled DGR-071 #55 blocked
|
||||
synced=55 next=DGR-030 dry_run=False
|
||||
Initializing Ralph TUI...
|
||||
Env filter: no vars matched exclusion patterns (*_API_KEY, *_SECRET_KEY, *_SECRET)
|
||||
|
||||
Session: b92af6fa-2fb6-4e92-9bf2-d571be5c8561
|
||||
Agent: claude
|
||||
Tracker: json
|
||||
PRD: .scratch/distributed-gguf-runtime/prd.json
|
||||
Max iterations: unlimited
|
||||
|
||||
[10:34:45] [INFO] [session] Session b92af6fa-2fb6-4e92-9bf2-d571be5c8561 created. Agent: claude, Tracker: json
|
||||
[10:34:45] [INFO] [engine] Ralph started. Total tasks: 42
|
||||
[10:34:45] [INFO] [progress] Iteration 1/∞: Working on DGR-030 - Add accelerator build presets and native CI matrix
|
||||
[10:34:51] [INFO] [agent] I'll start by reading the context files for this task.
|
||||
[10:34:52] [INFO] [agent] [94m[Bash][0m $ echo "--- RALPH-CONTEXT ---" && cat .scratch/distributed-gguf-runtime/RALPH-CONTEXT.md 2>/dev/null |...
|
||||
[10:34:53] [INFO] [agent] [94m[Bash][0m $ echo "--- ISSUE ---" && cat .scratch/distributed-gguf-runtime/issues/030-add-accelerator-build-prese...
|
||||
[10:34:54] [INFO] [agent] [94m[Bash][0m $ echo "--- PROGRESS ---" && cat .ralph-tui/progress.md 2>/dev/null | tail -300
|
||||
[10:34:58] [INFO] [agent] [94m[Bash][0m $ echo "--- DGR-029 evidence ---" && cat .scratch/distributed-gguf-runtime/evidence/DGR-029/README.md ...
|
||||
[10:34:59] [INFO] [agent] [94m[Bash][0m $ echo "--- find native/cmake dirs ---" && find . -iname "*CMakeLists*" -not -path "*/node_modules/*" ...
|
||||
[10:35:14] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/native/llama/UPSTREAM_LOCK.json[0m
|
||||
[10:35:14] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m
|
||||
[10:35:21] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/native/CMakeLists.txt[0m
|
||||
[10:35:23] [INFO] [agent] [94m[Bash][0m $ find /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime -maxdept...
|
||||
[10:35:30] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime find . -ma...
|
||||
[10:35:30] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_llama_cpp_dependency.py[0m
|
||||
[10:35:44] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -rn "...
|
||||
[10:35:44] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -rln ...
|
||||
[10:35:48] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -n "r...
|
||||
[10:35:49] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/tracker/meshnet_tracker/capability.py[0m
|
||||
[10:36:21] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -n -B...
|
||||
[10:36:21] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime grep -n -B...
|
||||
[10:37:10] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime ls build/l...
|
||||
[10:37:25] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime ls .venv/b...
|
||||
[10:37:29] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime timeout 8 ...
|
||||
[10:37:36] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime .venv/bin/...
|
||||
[10:37:41] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime timeout 90...
|
||||
[10:37:47] [INFO] [agent] Good, cmake is available now in `.venv/bin`. Now let's check for GPU SDK availability more thoroughly and check the c++ compiler.
|
||||
[10:37:48] [INFO] [agent] [94m[Bash][0m $ c++ --version 2>&1 | head -2 echo "---" which nvidia-smi rocminfo 2>&1 echo "--- CUDA_HOME / ROCM re...
|
||||
[10:37:55] [INFO] [agent] [94m[Bash][0m $ which glslc glslangValidator 2>&1 echo "---" find / -maxdepth 6 -iname "vulkan.h" 2>/dev/null | grep...
|
||||
[10:39:23] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime python3 -c...
|
||||
[10:39:27] [INFO] [agent] No separate shared-gates block; they're embedded per-story. Let's check the DGR-025 evidence README for the "isolated out-of-tree" and recipe-identity conventions, briefly, then start implementing.
|
||||
[10:39:28] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime cat packag...
|
||||
[10:39:31] [INFO] [agent] Now let's implement. First, update `UPSTREAM_LOCK.json` to add the accelerator presets.
|
||||
[10:39:35] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/packages/node/native/llama/UPSTREAM_LOCK.json[0m edit: " "smoke_binary": "bin/llama-gguf-hash",
|
||||
[10:39:35] [INFO] [agent] "sm..." → " "smoke_binary": "bin/llama-gguf-hash",
|
||||
[10:39:35] [INFO] [agent] "sm..."
|
||||
[10:39:39] [INFO] [agent] Now let's update `scripts/llama_cpp_dependency.py` with the accelerator lane primitives.
|
||||
[10:39:47] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " if lock["expected_source"] != {"git_tree": loc..." → " if lock["expected_source"] != {"git_tree": loc..."
|
||||
[10:39:51] [INFO] [agent] Now let's add the accelerator lane functions after `ctest_lane`:
|
||||
[10:40:07] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: "def ctest_lane(build_dir: pathlib.Path) -> None:
|
||||
[10:40:07] [INFO] [agent] ..." → "def ctest_lane(build_dir: pathlib.Path) -> None:
|
||||
[10:40:07] [INFO] [agent] ..."
|
||||
[10:40:10] [INFO] [agent] Now add the CLI subcommands:
|
||||
[10:40:13] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " ctest_parser = subcommands.add_parser("ctest")..." → " ctest_parser = subcommands.add_parser("ctest")..."
|
||||
[10:40:22] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " elif args.command == "ctest":
|
||||
[10:40:22] [INFO] [agent] ..." → " elif args.command == "ctest":
|
||||
[10:40:22] [INFO] [agent] ..."
|
||||
[10:40:29] [INFO] [agent] I need to check that this doesn't break the existing test whose fixture lock has `"build": {}`. Let me guard against that.
|
||||
[10:40:30] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m
|
||||
[10:40:36] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/llama_cpp_dependency.py[0m edit: " presets = lock.get("accelerator_presets", {})
|
||||
[10:40:36] [INFO] [agent] ..." → " presets = lock.get("accelerator_presets", {})
|
||||
[10:40:36] [INFO] [agent] ..."
|
||||
[10:40:40] [INFO] [agent] Now let's write the CI matrix script.
|
||||
[10:40:53] [INFO] [agent] [94m[Write][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/scripts/native_accelerator_matrix.py[0m "#!/usr/bin/env python3
|
||||
[10:40:53] [INFO] [agent] """DGR-030: native CI/build matrix over the CPU default plus accelerator lanes.
|
||||
[10:40:53] [INFO] [agent] Runs the exact deterministic CPU lane DGR-029 locked (unchanged), then probes
|
||||
[10:40:53] [INFO] [agent] each accelerator p... (4553 chars)"
|
||||
[10:40:59] [INFO] [agent] Now let's verify the Python compiles and existing tests still pass.
|
||||
[10:41:00] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime python3 -m...
|
||||
[10:41:03] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:41:10] [INFO] [agent] Existing tests pass. Now let's quickly test the CLI additions and the SDK-probe logic live.
|
||||
[10:41:11] [INFO] [agent] [94m[Bash][0m $ do echo "--- $lane ---" python3 scripts/llama_cpp_dependency.py accelerator-status --name "$lane...
|
||||
[10:41:14] [INFO] [agent] Now let's run the full matrix live end-to-end to verify the CPU lane still builds correctly and accelerator lanes report skipped, then verify the reverse-cleanup contract.
|
||||
[10:41:16] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:43:19] [INFO] [agent] It's compiling in the background (matches DGR-029's ~2m16s wall-clock for the CPU lane). I'll wait for it.
|
||||
[10:43:20] [INFO] [agent] [94m[Bash][0m $ sleep 90 && cat /tmp/claude-1000/-run-media-popov-d-DEV-repos-d-popov-com-AI--claude-worktrees-distr...
|
||||
[10:43:20] [INFO] [agent] [91m[Error: <tool_use_error>Blocked: sleep 90 followed by: cat /tmp/claude-1000/-run-media-popov-d-DEV-repos-d-popov-com-AI--claude-worktrees-distributed-gguf-runtime/972c183b-ce77-43c0-9fb1-d513602df13b/tasks/b1g4st9uy.output tail -100. To wait for a condition, use Monitor with an until-loop (e.g. `until <check>; do sleep 2; done`). To wait for a command you started, use run_in_background: true. Do not chain shorter sleeps to work around this block.</tool_use_error>][0m
|
||||
[10:43:25] [INFO] [agent] The build moved to background automatically; I'll get notified when it finishes. Meanwhile, let's write tests for the new accelerator functionality.
|
||||
[10:43:26] [INFO] [agent] [94m[Read][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_llama_cpp_dependency.py[0m
|
||||
[10:44:52] [INFO] [agent] [94m[Bash][0m $ tail -50 /tmp/claude-1000/-run-media-popov-d-DEV-repos-d-popov-com-AI--claude-worktrees-distributed-...
|
||||
[10:44:57] [INFO] [agent] The live matrix run succeeded: CPU lane built and CTest-passed exactly as before, and all four accelerator lanes correctly reported `skipped` with an explicit reason (no false success). Let's confirm the source checkout was properly reversed to pristine and the build dirs exist as expected.
|
||||
[10:44:58] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime git -C bui...
|
||||
[10:45:06] [INFO] [agent] Good — the checkout is pristine (matches the locked pin/tree), and only the CPU `build` dir was created (accelerator lanes never created their build dirs since they were skipped, as designed). Now let's write the unit tests.
|
||||
[10:45:26] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_llama_cpp_dependency.py[0m edit: " dependency._load_lock = lambda: _lock_with_reg..." → " dependency._load_lock = lambda: _lock_with_reg..."
|
||||
[10:45:30] [INFO] [agent] Now let's create the matrix orchestration test file.
|
||||
[10:45:51] [INFO] [agent] [94m[Write][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_native_accelerator_matrix.py[0m """"Offline behavior tests for DGR-030's native CI/build matrix orchestration.
|
||||
[10:45:51] [INFO] [agent] These tests never fetch or compile llama.cpp: `llama_cpp_dependency`'s fetch/
|
||||
[10:45:51] [INFO] [agent] apply/reverse/build/smoke/ctest_lane/accele... (7187 chars)"
|
||||
[10:46:02] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_native_accelerator_matrix.py[0m edit: " assert report["failed_lanes"] == []
|
||||
[10:46:02] [INFO] [agent] assert..." → " assert report["failed_lanes"] == []
|
||||
[10:46:02] [INFO] [agent] assert..."
|
||||
[10:46:17] [INFO] [agent] [94m[Bash][0m $ grep -n "workspace / \"build\"" tests/test_native_accelerator_matrix.py
|
||||
[10:46:21] [INFO] [agent] [94m[Edit][0m [95m/run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime/tests/test_native_accelerator_matrix.py[0m edit: " assert smoke_calls == [("smoke", (workspace / ..." → " assert smoke_calls == [("smoke", (workspace.re..."
|
||||
[10:46:28] [INFO] [agent] Now let's run the full test suite.
|
||||
[10:46:29] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:46:42] [INFO] [agent] All 19 tests pass. Now let's run the broader shared quality gates.
|
||||
[10:46:43] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:46:55] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime export PAT...
|
||||
[10:47:00] [INFO] [agent] [94m[Bash][0m $ cd /run/media/popov/d/DEV/repos/d-popov.com/AI/.claude/worktrees/distributed-gguf-runtime git stash ...
|
||||
|
||||
2
.ralph-tui/config.toml
Normal file
2
.ralph-tui/config.toml
Normal file
@@ -0,0 +1,2 @@
|
||||
autoCommit = true
|
||||
configVersion = "2.1"
|
||||
275
.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md
Normal file
275
.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md
Normal file
@@ -0,0 +1,275 @@
|
||||
# DGR-030 evidence — accelerator build presets and native CI/build matrix
|
||||
|
||||
**Status:** implementation complete, live-verified in this session (2026-07-23).
|
||||
**Authority:** local `prd.json` is authoritative; Gitea is a projection.
|
||||
**Upstream pin:** `e920c523e3b8a0163fe498af5bf90df35ff51d25` (`llama.cpp`, unchanged from DGR-027..029).
|
||||
|
||||
## What existed before this session
|
||||
|
||||
DGR-029 locked exactly one build lane — the deterministic CPU-only lane — in
|
||||
`UPSTREAM_LOCK.json`'s `build` section, plus `scripts/llama_cpp_dependency.py`'s
|
||||
`build()`/`smoke()`/`ctest_lane()`/`reproduce()`. There was no accelerator
|
||||
preset, no SDK-availability probing, and no matrix runner: only the one CPU
|
||||
lane existed, and there was no mechanism that could ever advertise a GPU
|
||||
backend as compiled or capable.
|
||||
|
||||
## What changed in this session
|
||||
|
||||
- `packages/node/native/llama/UPSTREAM_LOCK.json`: added a new top-level
|
||||
`accelerator_presets` object with one entry each for `cuda` (`GGML_CUDA`),
|
||||
`rocm` (`GGML_HIP`), `vulkan` (`GGML_VULKAN`), and `metal` (`GGML_METAL`).
|
||||
Each entry names only the one backend flag it flips and an `sdk_probe`
|
||||
(a binary to resolve on `PATH`, an optional env-var override, and — for
|
||||
Metal — a `platform_only: "darwin"` gate). **The existing `build` section
|
||||
— the deterministic CPU default DGR-029 locked — is untouched.**
|
||||
- `scripts/llama_cpp_dependency.py`:
|
||||
- `_load_lock()` now calls a new `_verify_accelerator_presets()`, which
|
||||
fail-closed-rejects any preset whose named backend flag is not `OFF` in
|
||||
the CPU default's `configure_flags` — structurally guaranteeing a preset
|
||||
can only ever *add* one backend on top of the untouched CPU baseline,
|
||||
never redefine it.
|
||||
- `accelerator_configure_flags(lock, name)` returns a **new** flag list —
|
||||
the CPU default's own `configure_flags` list is never mutated — with
|
||||
exactly the named preset's backend flag flipped `ON` and every other flag
|
||||
(including `GGML_CPU=ON`, the fallback ops backend GPU builds still need)
|
||||
left exactly as the CPU default declares it.
|
||||
- `_sdk_probe(probe)` / `accelerator_status(name, lock)` resolve a lane's
|
||||
SDK without ever raising: an absent SDK is returned as
|
||||
`{"available": false, "reason": "<binary> is unavailable on PATH"}` (or
|
||||
a platform-mismatch reason for Metal), so "unavailable" is data a caller
|
||||
reports, never an exception a caller has to remember to catch.
|
||||
- `accelerator_build(source, name, build_dir)` compiles one lane into its
|
||||
own out-of-tree `build_dir` (an isolated directory, never DGR-029's CPU
|
||||
`build_dir`), using the same patched-source verification and
|
||||
`native_targets` as the CPU lane, then writes a
|
||||
`meshnet-build-metadata.json` recording the exact `commit`/`commit_tree`,
|
||||
per-patch SHA-256 digests, the lane's overridden `configure_flags`, the
|
||||
resolved `cmake`/`cxx`/SDK-binary versions/paths, and explicit
|
||||
`model_downloads: false`, `hardware_execution: false`,
|
||||
`hardware_certified: false`, `semantic_certification: false` fields plus
|
||||
a `note` stating the lane is registered-dark until a real-hardware
|
||||
certification record exists. It **never** calls `smoke()`/`ctest_lane()`
|
||||
— running a binary linked against a real accelerator backend would touch
|
||||
real hardware, which this story deliberately keeps out of scope.
|
||||
- Added `accelerator-status --name <lane>` and
|
||||
`accelerator-build --name <lane> --source-dir --build-dir` CLI
|
||||
subcommands, mirroring the existing `ctest`/`build` subcommand pattern.
|
||||
- `scripts/native_accelerator_matrix.py` (new): the native CI/build matrix.
|
||||
`run_matrix(workspace)` fetches and applies the locked pin/patch stack once,
|
||||
runs the unchanged CPU lane (build → smoke → ctest, exactly DGR-029's
|
||||
contract), then for each `accelerator_presets` entry either reports
|
||||
`{"status": "skipped", "reason": ...}` (SDK absent) or compiles it via
|
||||
`accelerator_build` and reports `{"status": "built", ...}` — never silently
|
||||
treating a skip as a pass. Any `DependencyError` from a lane (CPU or
|
||||
accelerator) is caught per-lane and reported as `{"status": "failed", ...}`
|
||||
without aborting the remaining lanes or skipping cleanup. `reverse()` always
|
||||
runs in a `finally`, restoring the exact pristine pin/tree regardless of
|
||||
lane outcomes. The CLI prints a JSON report and exits non-zero only if any
|
||||
lane actually `failed` (a `skipped` lane never fails the run).
|
||||
- `tests/test_llama_cpp_dependency.py`: added 7 new tests —
|
||||
`test_accelerator_presets_isolate_one_backend_without_touching_the_cpu_default`
|
||||
(every preset flips exactly its own flag and the CPU default list is never
|
||||
mutated), `test_accelerator_configure_flags_rejects_an_unknown_lane`,
|
||||
`test_accelerator_status_reports_unavailable_sdks_without_raising` (asserts
|
||||
the exact reason string for cuda/rocm/vulkan/metal absence),
|
||||
`test_accelerator_status_honors_an_explicit_sdk_override`,
|
||||
`test_accelerator_status_rejects_an_unknown_lane`,
|
||||
`test_accelerator_build_refuses_to_compile_an_unavailable_lane` (asserts no
|
||||
build directory is created), and a `requires_cmake`-gated
|
||||
`test_accelerator_build_compiles_the_available_lane_with_isolated_evidence`,
|
||||
which builds a tiny synthetic CMake project (not the full llama.cpp tree) to
|
||||
prove `accelerator_build`'s "SDK present" path really configures with the
|
||||
overridden flag, compiles, and writes the registered-dark metadata — in
|
||||
about a second, without a real GPU SDK.
|
||||
- `tests/test_native_accelerator_matrix.py` (new): 3 offline tests exercising
|
||||
`run_matrix`'s orchestration with `llama_cpp_dependency`'s
|
||||
fetch/apply/reverse/build/smoke/ctest_lane/accelerator_status/
|
||||
accelerator_build stubbed out — proving unavailable SDKs are reported
|
||||
`skipped` (never a false pass), an available accelerator lane is compiled
|
||||
without ever calling `smoke`/`ctest_lane`, and a lane failure is reported
|
||||
per-lane without aborting sibling lanes or skipping the `reverse()` cleanup.
|
||||
|
||||
## Toolchain note
|
||||
|
||||
As in DGR-029, neither the ambient system Python nor `.venv-rocm` has `cmake`;
|
||||
this session's `.venv` also had no `cmake` (a prior session's install did not
|
||||
persist). This session ran `.venv/bin/python3 -m ensurepip --upgrade` (no
|
||||
`pip` was present in `.venv` either) and then
|
||||
`.venv/bin/python3 -m pip install cmake`, landing the same PyPI wheel
|
||||
(`cmake==4.4.0`) DGR-029 used, at `.venv/bin/cmake` / `.venv/bin/ctest`. All
|
||||
commands below were run with that `.venv/bin` prepended to `PATH`. No CUDA,
|
||||
ROCm, or Vulkan SDK (`nvcc`, `hipcc`, `glslc`) is installed in this
|
||||
environment, and the host platform is Linux, not `darwin` — so all four
|
||||
accelerator lanes are genuinely `skipped` in this environment's own live run
|
||||
below, which is real evidence for AC2 ("unavailable SDKs ... explicit
|
||||
unavailable/skipped lanes"), not a simulated one.
|
||||
|
||||
## Verification — live native CI/build matrix run
|
||||
|
||||
```text
|
||||
$ rm -rf build/llama.cpp/build build/llama.cpp/build-cuda build/llama.cpp/build-rocm build/llama.cpp/build-vulkan build/llama.cpp/build-metal
|
||||
$ python3 scripts/native_accelerator_matrix.py
|
||||
reused verified offline cache: .../build/llama.cpp/source
|
||||
usage: .../build/llama.cpp/build/bin/llama-gguf-hash [options] GGUF_IN
|
||||
...
|
||||
Test project .../build/llama.cpp/build
|
||||
Start 27: test-meshnet-range-ownership
|
||||
1/1 Test #27: test-meshnet-range-ownership ..... Passed 0.01 sec
|
||||
100% tests passed out of 1
|
||||
{
|
||||
"failed_lanes": [],
|
||||
"hardware_certified": false,
|
||||
"lanes": [
|
||||
{
|
||||
"build_dir": ".../build/llama.cpp/build",
|
||||
"lane": "cpu",
|
||||
"metadata": {
|
||||
"cmake": "cmake version 4.4.0",
|
||||
"commit": "e920c523e3b8a0163fe498af5bf90df35ff51d25",
|
||||
"commit_tree": "6c91a11407a3a3fb160f5dac705f9c59718f54f1",
|
||||
"configure_flags": [
|
||||
"-DCMAKE_BUILD_TYPE=Release", "-DLLAMA_BUILD_TESTS=ON",
|
||||
"-DLLAMA_BUILD_EXAMPLES=ON", "-DLLAMA_BUILD_SERVER=OFF",
|
||||
"-DLLAMA_BUILD_TOOLS=OFF", "-DLLAMA_BUILD_APP=OFF", "-DLLAMA_CURL=OFF",
|
||||
"-DGGML_CPU=ON", "-DGGML_BLAS=OFF", "-DGGML_CUDA=OFF",
|
||||
"-DGGML_HIP=OFF", "-DGGML_VULKAN=OFF", "-DGGML_METAL=OFF"
|
||||
],
|
||||
"cxx": "c++ (GCC) 15.2.1 20260123 (Red Hat 15.2.1-7)",
|
||||
"model_downloads": false,
|
||||
"patches": { "...": "... (5 entries, unchanged sha256 digests from DGR-029)" },
|
||||
"semantic_certification": false
|
||||
},
|
||||
"status": "built"
|
||||
},
|
||||
{"lane": "cuda", "reason": "nvcc is unavailable on PATH", "status": "skipped"},
|
||||
{"lane": "rocm", "reason": "hipcc is unavailable on PATH", "status": "skipped"},
|
||||
{"lane": "vulkan", "reason": "glslc is unavailable on PATH", "status": "skipped"},
|
||||
{"lane": "metal", "reason": "platform 'linux' is not 'darwin'", "status": "skipped"}
|
||||
],
|
||||
"note": "A `built` lane means it compiled with the exact recorded compiler/SDK/upstream-pin/patch-stack/build-option evidence — it never means an accelerator device was exercised. Every backend/model/recipe lane stays registered-dark until a separate real-hardware certification record exists."
|
||||
}
|
||||
$ echo $?
|
||||
0
|
||||
```
|
||||
|
||||
Wall-clock: `real 2m19.797s` — matches DGR-029's ~2m16s CPU-lane compile; no
|
||||
accelerator lane actually compiled in this environment (all four SDKs are
|
||||
genuinely absent), so this run's added cost over DGR-029's own CPU-only
|
||||
`reproduce()` is just the four fast SDK probes.
|
||||
|
||||
Post-run checks (source checkout left pristine by the matrix's `reverse()`):
|
||||
|
||||
```text
|
||||
$ git -C build/llama.cpp/source status --short --branch --untracked-files=all
|
||||
## HEAD (no branch)
|
||||
$ git -C build/llama.cpp/source rev-parse HEAD HEAD^{tree}
|
||||
e920c523e3b8a0163fe498af5bf90df35ff51d25
|
||||
6c91a11407a3a3fb160f5dac705f9c59718f54f1
|
||||
$ ls build/llama.cpp/ | grep build
|
||||
build
|
||||
```
|
||||
|
||||
Only the CPU lane's `build/` directory was created — no `build-cuda`,
|
||||
`build-rocm`, `build-vulkan`, or `build-metal` directory exists, because every
|
||||
accelerator lane was genuinely skipped rather than attempted.
|
||||
|
||||
## Verification — targeted test suites and shared gates
|
||||
|
||||
| Command | Result |
|
||||
| --- | --- |
|
||||
| `python3 -m pytest -q tests/test_llama_cpp_dependency.py tests/test_native_accelerator_matrix.py` | `19 passed` (9 pre-existing + 7 new accelerator-lane tests in `test_llama_cpp_dependency.py`, 3 new in `test_native_accelerator_matrix.py`; the `requires_cmake`-gated compile test ran for real, not skipped) |
|
||||
| `python3 -m compileall -q packages tests` | exit 0 |
|
||||
| `git diff --check -- packages/node/native/llama/UPSTREAM_LOCK.json scripts/llama_cpp_dependency.py tests/test_llama_cpp_dependency.py scripts/native_accelerator_matrix.py tests/test_native_accelerator_matrix.py` | exit 0 |
|
||||
| `python3 scripts/ralph_prd_schema.py validate .scratch/distributed-gguf-runtime/prd.json` | `OK: 55 stories validated.` |
|
||||
|
||||
`git diff --check` against the full working tree separately reports one
|
||||
pre-existing trailing-whitespace line in `.ralph-tui-run.log`, which was
|
||||
already modified before this session started (see the session's initial
|
||||
`git status`) and is unrelated to this story's scope; it is excluded above by
|
||||
naming this story's own changed files explicitly.
|
||||
|
||||
`python3 -m pytest -q tests/test_ralph_prd_schema.py` reports `55 failed, 53
|
||||
passed` in this session (all `test_render_issue_markdown_matches_committed_file`
|
||||
drift between `prd.json` and committed issue Markdown for other stories,
|
||||
e.g. `DGR-053`..`DGR-071`). `git stash`-ing this session's changes and rerunning
|
||||
reproduces `56 failed, 52 passed` identically — the same 56 failures minus the
|
||||
one this session's own `DGR-030` regeneration fixed, confirming the remaining
|
||||
55 predate this story and are out of scope to fix here. This session did
|
||||
regenerate `.scratch/distributed-gguf-runtime/issues/030-add-accelerator-
|
||||
build-presets-and-native-ci-matrix.md` via
|
||||
`python3 scripts/ralph_prd_schema.py render ... DGR-030` so DGR-030's own
|
||||
generated issue Markdown matches `prd.json` byte-for-byte (confirmed by the
|
||||
`test_render_issue_markdown_matches_committed_file[DGR-030]` case no longer
|
||||
appearing in the failure list).
|
||||
|
||||
## Ensuring build success does not advertise capability
|
||||
|
||||
- Every accelerator lane's `meshnet-build-metadata.json` explicitly records
|
||||
`hardware_execution: false`, `hardware_certified: false`, and
|
||||
`semantic_certification: false`, plus a `note` stating the lane is
|
||||
registered-dark until a separate real-hardware certification record exists
|
||||
— the same "artifact states this, not just prose" pattern DGR-029 used for
|
||||
the CPU lane's `model_downloads`/`semantic_certification` fields.
|
||||
- `accelerator_build` never runs `smoke()` or `ctest_lane()`: it only
|
||||
configures and compiles the exact `native_targets` DGR-029 already locked
|
||||
(`llama-gguf-hash`, `test-meshnet-range-ownership`) — no binary linked
|
||||
against a real accelerator backend is ever executed by this story's code.
|
||||
- `_verify_accelerator_presets()` structurally refuses any preset whose
|
||||
backend flag is not `OFF` in the locked CPU default, so a preset can never
|
||||
be defined in a way that redefines (rather than adds one backend on top of)
|
||||
DGR-029's deterministic CPU lane.
|
||||
- The matrix's top-level report always carries `"hardware_certified": false`
|
||||
regardless of how many lanes built, and its `note` field states this
|
||||
explicitly for any consumer reading only the report, not the per-lane
|
||||
metadata.
|
||||
|
||||
## Limitations
|
||||
|
||||
- This story proves accelerator lanes *compile* with correct, isolated
|
||||
flags and preserves exact evidence when a lane's SDK is present. It proves
|
||||
nothing about numerical correctness, performance, or any backend/model/
|
||||
recipe capability on real accelerator hardware — that is explicitly
|
||||
deferred to DGR-041 (capability registration), DGR-053 (real 2-4 stage
|
||||
certification), and DGR-067 (capability matrix certification), all of which
|
||||
remain unimplemented.
|
||||
- No CUDA, ROCm, or Vulkan SDK, and no macOS/Metal toolchain, is available in
|
||||
this session's environment, so the "compile an available accelerator lane"
|
||||
path is proven end-to-end only via the `requires_cmake`-gated synthetic-
|
||||
project unit test and the offline matrix-orchestration tests, not via a
|
||||
live compile of the real llama.cpp tree under `GGML_CUDA=ON` (etc.). A
|
||||
future session with a real SDK installed will exercise
|
||||
`accelerator_build`'s real-lane path against the genuine llama.cpp source
|
||||
for the first time; nothing in this story's design assumes that hasn't
|
||||
happened yet.
|
||||
- The accelerator lanes reuse the CPU lane's exact `native_targets`
|
||||
(`llama-gguf-hash`, `test-meshnet-range-ownership`), so a passing
|
||||
accelerator compile also proves the DGR-027/DGR-028 patch stack's
|
||||
range-ownership code compiles under that backend flag combination — but,
|
||||
per the point above, only structurally; it says nothing about GPU
|
||||
execution correctness.
|
||||
- `cmake`/`ctest` remain absent system-wide in this environment; this session
|
||||
reinstalled them into `.venv` exactly as DGR-029 did, and that install does
|
||||
not appear to persist across sessions (this session found `.venv` without
|
||||
`cmake` despite DGR-029's evidence recording its earlier install). A future
|
||||
session without a `cmake`-equipped `.venv` will see the same actionable
|
||||
"cmake is unavailable" failure DGR-029 demonstrated, not a silent pass, and
|
||||
the new `requires_cmake`-gated tests will be skipped rather than failing.
|
||||
- `git diff --check` and `tests/test_ralph_prd_schema.py` both carry
|
||||
pre-existing, out-of-scope failures unrelated to this story (see the gates
|
||||
table above); this story's own changed files pass both checks cleanly.
|
||||
|
||||
## Dependency handoff
|
||||
|
||||
DGR-053 (real 2-4 stage certification), DGR-067 (capability matrix
|
||||
certification), and DGR-068 (packaged releases) may rely on: four isolated,
|
||||
out-of-tree accelerator build presets (`cuda`/`rocm`/`vulkan`/`metal`) in
|
||||
`UPSTREAM_LOCK.json`'s `accelerator_presets`, each toggling exactly one
|
||||
backend flag on top of DGR-029's unchanged CPU default; a native CI/build
|
||||
matrix (`scripts/native_accelerator_matrix.py`) that compiles every
|
||||
SDK-available lane with full compiler/SDK/upstream-pin/patch-stack/build-
|
||||
option evidence and reports SDK-unavailable lanes as explicit `skipped`
|
||||
lanes, never a false pass; and a compile-only contract (no lane here ever
|
||||
runs a binary against real accelerator hardware). Real-hardware execution,
|
||||
numerical correctness, performance measurement, and backend/model/recipe
|
||||
certification for any accelerator remain entirely unimplemented and must not
|
||||
be assumed from any lane's green compile.
|
||||
@@ -1,7 +1,7 @@
|
||||
<!-- GENERATED FROM prd.json — DO NOT EDIT AS AN INDEPENDENT SOURCE. prd.json IS AUTHORITATIVE. -->
|
||||
# DGR-030: Add accelerator build presets and native CI matrix
|
||||
|
||||
- **Status / triage:** specification only; `ready-for-agent`; `passes: false`
|
||||
- **Status / triage:** completed; `passes: true`
|
||||
- **Execution mode:** `AFK`
|
||||
- **Milestone:** `M1`
|
||||
- **Dependencies:** `DGR-029`
|
||||
@@ -18,11 +18,11 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Add isolated out-of-tree presets for CUDA, ROCm, Vulkan, and Metal without changing the deterministic CPU default.
|
||||
- [ ] Add a native CI/build matrix that reports unavailable SDKs as explicit unavailable/skipped lanes rather than false success.
|
||||
- [ ] Compile each available lane and preserve exact compiler, SDK, upstream pin, patch-stack, and build-option evidence.
|
||||
- [ ] Keep every backend/model/recipe lane registered-dark until a separate real-hardware certification record exists.
|
||||
- [ ] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
- [x] Add isolated out-of-tree presets for CUDA, ROCm, Vulkan, and Metal without changing the deterministic CPU default.
|
||||
- [x] Add a native CI/build matrix that reports unavailable SDKs as explicit unavailable/skipped lanes rather than false success.
|
||||
- [x] Compile each available lane and preserve exact compiler, SDK, upstream pin, patch-stack, and build-option evidence.
|
||||
- [x] Keep every backend/model/recipe lane registered-dark until a separate real-hardware certification record exists.
|
||||
- [x] Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff.
|
||||
|
||||
## Shared quality gates
|
||||
|
||||
@@ -30,10 +30,7 @@ Fresh Ralph session: read `.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md`,
|
||||
- `git diff --check` passes.
|
||||
- Default tests are model-download-free, API-credit-free, and GPU-free.
|
||||
- Evidence README records exact changed files, commands/results, limitations, and dependency handoff; no fabricated evidence or inherited completion credit.
|
||||
- Native changes pass focused out-of-tree CMake build and CTest; patch changes verify clean apply/check/reverse against the exact llama.cpp pin.
|
||||
- Runs are opt-in and record exact artifact/split hashes, runtime/upstream pin, backend/driver, hardware, network, commands, and raw metrics. Model artifacts use configured mounted-drive storage and never `/home`.
|
||||
- Preserve existing Transformers behavior and backend-agnostic Tracker routing/load balancing/billing/relay semantics unless an explicit versioned contract says otherwise. One scoped story commit is expected during execution, but this specification-materialization change is not committed.
|
||||
|
||||
## Evidence handoff
|
||||
|
||||
Write and verify `.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md`. Until every criterion and applicable gate has real evidence, this story remains `passes: false`. Legacy evidence is provenance only, not completion credit.
|
||||
Verified evidence: `.scratch/distributed-gguf-runtime/evidence/DGR-030/README.md`. Legacy evidence remains provenance only and grants no implementation completion credit.
|
||||
|
||||
@@ -538,13 +538,14 @@
|
||||
"Keep every backend/model/recipe lane registered-dark until a separate real-hardware certification record exists.",
|
||||
"Applicable shared quality gates in `prd.json` pass, and the evidence handoff records exact commands/results, changed files, limitations, and dependency handoff."
|
||||
],
|
||||
"passes": false,
|
||||
"passes": true,
|
||||
"notes": "Generated source issue: .scratch/distributed-gguf-runtime/issues/030-add-accelerator-build-presets-and-native-ci-matrix.md; prd.json is authoritative.",
|
||||
"blocks": [
|
||||
"DGR-053",
|
||||
"DGR-067",
|
||||
"DGR-068"
|
||||
]
|
||||
],
|
||||
"completionNotes": "Completed by agent"
|
||||
},
|
||||
{
|
||||
"id": "DGR-031",
|
||||
@@ -2161,6 +2162,6 @@
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"updatedAt": "2026-07-22T06:44:18.107Z"
|
||||
"updatedAt": "2026-07-23T07:51:08.112Z"
|
||||
}
|
||||
}
|
||||
@@ -120,5 +120,5 @@ M1: Build system + protocol (DGR-021..033)
|
||||
|
||||
- Ralph runs headless: reads backlog, spawns fresh Claude Code per ticket, verifies, reports
|
||||
- DGR-019/020 marked `ready-for-human` — needs review before certifying
|
||||
- Changes left uncommitted for review per Ralph policy (unless explicitly pushed)
|
||||
- As of July 23, 2026: `autoCommit = true` in `.ralph-tui/config.toml` — the engine now commits after every completed task, and a supervisor process pushes each commit to `origin/ralph/distributed-gguf-runtime` immediately.
|
||||
- `ralph-tui resume` picks up where it left off
|
||||
@@ -52,6 +52,24 @@
|
||||
"smoke_output_token": "usage",
|
||||
"ctest_regex": "^test-meshnet-range-ownership$"
|
||||
},
|
||||
"accelerator_presets": {
|
||||
"cuda": {
|
||||
"backend_flag": "GGML_CUDA",
|
||||
"sdk_probe": {"binary": "nvcc", "env_var": "CUDACXX"}
|
||||
},
|
||||
"rocm": {
|
||||
"backend_flag": "GGML_HIP",
|
||||
"sdk_probe": {"binary": "hipcc", "env_var": "HIPCXX"}
|
||||
},
|
||||
"vulkan": {
|
||||
"backend_flag": "GGML_VULKAN",
|
||||
"sdk_probe": {"binary": "glslc", "env_var": "VULKAN_SDK_GLSLC"}
|
||||
},
|
||||
"metal": {
|
||||
"backend_flag": "GGML_METAL",
|
||||
"sdk_probe": {"binary": "xcrun", "platform_only": "darwin"}
|
||||
}
|
||||
},
|
||||
"required_upstream_blobs": {
|
||||
"CMakeLists.txt": "81f23d7e70b7378511af5d01be680c03aebc2b15"
|
||||
},
|
||||
|
||||
@@ -109,9 +109,41 @@ def _load_lock() -> dict[str, Any]:
|
||||
"workspace": "build/llama.cpp",
|
||||
}:
|
||||
raise DependencyError("retrieval must use the locked detached-commit build workspace")
|
||||
_verify_accelerator_presets(lock)
|
||||
return lock
|
||||
|
||||
|
||||
def _verify_accelerator_presets(lock: dict[str, Any]) -> None:
|
||||
"""Each preset must isolate one backend that the CPU default leaves OFF.
|
||||
|
||||
This is what keeps DGR-030's presets from ever being able to change the
|
||||
deterministic CPU default recorded in ``build.configure_flags``: a preset
|
||||
can only exist for a flag this lock already pins OFF, and
|
||||
``accelerator_configure_flags`` only ever returns a fresh list, never
|
||||
mutates ``build.configure_flags`` in place.
|
||||
"""
|
||||
presets = lock.get("accelerator_presets", {})
|
||||
if not isinstance(presets, dict):
|
||||
raise DependencyError("accelerator_presets must be a JSON object")
|
||||
if not presets:
|
||||
return
|
||||
base_flags = dict(flag[len("-D"):].split("=", 1) for flag in lock["build"]["configure_flags"])
|
||||
for name, preset in presets.items():
|
||||
if not isinstance(preset, dict):
|
||||
raise DependencyError(f"accelerator_presets.{name} must be a JSON object")
|
||||
backend_flag = preset.get("backend_flag")
|
||||
if not isinstance(backend_flag, str) or not backend_flag:
|
||||
raise DependencyError(f"accelerator_presets.{name} is missing backend_flag")
|
||||
if base_flags.get(backend_flag) != "OFF":
|
||||
raise DependencyError(
|
||||
f"accelerator_presets.{name} backend flag {backend_flag} must be OFF in "
|
||||
"the deterministic CPU default build.configure_flags"
|
||||
)
|
||||
probe = preset.get("sdk_probe")
|
||||
if not isinstance(probe, dict) or not isinstance(probe.get("binary"), str) or not probe["binary"]:
|
||||
raise DependencyError(f"accelerator_presets.{name} is missing an sdk_probe.binary")
|
||||
|
||||
|
||||
def _patches(lock: dict[str, Any]) -> list[pathlib.Path]:
|
||||
series = [line for line in (PATCH_DIR / "series").read_text().splitlines() if line]
|
||||
if series != lock["patch_series"] or series != sorted(series) or not series:
|
||||
@@ -507,6 +539,116 @@ def ctest_lane(build_dir: pathlib.Path) -> None:
|
||||
print(_run(_ctest(), "--test-dir", str(build_dir), "-R", regex, "--output-on-failure"))
|
||||
|
||||
|
||||
def _sdk_probe(probe: dict[str, Any]) -> str | None:
|
||||
"""Resolve one accelerator lane's SDK binary, or None if it is unavailable."""
|
||||
platform_only = probe.get("platform_only")
|
||||
if platform_only and sys.platform != platform_only:
|
||||
return None
|
||||
env_var = probe.get("env_var")
|
||||
if env_var:
|
||||
override = os.environ.get(env_var)
|
||||
if override:
|
||||
return override
|
||||
return shutil.which(probe["binary"])
|
||||
|
||||
|
||||
def accelerator_status(name: str, lock: dict[str, Any] | None = None) -> dict[str, Any]:
|
||||
"""Report whether lane `name`'s SDK is present, never raising for absence.
|
||||
|
||||
This is the single source of truth for DGR-030's "unavailable/skipped, not
|
||||
false success" contract: absence is reported as data, not swallowed and
|
||||
not escalated into a build attempt.
|
||||
"""
|
||||
lock = lock if lock is not None else _load_lock()
|
||||
presets = lock.get("accelerator_presets", {})
|
||||
if name not in presets:
|
||||
raise DependencyError(f"unknown accelerator lane: {name}")
|
||||
probe = presets[name]["sdk_probe"]
|
||||
resolved = _sdk_probe(probe)
|
||||
if resolved is None:
|
||||
platform_only = probe.get("platform_only")
|
||||
if platform_only and sys.platform != platform_only:
|
||||
reason = f"platform {sys.platform!r} is not {platform_only!r}"
|
||||
else:
|
||||
reason = f"{probe['binary']} is unavailable on PATH"
|
||||
return {"lane": name, "available": False, "reason": reason}
|
||||
return {"lane": name, "available": True, "sdk_binary": resolved}
|
||||
|
||||
|
||||
def accelerator_configure_flags(lock: dict[str, Any], name: str) -> list[str]:
|
||||
"""The CPU default's configure flags with exactly one backend flag flipped ON.
|
||||
|
||||
Returns a new list; `lock["build"]["configure_flags"]` (the deterministic
|
||||
CPU default DGR-029 locked) is never mutated.
|
||||
"""
|
||||
presets = lock.get("accelerator_presets", {})
|
||||
if name not in presets:
|
||||
raise DependencyError(f"unknown accelerator lane: {name}")
|
||||
backend_flag = presets[name]["backend_flag"]
|
||||
target = f"-D{backend_flag}="
|
||||
flags: list[str] = []
|
||||
replaced = False
|
||||
for flag in lock["build"]["configure_flags"]:
|
||||
if flag.startswith(target):
|
||||
flags.append(f"-D{backend_flag}=ON")
|
||||
replaced = True
|
||||
else:
|
||||
flags.append(flag)
|
||||
if not replaced:
|
||||
raise DependencyError(f"accelerator lane {name} backend flag {backend_flag} is not a locked base flag")
|
||||
return flags
|
||||
|
||||
|
||||
def accelerator_build(source: pathlib.Path, name: str, build_dir: pathlib.Path) -> pathlib.Path:
|
||||
"""Compile lane `name` into its own out-of-tree directory. Compile-only.
|
||||
|
||||
This never runs `smoke`/`ctest_lane`: exercising a binary linked against an
|
||||
accelerator backend would touch real hardware, and DGR-030 keeps every
|
||||
backend/model/recipe lane registered-dark (compiled, never certified)
|
||||
until a separate real-hardware certification record exists.
|
||||
"""
|
||||
lock = _load_lock()
|
||||
_patches(lock)
|
||||
_verify_source(source, lock, require_clean=False)
|
||||
_verify_patched_source(source, lock)
|
||||
expected_marker = source / "cmake/meshnet-patch-stack.cmake"
|
||||
if not expected_marker.is_file():
|
||||
raise DependencyError("patch stack is not applied: Meshnet CMake marker is absent")
|
||||
if build_dir.exists():
|
||||
raise DependencyError(f"accelerator build directory already exists; use a clean build dir: {build_dir}")
|
||||
status = accelerator_status(name, lock)
|
||||
if not status["available"]:
|
||||
raise DependencyError(f"accelerator lane {name} SDK is unavailable: {status['reason']}")
|
||||
flags = accelerator_configure_flags(lock, name)
|
||||
cmake = _cmake()
|
||||
_run(cmake, "-G", lock["build"]["generator"], "-S", str(source), "-B", str(build_dir), *flags)
|
||||
for target in lock["build"]["native_targets"]:
|
||||
_run(cmake, "--build", str(build_dir), "--target", target, "-j2")
|
||||
metadata = {
|
||||
"lane": name,
|
||||
"backend_flag": lock["accelerator_presets"][name]["backend_flag"],
|
||||
"commit": lock["commit"],
|
||||
"commit_tree": lock["commit_tree"],
|
||||
"patches": {patch.name: hashlib.sha256(patch.read_bytes()).hexdigest() for patch in _patches(lock)},
|
||||
"configure_flags": flags,
|
||||
"cmake": _run(cmake, "--version").splitlines()[0],
|
||||
"cxx": _run("c++", "--version").splitlines()[0],
|
||||
"sdk_binary": status["sdk_binary"],
|
||||
"model_downloads": False,
|
||||
"hardware_execution": False,
|
||||
"hardware_certified": False,
|
||||
"semantic_certification": False,
|
||||
"note": (
|
||||
"compiled only; no accelerator device was exercised or driven. "
|
||||
"Backend/model/recipe capability remains registered-dark until a "
|
||||
"separate real-hardware certification record exists (see "
|
||||
"DGR-041/053/067)."
|
||||
),
|
||||
}
|
||||
(build_dir / "meshnet-build-metadata.json").write_text(json.dumps(metadata, indent=2, sort_keys=True) + "\n")
|
||||
return build_dir
|
||||
|
||||
|
||||
def verify(workspace: pathlib.Path) -> None:
|
||||
"""Apply, verify, reverse, and leave the exact cached pin pristine."""
|
||||
source = fetch(workspace)
|
||||
@@ -562,6 +704,12 @@ def main() -> int:
|
||||
smoke_parser.add_argument("--binary", type=pathlib.Path, required=True)
|
||||
ctest_parser = subcommands.add_parser("ctest")
|
||||
ctest_parser.add_argument("--build-dir", type=pathlib.Path, required=True)
|
||||
accel_status_parser = subcommands.add_parser("accelerator-status")
|
||||
accel_status_parser.add_argument("--name", required=True)
|
||||
accel_build_parser = subcommands.add_parser("accelerator-build")
|
||||
accel_build_parser.add_argument("--name", required=True)
|
||||
accel_build_parser.add_argument("--source-dir", type=pathlib.Path, required=True)
|
||||
accel_build_parser.add_argument("--build-dir", type=pathlib.Path, required=True)
|
||||
reproduce_parser = subcommands.add_parser("reproduce")
|
||||
reproduce_parser.add_argument("--workspace", type=pathlib.Path, default=ROOT / "build/llama.cpp")
|
||||
args = parser.parse_args()
|
||||
@@ -582,6 +730,10 @@ def main() -> int:
|
||||
smoke(args.binary)
|
||||
elif args.command == "ctest":
|
||||
ctest_lane(args.build_dir)
|
||||
elif args.command == "accelerator-status":
|
||||
print(json.dumps(accelerator_status(args.name), indent=2, sort_keys=True))
|
||||
elif args.command == "accelerator-build":
|
||||
accelerator_build(args.source_dir, args.name, args.build_dir)
|
||||
else:
|
||||
reproduce(args.workspace)
|
||||
except DependencyError as error:
|
||||
|
||||
112
scripts/native_accelerator_matrix.py
Normal file
112
scripts/native_accelerator_matrix.py
Normal file
@@ -0,0 +1,112 @@
|
||||
#!/usr/bin/env python3
|
||||
"""DGR-030: native CI/build matrix over the CPU default plus accelerator lanes.
|
||||
|
||||
Runs the exact deterministic CPU lane DGR-029 locked (unchanged), then probes
|
||||
each accelerator preset (CUDA, ROCm, Vulkan, Metal) from `UPSTREAM_LOCK.json`
|
||||
and compiles the ones whose SDK is present on this machine into their own
|
||||
out-of-tree build directory.
|
||||
|
||||
A lane whose SDK is absent is reported as `skipped` with the exact probe
|
||||
reason, never treated as a false pass. A lane that compiles is reported as
|
||||
`built`, carrying exact compiler/SDK/upstream-pin/patch-stack/build-option
|
||||
evidence — never as a certified capability. This script never runs an
|
||||
accelerator binary and never certifies a backend/model/recipe: real-hardware
|
||||
certification is separate future work (DGR-041/053/067).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(ROOT / "scripts"))
|
||||
import llama_cpp_dependency as dep # noqa: E402
|
||||
|
||||
|
||||
def _cpu_lane(source: pathlib.Path, workspace: pathlib.Path) -> dict[str, Any]:
|
||||
build_dir = workspace.resolve() / "build"
|
||||
if build_dir.exists():
|
||||
return {
|
||||
"lane": "cpu",
|
||||
"status": "skipped",
|
||||
"reason": f"build directory already exists; remove for a clean rebuild: {build_dir}",
|
||||
}
|
||||
binary = dep.build(source, build_dir)
|
||||
dep.smoke(binary)
|
||||
dep.ctest_lane(build_dir)
|
||||
metadata = json.loads((build_dir / "meshnet-build-metadata.json").read_text())
|
||||
return {"lane": "cpu", "status": "built", "build_dir": str(build_dir), "metadata": metadata}
|
||||
|
||||
|
||||
def _accelerator_lane(source: pathlib.Path, workspace: pathlib.Path, name: str, lock: dict[str, Any]) -> dict[str, Any]:
|
||||
status = dep.accelerator_status(name, lock)
|
||||
if not status["available"]:
|
||||
return {"lane": name, "status": "skipped", "reason": status["reason"]}
|
||||
build_dir = workspace.resolve() / f"build-{name}"
|
||||
if build_dir.exists():
|
||||
return {
|
||||
"lane": name,
|
||||
"status": "skipped",
|
||||
"reason": f"build directory already exists; remove for a clean rebuild: {build_dir}",
|
||||
}
|
||||
dep.accelerator_build(source, name, build_dir)
|
||||
metadata = json.loads((build_dir / "meshnet-build-metadata.json").read_text())
|
||||
return {"lane": name, "status": "built", "build_dir": str(build_dir), "metadata": metadata}
|
||||
|
||||
|
||||
def run_matrix(workspace: pathlib.Path) -> dict[str, Any]:
|
||||
"""Fetch/apply once, run every lane, then always reverse the checkout."""
|
||||
source = dep.fetch(workspace)
|
||||
dep.apply(source)
|
||||
lanes: list[dict[str, Any]] = []
|
||||
try:
|
||||
lock = dep._load_lock()
|
||||
try:
|
||||
lanes.append(_cpu_lane(source, workspace))
|
||||
except dep.DependencyError as error:
|
||||
lanes.append({"lane": "cpu", "status": "failed", "reason": str(error)})
|
||||
for name in lock.get("accelerator_presets", {}):
|
||||
try:
|
||||
lanes.append(_accelerator_lane(source, workspace, name, lock))
|
||||
except dep.DependencyError as error:
|
||||
lanes.append({"lane": name, "status": "failed", "reason": str(error)})
|
||||
finally:
|
||||
dep.reverse(source)
|
||||
failed_lanes = [lane["lane"] for lane in lanes if lane["status"] == "failed"]
|
||||
return {
|
||||
"lanes": lanes,
|
||||
"hardware_certified": False,
|
||||
"note": (
|
||||
"A `built` lane means it compiled with the exact recorded compiler/SDK/"
|
||||
"upstream-pin/patch-stack/build-option evidence — it never means an "
|
||||
"accelerator device was exercised. Every backend/model/recipe lane "
|
||||
"stays registered-dark until a separate real-hardware certification "
|
||||
"record exists."
|
||||
),
|
||||
"failed_lanes": failed_lanes,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--workspace", type=pathlib.Path, default=ROOT / "build/llama.cpp")
|
||||
parser.add_argument("--out", type=pathlib.Path, default=None, help="also write the JSON report here")
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
report = run_matrix(args.workspace)
|
||||
except dep.DependencyError as error:
|
||||
print(f"DGR-030 dependency error: {error}", file=sys.stderr)
|
||||
return 2
|
||||
text = json.dumps(report, indent=2, sort_keys=True)
|
||||
print(text)
|
||||
if args.out:
|
||||
args.out.write_text(text + "\n")
|
||||
return 1 if report["failed_lanes"] else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -334,3 +334,157 @@ def test_ctest_lane_raises_an_actionable_error_for_a_failing_named_test(tmp_path
|
||||
assert "meshnet-fixture-fail" in str(error)
|
||||
else:
|
||||
raise AssertionError("a failing named CTest lane must raise DependencyError")
|
||||
|
||||
|
||||
def test_accelerator_presets_isolate_one_backend_without_touching_the_cpu_default() -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
presets = lock["accelerator_presets"]
|
||||
|
||||
assert set(presets) == {"cuda", "rocm", "vulkan", "metal"}
|
||||
base_flags = list(lock["build"]["configure_flags"])
|
||||
base_values = dict(flag[len("-D"):].split("=", 1) for flag in base_flags)
|
||||
|
||||
for name, preset in presets.items():
|
||||
flags = dependency.accelerator_configure_flags(lock, name)
|
||||
|
||||
# The CPU default's own flag list is never mutated by building a preset.
|
||||
assert lock["build"]["configure_flags"] == base_flags
|
||||
|
||||
new_values = dict(flag[len("-D"):].split("=", 1) for flag in flags)
|
||||
backend_flag = preset["backend_flag"]
|
||||
assert base_values[backend_flag] == "OFF"
|
||||
assert new_values[backend_flag] == "ON"
|
||||
for other_flag, value in base_values.items():
|
||||
if other_flag != backend_flag:
|
||||
assert new_values[other_flag] == value, f"{name}: {other_flag} drifted from the CPU default"
|
||||
|
||||
|
||||
def test_accelerator_configure_flags_rejects_an_unknown_lane() -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
try:
|
||||
dependency.accelerator_configure_flags(lock, "bogus")
|
||||
except dependency.DependencyError as error:
|
||||
assert "unknown accelerator lane" in str(error)
|
||||
else:
|
||||
raise AssertionError("an unknown accelerator lane must be refused")
|
||||
|
||||
|
||||
def test_accelerator_status_reports_unavailable_sdks_without_raising(monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
monkeypatch.setattr(dependency.shutil, "which", lambda name: None)
|
||||
|
||||
for name in ("cuda", "rocm", "vulkan"):
|
||||
env_var = lock["accelerator_presets"][name]["sdk_probe"]["env_var"]
|
||||
monkeypatch.delenv(env_var, raising=False)
|
||||
binary = lock["accelerator_presets"][name]["sdk_probe"]["binary"]
|
||||
assert dependency.accelerator_status(name, lock) == {
|
||||
"lane": name,
|
||||
"available": False,
|
||||
"reason": f"{binary} is unavailable on PATH",
|
||||
}
|
||||
|
||||
monkeypatch.setattr(dependency.sys, "platform", "linux")
|
||||
assert dependency.accelerator_status("metal", lock) == {
|
||||
"lane": "metal",
|
||||
"available": False,
|
||||
"reason": "platform 'linux' is not 'darwin'",
|
||||
}
|
||||
|
||||
|
||||
def test_accelerator_status_honors_an_explicit_sdk_override(tmp_path, monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
fake_nvcc = tmp_path / "nvcc"
|
||||
fake_nvcc.write_text("#!/bin/sh\nexit 0\n")
|
||||
fake_nvcc.chmod(0o755)
|
||||
monkeypatch.setenv("CUDACXX", str(fake_nvcc))
|
||||
|
||||
assert dependency.accelerator_status("cuda", lock) == {
|
||||
"lane": "cuda",
|
||||
"available": True,
|
||||
"sdk_binary": str(fake_nvcc),
|
||||
}
|
||||
|
||||
|
||||
def test_accelerator_status_rejects_an_unknown_lane() -> None:
|
||||
dependency = _load_dependency_module()
|
||||
lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
try:
|
||||
dependency.accelerator_status("bogus", lock)
|
||||
except dependency.DependencyError as error:
|
||||
assert "unknown accelerator lane" in str(error)
|
||||
else:
|
||||
raise AssertionError("an unknown accelerator lane must be refused")
|
||||
|
||||
|
||||
def test_accelerator_build_refuses_to_compile_an_unavailable_lane(tmp_path, monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
source = tmp_path / "source"
|
||||
(source / "cmake").mkdir(parents=True)
|
||||
(source / "cmake" / "meshnet-patch-stack.cmake").write_text("# marker\n")
|
||||
monkeypatch.setattr(dependency, "_verify_source", lambda *a, **k: None)
|
||||
monkeypatch.setattr(dependency, "_verify_patched_source", lambda *a, **k: None)
|
||||
monkeypatch.setattr(dependency.shutil, "which", lambda name: None)
|
||||
monkeypatch.delenv("CUDACXX", raising=False)
|
||||
|
||||
build_dir = tmp_path / "build-cuda"
|
||||
try:
|
||||
dependency.accelerator_build(source, "cuda", build_dir)
|
||||
except dependency.DependencyError as error:
|
||||
assert "SDK is unavailable" in str(error)
|
||||
else:
|
||||
raise AssertionError("accelerator_build must refuse to compile an unavailable lane")
|
||||
assert not build_dir.exists()
|
||||
|
||||
|
||||
@requires_cmake
|
||||
def test_accelerator_build_compiles_the_available_lane_with_isolated_evidence(tmp_path, monkeypatch) -> None:
|
||||
dependency = _load_dependency_module()
|
||||
|
||||
# A tiny synthetic project stands in for the patched llama.cpp checkout —
|
||||
# it only needs the meshnet patch-stack marker and one target, proving
|
||||
# accelerator_build's configure/build/evidence wiring without a multi-minute
|
||||
# llama.cpp compile or a real GPU SDK.
|
||||
source = tmp_path / "source"
|
||||
(source / "cmake").mkdir(parents=True)
|
||||
(source / "cmake" / "meshnet-patch-stack.cmake").write_text("# marker\n")
|
||||
(source / "CMakeLists.txt").write_text(
|
||||
"cmake_minimum_required(VERSION 3.14)\n"
|
||||
"project(accelerator_lane_fixture NONE)\n"
|
||||
"option(GGML_CUDA \"\" OFF)\n"
|
||||
"if(GGML_CUDA)\n"
|
||||
" file(WRITE ${CMAKE_BINARY_DIR}/lane-flag-on.txt \"on\")\n"
|
||||
"endif()\n"
|
||||
"add_custom_target(fixture-target ALL COMMAND ${CMAKE_COMMAND} -E true)\n"
|
||||
)
|
||||
|
||||
base_lock = json.loads((LLAMA_DIR / "UPSTREAM_LOCK.json").read_text())
|
||||
fake_lock = dict(base_lock)
|
||||
fake_lock["build"] = {
|
||||
**base_lock["build"],
|
||||
"generator": "Unix Makefiles",
|
||||
"configure_flags": ["-DGGML_CUDA=OFF"],
|
||||
"native_targets": ["fixture-target"],
|
||||
}
|
||||
monkeypatch.setattr(dependency, "_load_lock", lambda: fake_lock)
|
||||
monkeypatch.setattr(dependency, "_patches", lambda lock: [])
|
||||
monkeypatch.setattr(dependency, "_verify_source", lambda *a, **k: None)
|
||||
monkeypatch.setattr(dependency, "_verify_patched_source", lambda *a, **k: None)
|
||||
monkeypatch.setenv("CUDACXX", str(dependency._cmake()))
|
||||
|
||||
build_dir = tmp_path / "build-cuda"
|
||||
result = dependency.accelerator_build(source, "cuda", build_dir)
|
||||
|
||||
assert result == build_dir
|
||||
assert (build_dir / "lane-flag-on.txt").is_file()
|
||||
metadata = json.loads((build_dir / "meshnet-build-metadata.json").read_text())
|
||||
assert metadata["lane"] == "cuda"
|
||||
assert metadata["backend_flag"] == "GGML_CUDA"
|
||||
assert metadata["configure_flags"] == ["-DGGML_CUDA=ON"]
|
||||
assert metadata["hardware_execution"] is False
|
||||
assert metadata["hardware_certified"] is False
|
||||
assert metadata["semantic_certification"] is False
|
||||
assert "registered-dark" in metadata["note"]
|
||||
|
||||
171
tests/test_native_accelerator_matrix.py
Normal file
171
tests/test_native_accelerator_matrix.py
Normal file
@@ -0,0 +1,171 @@
|
||||
"""Offline behavior tests for DGR-030's native CI/build matrix orchestration.
|
||||
|
||||
These tests never fetch or compile llama.cpp: `llama_cpp_dependency`'s fetch/
|
||||
apply/reverse/build/smoke/ctest_lane/accelerator_status/accelerator_build are
|
||||
stubbed so the matrix's own lane-reporting and cleanup contract is exercised
|
||||
in isolation. The real compile path is covered separately by
|
||||
`tests/test_llama_cpp_dependency.py`'s `accelerator_build`/CPU-lane tests and
|
||||
by a live run recorded in the DGR-030 evidence README.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
|
||||
|
||||
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
||||
MATRIX_SCRIPT = ROOT / "scripts/native_accelerator_matrix.py"
|
||||
DEP_SCRIPT = ROOT / "scripts/llama_cpp_dependency.py"
|
||||
|
||||
|
||||
def _load_matrix_module(monkeypatch):
|
||||
"""Load private copies of both modules with a controllable `dep`.
|
||||
|
||||
`native_accelerator_matrix.py` does `import llama_cpp_dependency as dep`
|
||||
after inserting `scripts/` onto `sys.path`; pre-registering our own module
|
||||
instance under that name in `sys.modules` (undone by monkeypatch at
|
||||
teardown) makes the matrix module bind to the stub instead of importing a
|
||||
fresh copy of the real dependency module.
|
||||
"""
|
||||
dep_spec = importlib.util.spec_from_file_location("llama_cpp_dependency_matrix_dep", DEP_SCRIPT)
|
||||
dep = importlib.util.module_from_spec(dep_spec)
|
||||
dep_spec.loader.exec_module(dep)
|
||||
monkeypatch.setitem(sys.modules, "llama_cpp_dependency", dep)
|
||||
|
||||
matrix_spec = importlib.util.spec_from_file_location("native_accelerator_matrix", MATRIX_SCRIPT)
|
||||
matrix = importlib.util.module_from_spec(matrix_spec)
|
||||
matrix_spec.loader.exec_module(matrix)
|
||||
return matrix, dep
|
||||
|
||||
|
||||
def test_matrix_reports_unavailable_accelerator_sdks_as_skipped_not_false_success(tmp_path, monkeypatch) -> None:
|
||||
matrix, dep = _load_matrix_module(monkeypatch)
|
||||
|
||||
workspace = tmp_path / "llama.cpp"
|
||||
source = workspace / "source"
|
||||
source.mkdir(parents=True)
|
||||
calls: list = []
|
||||
|
||||
monkeypatch.setattr(dep, "fetch", lambda ws: source)
|
||||
monkeypatch.setattr(dep, "apply", lambda src: calls.append(("apply", src)))
|
||||
monkeypatch.setattr(dep, "reverse", lambda src: calls.append(("reverse", src)))
|
||||
monkeypatch.setattr(
|
||||
dep,
|
||||
"_load_lock",
|
||||
lambda: {"accelerator_presets": {"cuda": {}, "rocm": {}, "vulkan": {}, "metal": {}}},
|
||||
)
|
||||
|
||||
def _cpu_build(src, build_dir):
|
||||
build_dir.mkdir(parents=True)
|
||||
(build_dir / "meshnet-build-metadata.json").write_text(json.dumps({"lane": "cpu"}))
|
||||
return build_dir / "bin/llama-gguf-hash"
|
||||
|
||||
monkeypatch.setattr(dep, "build", _cpu_build)
|
||||
monkeypatch.setattr(dep, "smoke", lambda binary: calls.append(("smoke", binary)))
|
||||
monkeypatch.setattr(dep, "ctest_lane", lambda build_dir: calls.append(("ctest", build_dir)))
|
||||
monkeypatch.setattr(
|
||||
dep,
|
||||
"accelerator_status",
|
||||
lambda name, lock: {"lane": name, "available": False, "reason": f"{name} SDK is unavailable on PATH"},
|
||||
)
|
||||
|
||||
report = matrix.run_matrix(workspace)
|
||||
|
||||
assert report["lanes"][0] == {
|
||||
"lane": "cpu",
|
||||
"status": "built",
|
||||
"build_dir": str((workspace / "build").resolve()),
|
||||
"metadata": {"lane": "cpu"},
|
||||
}
|
||||
accelerator_lanes = {lane["lane"]: lane for lane in report["lanes"][1:]}
|
||||
assert set(accelerator_lanes) == {"cuda", "rocm", "vulkan", "metal"}
|
||||
for name, lane in accelerator_lanes.items():
|
||||
assert lane["status"] == "skipped"
|
||||
assert "unavailable" in lane["reason"]
|
||||
|
||||
assert report["failed_lanes"] == []
|
||||
assert report["hardware_certified"] is False
|
||||
assert ("reverse", source) in calls # cleanup always runs
|
||||
# Only the CPU lane is ever smoke-tested/ctested; skipped accelerator lanes are not.
|
||||
smoke_calls = [call for call in calls if call[0] == "smoke"]
|
||||
ctest_calls = [call for call in calls if call[0] == "ctest"]
|
||||
assert smoke_calls == [("smoke", (workspace.resolve() / "build" / "bin/llama-gguf-hash"))]
|
||||
assert ctest_calls == [("ctest", (workspace.resolve() / "build"))]
|
||||
|
||||
|
||||
def test_matrix_compiles_an_available_accelerator_lane_without_smoke_or_ctest(tmp_path, monkeypatch) -> None:
|
||||
matrix, dep = _load_matrix_module(monkeypatch)
|
||||
|
||||
workspace = tmp_path / "llama.cpp"
|
||||
source = workspace / "source"
|
||||
source.mkdir(parents=True)
|
||||
calls: list = []
|
||||
|
||||
monkeypatch.setattr(dep, "fetch", lambda ws: source)
|
||||
monkeypatch.setattr(dep, "apply", lambda src: None)
|
||||
monkeypatch.setattr(dep, "reverse", lambda src: calls.append("reverse"))
|
||||
monkeypatch.setattr(dep, "_load_lock", lambda: {"accelerator_presets": {"cuda": {}}})
|
||||
monkeypatch.setattr(
|
||||
matrix,
|
||||
"_cpu_lane",
|
||||
lambda src, ws: {"lane": "cpu", "status": "skipped", "reason": "pre-existing build dir"},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
dep, "accelerator_status", lambda name, lock: {"lane": name, "available": True, "sdk_binary": "/fake/nvcc"}
|
||||
)
|
||||
|
||||
def _accelerator_build(src, name, build_dir):
|
||||
calls.append(("accelerator_build", name))
|
||||
build_dir.mkdir(parents=True)
|
||||
(build_dir / "meshnet-build-metadata.json").write_text(
|
||||
json.dumps({"lane": name, "hardware_certified": False})
|
||||
)
|
||||
return build_dir
|
||||
|
||||
monkeypatch.setattr(dep, "accelerator_build", _accelerator_build)
|
||||
monkeypatch.setattr(dep, "smoke", lambda binary: calls.append(("smoke", binary)))
|
||||
monkeypatch.setattr(dep, "ctest_lane", lambda build_dir: calls.append(("ctest", build_dir)))
|
||||
|
||||
report = matrix.run_matrix(workspace)
|
||||
|
||||
assert report["lanes"][1]["lane"] == "cuda"
|
||||
assert report["lanes"][1]["status"] == "built"
|
||||
assert report["lanes"][1]["metadata"]["hardware_certified"] is False
|
||||
assert ("accelerator_build", "cuda") in calls
|
||||
assert not any(call[0] in ("smoke", "ctest") for call in calls if isinstance(call, tuple))
|
||||
assert "reverse" in calls
|
||||
|
||||
|
||||
def test_matrix_reports_a_lane_failure_without_aborting_the_others_or_skipping_reverse(tmp_path, monkeypatch) -> None:
|
||||
matrix, dep = _load_matrix_module(monkeypatch)
|
||||
|
||||
workspace = tmp_path / "llama.cpp"
|
||||
source = workspace / "source"
|
||||
source.mkdir(parents=True)
|
||||
calls: list = []
|
||||
|
||||
monkeypatch.setattr(dep, "fetch", lambda ws: source)
|
||||
monkeypatch.setattr(dep, "apply", lambda src: None)
|
||||
monkeypatch.setattr(dep, "reverse", lambda src: calls.append("reverse"))
|
||||
monkeypatch.setattr(dep, "_load_lock", lambda: {"accelerator_presets": {"cuda": {}, "vulkan": {}}})
|
||||
|
||||
def _cpu_lane_raises(src, ws):
|
||||
raise dep.DependencyError("simulated cpu compile failure")
|
||||
|
||||
monkeypatch.setattr(matrix, "_cpu_lane", _cpu_lane_raises)
|
||||
monkeypatch.setattr(
|
||||
dep,
|
||||
"accelerator_status",
|
||||
lambda name, lock: {"lane": name, "available": False, "reason": f"{name} SDK is unavailable on PATH"},
|
||||
)
|
||||
|
||||
report = matrix.run_matrix(workspace)
|
||||
|
||||
assert report["lanes"][0] == {"lane": "cpu", "status": "failed", "reason": "simulated cpu compile failure"}
|
||||
assert report["failed_lanes"] == ["cpu"]
|
||||
accelerator_statuses = {lane["lane"]: lane["status"] for lane in report["lanes"][1:]}
|
||||
assert accelerator_statuses == {"cuda": "skipped", "vulkan": "skipped"}
|
||||
assert "reverse" in calls
|
||||
Reference in New Issue
Block a user