Merge branch 'archived_ralph/dgr-001-performance-contract' into merge/all-branches-into-master

# Conflicts:
#	.claude/memory/MEMORY.md
#	.scratch/distributed-gguf-runtime/PRD.md
#	.scratch/distributed-gguf-runtime/RALPH-CONTEXT.md
#	.scratch/distributed-gguf-runtime/README.md
#	.scratch/distributed-gguf-runtime/architecture.md
#	.scratch/distributed-gguf-runtime/evidence/DGR-017/README.md
#	.scratch/distributed-gguf-runtime/implementation-strategy.md
#	.scratch/distributed-gguf-runtime/issues/07-add-isolated-concurrent-local-hot-kv-state.md
#	.scratch/distributed-gguf-runtime/issues/13-harden-failure-cancellation-and-restart-semantics.md
#	.scratch/distributed-gguf-runtime/milestones.md
#	.scratch/distributed-gguf-runtime/prd.json
#	docs/issues/distributed-gguf-runtime/01-lock-the-safetensors-versus-gguf-performance-contract.md
#	docs/issues/distributed-gguf-runtime/02-adopt-the-versioned-grpc-shard-protocol.md
#	docs/issues/distributed-gguf-runtime/03-define-exact-artifact-and-runtime-recipe-identity.md
#	docs/issues/distributed-gguf-runtime/05-implement-dense-llama-range-aware-gguf-ownership.md
#	docs/issues/distributed-gguf-runtime/06-implement-architecture-defined-boundary-input-output.md
This commit is contained in:
Dobromir Popov
2026-07-17 13:44:52 +03:00
124 changed files with 24939 additions and 91 deletions

View File

@@ -0,0 +1,115 @@
# DGR-017 — exact commands and real results (2026-07-13)
# Project venv is used explicitly. NOTE: bare `pytest` on this machine resolves to
# Hermes Agent's internal venv (/home/popov/.hermes/...), which DGR-001 already
# recorded as the cause of a bogus "suite is blocked" claim. Always use $VP.
VP=/run/media/popov/d/DEV/repos/d-popov.com/AI/.venv/bin/python # Python 3.14.6
# ---------------------------------------------------------------------------
# 1. Resolve the target from upstream metadata ONLY. No weight payload downloaded.
# Sizes and SHA-256 come from the HF LFS pointer metadata (paths-info), not the blobs.
# ---------------------------------------------------------------------------
curl -sS "https://huggingface.co/api/models/zai-org/GLM-5.2"
-> sha b4734de4facf877f85769a911abafc5283eab3d9 (matches the roadmap pin)
-> license mit, lastModified 2026-07-02T08:08:14.000Z
curl -sS "https://huggingface.co/api/models/unsloth/GLM-5.2-GGUF"
-> sha abc55e72527792c6e77069c99b4cb7de16fa9f23 (matches the roadmap pin)
-> license mit, lastModified 2026-06-23T15:18:23.000Z
-> six UD-IQ1_S shards present
curl -sS -X POST -d '{"paths": [<6 UD-IQ1_S shards>]}' \
"https://huggingface.co/api/models/unsloth/GLM-5.2-GGUF/paths-info/abc55e7..."
-> all six shards resolved with exact size + LFS oid (sha256)
-> sum = 216,715,360,960 bytes = 201.832 GiB = 216.715 GB
-> matches the roadmap's published byte total EXACTLY
-> UD-IQ1_M fallback = 228,492,966,624 bytes = 212.801 GiB (also matches)
curl -sS ".../resolve/b4734de4.../{config.json,chat_template.jinja,
generation_config.json,tokenizer_config.json}"
-> config.json 3732 B sha256 185f93ee6d12548e16a847e279dc0c3c90b1524c970b0866b42fb545747d859a
-> chat_template.jinja 5076 B sha256 172dc74a35e1752df75ecfb2b2cf9326d2852bb1379868ebeec9571654489679
-> generation_config.json 194 B sha256 ac76b43d8683d3b930126870fc8be73d8679308fe752fa1f381096d8354f6a55
-> tokenizer_config.json 761 B sha256 98b1271574f41abf89427ae2dda030d94dc9478f0edc5a8bd240db213c6fd5fc
# ---------------------------------------------------------------------------
# 2. Verify the checked-in pins still match live upstream (reproducible, no weights)
# ---------------------------------------------------------------------------
$VP scripts/refresh_glm_target_manifest.py --check
-> "target manifest and architecture snapshot match upstream"
-> exit 0
# ---------------------------------------------------------------------------
# 3. Upstream llama.cpp / donor status refresh (GitHub REST API, read-only)
# ---------------------------------------------------------------------------
curl -sS "https://api.github.com/repos/ggml-org/llama.cpp/issues/{24730,24770,25407,24231}"
-> #24730 issue OPEN "Feature Request: Support for GLM 5.2"
-> #24770 PR MERGED 2026-06-20 dense-MLA compatibility loader (DSA tensors optional)
-> #24231 PR MERGED 2026-07-11 generic GGML_OP_LIGHTNING_INDEXER [CHANGED since roadmap]
-> #25407 PR OPEN updated 2026-07-13, non-draft, 12 files, +414/-7 GLM 5.2 Indexer support
curl -sS "https://api.github.com/repos/Mesh-LLM/mesh-llm{,/branches/feat%2Fjianyang-glm-52}"
-> Apache-2.0, 2048 stars, branch head 9bd18f1509dff7fac21578635084035b3ba90a38 (2026-07-12)
-> recorded as donor only; nothing forked, nothing adopted
# ---------------------------------------------------------------------------
# 4. Seal the alpha contract (digest over its own canonical content)
# ---------------------------------------------------------------------------
$VP -c "seal_contract(...)" -> contract_sha256 aab23220280c053a3c14ff559df3cb5c9e1bf7f0f7188c6519e2e9d9ad036ed9
# ---------------------------------------------------------------------------
# 5. Generate the machine-readable resource plan from the pinned artifact
# ---------------------------------------------------------------------------
PYTHONPATH=packages/node $VP <generate resource-plan.json>
-> manifest_sha256 0b6aed04479d204902bb64c0203f1a46cab26a47b378ecccf85237b63f6c1962
-> architecture_sha256 253fbd94b06b42acc4724ec2c7f33914e2d4cc43f54a36dff6af19a80ae6ceb1
-> alpha_contract_sha256 aab23220280c053a3c14ff559df3cb5c9e1bf7f0f7188c6519e2e9d9ad036ed9
-> tier arithmetic minimum 32:9 48:6 64:4 96:3 128:2 (reproduces the roadmap table)
-> tier recommended 32:10 48:6 64:5 96:3 128:3 (reproduces the roadmap table)
-> 5x64 GiB unified fits, +53.28 GiB headroom
-> 3x96 GiB unified fits, +27.68 GiB headroom
-> 2x128 / 4x64 (fit probes) fit with only +2.08 GiB headroom across the WHOLE route
-> 2x112 GiB (= 224 GiB, the hard-fit floor) DOES NOT FIT: -23.52 GiB
-> 3x64 GiB does not fit: -49.12 GiB
# ---------------------------------------------------------------------------
# 6. Quality gates (project .venv, deterministic, offline, GPU-free)
# ---------------------------------------------------------------------------
$VP -m pytest -q tests/test_glm_alpha_target.py
-> 97 passed in 0.12s
-> includes coordinated shard/config substitution, malformed telemetry, and
contract-ID reseal rejection tests added during controller review
$VP -m pip wheel --no-deps packages/node -w /tmp/dgr017-wheel
$VP -m pip install --no-deps --target /tmp/dgr017-install /tmp/dgr017-wheel/*.whl
$VP -I -c "... from meshnet_node.glm_alpha import load_locked_target ..."
-> INSTALLED_WHEEL_PASS
-> packaged alpha-contract.json, target-manifest.json, and architecture-snapshot.json
load and cross-bind successfully outside the source tree
$VP -m compileall -q packages tests
-> exit 0
git diff --check
-> exit 0
$VP -m pytest -q # full deterministic suite
-> first final run: 1 failed, 851 passed, 13 skipped; only the tracker cancellation
race already documented by DGR-001/DGR-002 failed
$VP -m pytest -q tests/test_tracker_routing.py::test_tracker_dashboard_can_cancel_inflight_proxy
-> 1 passed, repeated 5/5 in isolation
$VP -m pytest -q # integrated rerun
-> 852 passed, 13 skipped in 253.30s (0:04:13)
# ---------------------------------------------------------------------------
# 7. Late independent-review repair (2026-07-14)
# ---------------------------------------------------------------------------
PYTHONPATH=packages/node $VP -m pytest -q tests/test_glm_alpha_target.py
-> 99 passed in 0.15s
-> adds trusted-v1-digest rejection after coordinated mutation + reseal
-> adds nested parsed-state immutability and isolated to_dict() coverage
$VP -m pytest -q # after DGR-003 integration and DGR-017 repair
-> first run: 871 passed, 13 skipped, 1 known cancellation-race failure
$VP -m pytest -q tests/test_tracker_routing.py::test_tracker_dashboard_can_cancel_inflight_proxy
-> 5/5 passed in isolation
$VP -m pytest -q # integrated rerun
-> 872 passed, 13 skipped in 253.46s (0:04:13)

View File

@@ -0,0 +1,255 @@
{
"generated_by": "DGR-017 meshnet_node.glm_alpha.planner",
"target": {
"gguf_repo_id": "unsloth/GLM-5.2-GGUF",
"gguf_revision": "abc55e72527792c6e77069c99b4cb7de16fa9f23",
"quantization": "UD-IQ1_S",
"total_bytes": 216715360960,
"total_gib": 201.832,
"total_gb": 216.715
},
"manifest_sha256": "0b6aed04479d204902bb64c0203f1a46cab26a47b378ecccf85237b63f6c1962",
"architecture_snapshot_sha256": "253fbd94b06b42acc4724ec2c7f33914e2d4cc43f54a36dff6af19a80ae6ceb1",
"alpha_contract_sha256": "aab23220280c053a3c14ff559df3cb5c9e1bf7f0f7188c6519e2e9d9ad036ed9",
"kv_assumptions": {
"dtype": "Q8_0",
"bytes_per_value": 1.0625,
"context_tokens": 16384,
"concurrency": 1,
"indexer_layout": "conservative",
"mla_values_per_token_per_layer": 576,
"backbone_layers": 78,
"indexer_full_layers": 21,
"note": "Alpha budgets indexer keys across all 78 layers (current experimental DSA layout), not only the 21 Full layers."
},
"kv_table_gib": {
"16384": {
"mla_only_q8_gib": 0.73,
"optimized_dsa_q8_gib": 0.77,
"conservative_dsa_q8_gib": 0.89,
"conservative_dsa_f16_gib": 1.68
},
"131072": {
"mla_only_q8_gib": 5.83,
"optimized_dsa_q8_gib": 6.18,
"conservative_dsa_q8_gib": 7.12,
"conservative_dsa_f16_gib": 13.41
},
"1048576": {
"mla_only_q8_gib": 46.62,
"optimized_dsa_q8_gib": 49.41,
"conservative_dsa_q8_gib": 56.98,
"conservative_dsa_f16_gib": 107.25
}
},
"aggregate_hard_fit_floor_gib": 224.0,
"aggregate_floor_class": "experimental_hard_fit_floor",
"placement_imbalance_factor": 1.1,
"tier_table": {
"32": {
"physical_usable_gib": 32.0,
"reserve_gib": 8.0,
"placement_budget_gib": 24.0,
"weight_gib": 201.832,
"kv_gib": 0.89,
"total_placement_gib": 202.722,
"arithmetic_minimum_nodes": 9,
"recommended_nodes": 10,
"imbalance_factor": 1.1
},
"48": {
"physical_usable_gib": 48.0,
"reserve_gib": 9.6,
"placement_budget_gib": 38.4,
"weight_gib": 201.832,
"kv_gib": 0.89,
"total_placement_gib": 202.722,
"arithmetic_minimum_nodes": 6,
"recommended_nodes": 6,
"imbalance_factor": 1.1
},
"64": {
"physical_usable_gib": 64.0,
"reserve_gib": 12.8,
"placement_budget_gib": 51.2,
"weight_gib": 201.832,
"kv_gib": 0.89,
"total_placement_gib": 202.722,
"arithmetic_minimum_nodes": 4,
"recommended_nodes": 5,
"imbalance_factor": 1.1
},
"96": {
"physical_usable_gib": 96.0,
"reserve_gib": 19.2,
"placement_budget_gib": 76.8,
"weight_gib": 201.832,
"kv_gib": 0.89,
"total_placement_gib": 202.722,
"arithmetic_minimum_nodes": 3,
"recommended_nodes": 3,
"imbalance_factor": 1.1
},
"128": {
"physical_usable_gib": 128.0,
"reserve_gib": 25.6,
"placement_budget_gib": 102.4,
"weight_gib": 201.832,
"kv_gib": 0.89,
"total_placement_gib": 202.722,
"arithmetic_minimum_nodes": 2,
"recommended_nodes": 3,
"imbalance_factor": 1.1
}
},
"routes": {
"recommended_5x64_unified": {
"node_count": 5,
"aggregate_usable_gib": 320.0,
"aggregate_placement_budget_gib": 256.0,
"required_placement_gib": 202.722,
"fits": true,
"meets_hard_fit_floor": true,
"no_single_node_can_admit_target": true,
"headroom_gib": 53.278,
"reasons": []
},
"recommended_3x96_unified": {
"node_count": 3,
"aggregate_usable_gib": 288.0,
"aggregate_placement_budget_gib": 230.4,
"required_placement_gib": 202.722,
"fits": true,
"meets_hard_fit_floor": true,
"no_single_node_can_admit_target": true,
"headroom_gib": 27.678,
"reasons": []
},
"recommended_3x128_unified": {
"node_count": 3,
"aggregate_usable_gib": 384.0,
"aggregate_placement_budget_gib": 307.2,
"required_placement_gib": 202.722,
"fits": true,
"meets_hard_fit_floor": true,
"no_single_node_can_admit_target": true,
"headroom_gib": 104.478,
"reasons": []
},
"fit_probe_2x128_unified": {
"node_count": 2,
"aggregate_usable_gib": 256.0,
"aggregate_placement_budget_gib": 204.8,
"required_placement_gib": 202.722,
"fits": true,
"meets_hard_fit_floor": true,
"no_single_node_can_admit_target": true,
"headroom_gib": 2.078,
"reasons": []
},
"fit_probe_4x64_unified": {
"node_count": 4,
"aggregate_usable_gib": 256.0,
"aggregate_placement_budget_gib": 204.8,
"required_placement_gib": 202.722,
"fits": true,
"meets_hard_fit_floor": true,
"no_single_node_can_admit_target": true,
"headroom_gib": 2.078,
"reasons": []
},
"hard_fit_floor_2x112_unified": {
"node_count": 2,
"aggregate_usable_gib": 224.0,
"aggregate_placement_budget_gib": 179.2,
"required_placement_gib": 202.722,
"fits": false,
"meets_hard_fit_floor": true,
"no_single_node_can_admit_target": true,
"headroom_gib": -23.522,
"reasons": [
"aggregate placement budget 179.2 GiB is below the 202.7 GiB the target needs after each node's reserve"
]
},
"insufficient_3x64_unified": {
"node_count": 3,
"aggregate_usable_gib": 192.0,
"aggregate_placement_budget_gib": 153.6,
"required_placement_gib": 202.722,
"fits": false,
"meets_hard_fit_floor": false,
"no_single_node_can_admit_target": true,
"headroom_gib": -49.122,
"reasons": [
"aggregate placement budget 153.6 GiB is below the 202.7 GiB the target needs after each node's reserve",
"aggregate usable memory 192.0 GiB is below the 224 GiB experimental hard-fit floor"
]
}
},
"seams": {
"3_nodes_2.5gbe": {
"node_count": 3,
"seam_count": 2,
"hidden_size": 6144,
"bytes_per_token_per_seam": 12288,
"prefill_bytes_per_seam": 201326592,
"decode_bytes_per_seam_per_token": 12288,
"dsa_sideband_bytes_per_query": 8192,
"link_rate_gbps": 2.5,
"meets_alpha_minimum": true,
"is_recommended_link": false,
"decode_serialization_ms_per_token": 0.0786,
"decode_latency_ms_per_token": 1.0,
"decode_bandwidth_share_ms_per_token": 0.0786,
"prefill_serialization_ms": 1288.49
},
"3_nodes_10.0gbe": {
"node_count": 3,
"seam_count": 2,
"hidden_size": 6144,
"bytes_per_token_per_seam": 12288,
"prefill_bytes_per_seam": 201326592,
"decode_bytes_per_seam_per_token": 12288,
"dsa_sideband_bytes_per_query": 8192,
"link_rate_gbps": 10.0,
"meets_alpha_minimum": true,
"is_recommended_link": true,
"decode_serialization_ms_per_token": 0.0197,
"decode_latency_ms_per_token": 1.0,
"decode_bandwidth_share_ms_per_token": 0.0197,
"prefill_serialization_ms": 322.123
},
"5_nodes_2.5gbe": {
"node_count": 5,
"seam_count": 4,
"hidden_size": 6144,
"bytes_per_token_per_seam": 12288,
"prefill_bytes_per_seam": 201326592,
"decode_bytes_per_seam_per_token": 12288,
"dsa_sideband_bytes_per_query": 8192,
"link_rate_gbps": 2.5,
"meets_alpha_minimum": true,
"is_recommended_link": false,
"decode_serialization_ms_per_token": 0.1573,
"decode_latency_ms_per_token": 2.0,
"decode_bandwidth_share_ms_per_token": 0.1573,
"prefill_serialization_ms": 2576.98
},
"5_nodes_10.0gbe": {
"node_count": 5,
"seam_count": 4,
"hidden_size": 6144,
"bytes_per_token_per_seam": 12288,
"prefill_bytes_per_seam": 201326592,
"decode_bytes_per_seam_per_token": 12288,
"dsa_sideband_bytes_per_query": 8192,
"link_rate_gbps": 10.0,
"meets_alpha_minimum": true,
"is_recommended_link": true,
"decode_serialization_ms_per_token": 0.0393,
"decode_latency_ms_per_token": 2.0,
"decode_bandwidth_share_ms_per_token": 0.0393,
"prefill_serialization_ms": 644.245
}
}
}

View File

@@ -0,0 +1,89 @@
{
"observed_at": "2026-07-13",
"observed_by": "DGR-017",
"method": "GitHub REST API (api.github.com), read-only; no fork, no clone, no patch adopted",
"refresh_note": "The roadmap's 2026-07-13 observations were re-verified against live upstream. One item changed: PR #24231 is now MERGED (2026-07-11), which the roadmap already anticipated as 'generic CPU lightning-indexer support is merged'.",
"llama_cpp": {
"repo": "ggml-org/llama.cpp",
"items": [
{
"ref": "issue #24730",
"url": "https://github.com/ggml-org/llama.cpp/issues/24730",
"title": "Feature Request: Support for GLM 5.2",
"type": "issue",
"state": "open",
"updated_at": "2026-07-03T22:02:15Z",
"meaning": "The umbrella GLM-5.2 support request is still open. GLM-5.2 is not fully supported upstream."
},
{
"ref": "PR #24770",
"url": "https://github.com/ggml-org/llama.cpp/pull/24770",
"title": "model : glm-dsa load DSA indexer tensors as optional",
"type": "pull_request",
"state": "closed",
"merged_at": "2026-06-20T10:48:24Z",
"meaning": "MERGED. GLM-5.2 loads, but through a dense-MLA compatibility path with DSA indexer tensors treated as optional. This is the fallback the alpha contract explicitly refuses: it can produce text without performing DSA/IndexShare computation."
},
{
"ref": "PR #24231",
"url": "https://github.com/ggml-org/llama.cpp/pull/24231",
"title": "New GGML_OP_LIGHTNING_INDEXER that implements DeepSeek V3.2/V4 lightning indexer",
"type": "pull_request",
"state": "closed",
"merged_at": "2026-07-11T09:39:07Z",
"meaning": "MERGED since the roadmap was written. A generic lightning-indexer op now exists in GGML. Backend coverage beyond CPU remains uneven and must be verified per backend by DGR-018, not assumed."
},
{
"ref": "PR #25407",
"url": "https://github.com/ggml-org/llama.cpp/pull/25407",
"title": "GLM 5.2 Indexer support",
"type": "pull_request",
"state": "open",
"draft": false,
"mergeable_state": "unstable",
"head_sha": "8dedd06415f36f10fc6091241a39b23c1bf0ee11",
"base": "master",
"commits": 6,
"changed_files": 12,
"additions": 414,
"deletions": 7,
"updated_at": "2026-07-13T15:28:51Z",
"meaning": "OPEN and actively moving (updated today). This is the real DSA/IndexShare implementation alpha needs. It is narrow — 12 files, +414/-7 — which is the single most important finding for donor policy: the semantics alpha requires are reviewable and trackable upstream, not a 261-patch fork."
}
]
},
"capability_status_for_alpha": {
"gguf_load_of_UD-IQ1_S": "expected via merged #24770, unverified by this project; DGR-018 must prove it against the exact pinned artifact",
"moe_routing_and_shared_expert": "expected supported; unverified here",
"compressed_mla_kv": "supported via the dense-MLA compatibility path",
"dsa_lightning_indexer": "generic GGML op merged (#24231); GLM-5.2 wiring still open (#25407)",
"indexshare_full_shared_roles": "NOT upstream; only in open PR #25407",
"mtp_nextn": "not required for alpha; NextN tensors must be explicitly loaded or excluded, never silently reinterpreted",
"conclusion": "As of 2026-07-13 no released upstream llama.cpp performs native GLM-5.2 DSA + IndexShare. A stock pin today would satisfy 'it emits text' via the dense fallback and would FAIL the alpha semantic-correctness contract. This is the gating technical risk for DGR-004 and DGR-018."
},
"donor": {
"repo": "Mesh-LLM/mesh-llm",
"url": "https://github.com/Mesh-LLM/mesh-llm",
"license": "Apache-2.0",
"stars_observed": 2048,
"pushed_at": "2026-07-13T06:45:51Z",
"glm_branch": "feat/jianyang-glm-52",
"glm_branch_head": "9bd18f1509dff7fac21578635084035b3ba90a38",
"glm_branch_head_date": "2026-07-12T06:37:43Z",
"policy": "TEST AND PATCH DONOR ONLY. Do not adopt the fork, its scheduler, discovery, routing, public mesh, or package manager. Meshnet remains the sole control plane (RALPH-CONTEXT runtime decision, ADR-0020).",
"focused_candidates": [
"GLM DSA graph semantics",
"lightning indexer and sparse-attention tests",
"IndexShare metadata and Full/Shared role validation",
"top-k sideband shape and lifecycle",
"stage-local KV filtering",
"target parity and performance fixtures"
],
"adoption_state": "none adopted in DGR-017. This story reads and records upstream state; it takes no patch and forks nothing."
},
"recommendation_for_dgr_004_and_dgr_018": [
"Track upstream PR #25407 rather than forking Mesh-LLM. At 12 files and +414/-7 it is small enough to review, reproduce, and carry as a numbered patch in the project's own pinned stack.",
"Any pin chosen before #25407 merges will load GLM-5.2 through the dense-MLA compatibility path. DGR-018 must therefore prove DSA/IndexShare are ACTIVE, not merely that the model emits text — the alpha contract already forbids the fallback.",
"Verify lightning-indexer backend coverage (#24231) on the specific backend the route will use. CPU support being merged says nothing about ROCm/HIP."
]
}