DGR-019 Lock alpha/beta performance contracts (evidence + contract framework) DGR-020 Run controlled whole-model GGUF baseline (benchmark results & contracts) DGR-024 Real generated-gRPC protocol harness (shard_runtime_server.py + tests) DGR-026 split-GGUF provisioning outside /home (provision script + manifest + tests) DGR-028 Numbered patch-stack apply & verify (llama_cpp_dependency.py + UPSTREAM_LOCK.json) DGR-029 Native CMake skeleton + deterministic CPU lane (UPSTREAM_LOCK.json + cmake gating) New modules: packages/node/meshnet_node/dgr_performance/ — performance contract framework packages/node/meshnet_node/split_gguf/ — split-GGUF manifest & provisioning scripts/provision_split_gguf.py — artifact provisioning CLI tests/test_dgr_performance_contract.py — contract validation tests tests/test_split_gguf_manifest.py — manifest tests tests/test_split_gguf_provision.py — provisioning tests tests/test_shard_runtime_harness.py — gRPC harness tests
72 lines
2.3 KiB
JSON
72 lines
2.3 KiB
JSON
{
|
|
"contract_version": 1,
|
|
"fit_benefit": true,
|
|
"plan_id": "dgr-001-controlled-whole-model-baseline-v1",
|
|
"quality_lane_pass": false,
|
|
"rationale": [
|
|
"the near-lossless quality lane failed: the GGUF runtime disagrees with the safetensors reference beyond what near-lossless weights can explain",
|
|
"a meaningful speed benefit was measured",
|
|
"a meaningful fit benefit was measured"
|
|
],
|
|
"recipes": [
|
|
{
|
|
"comparable": true,
|
|
"failures": 0,
|
|
"fit_benefit": false,
|
|
"incomparable_reason": "",
|
|
"lane": "quality",
|
|
"measurements": {
|
|
"aggregate_concurrency": 4,
|
|
"aggregate_throughput_speedup": 4.465,
|
|
"artifact_size_ratio": 0.9946,
|
|
"artifact_size_win": false,
|
|
"compared_prompts": 3,
|
|
"decode_speedup": 2.0171,
|
|
"exact_match_rate": 0.3333,
|
|
"expected_prompts": 3,
|
|
"failure_rate": 0.0,
|
|
"mean_similarity": 0.9471,
|
|
"resident_memory_ratio": 0.5742,
|
|
"ttft_ratio": 0.4586
|
|
},
|
|
"quality_pass": false,
|
|
"reasons": [
|
|
"single-request decode 2.02x reference (>= 1.25x) at TTFT ratio 0.46",
|
|
"aggregate throughput at concurrency 4 is 4.46x reference (>= 1.25x)",
|
|
"peak resident memory is 0.57x reference (<= 0.75x)",
|
|
"quality lane exact-match 0.33 / similarity 0.947 versus the reference (fail)"
|
|
],
|
|
"recipe_id": "llama-cpp-near-lossless-quality",
|
|
"speed_benefit": false
|
|
},
|
|
{
|
|
"comparable": true,
|
|
"failures": 0,
|
|
"fit_benefit": true,
|
|
"incomparable_reason": "",
|
|
"lane": "performance-fit",
|
|
"measurements": {
|
|
"aggregate_concurrency": 4,
|
|
"aggregate_throughput_speedup": 4.825,
|
|
"artifact_size_ratio": 0.398,
|
|
"artifact_size_win": true,
|
|
"decode_speedup": 4.1931,
|
|
"failure_rate": 0.0,
|
|
"resident_memory_ratio": 0.2802,
|
|
"ttft_ratio": 0.5251
|
|
},
|
|
"quality_pass": null,
|
|
"reasons": [
|
|
"single-request decode 4.19x reference (>= 1.25x) at TTFT ratio 0.53",
|
|
"aggregate throughput at concurrency 4 is 4.83x reference (>= 1.25x)",
|
|
"peak resident memory is 0.28x reference (<= 0.75x)"
|
|
],
|
|
"recipe_id": "llama-cpp-quantized-performance-fit",
|
|
"speed_benefit": true
|
|
}
|
|
],
|
|
"speed_benefit": true,
|
|
"stop_condition_met": true,
|
|
"verdict": "stop"
|
|
}
|