distributed-gguf-runtime: add CMake skeleton, gRPC harness, split-GGUF provisioning, performance contracts
DGR-019 Lock alpha/beta performance contracts (evidence + contract framework) DGR-020 Run controlled whole-model GGUF baseline (benchmark results & contracts) DGR-024 Real generated-gRPC protocol harness (shard_runtime_server.py + tests) DGR-026 split-GGUF provisioning outside /home (provision script + manifest + tests) DGR-028 Numbered patch-stack apply & verify (llama_cpp_dependency.py + UPSTREAM_LOCK.json) DGR-029 Native CMake skeleton + deterministic CPU lane (UPSTREAM_LOCK.json + cmake gating) New modules: packages/node/meshnet_node/dgr_performance/ — performance contract framework packages/node/meshnet_node/split_gguf/ — split-GGUF manifest & provisioning scripts/provision_split_gguf.py — artifact provisioning CLI tests/test_dgr_performance_contract.py — contract validation tests tests/test_split_gguf_manifest.py — manifest tests tests/test_split_gguf_provision.py — provisioning tests tests/test_shard_runtime_harness.py — gRPC harness tests
This commit is contained in:
@@ -0,0 +1,87 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"contract_version": 1,
|
||||
"locked_at": "2026-07-13T00:00:00Z",
|
||||
"locked_by": "DGR-001",
|
||||
"plan_id": "dgr-001-controlled-whole-model-baseline-v1",
|
||||
"thresholds": {
|
||||
"min_decode_speedup": 1.25,
|
||||
"max_ttft_ratio": 1.25,
|
||||
"min_aggregate_throughput_speedup": 1.25,
|
||||
"max_resident_memory_ratio": 0.75,
|
||||
"max_artifact_size_ratio": 0.6,
|
||||
"min_quality_exact_match_rate": 0.9,
|
||||
"min_quality_mean_similarity": 0.97,
|
||||
"max_failure_rate": 0.0
|
||||
},
|
||||
"baseline": {
|
||||
"status": "pending-real-evidence",
|
||||
"required_evidence_class": "local-real",
|
||||
"required_recipes": [
|
||||
"transformers-safetensors-reference",
|
||||
"llama-cpp-near-lossless-quality",
|
||||
"llama-cpp-quantized-performance-fit"
|
||||
],
|
||||
"required_concurrency_levels": [
|
||||
1,
|
||||
4
|
||||
],
|
||||
"required_controlled_variables": [
|
||||
"model architecture",
|
||||
"model revision",
|
||||
"machine and device",
|
||||
"formatted prompts and context lengths",
|
||||
"output length and greedy sampling policy"
|
||||
],
|
||||
"required_plan_sha256": "efe24690a9a7164bac6ab3fd0a6b22f078fc08aaefcfb96210ddf154e6050570",
|
||||
"minimum_prompt_count": 3,
|
||||
"minimum_repeats": 3,
|
||||
"minimum_output_tokens": 32,
|
||||
"required_device": "cpu",
|
||||
"required_config_sha256": "00b2cce3e2f281bdf92fc5304ba5cac915a178ffccd3b9a25995ce39c00b90d3",
|
||||
"required_signer_public_key": "zQ/qRMwF/ydazzaxEI24Xvnrl5bZxzw16JYpP0bfRuI=",
|
||||
"required_artifact_sha256": {
|
||||
"transformers-safetensors-reference": "e596e9d6205fdc9177569cccd7f8b471b058f66e3630c8e4326d5aad52bd18b6",
|
||||
"llama-cpp-near-lossless-quality": "e842fdc35d7f00fda95a54e1b51731ba1d196aea45065cc9f46925fdc1d6f862",
|
||||
"llama-cpp-quantized-performance-fit": "a88e3f570e2efeaf06b50df9859db2c70d8646aa3a2c94a14e14d5797a2921a5"
|
||||
},
|
||||
"required_recipe_runtime": {
|
||||
"transformers-safetensors-reference": {
|
||||
"runtime": "transformers-5.13.0",
|
||||
"weight_format": "safetensors",
|
||||
"weight_quantization": "bfloat16",
|
||||
"device": "cpu"
|
||||
},
|
||||
"llama-cpp-near-lossless-quality": {
|
||||
"runtime": "llama.cpp-9991-e920c523",
|
||||
"weight_format": "gguf",
|
||||
"weight_quantization": "bfloat16",
|
||||
"device": "cpu"
|
||||
},
|
||||
"llama-cpp-quantized-performance-fit": {
|
||||
"runtime": "llama.cpp-9991-e920c523",
|
||||
"weight_format": "gguf",
|
||||
"weight_quantization": "Q4_K_M",
|
||||
"device": "cpu"
|
||||
}
|
||||
},
|
||||
"required_backend_detail": {
|
||||
"transformers-safetensors-reference": "torch 2.10.0+rocm7.13.0a20260513; dtype bfloat16; device cpu; intra-op threads 16",
|
||||
"llama-cpp-near-lossless-quality": "version: 9991 (e920c523) | built with GNU 15.2.1 for Linux x86_64; binary sha256 fd8fe612970f23e447f2e717cfa51665be06b8d7315ba60556e010f6bca510dd; threads 16; parallel slots 4; ctx/slot 512; gpu layers 0",
|
||||
"llama-cpp-quantized-performance-fit": "version: 9991 (e920c523) | built with GNU 15.2.1 for Linux x86_64; binary sha256 fd8fe612970f23e447f2e717cfa51665be06b8d7315ba60556e010f6bca510dd; threads 16; parallel slots 4; ctx/slot 512; gpu layers 0"
|
||||
},
|
||||
"required_host_identity": {
|
||||
"python": "3.12.13",
|
||||
"torch_version": "2.10.0+rocm7.13.0a20260513",
|
||||
"transformers_version": "5.13.0",
|
||||
"llama_server_identities": {
|
||||
"/run/media/popov/d/DEV/llamacpp/llama.cpp/build/bin/llama-server": {
|
||||
"sha256": "fd8fe612970f23e447f2e717cfa51665be06b8d7315ba60556e010f6bca510dd",
|
||||
"version": "version: 9991 (e920c523) | built with GNU 15.2.1 for Linux x86_64"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"stop_condition": "Stop the native llama.cpp/GGUF track when, on the same machine and device as the Transformers/safetensors reference and under this plan, no performance-fit GGUF recipe delivers either a meaningful speed benefit (>=25% higher single-request decode tokens/sec without a >25% worse TTFT, or >=25% higher aggregate throughput under concurrency) or a meaningful fit benefit (>=25% lower peak resident memory), or when the near-lossless quality lane fails, which indicates a broken runtime rather than a quantization trade-off.",
|
||||
"notes": "Quantized performance-fit output drift is reported as advisory only. It is not numerical-equivalence evidence. DGR-014 consumes this immutable v1 contract. Non-synthetic evidence must be Ed25519-signed by the pinned key and match the exact locked config, artifacts, runtimes, backends, and host runtime identity."
|
||||
}
|
||||
Reference in New Issue
Block a user