{"apiVersion":"v1","registryVersion":"1.0.0","plan":{"schemaVersion":"1.0.0","kind":"deployment_validation_plan","scenarioKey":"kimi-k3--h200--vllm--8x8--r1300--i4096--o1024--c1--t1000--p50--single","artifact":{"modelId":"kimi-k3","modelName":"Kimi K3","repository":"moonshotai/Kimi-K3","revision":"9f62e4e9fffbd0a83ddd60e1c209d828994b3569","manifestUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/9f62e4e9fffbd0a83ddd60e1c209d828994b3569/model.safetensors.index.json"},"target":{"hardwareId":"h200","hardwareName":"NVIDIA H200 SXM","runtime":"vllm","requestedAccelerators":8,"acceleratorsPerNode":8,"reserveBasisPoints":1300},"workload":{"inputTokens":4096,"outputTokens":1024,"concurrency":1,"targetTtftMs":1000,"targetInterTokenMs":50,"availability":"single"},"evidencePolicy":{"resultClass":"calculated","measuredPerformanceAvailable":false,"statement":"This plan contains a deterministic static-residency calculation and an execution protocol. It contains no measured performance result."},"readiness":{"status":"blocked","reason":"The requested topology fails the static checkpoint-residency floor.","staticResidency":"fail","runtimeCompatibility":"unknown"},"dimensions":{"load":{"status":"blocked","evidenceClass":"calculated","summary":"The requested topology cannot hold the static checkpoint tensors under the declared reserve.","known":["13 accelerator(s) are the mathematical checkpoint-residency minimum.","16 accelerator(s) are required after 8-device node rounding."],"unknown":["Static checkpoint tensor bytes only; runtime allocations are not included.","Framework, kernel, interconnect, quantization, and sharding support are not asserted.","KV cache, activations, routing buffers, allocator fragmentation, and host memory are not modeled."]},"run":{"status":"blocked","evidenceClass":"unknown","summary":"Latency, throughput, quality retention, and complete memory are not measured.","known":[],"unknown":["runtime load success","peak resident memory and KV capacity","P95 time to first token and inter-token latency","quality retention under the chosen serving configuration"]},"scale":{"status":"blocked","evidenceClass":"unknown","summary":"Concurrency, routing balance, communication pressure, and resilience require execution.","known":[],"unknown":["SLA throughput frontier","expert-load distribution","all-to-all communication share","failure and recovery behavior"]},"economics":{"status":"blocked","evidenceClass":"unknown","summary":"Cost requires successful-token throughput, utilization, power, and price evidence.","known":[],"unknown":["cost per one million successful output tokens","energy per successful token","utilization-adjusted monthly cost"]}},"risks":["Requested accelerator capacity is below the static checkpoint-residency floor.","No reviewed vllm compatibility result proves this exact configuration.","Static checkpoint fit excludes KV cache, activations, routing buffers, allocator fragmentation, and host memory.","No controlled workload result currently supports a performance or cost prediction."],"validationGates":[{"id":"artifact_integrity","order":1,"status":"required","title":"Verify the artifact","objective":"Bind every result to the exact immutable checkpoint bytes under evaluation.","acceptanceCriteria":["Resolve moonshotai/Kimi-K3@9f62e4e9fffbd0a83ddd60e1c209d828994b3569.","Record manifest and checkpoint digests before execution."],"requiredEvidence":["artifact manifest","revision identity","content digests"]},{"id":"static_residency","order":2,"status":"blocked","title":"Clear the static residency floor","objective":"Reject topologies that cannot hold the pinned checkpoint tensor bytes.","acceptanceCriteria":["Checkpoint tensor bytes fit inside 8 accelerator(s) after a 13.00% reserve."],"requiredEvidence":["artifact tensor bytes","advertised accelerator memory","declared reserve"]},{"id":"runtime_load","order":3,"status":"blocked","title":"Prove runtime load compatibility","objective":"Load the exact artifact with vllm on the declared topology.","acceptanceCriteria":["Process reaches a ready state without unsupported-architecture or out-of-memory errors.","Record runtime, container, kernels, drivers, sharding, precision, and startup command."],"requiredEvidence":["runtime logs","environment manifest","startup result"]},{"id":"memory_envelope","order":4,"status":"blocked","title":"Measure the complete memory envelope","objective":"Replace the static lower bound with observed resident and peak memory.","acceptanceCriteria":["Capture idle, warm, and peak accelerator and host memory.","Exercise 4,096 input and 1,024 output tokens at concurrency 1.","Retain allocator, KV-cache, activation, routing-buffer, and fragmentation evidence."],"requiredEvidence":["memory telemetry","workload manifest","raw time series"]},{"id":"workload_sla","order":5,"status":"blocked","title":"Replay the target workload","objective":"Measure the latency and throughput frontier under the stated workload.","acceptanceCriteria":["P95 time to first token is at or below 1000 ms.","P95 inter-token latency is at or below 50 ms.","Sustain concurrency 1 without request loss or invalid output."],"requiredEvidence":["request-level timings","throughput series","error and quality results"]},{"id":"routing_health","order":6,"status":"blocked","title":"Inspect expert routing health","objective":"Expose MoE-specific imbalance and communication bottlenecks hidden by aggregate throughput.","acceptanceCriteria":["Report expert-load distribution, hot and dead experts, and dropped-token behavior.","Quantify routing and all-to-all communication share where runtime telemetry permits."],"requiredEvidence":["expert counters","routing telemetry","communication trace"]},{"id":"economics_resilience","order":7,"status":"blocked","title":"Validate economics and resilience","objective":"Attach cost and failure behavior to successful workload outcomes.","acceptanceCriteria":["Report infrastructure cost per one million successful output tokens.","Record restart time and data-plane interruption for one controlled worker failure."],"requiredEvidence":["price assumptions","successful-token count","failure timeline"]}],"nextAction":{"code":"increase_capacity","summary":"Increase the target to at least 13 accelerator(s); use 16 for complete 8-device nodes."},"reproducibility":{"cli":"moemodels plan kimi-k3 h200 --devices 8 --devices-per-node 8 --reserve-bps 1300 --runtime vllm --input-tokens 4096 --output-tokens 1024 --concurrency 1 --target-ttft-ms 1000 --target-inter-token-ms 50 --availability single --json","apiPath":"/api/v1/plan?model=kimi-k3&hardware=h200&devices=8&devicesPerNode=8&reserveBps=1300&runtime=vllm&inputTokens=4096&outputTokens=1024&concurrency=1&targetTtftMs=1000&targetInterTokenMs=50&availability=single","assumptions":["Accelerator capacity uses decimal gigabytes from the versioned hardware registry.","The declared reserve is applied before integer checkpoint-residency division.","Node rounding is a planning convention, not proof of a supported sharding topology."]}}}