{"schema":"sizeof-estimate/v1","model":{"id":"Qwen/Qwen2.5-1.5B-Instruct","owner":"Qwen","name":"Qwen2.5-1.5B-Instruct","kind":"language","parametersB":1.543714304,"layers":28,"maxContext":32768,"estimateConfidence":"runtime-specific","sourceUrl":"https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct","repositoryUpdatedAt":"2024-09-25T12:32:50.000Z"},"configuration":{"quantization":"q4_k_m","context":4096,"kvPrecision":"fp16","mlaCacheMode":"expanded","source":"estimated","artifactId":null},"hardware":{"profile":"discrete-gpu","capacityGiB":16},"result":{"state":"lower-bound","estimate":{"baseWeightsGiB":0.871603187918663,"addonWeightsGiB":0,"weightsGiB":0.871603187918663,"kvCacheGiB":0.109375,"runtimeGiB":0.5980978187918663,"totalGiB":1.5790760067105292,"exceedsNativeContext":false,"isLowerBound":true},"fit":null,"headroomGiB":14.42092399328947,"reason":null},"planner":{"maximumSafeContext":{"kind":"unavailable","reason":"runtime-specific"},"highestPrecisionFit":{"kind":"unavailable","reason":"runtime-specific"},"adjustments":[]},"evidence":[{"id":"model-specification","label":"Published model specification","kind":"verified","detail":"Published model parameters, context limit, and cache geometry used by this calculator.","sourceUrl":"https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct","repositoryUpdatedAt":"2024-09-25T12:32:50.000Z"},{"id":"weights-estimate","label":"Estimated model weights","kind":"derived","detail":"Weight memory is derived from the selected bit precision and published parameter count.","sourceUrl":"https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct"},{"id":"memory-formula","label":"Derived memory estimate","kind":"derived","detail":"KV cache and runtime reserve are calculated from the selected context and precision settings."},{"id":"estimate-confidence","label":"Runtime-specific memory factors","kind":"unknown","detail":"This architecture has engine-dependent runtime state, so the total cannot be treated as a safe fit claim."}],"provenance":{"modelSource":"https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct","configSourceId":null,"parameterCountKind":"logical","artifact":null},"serving":null,"sourceUrl":"https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct","detailUrl":"https://testnet.sizeof.ai/Qwen/Qwen2.5-1.5B-Instruct?state=1&quant=q4_k_m&ctx=4096&kv=fp16&mla=expanded&vram=16&source=estimated&variant=none","apiReproductionUrl":"https://testnet.sizeof.ai/api/v1/estimate?model=Qwen%2FQwen2.5-1.5B-Instruct&quant=q4_k_m&context=4096&kv=fp16&mla=expanded&vram=16&profile=discrete-gpu&source=estimated&prompt=2048&generated=256&concurrency=1","generatedAt":"2026-09-08T12:55:45.379Z","disclaimer":"Estimate only. Verify on the target runtime and hardware; engines, drivers, batching, offload, and workload can change actual memory use."}