{"inputs":{"model":"qwen3-30b-a3b","gpu":"rtx-4090","contextTokens":8192,"kvBytesPerElement":2,"batch":1},"result":{"model":{"slug":"qwen3-30b-a3b","name":"Qwen3 30B-A3B","family":"Qwen","repo":"Qwen/Qwen3-30B-A3B","canonical":"Qwen/Qwen3-30B-A3B","modelType":"qwen3_moe","params":30532122624,"parametersByDtype":{"BF16":30532122624},"checkpointBytes":61066575648,"checkpointFiles":16,"dtype":"BF16","layers":48,"hiddenSize":2048,"heads":32,"kvHeads":4,"headDim":128,"vocabSize":151936,"intermediateSize":6144,"maxPositionEmbeddings":40960,"tieWordEmbeddings":false,"hasVisionTower":false,"moe":{"experts":128,"expertsPerTok":8,"moeIntermediateSize":768},"sliding":null,"downloads":2840907,"likes":917,"sources":{"params":"https://huggingface.co/api/models/Qwen/Qwen3-30B-A3B","config":"https://huggingface.co/Qwen/Qwen3-30B-A3B/resolve/main/config.json"},"gguf":{"repo":"unsloth/Qwen3-30B-A3B-GGUF","url":"https://huggingface.co/unsloth/Qwen3-30B-A3B-GGUF","files":{"Q3_K_M":{"bytes":14711847488,"parts":1},"Q4_K_M":{"bytes":18556686912,"parts":1},"Q5_K_M":{"bytes":21725581888,"parts":1},"Q6_K":{"bytes":25092532800,"parts":1},"Q8_0":{"bytes":32483932736,"parts":1}}}},"gpu":{"slug":"rtx-4090","name":"GeForce RTX 4090","serpName":"RTX 4090","vendor":"NVIDIA","memoryGiB":24,"memoryType":"GDDR6X","bandwidthGBs":1008,"bus":{"bits":384,"dataRateGbps":21},"memoryClass":"dedicated","segment":"consumer","year":2022,"sourceUrl":"https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4090/"},"usableBytes":23708219473.920002,"quants":[{"quant":{"slug":"f16","label":"FP16 / BF16","bitsPerWeight":16,"exact":true,"ggufToken":null,"note":"Unquantized. Two bytes per parameter, exactly."},"weightBytes":61066575648,"weightBasis":"measured","bitsPerWeight":16.000610609364752,"headroomBytes":-37358356174.08,"fits":false,"maxContextTokens":0,"contextCappedByModel":false,"ceilingTokensPerSecond":null,"bytesReadPerToken":7511252126.890015,"counterfactualContextTokens":8192,"offloadBytes":37358356174.08,"offloadCeilingTokensPerSecond":17.508980418194096},{"quant":{"slug":"q8_0","label":"Q8_0","bitsPerWeight":8.5,"exact":false,"ggufToken":"Q8_0","note":"Effectively lossless; the usual reference point for quantized quality."},"weightBytes":32483932736,"weightBasis":"measured","bitsPerWeight":8.511411574239064,"headroomBytes":-8775713262.079998,"fits":false,"maxContextTokens":0,"contextCappedByModel":false,"ceilingTokensPerSecond":null,"bytesReadPerToken":4372486755.167854,"counterfactualContextTokens":8192,"offloadBytes":8775713262.079998,"offloadCeilingTokensPerSecond":44.96730918318476},{"quant":{"slug":"q6_k","label":"Q6_K","bitsPerWeight":6.56,"exact":false,"ggufToken":"Q6_K","note":"Quality loss is hard to measure on most benchmarks."},"weightBytes":25092532800,"weightBasis":"measured","bitsPerWeight":6.574723443636593,"headroomBytes":-1384313326.079998,"fits":false,"maxContextTokens":0,"contextCappedByModel":false,"ceilingTokensPerSecond":null,"bytesReadPerToken":3560809884.029879,"counterfactualContextTokens":8192,"offloadBytes":1384313326.079998,"offloadCeilingTokensPerSecond":75.64442047134705},{"quant":{"slug":"q5_k_m","label":"Q5_K_M","bitsPerWeight":5.67,"exact":false,"ggufToken":"Q5_K_M","note":"A middle point when Q4 fits with too little room for context."},"weightBytes":21725581888,"weightBasis":"measured","bitsPerWeight":5.69251791774803,"headroomBytes":1982637585.920002,"fits":true,"maxContextTokens":20168,"contextCappedByModel":false,"ceilingTokensPerSecond":315.88125004768403,"bytesReadPerToken":3191072594.0455055,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},{"quant":{"slug":"q4_k_m","label":"Q4_K_M","bitsPerWeight":4.83,"exact":false,"ggufToken":"Q4_K_M","note":"The default choice for local inference — the best size/quality knee."},"weightBytes":18556686912,"weightBasis":"measured","bitsPerWeight":4.862206834558795,"headroomBytes":5151532561.920002,"fits":true,"maxContextTokens":40960,"contextCappedByModel":true,"ceilingTokensPerSecond":354.5445026340323,"bytesReadPerToken":2843084556.4131536,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},{"quant":{"slug":"q3_k_m","label":"Q3_K_M","bitsPerWeight":3.91,"exact":false,"ggufToken":"Q3_K_M","note":"Measurable quality loss. Worth it only to make a model fit at all."},"weightBytes":14711847488,"weightBasis":"measured","bitsPerWeight":3.8547853797588627,"headroomBytes":8996371985.920002,"fits":true,"maxContextTokens":40960,"contextCappedByModel":true,"ceilingTokensPerSecond":416.379481954332,"bytesReadPerToken":2420868567.463553,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null}],"best":{"quant":{"slug":"q5_k_m","label":"Q5_K_M","bitsPerWeight":5.67,"exact":false,"ggufToken":"Q5_K_M","note":"A middle point when Q4 fits with too little room for context."},"weightBytes":21725581888,"weightBasis":"measured","bitsPerWeight":5.69251791774803,"headroomBytes":1982637585.920002,"fits":true,"maxContextTokens":20168,"contextCappedByModel":false,"ceilingTokensPerSecond":315.88125004768403,"bytesReadPerToken":3191072594.0455055,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},"fullPrecision":{"quant":{"slug":"f16","label":"FP16 / BF16","bitsPerWeight":16,"exact":true,"ggufToken":null,"note":"Unquantized. Two bytes per parameter, exactly."},"weightBytes":61066575648,"weightBasis":"measured","bitsPerWeight":16.000610609364752,"headroomBytes":-37358356174.08,"fits":false,"maxContextTokens":0,"contextCappedByModel":false,"ceilingTokensPerSecond":null,"bytesReadPerToken":7511252126.890015,"counterfactualContextTokens":8192,"offloadBytes":37358356174.08,"offloadCeilingTokensPerSecond":17.508980418194096},"verdict":"quantized-only"},"provenance":{"method":"Weight bytes are the MEASURED size of a published GGUF file wherever one exists and params × bits/8 only where none does — the two differ by 33% on sub-1B models and by 89% on natively-quantized releases, so the basis travels with every figure. KV cache is layers × kv_heads × head_dim × 2 × bytes-per-element per token, honouring sliding-window layers. Usable VRAM is a fixed fraction of nameplate memory (different for dedicated and unified). The decode ceiling is bandwidth ÷ bytes-read-per-token, which no runtime beats.","canonicalUrl":"https://makerportal.ai/lab/llm-vram/qwen3-30b-a3b/rtx-4090"},"license":"free-with-attribution"}