{"inputs":{"model":"llama-3-1-8b-instruct","gpu":"rtx-4090","contextTokens":8192,"kvBytesPerElement":2,"batch":1},"result":{"model":{"slug":"llama-3-1-8b-instruct","name":"Llama 3.1 8B Instruct","family":"Llama","repo":"unsloth/Meta-Llama-3.1-8B-Instruct","canonical":"meta-llama/Llama-3.1-8B-Instruct","modelType":"llama","params":8030261248,"parametersByDtype":{"BF16":8030261248},"checkpointBytes":16060556376,"checkpointFiles":4,"dtype":"BF16","layers":32,"hiddenSize":4096,"heads":32,"kvHeads":8,"headDim":128,"vocabSize":128256,"intermediateSize":14336,"maxPositionEmbeddings":131072,"tieWordEmbeddings":false,"hasVisionTower":false,"moe":null,"sliding":null,"downloads":389806,"likes":98,"sources":{"params":"https://huggingface.co/api/models/unsloth/Meta-Llama-3.1-8B-Instruct","config":"https://huggingface.co/unsloth/Meta-Llama-3.1-8B-Instruct/resolve/main/config.json"},"gguf":{"repo":"bartowski/Meta-Llama-3.1-8B-Instruct-GGUF","url":"https://huggingface.co/bartowski/Meta-Llama-3.1-8B-Instruct-GGUF","files":{"Q3_K_M":{"bytes":4018922912,"parts":1},"Q4_K_M":{"bytes":4920739232,"parts":1},"Q5_K_M":{"bytes":5732992416,"parts":1},"Q6_K":{"bytes":6596011424,"parts":1},"Q8_0":{"bytes":8540775840,"parts":1}}}},"gpu":{"slug":"rtx-4090","name":"GeForce RTX 4090","serpName":"RTX 4090","vendor":"NVIDIA","memoryGiB":24,"memoryType":"GDDR6X","bandwidthGBs":1008,"bus":{"bits":384,"dataRateGbps":21},"memoryClass":"dedicated","segment":"consumer","year":2022,"sourceUrl":"https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4090/"},"usableBytes":23708219473.920002,"quants":[{"quant":{"slug":"f16","label":"FP16 / BF16","bitsPerWeight":16,"exact":true,"ggufToken":null,"note":"Unquantized. Two bytes per parameter, exactly."},"weightBytes":16060556376,"weightBasis":"measured","bitsPerWeight":16.00003375232656,"headroomBytes":7647663097.920002,"fits":true,"maxContextTokens":58347,"contextCappedByModel":false,"ceilingTokensPerSecond":58.829371838526775,"bytesReadPerToken":17134298200,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},{"quant":{"slug":"q8_0","label":"Q8_0","bitsPerWeight":8.5,"exact":false,"ggufToken":"Q8_0","note":"Effectively lossless; the usual reference point for quantized quality."},"weightBytes":8540775840,"weightBasis":"measured","bitsPerWeight":8.508590768079578,"headroomBytes":15167443633.920002,"fits":true,"maxContextTokens":115718,"contextCappedByModel":false,"ceilingTokensPerSecond":104.84145281403895,"bytesReadPerToken":9614517664,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},{"quant":{"slug":"q6_k","label":"Q6_K","bitsPerWeight":6.56,"exact":false,"ggufToken":"Q6_K","note":"Quality loss is hard to measure on most benchmarks."},"weightBytes":6596011424,"weightBasis":"measured","bitsPerWeight":6.571155005093055,"headroomBytes":17112208049.920002,"fits":true,"maxContextTokens":130555,"contextCappedByModel":false,"ceilingTokensPerSecond":131.4253493439115,"bytesReadPerToken":7669753248,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},{"quant":{"slug":"q5_k_m","label":"Q5_K_M","bitsPerWeight":5.67,"exact":false,"ggufToken":"Q5_K_M","note":"A middle point when Q4 fits with too little room for context."},"weightBytes":5732992416,"weightBasis":"measured","bitsPerWeight":5.711388199160118,"headroomBytes":17975227057.920002,"fits":true,"maxContextTokens":131072,"contextCappedByModel":true,"ceilingTokensPerSecond":148.08863758429916,"bytesReadPerToken":6806734240,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},{"quant":{"slug":"q4_k_m","label":"Q4_K_M","bitsPerWeight":4.83,"exact":false,"ggufToken":"Q4_K_M","note":"The default choice for local inference — the best size/quality knee."},"weightBytes":4920739232,"weightBasis":"measured","bitsPerWeight":4.902195911223236,"headroomBytes":18787480241.920002,"fits":true,"maxContextTokens":131072,"contextCappedByModel":true,"ceilingTokensPerSecond":168.15467270366497,"bytesReadPerToken":5994481056,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},{"quant":{"slug":"q3_k_m","label":"Q3_K_M","bitsPerWeight":3.91,"exact":false,"ggufToken":"Q3_K_M","note":"Measurable quality loss. Worth it only to make a model fit at all."},"weightBytes":4018922912,"weightBasis":"measured","bitsPerWeight":4.0037779971364635,"headroomBytes":19689296561.920002,"fits":true,"maxContextTokens":131072,"contextCappedByModel":true,"ceilingTokensPerSecond":197.9317414858389,"bytesReadPerToken":5092664736,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null}],"best":{"quant":{"slug":"f16","label":"FP16 / BF16","bitsPerWeight":16,"exact":true,"ggufToken":null,"note":"Unquantized. Two bytes per parameter, exactly."},"weightBytes":16060556376,"weightBasis":"measured","bitsPerWeight":16.00003375232656,"headroomBytes":7647663097.920002,"fits":true,"maxContextTokens":58347,"contextCappedByModel":false,"ceilingTokensPerSecond":58.829371838526775,"bytesReadPerToken":17134298200,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},"fullPrecision":{"quant":{"slug":"f16","label":"FP16 / BF16","bitsPerWeight":16,"exact":true,"ggufToken":null,"note":"Unquantized. Two bytes per parameter, exactly."},"weightBytes":16060556376,"weightBasis":"measured","bitsPerWeight":16.00003375232656,"headroomBytes":7647663097.920002,"fits":true,"maxContextTokens":58347,"contextCappedByModel":false,"ceilingTokensPerSecond":58.829371838526775,"bytesReadPerToken":17134298200,"counterfactualContextTokens":8192,"offloadBytes":0,"offloadCeilingTokensPerSecond":null},"verdict":"comfortable"},"provenance":{"method":"Weight bytes are the MEASURED size of a published GGUF file wherever one exists and params × bits/8 only where none does — the two differ by 33% on sub-1B models and by 89% on natively-quantized releases, so the basis travels with every figure. KV cache is layers × kv_heads × head_dim × 2 × bytes-per-element per token, honouring sliding-window layers. Usable VRAM is a fixed fraction of nameplate memory (different for dedicated and unified). The decode ceiling is bandwidth ÷ bytes-read-per-token, which no runtime beats.","canonicalUrl":"https://makerportal.ai/lab/llm-vram/llama-3-1-8b-instruct/rtx-4090"},"license":"free-with-attribution"}