{
    "run": {
        "model_name": "Qwen3.8-27B",
        "repo": "ggml-org/Qwen3.8-27B-GGUF",
        "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
        "threshold_tps": 10,
        "threshold_source": "3060/out/bench.json",
        "text_repo": "datasets/ggml-org/ci",
        "text_revision": "927b3642933080f1b0e811e2f916e14c292992f9",
        "text_member": "wikitext-2-raw/wiki.test.raw",
        "text_sha256": "173c87a53759e0201f33e0ccf978e510c2042d7f2cb78229d9a50d79b9e7dd08"
    },
    "cards": [
        {
            "dir": "3060",
            "card": "NVIDIA GeForce RTX 3060",
            "llama_commit": "df750f76b",
            "llama_build": "b10871",
            "llama_source": "3060/m1/matrix.json",
            "protocol": "bench-runner/1",
            "flags": "-p 512 -n 128 -r 3 -t 8",
            "taken_on": "2026-09-09",
            "price": "$0.221/hr",
            "price_source": "run notes",
            "instance_id": "50375466",
            "machine_id": "146697 (host 613713)",
            "file": "3060/out/bench.json"
        },
        {
            "dir": "3090",
            "card": "NVIDIA GeForce RTX 3090",
            "llama_commit": "df750f76b",
            "llama_build": "b10871",
            "llama_source": "3090/bench.json",
            "protocol": "bench-runner/1",
            "flags": "-p 512 -n 128 -r 3 -t 8",
            "taken_on": "2026-09-09",
            "price": "$0.172/hr",
            "price_source": "run notes",
            "instance_id": "50354495",
            "machine_id": "149874 (host 623721)",
            "file": "3090/bench.json"
        },
        {
            "dir": "4070tis",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "llama_commit": "df750f76b",
            "llama_build": "b10871",
            "llama_source": "4070tis/m2/matrix.json",
            "protocol": "bench-runner/1",
            "flags": "-p 512 -n 128 -r 3 -t 8",
            "taken_on": "2026-09-09",
            "price": "$0.131/hr",
            "price_source": "instance panel frame",
            "instance_id": "50403508",
            "machine_id": "93269 (host 435128)",
            "file": "4070tis/out/bench.json"
        },
        {
            "dir": "4090",
            "card": "NVIDIA GeForce RTX 4090",
            "llama_commit": "df750f76b",
            "llama_build": "b10871",
            "llama_source": "4090/bench.json",
            "protocol": "bench-runner/1",
            "flags": "-p 512 -n 128 -r 3 -t 8",
            "taken_on": "2026-09-09",
            "price": "$0.390/hr",
            "price_source": "run notes",
            "instance_id": "50361969",
            "machine_id": "57808 (host 370680)",
            "file": "4090/bench.json"
        },
        {
            "dir": "5090",
            "card": "NVIDIA GeForce RTX 5090",
            "llama_commit": "df750f76b",
            "llama_build": "b10871",
            "llama_source": "5090/out/bench.json",
            "protocol": "bench-runner/1",
            "flags": "-p 512 -n 128 -r 3 -t 8",
            "taken_on": "2026-09-09",
            "price": "$0.074/hr",
            "price_source": "run notes",
            "instance_id": "50387901",
            "machine_id": "70996 (host 343861)",
            "file": "5090/out/bench.json"
        }
    ],
    "units": {
        "prompt_processing_tps": "TPS",
        "token_generation_tps": "TPS",
        "peak_vram_gib": "GiB",
        "peak_ram_gib": "GiB"
    },
    "points": [
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "8k",
            "tokens": 8192,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 0.1,
            "peak_ram_gib": 18,
            "status": "oom_load",
            "failure": "17402.38 MiB short: failed to allocate CUDA0 buffer of size 18247720960"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "32k",
            "tokens": 32768,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "96k",
            "tokens": 98304,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "256k",
            "tokens": 262144,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "8k",
            "tokens": 8192,
            "prompt_processing_tps": 1213.3,
            "prompt_processing_stddev": 20.6,
            "token_generation_tps": 37.1,
            "token_generation_stddev": 0,
            "peak_vram_gib": 18.5,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "32k",
            "tokens": 32768,
            "prompt_processing_tps": 955.1,
            "prompt_processing_stddev": 12.6,
            "token_generation_tps": 34.3,
            "token_generation_stddev": 0,
            "peak_vram_gib": 20,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "96k",
            "tokens": 98304,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.3,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "oom_prefill"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 3090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "256k",
            "tokens": 262144,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "8k",
            "tokens": 8192,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 0.2,
            "peak_ram_gib": 18,
            "status": "oom_load",
            "failure": "17402.38 MiB short: failed to allocate CUDA0 buffer of size 18247720960"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "32k",
            "tokens": 32768,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "96k",
            "tokens": 98304,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "256k",
            "tokens": 262144,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "8k",
            "tokens": 8192,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "oom_prefill",
            "failure": "149.62 MiB short: failed to allocate CUDA0 buffer of size 156893184"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "32k",
            "tokens": 32768,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "96k",
            "tokens": 98304,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "256k",
            "tokens": 262144,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "8k",
            "tokens": 8192,
            "prompt_processing_tps": 2828.8,
            "prompt_processing_stddev": 129.7,
            "token_generation_tps": 42.9,
            "token_generation_stddev": 0.3,
            "peak_vram_gib": 18.6,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "32k",
            "tokens": 32768,
            "prompt_processing_tps": 2343,
            "prompt_processing_stddev": 83.9,
            "token_generation_tps": 40,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 20.1,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "96k",
            "tokens": 98304,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.4,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "oom_prefill"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 4090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "256k",
            "tokens": 262144,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": null,
            "peak_ram_gib": null,
            "status": "skipped",
            "failure": "skipped"
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "8k",
            "tokens": 8192,
            "prompt_processing_tps": 3696.2,
            "prompt_processing_stddev": 177.1,
            "token_generation_tps": 67.7,
            "token_generation_stddev": 0.2,
            "peak_vram_gib": 18.8,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "32k",
            "tokens": 32768,
            "prompt_processing_tps": 2903.8,
            "prompt_processing_stddev": 105,
            "token_generation_tps": 63.5,
            "token_generation_stddev": 0.2,
            "peak_vram_gib": 20.2,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "96k",
            "tokens": 98304,
            "prompt_processing_tps": 1536,
            "prompt_processing_stddev": 15.5,
            "token_generation_tps": 54.5,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 24.2,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "ladder",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "256k",
            "tokens": 262144,
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.5,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "16416 MiB short: failed to allocate CUDA0 buffer of size 17213423616"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "-1",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 0.1,
            "peak_ram_gib": 18,
            "status": "oom_load",
            "failure": "17402.38 MiB short: failed to allocate CUDA0 buffer of size 18247720960"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "64",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 0.1,
            "peak_ram_gib": 16.7,
            "status": "oom_load",
            "failure": "17141.38 MiB short: failed to allocate CUDA0 buffer of size 17974035456"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "56",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 0.1,
            "peak_ram_gib": 15.2,
            "status": "oom_load",
            "failure": "15090.41 MiB short: failed to allocate CUDA0 buffer of size 15823440896"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "48",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 0.1,
            "peak_ram_gib": 18,
            "status": "oom_load",
            "failure": "13039.44 MiB short: failed to allocate CUDA0 buffer of size 13672846336"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "40",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 10.8,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "513.51 MiB short: failed to allocate CUDA0 buffer of size 538454144"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "32",
            "depth": "8k",
            "prompt_processing_tps": 273.1,
            "prompt_processing_stddev": 0.7,
            "token_generation_tps": 2.6,
            "token_generation_stddev": 0,
            "peak_vram_gib": 9.8,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "24",
            "depth": "8k",
            "prompt_processing_tps": 244.3,
            "prompt_processing_stddev": 0.6,
            "token_generation_tps": 2.1,
            "token_generation_stddev": 0,
            "peak_vram_gib": 7.7,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "16",
            "depth": "8k",
            "prompt_processing_tps": 223,
            "prompt_processing_stddev": 0.6,
            "token_generation_tps": 1.8,
            "token_generation_stddev": 0,
            "peak_vram_gib": 5.6,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "8",
            "depth": "8k",
            "prompt_processing_tps": 202.9,
            "prompt_processing_stddev": 0.1,
            "token_generation_tps": 1.6,
            "token_generation_stddev": 0,
            "peak_vram_gib": 3.5,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "0",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 1.8,
            "peak_ram_gib": 19.5,
            "status": "failed",
            "failure": "failed"
        },
        {
            "series": "sweep",
            "sweep": "Single point",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "depth": "8k",
            "prompt_processing_tps": 185.5,
            "prompt_processing_stddev": 2.7,
            "token_generation_tps": 1.4,
            "token_generation_stddev": 0,
            "peak_vram_gib": 1.8,
            "peak_ram_gib": 19.5,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "512",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 10.8,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "513.51 MiB short: failed to allocate CUDA0 buffer of size 538454144"
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "256",
            "depth": "8k",
            "prompt_processing_tps": 220.2,
            "prompt_processing_stddev": 0.1,
            "token_generation_tps": 3.3,
            "token_generation_stddev": 0,
            "peak_vram_gib": 11.6,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "128",
            "depth": "8k",
            "prompt_processing_tps": 142.7,
            "prompt_processing_stddev": 0,
            "token_generation_tps": 3.3,
            "token_generation_stddev": 0,
            "peak_vram_gib": 11.5,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "64",
            "depth": "8k",
            "prompt_processing_tps": 80.9,
            "prompt_processing_stddev": 0.7,
            "token_generation_tps": 3.3,
            "token_generation_stddev": 0,
            "peak_vram_gib": 11.4,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "f16",
            "depth": "8k",
            "prompt_processing_tps": 221.7,
            "prompt_processing_stddev": 1.5,
            "token_generation_tps": 3.2,
            "token_generation_stddev": 0,
            "peak_vram_gib": 11.6,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "q8_0",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 11.6,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "oom_prefill"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "q4_0",
            "depth": "8k",
            "prompt_processing_tps": 212.7,
            "prompt_processing_stddev": 0.2,
            "token_generation_tps": 2.9,
            "token_generation_stddev": 0,
            "peak_vram_gib": 11.6,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "f16",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 10.8,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "1300 MiB short: failed to allocate CUDA0 buffer of size 1363148800"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "q8_0",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 10.8,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "995.31 MiB short: failed to allocate CUDA0 buffer of size 1043660800"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "q4_0",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 10.8,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "832.81 MiB short: failed to allocate CUDA0 buffer of size 873267200"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "-1",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 0.1,
            "peak_ram_gib": 18,
            "status": "oom_load",
            "failure": "17402.38 MiB short: failed to allocate CUDA0 buffer of size 18247720960"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "36",
            "depth": "32k",
            "prompt_processing_tps": 185.2,
            "prompt_processing_stddev": 0.2,
            "token_generation_tps": 2.6,
            "token_generation_stddev": 0,
            "peak_vram_gib": 10.5,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "32",
            "depth": "32k",
            "prompt_processing_tps": 171.8,
            "prompt_processing_stddev": 0.5,
            "token_generation_tps": 2.3,
            "token_generation_stddev": 0,
            "peak_vram_gib": 9.5,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 3060",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "28",
            "depth": "32k",
            "prompt_processing_tps": 161.1,
            "prompt_processing_stddev": 0.3,
            "token_generation_tps": 2.1,
            "token_generation_stddev": 0,
            "peak_vram_gib": 8.5,
            "peak_ram_gib": 18,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "512",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "oom_prefill",
            "failure": "149.62 MiB short: failed to allocate CUDA0 buffer of size 156893184"
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "256",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "oom_prefill",
            "failure": "149.62 MiB short: failed to allocate CUDA0 buffer of size 156893184"
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "128",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "oom_prefill",
            "failure": "149.62 MiB short: failed to allocate CUDA0 buffer of size 156893184"
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "64",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "oom_prefill",
            "failure": "149.62 MiB short: failed to allocate CUDA0 buffer of size 156893184"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "-1",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "oom_prefill",
            "failure": "149.62 MiB short: failed to allocate CUDA0 buffer of size 156893184"
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "62",
            "depth": "8k",
            "prompt_processing_tps": 1313,
            "prompt_processing_stddev": 26.7,
            "token_generation_tps": 20.8,
            "token_generation_stddev": 0,
            "peak_vram_gib": 15.5,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "60",
            "depth": "8k",
            "prompt_processing_tps": 1193.5,
            "prompt_processing_stddev": 20.4,
            "token_generation_tps": 15.5,
            "token_generation_stddev": 0,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Layers on the GPU",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ngl": "56",
            "depth": "8k",
            "prompt_processing_tps": 1020.3,
            "prompt_processing_stddev": 13,
            "token_generation_tps": 10.8,
            "token_generation_stddev": 0,
            "peak_vram_gib": 14.1,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Prefill depth",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "d": "0",
            "depth": "0",
            "prompt_processing_tps": 1772.5,
            "prompt_processing_stddev": 59.5,
            "token_generation_tps": 37.8,
            "token_generation_stddev": 0,
            "peak_vram_gib": 14.5,
            "peak_ram_gib": 14.5,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Prefill depth",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "d": "8192",
            "depth": "8k",
            "prompt_processing_tps": 1650.6,
            "prompt_processing_stddev": 39.4,
            "token_generation_tps": 36.6,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 15,
            "peak_ram_gib": 14.5,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Prefill depth",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "d": "32768",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 13.7,
            "peak_ram_gib": 14.5,
            "status": "oom_prefill",
            "failure": "2080 MiB short: failed to allocate CUDA0 buffer of size 2181038080"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "ctk": "f16",
            "ctv": "f16",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 13.7,
            "peak_ram_gib": 14.5,
            "status": "oom_prefill",
            "failure": "2080 MiB short: failed to allocate CUDA0 buffer of size 2181038080"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "ctk": "q8_0",
            "ctv": "f16",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 13.7,
            "peak_ram_gib": 14.5,
            "status": "oom_prefill",
            "failure": "1758.27 MiB short: failed to allocate CUDA0 buffer of size 1843677312"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "ctk": "q4_0",
            "ctv": "f16",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 13.7,
            "peak_ram_gib": 14.5,
            "status": "oom_prefill",
            "failure": "1758.27 MiB short: failed to allocate CUDA0 buffer of size 1843677312"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "ctk": "f16",
            "ctv": "q8_0",
            "depth": "32k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 13.7,
            "peak_ram_gib": 14.5,
            "status": "oom_prefill",
            "failure": "505 MiB short: failed to allocate CUDA0 buffer of size 529532928"
        },
        {
            "series": "sweep",
            "sweep": "Single point",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "depth": "32k",
            "prompt_processing_tps": 1325.2,
            "prompt_processing_stddev": 2.7,
            "token_generation_tps": 33,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 15.2,
            "peak_ram_gib": 14.5,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Flash attention",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "fa": "auto",
            "depth": "8k",
            "prompt_processing_tps": 1655.7,
            "prompt_processing_stddev": 46.7,
            "token_generation_tps": 36.6,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 15,
            "peak_ram_gib": 14.5,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Flash attention",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "fa": "on",
            "depth": "8k",
            "prompt_processing_tps": 1652.3,
            "prompt_processing_stddev": 44,
            "token_generation_tps": 36.6,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 15,
            "peak_ram_gib": 14.5,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Flash attention",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "fa": "off",
            "depth": "8k",
            "prompt_processing_tps": 1413.5,
            "prompt_processing_stddev": 30.6,
            "token_generation_tps": 35.6,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 15.2,
            "peak_ram_gib": 14.5,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Threads",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "t": "8",
            "depth": "8k",
            "prompt_processing_tps": 1306,
            "prompt_processing_stddev": 26.7,
            "token_generation_tps": 20.7,
            "token_generation_stddev": 0.1,
            "peak_vram_gib": 15.5,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Threads",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "t": "1",
            "depth": "8k",
            "prompt_processing_tps": 1303.9,
            "prompt_processing_stddev": 25.3,
            "token_generation_tps": 14.6,
            "token_generation_stddev": 0,
            "peak_vram_gib": 15.5,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Threads",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "t": "2",
            "depth": "8k",
            "prompt_processing_tps": 1301.4,
            "prompt_processing_stddev": 24.1,
            "token_generation_tps": 19.2,
            "token_generation_stddev": 0,
            "peak_vram_gib": 15.5,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Threads",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "t": "4",
            "depth": "8k",
            "prompt_processing_tps": 1301.7,
            "prompt_processing_stddev": 26.4,
            "token_generation_tps": 20.7,
            "token_generation_stddev": 0,
            "peak_vram_gib": 15.5,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Threads",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "t": "16",
            "depth": "8k",
            "prompt_processing_tps": 1293.9,
            "prompt_processing_stddev": 26.9,
            "token_generation_tps": 18.8,
            "token_generation_stddev": 0.2,
            "peak_vram_gib": 15.5,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Tensor overrides",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ot": "default",
            "depth": "8k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 15,
            "peak_ram_gib": 15.8,
            "status": "oom_prefill",
            "failure": "149.62 MiB short: failed to allocate CUDA0 buffer of size 156893184"
        },
        {
            "series": "sweep",
            "sweep": "Tensor overrides",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ot": "blk\\.(3|7|11|15|19|23|27|31|35|39|43|47|51|55|59|63)\\..*=CPU",
            "depth": "8k",
            "prompt_processing_tps": 894.4,
            "prompt_processing_stddev": 2.7,
            "token_generation_tps": 9,
            "token_generation_stddev": 0,
            "peak_vram_gib": 12.9,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Tensor overrides",
            "card": "NVIDIA GeForce RTX 4070 Ti SUPER",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ot": "\\.ffn_down\\.=CPU",
            "depth": "8k",
            "prompt_processing_tps": 849.8,
            "prompt_processing_stddev": 1.9,
            "token_generation_tps": 8.4,
            "token_generation_stddev": 0,
            "peak_vram_gib": 12.6,
            "peak_ram_gib": 15.8,
            "status": "ok",
            "failure": null
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "512",
            "depth": "256k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.5,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "16416 MiB short: failed to allocate CUDA0 buffer of size 17213423616"
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "256",
            "depth": "256k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.5,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "16416 MiB short: failed to allocate CUDA0 buffer of size 17213423616"
        },
        {
            "series": "sweep",
            "sweep": "Micro-batch",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ub": "128",
            "depth": "256k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.5,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "16416 MiB short: failed to allocate CUDA0 buffer of size 17213423616"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "f16",
            "ctv": "f16",
            "depth": "256k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.5,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "16416 MiB short: failed to allocate CUDA0 buffer of size 17213423616"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "q8_0",
            "ctv": "f16",
            "depth": "256k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.5,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "13406.27 MiB short: failed to allocate CUDA0 buffer of size 14057490560"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "q4_0",
            "ctv": "f16",
            "depth": "256k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 17.5,
            "peak_ram_gib": 18,
            "status": "oom_prefill",
            "failure": "13406.27 MiB short: failed to allocate CUDA0 buffer of size 14057490560"
        },
        {
            "series": "sweep",
            "sweep": "KV cache type",
            "card": "NVIDIA GeForce RTX 5090",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ctk": "f16",
            "ctv": "q8_0",
            "depth": "256k",
            "prompt_processing_tps": null,
            "prompt_processing_stddev": null,
            "token_generation_tps": null,
            "token_generation_stddev": null,
            "peak_vram_gib": 30.5,
            "peak_ram_gib": 18,
            "status": "timeout",
            "failure": "timeout"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-IQ4_XS.gguf",
            "ppl": 6.869776,
            "ppl_unc": 0.075071,
            "ppl_ratio": 1.012824,
            "mean_kld": 0.023808,
            "median_kld": 0.010319,
            "rms_dp": 4.34,
            "same_top_p": 93.151,
            "vram_peak_gib": 15.3,
            "status": "ok"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-Q2_K.gguf",
            "ppl": 8.455446,
            "ppl_unc": 0.09406,
            "ppl_ratio": 1.246602,
            "mean_kld": 0.294271,
            "median_kld": 0.148913,
            "rms_dp": 15.772,
            "same_top_p": 77.227,
            "vram_peak_gib": 11.4,
            "status": "ok"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-Q3_K_M.gguf",
            "ppl": 7.173957,
            "ppl_unc": 0.079915,
            "ppl_ratio": 1.05767,
            "mean_kld": 0.071103,
            "median_kld": 0.032325,
            "rms_dp": 7.458,
            "same_top_p": 88.553,
            "vram_peak_gib": 13.7,
            "status": "ok"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-Q4_0.gguf",
            "ppl": 6.945883,
            "ppl_unc": 0.076069,
            "ppl_ratio": 1.024045,
            "mean_kld": 0.036346,
            "median_kld": 0.016208,
            "rms_dp": 5.175,
            "same_top_p": 91.612,
            "vram_peak_gib": 15.5,
            "status": "ok"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-Q4_K_M.gguf",
            "ppl": 6.814875,
            "ppl_unc": 0.074022,
            "ppl_ratio": 1.00473,
            "mean_kld": 0.023021,
            "median_kld": 0.009979,
            "rms_dp": 4.164,
            "same_top_p": 93.327,
            "vram_peak_gib": 16.5,
            "status": "ok"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-Q5_K_M.gguf",
            "ppl": 6.820789,
            "ppl_unc": 0.074328,
            "ppl_ratio": 1.005602,
            "mean_kld": 0.008207,
            "median_kld": 0.003519,
            "rms_dp": 2.501,
            "same_top_p": 96.008,
            "vram_peak_gib": 18.9,
            "status": "ok"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-Q6_K.gguf",
            "ppl": 6.788637,
            "ppl_unc": 0.073886,
            "ppl_ratio": 1.000862,
            "mean_kld": 0.002654,
            "median_kld": 0.0012,
            "rms_dp": 1.458,
            "same_top_p": 97.698,
            "vram_peak_gib": 21.4,
            "status": "ok"
        },
        {
            "series": "quality",
            "quant": "Qwen3.8-27B-Q8_0.gguf",
            "ppl": 6.790667,
            "ppl_unc": 0.073943,
            "ppl_ratio": 1.001161,
            "mean_kld": 0.000797,
            "median_kld": 0.000326,
            "rms_dp": 0.814,
            "same_top_p": 98.71,
            "vram_peak_gib": 27.2,
            "status": "ok"
        }
    ]
}
