{
  "metadata": {
    "measurement_date": "2026-09-05",
    "snapshot_date": "2026-09-12",
    "verified_at": "2026-09-12T19:03:13.523320+00:00",
    "citation_url": "https://llm-speed.com/blog/qwen3-8-27b-vs-gemma-4-12b-rtx-5090",
    "data_license": "CC BY 4.0",
    "hardware": "RTX 5090 (32GB); AMD Ryzen 7 9850X3D; 30GB system RAM reported by WSL",
    "runtime_commit": "9725a313be0528214c4a02fed906ddaf7b3f712e",
    "configured_context_tokens": 16384,
    "active_requests": 1,
    "model_state": "warm",
    "thinking": "off",
    "prompt_cache": "disabled",
    "temperature": 0,
    "seed": 42,
    "output_caps": {
      "chat-short": 256,
      "chat-long": 1024
    },
    "limitations": [
      "Different model sizes, quantization schemes, tokenizers and actual output lengths",
      "No quality, vision, cold-start, peak whole-device memory, full-context or multi-user measurement",
      "Other idle services remained GPU-resident",
      "Only the six accepted source runs are included; five earlier affected submissions excluded",
      "Isolated 0.0.5 client had usage/quantization fixes; server settings alone do not supply full end-to-end reproduction"
    ],
    "artifacts": [
      {
        "model": "Qwen3.8-27B",
        "quantization": "Q4_K_M",
        "repository": "ggml-org/Qwen3.8-27B-GGUF",
        "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
        "filename": "Qwen3.8-27B-Q4_K_M.gguf",
        "sha256": "31629f53165ab6a7dad8c9847dcfd1fdf55829dac1e6e748f4a68581b0033d34",
        "file_bytes": 18973870432,
        "runtime_allocated_gpu_mib": 19071,
        "gpu_layers": 65
      },
      {
        "model": "gemma-4-12b-it-qat",
        "quantization": "Q4_0",
        "repository": "google/gemma-4-12B-it-qat-q4_0-gguf",
        "revision": "29d097773436b69ff9feafd636ab4cf873786537",
        "filename": "gemma-4-12b-it-qat-q4_0.gguf",
        "sha256": "93567e57a8fe10b23569b9d9ec38cd005deedf71e29477c421a4b83f418a538b",
        "file_bytes": 6975879296,
        "runtime_allocated_gpu_mib": 7931,
        "gpu_layers": 49
      }
    ]
  },
  "measurements": [
    {
      "run_id": "r_7mod38qsldj",
      "source_url": "https://llm-speed.com/r/r_7mod38qsldj",
      "received_at": "2026-09-05 20:41:27",
      "model": "Qwen3.8-27B",
      "quantization": "Q4_K_M",
      "workload": "chat-short",
      "input_tokens": 114,
      "output_tokens": 256,
      "decode_tps": 65.42599106488873,
      "ttft_ms": 131.74251198768616
    },
    {
      "run_id": "r_7mod38qsldj",
      "source_url": "https://llm-speed.com/r/r_7mod38qsldj",
      "received_at": "2026-09-05 20:41:27",
      "model": "Qwen3.8-27B",
      "quantization": "Q4_K_M",
      "workload": "chat-long",
      "input_tokens": 3184,
      "output_tokens": 874,
      "decode_tps": 65.10200190331615,
      "ttft_ms": 909.4263269901276
    },
    {
      "run_id": "r_c3ps9wjygi9",
      "source_url": "https://llm-speed.com/r/r_c3ps9wjygi9",
      "received_at": "2026-09-05 20:41:46",
      "model": "Qwen3.8-27B",
      "quantization": "Q4_K_M",
      "workload": "chat-short",
      "input_tokens": 114,
      "output_tokens": 256,
      "decode_tps": 65.72733223796655,
      "ttft_ms": 117.09531092643738
    },
    {
      "run_id": "r_c3ps9wjygi9",
      "source_url": "https://llm-speed.com/r/r_c3ps9wjygi9",
      "received_at": "2026-09-05 20:41:46",
      "model": "Qwen3.8-27B",
      "quantization": "Q4_K_M",
      "workload": "chat-long",
      "input_tokens": 3184,
      "output_tokens": 874,
      "decode_tps": 65.09428452837453,
      "ttft_ms": 908.02221596241
    },
    {
      "run_id": "r_dz5a3lzxhrj",
      "source_url": "https://llm-speed.com/r/r_dz5a3lzxhrj",
      "received_at": "2026-09-05 20:42:06",
      "model": "Qwen3.8-27B",
      "quantization": "Q4_K_M",
      "workload": "chat-short",
      "input_tokens": 114,
      "output_tokens": 256,
      "decode_tps": 65.66459008761952,
      "ttft_ms": 117.54690706729889
    },
    {
      "run_id": "r_dz5a3lzxhrj",
      "source_url": "https://llm-speed.com/r/r_dz5a3lzxhrj",
      "received_at": "2026-09-05 20:42:06",
      "model": "Qwen3.8-27B",
      "quantization": "Q4_K_M",
      "workload": "chat-long",
      "input_tokens": 3184,
      "output_tokens": 874,
      "decode_tps": 65.09837588987713,
      "ttft_ms": 910.5457819700241
    },
    {
      "run_id": "r_8xmm65n1fhq",
      "source_url": "https://llm-speed.com/r/r_8xmm65n1fhq",
      "received_at": "2026-09-05 20:43:41",
      "model": "gemma-4-12b-it-qat",
      "quantization": "Q4_0",
      "workload": "chat-short",
      "input_tokens": 121,
      "output_tokens": 256,
      "decode_tps": 137.26866341460945,
      "ttft_ms": 52.23644304275513
    },
    {
      "run_id": "r_8xmm65n1fhq",
      "source_url": "https://llm-speed.com/r/r_8xmm65n1fhq",
      "received_at": "2026-09-05 20:43:41",
      "model": "gemma-4-12b-it-qat",
      "quantization": "Q4_0",
      "workload": "chat-long",
      "input_tokens": 3188,
      "output_tokens": 620,
      "decode_tps": 135.6114752156515,
      "ttft_ms": 577.1910140514374
    },
    {
      "run_id": "r_c2podhe9qwc",
      "source_url": "https://llm-speed.com/r/r_c2podhe9qwc",
      "received_at": "2026-09-05 20:43:49",
      "model": "gemma-4-12b-it-qat",
      "quantization": "Q4_0",
      "workload": "chat-short",
      "input_tokens": 121,
      "output_tokens": 256,
      "decode_tps": 139.50260140767685,
      "ttft_ms": 61.77092409133911
    },
    {
      "run_id": "r_c2podhe9qwc",
      "source_url": "https://llm-speed.com/r/r_c2podhe9qwc",
      "received_at": "2026-09-05 20:43:49",
      "model": "gemma-4-12b-it-qat",
      "quantization": "Q4_0",
      "workload": "chat-long",
      "input_tokens": 3188,
      "output_tokens": 620,
      "decode_tps": 137.1097358754115,
      "ttft_ms": 622.9023209810257
    },
    {
      "run_id": "r_v5arqmazf31",
      "source_url": "https://llm-speed.com/r/r_v5arqmazf31",
      "received_at": "2026-09-05 20:43:57",
      "model": "gemma-4-12b-it-qat",
      "quantization": "Q4_0",
      "workload": "chat-short",
      "input_tokens": 121,
      "output_tokens": 256,
      "decode_tps": 141.11711710968086,
      "ttft_ms": 59.66701304912567
    },
    {
      "run_id": "r_v5arqmazf31",
      "source_url": "https://llm-speed.com/r/r_v5arqmazf31",
      "received_at": "2026-09-05 20:43:57",
      "model": "gemma-4-12b-it-qat",
      "quantization": "Q4_0",
      "workload": "chat-long",
      "input_tokens": 3188,
      "output_tokens": 620,
      "decode_tps": 135.64206344798518,
      "ttft_ms": 619.3108520507812
    }
  ],
  "summary": [
    {
      "model": "gemma-4-12b-it-qat",
      "workload": "chat-short",
      "n": 3,
      "decode_tps": {
        "median": 139.50260140767685,
        "min": 137.26866341460945,
        "max": 141.11711710968086
      },
      "ttft_ms": {
        "median": 59.66701304912567,
        "min": 52.23644304275513,
        "max": 61.77092409133911
      },
      "input_tokens": {
        "median": 121,
        "min": 121,
        "max": 121
      },
      "output_tokens": {
        "median": 256,
        "min": 256,
        "max": 256
      }
    },
    {
      "model": "Qwen3.8-27B",
      "workload": "chat-short",
      "n": 3,
      "decode_tps": {
        "median": 65.66459008761952,
        "min": 65.42599106488873,
        "max": 65.72733223796655
      },
      "ttft_ms": {
        "median": 117.54690706729889,
        "min": 117.09531092643738,
        "max": 131.74251198768616
      },
      "input_tokens": {
        "median": 114,
        "min": 114,
        "max": 114
      },
      "output_tokens": {
        "median": 256,
        "min": 256,
        "max": 256
      }
    },
    {
      "model": "gemma-4-12b-it-qat",
      "workload": "chat-long",
      "n": 3,
      "decode_tps": {
        "median": 135.64206344798518,
        "min": 135.6114752156515,
        "max": 137.1097358754115
      },
      "ttft_ms": {
        "median": 619.3108520507812,
        "min": 577.1910140514374,
        "max": 622.9023209810257
      },
      "input_tokens": {
        "median": 3188,
        "min": 3188,
        "max": 3188
      },
      "output_tokens": {
        "median": 620,
        "min": 620,
        "max": 620
      }
    },
    {
      "model": "Qwen3.8-27B",
      "workload": "chat-long",
      "n": 3,
      "decode_tps": {
        "median": 65.09837588987713,
        "min": 65.09428452837453,
        "max": 65.10200190331615
      },
      "ttft_ms": {
        "median": 909.4263269901276,
        "min": 908.02221596241,
        "max": 910.5457819700241
      },
      "input_tokens": {
        "median": 3184,
        "min": 3184,
        "max": 3184
      },
      "output_tokens": {
        "median": 874,
        "min": 874,
        "max": 874
      }
    }
  ]
}
