{
  "$comment": "GENERATED for curated-public rows by scripts/seed-curated-benchmarks.ts. Community submissions are merged in by scripts/export-measured-benchmarks.ts and carry provenance \"community-submitted\".",
  "schemaVersion": "2.0",
  "schemaDescription": "Each entry is one benchmark run of one model on one accelerator at one quantization and context length. tokS is decode throughput. Fields added in v2 (promptTokS, vramUsedGb, measuredWatts + wattsMethod, offloadState, nGpuLayers, kvCacheType, batchSize, reviewedAt) are optional and null where no figure exists. A record carrying measuredWatts always carries wattsMethod; a community-submitted record always carries reviewedAt.",
  "generatedAt": "2026-09-08T15:39:01.768Z",
  "license": "CC BY 4.0",
  "attribution": "LLM Configurator — https://llmconfigurator.com/en/benchmarks?utm_source=dataset&utm_medium=referral&utm_campaign=measured_benchmarks_export",
  "provenanceNote": "Rows marked curated-public are compiled from third-party published benchmarks and are credited to their publisher with a source URL; they are not submissions to this site. Rows marked community-submitted come from readers and carry a sample count.",
  "entries": [
    {
      "id": "hc-4090-qwen3-30b-a3b-4k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-30b-a3b",
      "quant": "Q4_K_XL",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 4096,
      "os": "unspecified",
      "tokS": 195.78,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/rtx-4090-llm-benchmarks/",
      "modelIdentity": "confirmed",
      "note": "Source states the 30B MoE is 16.47 GB at Q4_K_XL, which matches Qwen3-30B-A3B and fixes the file size directly rather than deriving it from a bits-per-weight table.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-qwen3-30b-a3b-57k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-30b-a3b",
      "quant": "Q4_K_XL",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 57344,
      "os": "unspecified",
      "tokS": 74.64,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/rtx-4090-llm-benchmarks/",
      "modelIdentity": "confirmed",
      "note": "57K is the largest context the source reports fitting on a 24 GB card for this model.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-qwen3-8b-4k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-8b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 4096,
      "os": "unspecified",
      "tokS": 141.3,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/gpu-llm-benchmarks/rtx-4090/",
      "modelIdentity": "confirmed",
      "note": "Named Qwen3-8B Q4_K context-scaling series, five points from 4K to 128K.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-qwen3-8b-16k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-8b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 16384,
      "os": "unspecified",
      "tokS": 108,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/gpu-llm-benchmarks/rtx-4090/",
      "modelIdentity": "confirmed",
      "note": null,
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-qwen3-8b-32k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-8b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 32768,
      "os": "unspecified",
      "tokS": 82.3,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/gpu-llm-benchmarks/rtx-4090/",
      "modelIdentity": "confirmed",
      "note": null,
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-qwen3-8b-64k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-8b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 65536,
      "os": "unspecified",
      "tokS": 56.1,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/gpu-llm-benchmarks/rtx-4090/",
      "modelIdentity": "confirmed",
      "note": null,
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-qwen3-8b-128k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-8b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 131072,
      "os": "unspecified",
      "tokS": 33.8,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/gpu-llm-benchmarks/rtx-4090/",
      "modelIdentity": "confirmed",
      "note": null,
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-8b-q4kxl-4k",
      "gpuId": "rtx-4090",
      "variantId": "llama-3.1-8b",
      "quant": "Q4_K_XL",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 4096,
      "os": "unspecified",
      "tokS": 131,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/rtx-4090-llm-benchmarks/",
      "modelIdentity": "inferred",
      "note": "Source reports this row as '8B' without naming the checkpoint. Attributed to Llama 3.1 8B; its 32-layer/8-KV-head geometry differs slightly from Qwen3-8B's 36 layers, so the identity is marked inferred.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-8b-q4kxl-65k",
      "gpuId": "rtx-4090",
      "variantId": "llama-3.1-8b",
      "quant": "Q4_K_XL",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 65536,
      "os": "unspecified",
      "tokS": 53.07,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/rtx-4090-llm-benchmarks/",
      "modelIdentity": "inferred",
      "note": "Pairs with the 4K row; the 59% drop between them is the headline context-scaling figure.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-14b-q4kxl-4k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-14b",
      "quant": "Q4_K_XL",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 4096,
      "os": "unspecified",
      "tokS": 82.82,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/rtx-4090-llm-benchmarks/",
      "modelIdentity": "inferred",
      "note": "Source reports '14B' only. Attributed to Qwen3-14B because the publisher's confirmed rows on the same page (8B, 30B-A3B) are Qwen3.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "hc-4090-14b-q4kxl-65k",
      "gpuId": "rtx-4090",
      "variantId": "qwen3-14b",
      "quant": "Q4_K_XL",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 65536,
      "os": "unspecified",
      "tokS": 38.73,
      "provenance": "curated-public",
      "publisher": "Hardware Corner",
      "sourceUrl": "https://www.hardware-corner.net/rtx-4090-llm-benchmarks/",
      "modelIdentity": "inferred",
      "note": null,
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "mustafa-4090-7b",
      "gpuId": "rtx-4090",
      "variantId": "qwen-2.5-7b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 4096,
      "os": "unspecified",
      "tokS": 135,
      "provenance": "curated-public",
      "publisher": "Mustafa.net",
      "sourceUrl": "https://mustafa.net/llm-tokens-per-second-benchmarks/",
      "modelIdentity": "inferred",
      "note": "Source reports a generic '7B Q4_K_M, single batch, warm model' row with no checkpoint and no stated context. Qwen 2.5 7B is used as a standard-geometry 7B stand-in and 4K assumed.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "mustafa-3090-7b",
      "gpuId": "rtx-3090",
      "variantId": "qwen-2.5-7b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 4096,
      "os": "unspecified",
      "tokS": 95,
      "provenance": "curated-public",
      "publisher": "Mustafa.net",
      "sourceUrl": "https://mustafa.net/llm-tokens-per-second-benchmarks/",
      "modelIdentity": "inferred",
      "note": "THE ONLY AMPERE ANCHOR IN THE SET. eta for nvidia-ampere rests entirely on this one inferred-identity point; treat the Ampere constant as materially weaker than the Ada one.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    },
    {
      "id": "mustafa-m3max-7b",
      "gpuId": "apple-m3-max",
      "variantId": "qwen-2.5-7b",
      "quant": "Q4_K_M",
      "runtime": "llama.cpp llama-bench",
      "contextLength": 4096,
      "os": "unspecified",
      "tokS": 40,
      "provenance": "curated-public",
      "publisher": "Mustafa.net",
      "sourceUrl": "https://mustafa.net/llm-tokens-per-second-benchmarks/",
      "modelIdentity": "inferred",
      "note": "THE ONLY APPLE ANCHOR IN THE SET, and the only Metal row. With one point, eta and t_fixed are not jointly identifiable, so the fit solves eta only and leaves the Metal t_fixed at its seed value, reported unfitted.",
      "date": "2026-08-17",
      "offloadState": "fully-resident",
      "reviewedAt": null,
      "promptTokS": null,
      "vramUsedGb": null,
      "measuredWatts": null,
      "wattsMethod": null,
      "measuredNoiseDba": null,
      "measuredDeltaC": null,
      "nGpuLayers": null,
      "kvCacheType": null,
      "batchSize": null,
      "submittedBy": null
    }
  ]
}
