{
  "schemaVersion": 1,
  "license": "MIT; see /data/LICENSE.txt",
  "reviewed": "2026-09-10",
  "records": [
    {
      "id": "intel-arc-pro-b70-qwen3-6-35b-a3b-ud-q4-k-m-sycl",
      "sourceRecordId": "intel-arc-pro-b70-qwen3-6-35b-a3b-ud-q4-k-m-sycl",
      "model": "Qwen 3.6 35B-A3B",
      "modelId": "qwen3-6-35b-a3b",
      "modelRevision": null,
      "architecture": "MoE",
      "quantization": "UD-Q4_K_M",
      "gpus": 1,
      "cohort": "april-2026",
      "tested": "2026-04-21",
      "reviewed": "2026-09-10",
      "status": "Historical",
      "backend": "llama.cpp / SYCL",
      "commit": "ec6f7a6a5c",
      "build": "b8840-12-gec6f7a6a5-dirty",
      "dirty": true,
      "localPatches": null,
      "weightsGiB": 20.61,
      "contextConfigured": 4096,
      "prefillTokens": 512,
      "decodeTokens": 128,
      "decodePromptTokens": null,
      "decodeDepth": null,
      "kvK": "f16",
      "kvV": "f16",
      "flashAttention": true,
      "threads": 1,
      "batch": null,
      "microbatch": null,
      "concurrency": null,
      "warmup": null,
      "repetitions": null,
      "environment": {
        "os": "Ubuntu 26.04 (cohort report)",
        "kernel": "7.0.0-10-generic (cohort report)",
        "driver": "xe / compute-runtime 26.09 (cohort report)",
        "runtime": "oneAPI 2025.3.3 (cohort report)",
        "cpu": "Ryzen 5 9600X (cohort report)",
        "pcie": null
      },
      "metrics": {
        "decode": 54.65,
        "prefill": 615.3,
        "decodeStdDev": 0.03,
        "prefillStdDev": 2.82
      },
      "source": {
        "label": "Original benchmark JSON",
        "url": "https://github.com/PMZFX/intel-arc-pro-b70-benchmarks/blob/daddb701fe116fb73ed9884f3fc566af51c2ad40/data/llm/intel-arc-pro-b70-qwen3-6-35b-a3b-ud-q4-k-m-sycl.json",
        "sha256": "2ec17f8960d7032c7d46b515ec7975d62f816e19e777ca1c6e6054fc9e61d15a",
        "file": "intel-arc-pro-b70-qwen3-6-35b-a3b-ud-q4-k-m-sycl.json"
      },
      "limitations": [
        "Dirty build: the exact local patch diff has not been recovered.",
        "One thread is recorded in this result; methodology prose says six. Per-result metadata is used.",
        "Warmup and five repetitions are described by methodology, but not recorded per result.",
        "Configured 4K context is not a full-context generation test.",
        "Energy and VRAM telemetry are excluded: device inclusion and measurement windows are unresolved.",
        "Environment is reported by cohort documentation, not captured in this result. PCIe topology conflicts remain unresolved."
      ]
    },
    {
      "id": "intel-arc-pro-b70-qwen3-coder-next-80b-a3b-q4-k-m-sycl-2gpu",
      "sourceRecordId": "intel-arc-pro-b70-qwen3-coder-next-80b-a3b-q4-k-m-sycl-2gpu",
      "model": "Qwen3-Coder-Next 80B-A3B",
      "modelId": "qwen3-coder-next-80b-a3b",
      "modelRevision": null,
      "architecture": "MoE",
      "quantization": "Q4_K_M",
      "gpus": 2,
      "cohort": "april-2026",
      "tested": "2026-04-21",
      "reviewed": "2026-09-10",
      "status": "Historical",
      "backend": "llama.cpp / SYCL",
      "commit": "ec6f7a6a5c",
      "build": "b8840-12-gec6f7a6a5-dirty",
      "dirty": true,
      "localPatches": null,
      "weightsGiB": null,
      "contextConfigured": 4096,
      "prefillTokens": 512,
      "decodeTokens": 128,
      "decodePromptTokens": null,
      "decodeDepth": null,
      "kvK": "f16",
      "kvV": "f16",
      "flashAttention": true,
      "threads": 1,
      "batch": null,
      "microbatch": null,
      "concurrency": null,
      "warmup": null,
      "repetitions": null,
      "environment": {
        "os": "Ubuntu 26.04 (cohort report)",
        "kernel": "7.0.0-10-generic (cohort report)",
        "driver": "xe / compute-runtime 26.09 (cohort report)",
        "runtime": "oneAPI 2025.3.3 (cohort report)",
        "cpu": "Ryzen 5 9600X (cohort report)",
        "pcie": null
      },
      "metrics": {
        "decode": 43.35,
        "prefill": 304.98,
        "decodeStdDev": 0.07,
        "prefillStdDev": 2.22
      },
      "source": {
        "label": "Original benchmark JSON",
        "url": "https://github.com/PMZFX/intel-arc-pro-b70-benchmarks/blob/daddb701fe116fb73ed9884f3fc566af51c2ad40/data/llm/intel-arc-pro-b70-qwen3-coder-next-80b-a3b-q4-k-m-sycl-2gpu.json",
        "sha256": "e99cbe3e1dbb76154c48f16eabd3811f91844b4de1849f1a063470222ba7f902",
        "file": "intel-arc-pro-b70-qwen3-coder-next-80b-a3b-q4-k-m-sycl-2gpu.json"
      },
      "limitations": [
        "Dirty build: the exact local patch diff has not been recovered.",
        "One thread is recorded in this result; methodology prose says six. Per-result metadata is used.",
        "Warmup and five repetitions are described by methodology, but not recorded per result.",
        "Configured 4K context is not a full-context generation test.",
        "Energy and VRAM telemetry are excluded: device inclusion and measurement windows are unresolved.",
        "Environment is reported by cohort documentation, not captured in this result. PCIe topology conflicts remain unresolved.",
        "Weight size unresolved: this JSON says 14.46 GiB while the report says about 45.1 GiB. Neither is used as an audited weight-size claim."
      ]
    },
    {
      "id": "intel-arc-pro-b70-deepseek-r1-distill-llama-70b-q4-k-m-sycl-2gpu",
      "sourceRecordId": "intel-arc-pro-b70-deepseek-r1-distill-llama-70b-q4-k-m-sycl-2gpu",
      "model": "DeepSeek R1 Distill 70B",
      "modelId": "deepseek-r1-distill-llama-70b",
      "modelRevision": null,
      "architecture": "Dense",
      "quantization": "Q4_K_M",
      "gpus": 2,
      "cohort": "april-2026",
      "tested": "2026-04-21",
      "reviewed": "2026-09-10",
      "status": "Historical",
      "backend": "llama.cpp / SYCL",
      "commit": "ec6f7a6a5c",
      "build": "b8840-12-gec6f7a6a5-dirty",
      "dirty": true,
      "localPatches": null,
      "weightsGiB": 39.6,
      "contextConfigured": 4096,
      "prefillTokens": 512,
      "decodeTokens": 128,
      "decodePromptTokens": null,
      "decodeDepth": null,
      "kvK": "f16",
      "kvV": "f16",
      "flashAttention": true,
      "threads": 1,
      "batch": null,
      "microbatch": null,
      "concurrency": null,
      "warmup": null,
      "repetitions": null,
      "environment": {
        "os": "Ubuntu 26.04 (cohort report)",
        "kernel": "7.0.0-10-generic (cohort report)",
        "driver": "xe / compute-runtime 26.09 (cohort report)",
        "runtime": "oneAPI 2025.3.3 (cohort report)",
        "cpu": "Ryzen 5 9600X (cohort report)",
        "pcie": null
      },
      "metrics": {
        "decode": 11.47,
        "prefill": 336.06,
        "decodeStdDev": 0,
        "prefillStdDev": 3.82
      },
      "source": {
        "label": "Original benchmark JSON",
        "url": "https://github.com/PMZFX/intel-arc-pro-b70-benchmarks/blob/daddb701fe116fb73ed9884f3fc566af51c2ad40/data/llm/intel-arc-pro-b70-deepseek-r1-distill-llama-70b-q4-k-m-sycl-2gpu.json",
        "sha256": "24e822a19618191617fca349f133dba1864bd5fee900fa783a1868d7f537e466",
        "file": "intel-arc-pro-b70-deepseek-r1-distill-llama-70b-q4-k-m-sycl-2gpu.json"
      },
      "limitations": [
        "Dirty build: the exact local patch diff has not been recovered.",
        "One thread is recorded in this result; methodology prose says six. Per-result metadata is used.",
        "Warmup and five repetitions are described by methodology, but not recorded per result.",
        "Configured 4K context is not a full-context generation test.",
        "Energy and VRAM telemetry are excluded: device inclusion and measurement windows are unresolved.",
        "Environment is reported by cohort documentation, not captured in this result. PCIe topology conflicts remain unresolved."
      ]
    },
    {
      "id": "intel-arc-pro-b70-qwen3-5-27b-q4-k-m-sycl",
      "sourceRecordId": "intel-arc-pro-b70-qwen3-5-27b-q4-k-m-sycl",
      "model": "Qwen 3.5 27B",
      "modelId": "qwen3-5-27b",
      "modelRevision": null,
      "architecture": "Dense",
      "quantization": "Q4_K_M",
      "gpus": 1,
      "cohort": "april-2026",
      "tested": "2026-04-21",
      "reviewed": "2026-09-10",
      "status": "Historical",
      "backend": "llama.cpp / SYCL",
      "commit": "ec6f7a6a5c",
      "build": "b8840-12-gec6f7a6a5-dirty",
      "dirty": true,
      "localPatches": null,
      "weightsGiB": 15.59,
      "contextConfigured": 4096,
      "prefillTokens": 512,
      "decodeTokens": 128,
      "decodePromptTokens": null,
      "decodeDepth": null,
      "kvK": "f16",
      "kvV": "f16",
      "flashAttention": true,
      "threads": 1,
      "batch": null,
      "microbatch": null,
      "concurrency": null,
      "warmup": null,
      "repetitions": null,
      "environment": {
        "os": "Ubuntu 26.04 (cohort report)",
        "kernel": "7.0.0-10-generic (cohort report)",
        "driver": "xe / compute-runtime 26.09 (cohort report)",
        "runtime": "oneAPI 2025.3.3 (cohort report)",
        "cpu": "Ryzen 5 9600X (cohort report)",
        "pcie": null
      },
      "metrics": {
        "decode": 20.35,
        "prefill": 718.21,
        "decodeStdDev": 0.03,
        "prefillStdDev": 3.16
      },
      "source": {
        "label": "Original benchmark JSON",
        "url": "https://github.com/PMZFX/intel-arc-pro-b70-benchmarks/blob/daddb701fe116fb73ed9884f3fc566af51c2ad40/data/llm/intel-arc-pro-b70-qwen3-5-27b-q4-k-m-sycl.json",
        "sha256": "e24e66b6dd5965219902775116f2fe733d81c87ba75cfc4dcedd5a740150a196",
        "file": "intel-arc-pro-b70-qwen3-5-27b-q4-k-m-sycl.json"
      },
      "limitations": [
        "Dirty build: the exact local patch diff has not been recovered.",
        "One thread is recorded in this result; methodology prose says six. Per-result metadata is used.",
        "Warmup and five repetitions are described by methodology, but not recorded per result.",
        "Configured 4K context is not a full-context generation test.",
        "Energy and VRAM telemetry are excluded: device inclusion and measurement windows are unresolved.",
        "Environment is reported by cohort documentation, not captured in this result. PCIe topology conflicts remain unresolved."
      ]
    },
    {
      "id": "intel-arc-pro-b70-gemma-4-31b-q4-k-m-sycl",
      "sourceRecordId": "intel-arc-pro-b70-gemma-4-31b-q4-k-m-sycl",
      "model": "Gemma 4 31B",
      "modelId": "gemma-4-31b",
      "modelRevision": null,
      "architecture": "Dense",
      "quantization": "Q4_K_M",
      "gpus": 1,
      "cohort": "april-2026",
      "tested": "2026-04-21",
      "reviewed": "2026-09-10",
      "status": "Historical",
      "backend": "llama.cpp / SYCL",
      "commit": "ec6f7a6a5c",
      "build": "b8840-12-gec6f7a6a5-dirty",
      "dirty": true,
      "localPatches": null,
      "weightsGiB": 17.07,
      "contextConfigured": 4096,
      "prefillTokens": 512,
      "decodeTokens": 128,
      "decodePromptTokens": null,
      "decodeDepth": null,
      "kvK": "f16",
      "kvV": "f16",
      "flashAttention": true,
      "threads": 1,
      "batch": null,
      "microbatch": null,
      "concurrency": null,
      "warmup": null,
      "repetitions": null,
      "environment": {
        "os": "Ubuntu 26.04 (cohort report)",
        "kernel": "7.0.0-10-generic (cohort report)",
        "driver": "xe / compute-runtime 26.09 (cohort report)",
        "runtime": "oneAPI 2025.3.3 (cohort report)",
        "cpu": "Ryzen 5 9600X (cohort report)",
        "pcie": null
      },
      "metrics": {
        "decode": 21.7,
        "prefill": 600.78,
        "decodeStdDev": 0.03,
        "prefillStdDev": 0.87
      },
      "source": {
        "label": "Original benchmark JSON",
        "url": "https://github.com/PMZFX/intel-arc-pro-b70-benchmarks/blob/daddb701fe116fb73ed9884f3fc566af51c2ad40/data/llm/intel-arc-pro-b70-gemma-4-31b-q4-k-m-sycl.json",
        "sha256": "8d0b5ebabde9a515fcdd38538c435fd660dec076b3d6a98758aab58ded890470",
        "file": "intel-arc-pro-b70-gemma-4-31b-q4-k-m-sycl.json"
      },
      "limitations": [
        "Dirty build: the exact local patch diff has not been recovered.",
        "One thread is recorded in this result; methodology prose says six. Per-result metadata is used.",
        "Warmup and five repetitions are described by methodology, but not recorded per result.",
        "Configured 4K context is not a full-context generation test.",
        "Energy and VRAM telemetry are excluded: device inclusion and measurement windows are unresolved.",
        "Environment is reported by cohort documentation, not captured in this result. PCIe topology conflicts remain unresolved."
      ]
    },
    {
      "id": "june-qwen36-35b-q4",
      "sourceRecordId": "20260613T085749Z_llamacpp_qwen36-moe-35b-q4_single_32768_q4_0",
      "model": "Qwen 3.6 35B-A3B",
      "modelId": "qwen3-6-35b-a3b",
      "modelRevision": null,
      "architecture": "MoE",
      "quantization": "UD-Q4_K_M",
      "gpus": 1,
      "cohort": "june-2026",
      "tested": "2026-06-13",
      "reviewed": "2026-09-10",
      "status": "Historical",
      "backend": "llama.cpp / SYCL",
      "commit": "d8a24cc",
      "build": null,
      "dirty": null,
      "localPatches": null,
      "weightsGiB": 20.604151248931885,
      "contextConfigured": 32768,
      "prefillTokens": 512,
      "decodeTokens": 128,
      "decodePromptTokens": 0,
      "decodeDepth": 0,
      "kvK": "q4_0",
      "kvV": "q4_0",
      "flashAttention": null,
      "threads": 6,
      "batch": 2048,
      "microbatch": 512,
      "concurrency": null,
      "warmup": null,
      "repetitions": 3,
      "environment": {
        "os": null,
        "kernel": null,
        "driver": null,
        "runtime": "oneAPI 2026.0 (cohort report)",
        "cpu": "AMD Ryzen 5 9600X 6-Core Processor",
        "pcie": null
      },
      "metrics": {
        "decode": 68.880402,
        "prefill": 1030.696373,
        "decodeStdDev": 0.411744,
        "prefillStdDev": 6.635591
      },
      "source": {
        "label": "Reviewed June measurement excerpt",
        "url": "/data/june-evidence.json",
        "sha256": "9558dbe958e1f04cd61190a0be5341377a380749dab2c5720b64f17942dafdf4",
        "file": "20260613T085749Z_llamacpp_qwen36-moe-35b-q4_single_32768_q4_0.jsonl"
      },
      "limitations": [
        "Decode is a separate n_prompt=0, n_gen=128, n_depth=0 row. It is not generation after filling 32K context.",
        "The 32K prefill row measures 747.127922 tok/s; explorer prefill consistently shows pp512.",
        "Build cleanliness, model revision, OS, kernel, driver and warmup are not established by this raw file.",
        "Flash attention records -1 (automatic); effective enablement is unknown.",
        "This is a separate cohort with different cache and software settings. Do not calculate a controlled speedup against April."
      ]
    }
  ]
}