{
  "source_sha256": "9a357bf482825e3c8e9af84f6ad273377d248cda5594d19d1818ea1f9eaa44da",
  "plot_script_sha256": "f6d7fb1dabf9e27cdf90e8f4db79aedd46ee6aa363f0f9951479af088444dc0e",
  "matplotlib_version": "3.11.2",
  "numpy_version": "2.5.3",
  "design_status": "Post-analysis descriptive figures, not preregistered axes. Fixed full-domain axes and complete declared subsets; no outcome-selected cell removal.",
  "charts": [
    {
      "name": "01-ttft",
      "data": "../matrix.json",
      "axes": "All 13 cells in pinned order; logarithmic seconds, 0.2\u20131500; compact/full paired within each cell.",
      "note": "C13 * has 86 pre-interaction requests, one confirmed overlap, three recovery-unknown. Timings include truncated outputs; not useful-answer latency.",
      "files": [
        "01-ttft.svg",
        "01-ttft.pdf",
        "01-ttft.png"
      ]
    },
    {
      "name": "02-semantic",
      "data": "../matrix.json",
      "axes": "All 13 cells; stacked integer task counts, 0\u201336; no collapsed benchmark score.",
      "note": "C13 diagnostics all precede recorded interaction; 28 truncated diagnostic outputs and 7 other unknowns. Different context/KV settings must not be treated as weight-only comparisons.",
      "files": [
        "02-semantic.svg",
        "02-semantic.pdf",
        "02-semantic.png"
      ]
    },
    {
      "name": "03-generation",
      "data": "../matrix.json",
      "axes": "All 13 cells; linear zero-based tokens/s axis; diagnostic/full conditions separately shown.",
      "note": "C13 * includes truncations; throughput is not answer quality. Generation limit is 2048. Different input lengths are explicit, so this is not a pure backend-speed ranking.",
      "files": [
        "03-generation.svg",
        "03-generation.pdf",
        "03-generation.png"
      ]
    },
    {
      "name": "04-prefill",
      "data": "../matrix.json",
      "axes": "All 13 cells; linear zero-based tokens/s axis; diagnostic/full conditions separately shown.",
      "note": "C13 * includes truncations; throughput is not answer quality. Generation limit is 2048. Different input lengths are explicit, so this is not a pure backend-speed ranking.",
      "files": [
        "04-prefill.svg",
        "04-prefill.pdf",
        "04-prefill.png"
      ]
    },
    {
      "name": "05-weight-bytes",
      "data": "../matrix.json",
      "axes": "Nine unique weight artifacts, ordered by size class and precision; zero-based 0\u201322 GiB.",
      "note": "Do not read unused chart area as model headroom. Total GPU allocation, exact KV allocation and usable headroom are NOT_MEASURED.",
      "files": [
        "05-weight-bytes.svg",
        "05-weight-bytes.pdf",
        "05-weight-bytes.png"
      ]
    },
    {
      "name": "06-speed-capability",
      "data": "../matrix.json",
      "axes": "Nine 4K/f16 cells only; linear x 0\u201355 tokens/s, y 0\u2013100%; per-point UNKNOWN explicit.",
      "note": "Descriptive observed frontier only; one host/session, no statistical equivalence or causal quantization claim. 32B Q8 and other untested configurations are not inferred.",
      "files": [
        "06-speed-capability.svg",
        "06-speed-capability.pdf",
        "06-speed-capability.png"
      ]
    }
  ]
}
