{
 "attribution": "AGmind Systems Lab (agmind.ai)",
 "id": "strix.qwen36.docsession.c1.ttft-q2-8k-cache",
 "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first token, second question over an 8k document, cache on: 660 ms",
 "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first token, second question over an 8k document, cache on was 660 ms (median over valid requests for this item). The measurement was taken under the frozen doc-session-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
 "value": "660",
 "unit": "ms",
 "metric": "time to first token, second question over an 8k document, cache on",
 "concurrency": 1,
 "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
 "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
 "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
 "units_measured": 1,
 "statement": "Median time to first token for the second question over the same 8k-token document with the server prompt cache enabled.",
 "scope": "doc-session-v1@2026-08-03",
 "aggregation": "median over valid requests for this item",
 "limitations": "Three repeated runs on one unit, reasoning disabled, single stream. Cache benefit requires a byte-identical document prefix in the same server session.",
 "evidence_level": "lab_repeated",
 "status": "active",
 "run_ids": [
  "run-20260803-ds-cache-a",
  "run-20260803-ds-cache-a-r2",
  "run-20260803-ds-cache-a-r3"
 ],
 "runs": [
  "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a",
  "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r2",
  "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r3"
 ],
 "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-ds-en-8k-q2.sql",
 "permalink": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache/",
 "json": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache.json",
 "markdown": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache.md",
 "cite": "AGmind Systems Lab. \"Median time to first token for the second question over the same 8k-token document with the server prompt cache enabled.\" Claim strix.qwen36.docsession.c1.ttft-q2-8k-cache (660 ms), evidence level lab_repeated, scope doc-session-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache/",
 "license": "https://creativecommons.org/licenses/by/4.0/",
 "corrections": "https://agmind.ai/errata/"
}