{
 "title": "AGmind Systems Lab — claim registry",
 "description": "Every measured number published on agmind.ai. Values are re-derived from raw run records in CI; statements are EN-canonical. Raw evidence: https://github.com/botAGI/agmind-lab",
 "license": "https://creativecommons.org/licenses/by/4.0/",
 "attribution": "AGmind Systems Lab (agmind.ai)",
 "count": 40,
 "claims": [
  {
   "id": "strix.gemma4.interactive2.c1.answerless-default",
   "headline": "gemma-4-26B-A4B-it Q4_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — answerless (empty) responses in default mode: 8.3 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0), the measured answerless (empty) responses in default mode was 8.3 % of requests (share of all issued requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "8.3",
   "unit": "% of requests",
   "metric": "answerless (empty) responses in default mode",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0)",
   "units_measured": 1,
   "statement": "In its default operating mode at a 1024-token budget, this share of everyday requests returned HTTP 200 with an empty answer: the reasoning pass consumed the entire budget. A second model family reproduces the failure mode.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit. Default operating mode of the model (reasoning emitted); disabling reasoning via the chat template was not tested for this model family.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-gm-v2-a",
    "run-20260803-gm-v2-a-r2",
    "run-20260803-gm-v2-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/answerless-rate-1k.sql",
   "permalink": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.answerless-default/",
   "json": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.answerless-default.json",
   "markdown": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.answerless-default.md",
   "cite": "AGmind Systems Lab. \"In its default operating mode at a 1024-token budget, this share of everyday requests returned HTTP 200 with an empty answer: the reasoning pass consumed the entire budget. A second model family reproduces the failure mode.\" Claim strix.gemma4.interactive2.c1.answerless-default (8.3 % of requests), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.gemma4.interactive2.c1.answerless-default/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.gemma4.interactive2.c1.ttfa-default",
   "headline": "gemma-4-26B-A4B-it Q4_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first answer token in default mode: 10768 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0), the measured time to first answer token in default mode was 10768 ms (median over answered requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "10768",
   "unit": "ms",
   "metric": "time to first answer token in default mode",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0)",
   "units_measured": 1,
   "statement": "Median client-side time to the first token of the actual answer in the default operating mode (reasoning streams first).",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over answered requests",
   "limitations": "Three repeated runs on one unit. Default operating mode of the model (reasoning emitted); disabling reasoning via the chat template was not tested for this model family.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-gm-v2-a",
    "run-20260803-gm-v2-a-r2",
    "run-20260803-gm-v2-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-thinking-1k.sql",
   "permalink": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttfa-default/",
   "json": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttfa-default.json",
   "markdown": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttfa-default.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first token of the actual answer in the default operating mode (reasoning streams first).\" Claim strix.gemma4.interactive2.c1.ttfa-default (10768 ms), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttfa-default/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.gemma4.interactive2.c1.ttft-any-token",
   "headline": "gemma-4-26B-A4B-it Q4_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to the first token of any output: 282 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0), the measured time to the first token of any output was 282 ms (median over answered requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "282",
   "unit": "ms",
   "metric": "time to the first token of any output",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0)",
   "units_measured": 1,
   "statement": "Median client-side time to the first streamed token of any kind (reasoning included) in the default operating mode.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over answered requests",
   "limitations": "Three repeated runs on one unit. Default operating mode of the model (reasoning emitted); disabling reasoning via the chat template was not tested for this model family.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-gm-v2-a",
    "run-20260803-gm-v2-a-r2",
    "run-20260803-gm-v2-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-v2-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-any-token.sql",
   "permalink": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttft-any-token/",
   "json": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttft-any-token.json",
   "markdown": "https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttft-any-token.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first streamed token of any kind (reasoning included) in the default operating mode.\" Claim strix.gemma4.interactive2.c1.ttft-any-token (282 ms), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.gemma4.interactive2.c1.ttft-any-token/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.gemma4.longctx.c1.control-success",
   "headline": "gemma-4-26B-A4B-it Q4_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — unanswerable-control honesty: 75.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0), the measured unanswerable-control honesty was 75.0 % of requests (share of all control requests). The measurement was taken under the frozen long-context-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "75.0",
   "unit": "% of requests",
   "metric": "unanswerable-control honesty",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0)",
   "units_measured": 1,
   "statement": "Share of unanswerable-control requests answered with an explicit admission that the answer is absent. Every failed control returned an EMPTY answer — the reasoning pass consumed the token budget before any text was produced; no run fabricated a code.",
   "scope": "long-context-v1@2026-08-03",
   "aggregation": "share of all control requests",
   "limitations": "Three repeated runs on one unit. Default operating mode of the model (reasoning emitted); disabling reasoning via the chat template was not tested for this model family. The failed controls are empty answers under the 1024-token budget in the default reasoning mode, not fabrications — the distinction is visible per-request in quality.jsonl.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-gm-lc-a",
    "run-20260803-gm-lc-a-r2",
    "run-20260803-gm-lc-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-lc-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-lc-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-lc-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/control-success.sql",
   "permalink": "https://agmind.ai/claims/strix.gemma4.longctx.c1.control-success/",
   "json": "https://agmind.ai/claims/strix.gemma4.longctx.c1.control-success.json",
   "markdown": "https://agmind.ai/claims/strix.gemma4.longctx.c1.control-success.md",
   "cite": "AGmind Systems Lab. \"Share of unanswerable-control requests answered with an explicit admission that the answer is absent. Every failed control returned an EMPTY answer — the reasoning pass consumed the token budget before any text was produced; no run fabricated a code.\" Claim strix.gemma4.longctx.c1.control-success (75.0 % of requests), evidence level lab_repeated, scope long-context-v1@2026-08-03. https://agmind.ai/claims/strix.gemma4.longctx.c1.control-success/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.gemma4.longctx.c1.needle-success",
   "headline": "gemma-4-26B-A4B-it Q4_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — needle retrieval success: 95.8 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0), the measured needle retrieval success was 95.8 % of requests (share of all needle requests). The measurement was taken under the frozen long-context-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "95.8",
   "unit": "% of requests",
   "metric": "needle retrieval success",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0)",
   "units_measured": 1,
   "statement": "Share of requests where the model retrieved the embedded fact across a 2k-32k-token ladder, EN and RU, default operating mode.",
   "scope": "long-context-v1@2026-08-03",
   "aggregation": "share of all needle requests",
   "limitations": "Three repeated runs on one unit. Default operating mode of the model (reasoning emitted); disabling reasoning via the chat template was not tested for this model family.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-gm-lc-a",
    "run-20260803-gm-lc-a-r2",
    "run-20260803-gm-lc-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-lc-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-lc-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-lc-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/needle-success.sql",
   "permalink": "https://agmind.ai/claims/strix.gemma4.longctx.c1.needle-success/",
   "json": "https://agmind.ai/claims/strix.gemma4.longctx.c1.needle-success.json",
   "markdown": "https://agmind.ai/claims/strix.gemma4.longctx.c1.needle-success.md",
   "cite": "AGmind Systems Lab. \"Share of requests where the model retrieved the embedded fact across a 2k-32k-token ladder, EN and RU, default operating mode.\" Claim strix.gemma4.longctx.c1.needle-success (95.8 % of requests), evidence level lab_repeated, scope long-context-v1@2026-08-03. https://agmind.ai/claims/strix.gemma4.longctx.c1.needle-success/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.gemma4.structured.c1.task-success",
   "headline": "gemma-4-26B-A4B-it Q4_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — strict-JSON task success: 93.8 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0), the measured strict-JSON task success was 93.8 % of requests (share of all issued requests). The measurement was taken under the frozen structured-agent-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "93.8",
   "unit": "% of requests",
   "metric": "strict-JSON task success",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/gemma-4-26B-A4B-it-GGUF @ bb4531cda34d (Q4_0)",
   "units_measured": 1,
   "statement": "Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, default operating mode.",
   "scope": "structured-agent-v1@2026-08-03",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit. Default operating mode of the model (reasoning emitted); disabling reasoning via the chat template was not tested for this model family. Failures were unparseable outputs on RU generation tasks, not wrong values.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-gm-sa-a",
    "run-20260803-gm-sa-a-r2",
    "run-20260803-gm-sa-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-sa-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-sa-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-gm-sa-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/json-task-success.sql",
   "permalink": "https://agmind.ai/claims/strix.gemma4.structured.c1.task-success/",
   "json": "https://agmind.ai/claims/strix.gemma4.structured.c1.task-success.json",
   "markdown": "https://agmind.ai/claims/strix.gemma4.structured.c1.task-success.md",
   "cite": "AGmind Systems Lab. \"Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, default operating mode.\" Claim strix.gemma4.structured.c1.task-success (93.8 % of requests), evidence level lab_repeated, scope structured-agent-v1@2026-08-03. https://agmind.ai/claims/strix.gemma4.structured.c1.task-success/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.docsession.c1.ttft-q1-32k",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first token, first question over a 32k document: 33940 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first token, first question over a 32k document was 33940 ms (median over valid requests for this item). The measurement was taken under the frozen doc-session-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "33940",
   "unit": "ms",
   "metric": "time to first token, first question over a 32k document",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median time to first token for the FIRST question over a 32k-token English document — the prefill every fresh document pays, prompt cache enabled.",
   "scope": "doc-session-v1@2026-08-03",
   "aggregation": "median over valid requests for this item",
   "limitations": "Three repeated runs on one unit, reasoning disabled, single stream. Cache benefit requires a byte-identical document prefix in the same server session.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-ds-cache-a",
    "run-20260803-ds-cache-a-r2",
    "run-20260803-ds-cache-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-ds-en-32k-q1.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q1-32k/",
   "json": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q1-32k.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q1-32k.md",
   "cite": "AGmind Systems Lab. \"Median time to first token for the FIRST question over a 32k-token English document — the prefill every fresh document pays, prompt cache enabled.\" Claim strix.qwen36.docsession.c1.ttft-q1-32k (33940 ms), evidence level lab_repeated, scope doc-session-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q1-32k/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.docsession.c1.ttft-q2-32k-cache",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first token, second question with prompt cache on: 860 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first token, second question with prompt cache on was 860 ms (median over valid requests for this item). The measurement was taken under the frozen doc-session-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "860",
   "unit": "ms",
   "metric": "time to first token, second question with prompt cache on",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median time to first token for the SECOND question over the same 32k-token document with the server prompt cache enabled.",
   "scope": "doc-session-v1@2026-08-03",
   "aggregation": "median over valid requests for this item",
   "limitations": "Three repeated runs on one unit, reasoning disabled, single stream. Cache benefit requires a byte-identical document prefix in the same server session.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-ds-cache-a",
    "run-20260803-ds-cache-a-r2",
    "run-20260803-ds-cache-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-ds-en-32k-q2.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-cache/",
   "json": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-cache.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-cache.md",
   "cite": "AGmind Systems Lab. \"Median time to first token for the SECOND question over the same 32k-token document with the server prompt cache enabled.\" Claim strix.qwen36.docsession.c1.ttft-q2-32k-cache (860 ms), evidence level lab_repeated, scope doc-session-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-cache/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.docsession.c1.ttft-q2-32k-nocache",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first token, second question with prompt cache off: 33728 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first token, second question with prompt cache off was 33728 ms (median over valid requests for this item). The measurement was taken under the frozen doc-session-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "33728",
   "unit": "ms",
   "metric": "time to first token, second question with prompt cache off",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median time to first token for the second question over the same 32k-token document with the prompt cache disabled — the full prefill is paid again.",
   "scope": "doc-session-v1@2026-08-03",
   "aggregation": "median over valid requests for this item",
   "limitations": "Three repeated runs on one unit, reasoning disabled, single stream.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-ds-nocache-a",
    "run-20260803-ds-nocache-a-r2",
    "run-20260803-ds-nocache-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-nocache-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-nocache-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-nocache-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-ds-en-32k-q2.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-nocache/",
   "json": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-nocache.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-nocache.md",
   "cite": "AGmind Systems Lab. \"Median time to first token for the second question over the same 32k-token document with the prompt cache disabled — the full prefill is paid again.\" Claim strix.qwen36.docsession.c1.ttft-q2-32k-nocache (33728 ms), evidence level lab_repeated, scope doc-session-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-32k-nocache/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.docsession.c1.ttft-q2-8k-cache",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first token, second question over an 8k document, cache on: 660 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first token, second question over an 8k document, cache on was 660 ms (median over valid requests for this item). The measurement was taken under the frozen doc-session-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "660",
   "unit": "ms",
   "metric": "time to first token, second question over an 8k document, cache on",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median time to first token for the second question over the same 8k-token document with the server prompt cache enabled.",
   "scope": "doc-session-v1@2026-08-03",
   "aggregation": "median over valid requests for this item",
   "limitations": "Three repeated runs on one unit, reasoning disabled, single stream. Cache benefit requires a byte-identical document prefix in the same server session.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-ds-cache-a",
    "run-20260803-ds-cache-a-r2",
    "run-20260803-ds-cache-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-ds-cache-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-ds-en-8k-q2.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache/",
   "json": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache.md",
   "cite": "AGmind Systems Lab. \"Median time to first token for the second question over the same 8k-token document with the server prompt cache enabled.\" Claim strix.qwen36.docsession.c1.ttft-q2-8k-cache (660 ms), evidence level lab_repeated, scope doc-session-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.docsession.c1.ttft-q2-8k-cache/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.endurance.c4.completion-180m",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — request completion over three hours, 4 concurrent requests: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured request completion over three hours, 4 concurrent requests was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen endurance-30m-v1@2026-08-03 workload; evidence level lab_single_run, 2 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "request completion over three hours",
   "concurrency": 4,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Share of requests that completed with a non-empty answer over the full 3-hour sustained pass, both units pooled.",
   "scope": "endurance-30m-v1@2026-08-03",
   "aggregation": "share of all issued requests",
   "limitations": "One 3-hour pass per unit — the pattern reproduced on both commercially identical units, but repeats within a unit are pending. Closed-loop concurrency 4, reasoning disabled, this corpus — not a general thermals verdict. Ambient not instrumented; die temperatures documented in the run notes.",
   "evidence_level": "lab_single_run",
   "status": "active",
   "run_ids": [
    "run-20260803-end-c4-a",
    "run-20260803-end-c4-b"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-end-c4-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-end-c4-b"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/completion-share.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.endurance.c4.completion-180m/",
   "json": "https://agmind.ai/claims/strix.qwen36.endurance.c4.completion-180m.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.endurance.c4.completion-180m.md",
   "cite": "AGmind Systems Lab. \"Share of requests that completed with a non-empty answer over the full 3-hour sustained pass, both units pooled.\" Claim strix.qwen36.endurance.c4.completion-180m (100.0 % of requests), evidence level lab_single_run, scope endurance-30m-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.endurance.c4.completion-180m/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.endurance.c4.itl-drift-180m",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — decode-pace drift over three hours, 4 concurrent requests: 1.6 %",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured decode-pace drift over three hours, 4 concurrent requests was 1.6 % (last window vs first window, pooled over both units). The measurement was taken under the frozen endurance-30m-v1@2026-08-03 workload; evidence level lab_single_run, 2 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "1.6",
   "unit": "%",
   "metric": "decode-pace drift over three hours",
   "concurrency": 4,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Relative change of the median inter-token latency between the first five minutes and minutes 175-180 of a continuous 3-hour closed-loop pass, across both units.",
   "scope": "endurance-30m-v1@2026-08-03",
   "aggregation": "last window vs first window, pooled over both units",
   "limitations": "One 3-hour pass per unit — the pattern reproduced on both commercially identical units, but repeats within a unit are pending. Closed-loop concurrency 4, reasoning disabled, this corpus — not a general thermals verdict. Ambient not instrumented; die temperatures documented in the run notes.",
   "evidence_level": "lab_single_run",
   "status": "active",
   "run_ids": [
    "run-20260803-end-c4-a",
    "run-20260803-end-c4-b"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-end-c4-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-end-c4-b"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/itl-drift-180m.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-drift-180m/",
   "json": "https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-drift-180m.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-drift-180m.md",
   "cite": "AGmind Systems Lab. \"Relative change of the median inter-token latency between the first five minutes and minutes 175-180 of a continuous 3-hour closed-loop pass, across both units.\" Claim strix.qwen36.endurance.c4.itl-drift-180m (1.6 %), evidence level lab_single_run, scope endurance-30m-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-drift-180m/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.endurance.c4.itl-median",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — inter-token latency, 4 concurrent requests: 30.0 ms/token",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured inter-token latency, 4 concurrent requests was 30.0 ms/token (median over valid requests). The measurement was taken under the frozen endurance-30m-v1@2026-08-03 workload; evidence level lab_single_run, 2 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "30.0",
   "unit": "ms/token",
   "metric": "inter-token latency",
   "concurrency": 4,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Median inter-token latency sustained over the full 3-hour pass at closed-loop concurrency 4, both units pooled.",
   "scope": "endurance-30m-v1@2026-08-03",
   "aggregation": "median over valid requests",
   "limitations": "One 3-hour pass per unit — the pattern reproduced on both commercially identical units, but repeats within a unit are pending. Closed-loop concurrency 4, reasoning disabled, this corpus — not a general thermals verdict. Ambient not instrumented; die temperatures documented in the run notes.",
   "evidence_level": "lab_single_run",
   "status": "active",
   "run_ids": [
    "run-20260803-end-c4-a",
    "run-20260803-end-c4-b"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-end-c4-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-end-c4-b"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/itl-median.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-median/",
   "json": "https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-median.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-median.md",
   "cite": "AGmind Systems Lab. \"Median inter-token latency sustained over the full 3-hour pass at closed-loop concurrency 4, both units pooled.\" Claim strix.qwen36.endurance.c4.itl-median (30.0 ms/token), evidence level lab_single_run, scope endurance-30m-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.endurance.c4.itl-median/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive.c1.answerless-1k",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — answerless (empty) responses at a 1k budget: 75.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured answerless (empty) responses at a 1k budget was 75.0 % of requests (share of all issued requests). The measurement was taken under the frozen interactive-assistant-v1@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "75.0",
   "unit": "% of requests",
   "metric": "answerless (empty) responses at a 1k budget",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "At a 1024-token budget with reasoning enabled, this share of requests returned HTTP 200 with an empty answer: the reasoning pass consumed the entire budget.",
   "scope": "interactive-assistant-v1@2026-08-02",
   "aggregation": "share of all issued requests",
   "limitations": "Three runs of 16 requests each (48 total) on one unit — repeated, not unit-replicated. Specific to a 1024-token budget and this corpus. Failed requests are counted in the denominator, as required.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260802-gonogo-a2",
    "run-20260802-budget1k-a-r2",
    "run-20260802-budget1k-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-gonogo-a2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget1k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget1k-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/answerless-rate-1k.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive.c1.answerless-1k/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive.c1.answerless-1k.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive.c1.answerless-1k.md",
   "cite": "AGmind Systems Lab. \"At a 1024-token budget with reasoning enabled, this share of requests returned HTTP 200 with an empty answer: the reasoning pass consumed the entire budget.\" Claim strix.qwen36.interactive.c1.answerless-1k (75.0 % of requests), evidence level lab_repeated, scope interactive-assistant-v1@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive.c1.answerless-1k/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive.c1.ttfa-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first answer token: 220 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first answer token was 220 ms (median over valid requests). The measurement was taken under the frozen interactive-assistant-v1@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "220",
   "unit": "ms",
   "metric": "time to first answer token",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "With the reasoning block disabled, the first answer token arrives a median of this many milliseconds after the request — the answer starts immediately instead of after the reasoning pass.",
   "scope": "interactive-assistant-v1@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit. Disabling reasoning is an operating setting; answer quality under it was gated only for format, language and repetition. Concurrency 1 only.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260802-nothink-a",
    "run-20260802-nothink-a-r2",
    "run-20260802-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-nothink.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-nothink.md",
   "cite": "AGmind Systems Lab. \"With the reasoning block disabled, the first answer token arrives a median of this many milliseconds after the request — the answer starts immediately instead of after the reasoning pass.\" Claim strix.qwen36.interactive.c1.ttfa-nothink (220 ms), evidence level lab_repeated, scope interactive-assistant-v1@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive.c1.ttfa-thinking",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first answer token with reasoning on: 23453 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first answer token with reasoning on was 23453 ms (median over valid requests). The measurement was taken under the frozen interactive-assistant-v1@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "23453",
   "unit": "ms",
   "metric": "time to first answer token with reasoning on",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "With the reasoning block enabled, the first token of the actual answer arrives a median of this many milliseconds after the request — two orders of magnitude later than the conventional time-to-first-token figure.",
   "scope": "interactive-assistant-v1@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit — diagnostic depth, not a cross-unit qualification. Token budget 4096; a smaller budget changes the outcome entirely. Concurrency 1 only.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260802-budget4k-a",
    "run-20260802-budget4k-a-r2",
    "run-20260802-budget4k-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget4k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget4k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget4k-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-thinking-1k.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-thinking/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-thinking.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-thinking.md",
   "cite": "AGmind Systems Lab. \"With the reasoning block enabled, the first token of the actual answer arrives a median of this many milliseconds after the request — two orders of magnitude later than the conventional time-to-first-token figure.\" Claim strix.qwen36.interactive.c1.ttfa-thinking (23453 ms), evidence level lab_repeated, scope interactive-assistant-v1@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive.c1.ttfa-thinking/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive.c1.ttft-any-token",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to the first token of any output: 221 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to the first token of any output was 221 ms (median over all valid requests of three cells). The measurement was taken under the frozen interactive-assistant-v1@2026-08-02 workload; evidence level lab_repeated, 9 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "221",
   "unit": "ms",
   "metric": "time to the first token of any output",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "On the tested Strix Halo configuration, the first streamed token of any kind arrives in a median of this many milliseconds — but on this model that first token is reasoning, not the answer.",
   "scope": "interactive-assistant-v1@2026-08-02",
   "aggregation": "median over all valid requests of three cells",
   "limitations": "Nine runs across three cells on one unit — repeated, not unit-replicated. Measured at concurrency 1 on a quiesced node with a desktop session present. Says nothing about when a usable answer starts; see the answer-token claim.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260802-nothink-a",
    "run-20260802-nothink-a-r2",
    "run-20260802-nothink-a-r3",
    "run-20260802-budget4k-a",
    "run-20260802-budget4k-a-r2",
    "run-20260802-budget4k-a-r3",
    "run-20260802-gonogo-a2",
    "run-20260802-budget1k-a-r2",
    "run-20260802-budget1k-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-nothink-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget4k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget4k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget4k-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-gonogo-a2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget1k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-budget1k-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-any-token.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttft-any-token/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttft-any-token.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive.c1.ttft-any-token.md",
   "cite": "AGmind Systems Lab. \"On the tested Strix Halo configuration, the first streamed token of any kind arrives in a median of this many milliseconds — but on this model that first token is reasoning, not the answer.\" Claim strix.qwen36.interactive.c1.ttft-any-token (221 ms), evidence level lab_repeated, scope interactive-assistant-v1@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive.c1.ttft-any-token/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c1.answerless-1k",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — answerless (empty) responses at a 1k budget: 58.3 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured answerless (empty) responses at a 1k budget was 58.3 % of requests (share of all issued requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_unit_replicated, 6 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "58.3",
   "unit": "% of requests",
   "metric": "answerless (empty) responses at a 1k budget",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "At a 1024-token budget with reasoning enabled, this share of everyday-task requests returned HTTP 200 with an empty answer: the reasoning pass consumed the entire budget.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "share of all issued requests",
   "limitations": "Three runs of 16 requests each (48 total) across two commercially identical units — repeated, not unit-replicated. Specific to a 1024-token budget and this corpus. Failed requests are counted in the denominator, as required.",
   "evidence_level": "lab_unit_replicated",
   "status": "active",
   "run_ids": [
    "run-20260802-v2-budget1k-a",
    "run-20260802-v2-budget1k-a-r2",
    "run-20260802-v2-budget1k-a-r3",
    "run-20260803-v2-budget1k-b",
    "run-20260803-v2-budget1k-b-r2",
    "run-20260803-v2-budget1k-b-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget1k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget1k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget1k-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget1k-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget1k-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget1k-b-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/answerless-rate-1k.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.answerless-1k/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.answerless-1k.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.answerless-1k.md",
   "cite": "AGmind Systems Lab. \"At a 1024-token budget with reasoning enabled, this share of everyday-task requests returned HTTP 200 with an empty answer: the reasoning pass consumed the entire budget.\" Claim strix.qwen36.interactive2.c1.answerless-1k (58.3 % of requests), evidence level lab_unit_replicated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c1.answerless-1k/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c1.completion-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — request completion: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured request completion was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_unit_replicated, 6 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "request completion",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Share of everyday requests that completed with a non-empty answer with the reasoning block disabled.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "share of all issued requests",
   "limitations": "Six repeated runs across two commercially identical units.",
   "evidence_level": "lab_unit_replicated",
   "status": "active",
   "run_ids": [
    "run-20260802-v2-nothink-a",
    "run-20260802-v2-nothink-a-r2",
    "run-20260802-v2-nothink-a-r3",
    "run-20260803-v2-nothink-b",
    "run-20260803-v2-nothink-b-r2",
    "run-20260803-v2-nothink-b-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/completion-share.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.completion-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.completion-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.completion-nothink.md",
   "cite": "AGmind Systems Lab. \"Share of everyday requests that completed with a non-empty answer with the reasoning block disabled.\" Claim strix.qwen36.interactive2.c1.completion-nothink (100.0 % of requests), evidence level lab_unit_replicated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c1.completion-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c1.ttfa-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first answer token: 210 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first answer token was 210 ms (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_unit_replicated, 6 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "210",
   "unit": "ms",
   "metric": "time to first answer token",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Median client-side time to the first token of the answer with the reasoning block disabled, human-task corpus.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs across two commercially identical units.",
   "evidence_level": "lab_unit_replicated",
   "status": "active",
   "run_ids": [
    "run-20260802-v2-nothink-a",
    "run-20260802-v2-nothink-a-r2",
    "run-20260802-v2-nothink-a-r3",
    "run-20260803-v2-nothink-b",
    "run-20260803-v2-nothink-b-r2",
    "run-20260803-v2-nothink-b-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-nothink.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-nothink.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first token of the answer with the reasoning block disabled, human-task corpus.\" Claim strix.qwen36.interactive2.c1.ttfa-nothink (210 ms), evidence level lab_unit_replicated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c1.ttfa-thinking",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first answer token with reasoning on: 20191 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first answer token with reasoning on was 20191 ms (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_unit_replicated, 6 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "20191",
   "unit": "ms",
   "metric": "time to first answer token with reasoning on",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Median client-side time to the first token of the actual answer with reasoning enabled at a 4096-token budget, human-task corpus.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs across two commercially identical units — diagnostic depth, not a cross-unit qualification.",
   "evidence_level": "lab_unit_replicated",
   "status": "active",
   "run_ids": [
    "run-20260802-v2-budget4k-a",
    "run-20260802-v2-budget4k-a-r2",
    "run-20260802-v2-budget4k-a-r3",
    "run-20260803-v2-budget4k-b",
    "run-20260803-v2-budget4k-b-r2",
    "run-20260803-v2-budget4k-b-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget4k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget4k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget4k-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget4k-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget4k-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget4k-b-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-thinking-1k.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-thinking/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-thinking.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-thinking.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first token of the actual answer with reasoning enabled at a 4096-token budget, human-task corpus.\" Claim strix.qwen36.interactive2.c1.ttfa-thinking (20191 ms), evidence level lab_unit_replicated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttfa-thinking/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c1.ttft-any-token",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to the first token of any output: 212 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to the first token of any output was 212 ms (median over all valid requests of three cells). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_unit_replicated, 18 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "212",
   "unit": "ms",
   "metric": "time to the first token of any output",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Median client-side time to the first streamed token of any kind, across all three operating settings on the human-task corpus.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over all valid requests of three cells",
   "limitations": "Nine runs across three cells across two commercially identical units — repeated, not unit-replicated.",
   "evidence_level": "lab_unit_replicated",
   "status": "active",
   "run_ids": [
    "run-20260802-v2-nothink-a",
    "run-20260802-v2-nothink-a-r2",
    "run-20260802-v2-nothink-a-r3",
    "run-20260802-v2-budget4k-a",
    "run-20260802-v2-budget4k-a-r2",
    "run-20260802-v2-budget4k-a-r3",
    "run-20260802-v2-budget1k-a",
    "run-20260802-v2-budget1k-a-r2",
    "run-20260802-v2-budget1k-a-r3",
    "run-20260803-v2-nothink-b",
    "run-20260803-v2-nothink-b-r2",
    "run-20260803-v2-nothink-b-r3",
    "run-20260803-v2-budget1k-b",
    "run-20260803-v2-budget1k-b-r2",
    "run-20260803-v2-budget1k-b-r3",
    "run-20260803-v2-budget4k-b",
    "run-20260803-v2-budget4k-b-r2",
    "run-20260803-v2-budget4k-b-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget4k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget4k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget4k-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget1k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget1k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-budget1k-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-nothink-b-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget1k-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget1k-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget1k-b-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget4k-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget4k-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-budget4k-b-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-any-token.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttft-any-token/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttft-any-token.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttft-any-token.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first streamed token of any kind, across all three operating settings on the human-task corpus.\" Claim strix.qwen36.interactive2.c1.ttft-any-token (212 ms), evidence level lab_unit_replicated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c1.ttft-any-token/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c4.completion-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — request completion, 4 concurrent requests: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured request completion, 4 concurrent requests was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "request completion",
   "concurrency": 4,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Share of requests that completed with a non-empty answer at closed-loop concurrency 4, reasoning disabled.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit. Closed-loop concurrency 4 — not an arrival-rate/SLO capacity claim.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-v2-c4-nothink-a",
    "run-20260803-v2-c4-nothink-a-r2",
    "run-20260803-v2-c4-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c4-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c4-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c4-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/completion-share.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c4.completion-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c4.completion-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c4.completion-nothink.md",
   "cite": "AGmind Systems Lab. \"Share of requests that completed with a non-empty answer at closed-loop concurrency 4, reasoning disabled.\" Claim strix.qwen36.interactive2.c4.completion-nothink (100.0 % of requests), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c4.completion-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c4.ttfa-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first answer token, 4 concurrent requests: 338 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first answer token, 4 concurrent requests was 338 ms (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "338",
   "unit": "ms",
   "metric": "time to first answer token",
   "concurrency": 4,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median client-side time to the first answer token with the reasoning block disabled, while 4 closed-loop streams run concurrently on the unit (each request measured from its own client).",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit. Closed-loop concurrency 4 — not an arrival-rate/SLO capacity claim. Context split across server slots (8192 per slot).",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-v2-c4-nothink-a",
    "run-20260803-v2-c4-nothink-a-r2",
    "run-20260803-v2-c4-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c4-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c4-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c4-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-nothink.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c4.ttfa-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c4.ttfa-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c4.ttfa-nothink.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first answer token with the reasoning block disabled, while 4 closed-loop streams run concurrently on the unit (each request measured from its own client).\" Claim strix.qwen36.interactive2.c4.ttfa-nothink (338 ms), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c4.ttfa-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c8.completion-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — request completion, 8 concurrent requests: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured request completion, 8 concurrent requests was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "request completion",
   "concurrency": 8,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Share of requests that completed with a non-empty answer at closed-loop concurrency 8, reasoning disabled.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit. Closed-loop concurrency 8 — not an arrival-rate/SLO capacity claim.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-v2-c8-nothink-a",
    "run-20260803-v2-c8-nothink-a-r2",
    "run-20260803-v2-c8-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c8-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c8-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c8-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/completion-share.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c8.completion-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c8.completion-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c8.completion-nothink.md",
   "cite": "AGmind Systems Lab. \"Share of requests that completed with a non-empty answer at closed-loop concurrency 8, reasoning disabled.\" Claim strix.qwen36.interactive2.c8.completion-nothink (100.0 % of requests), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c8.completion-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.interactive2.c8.ttfa-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first answer token, 8 concurrent requests: 865 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first answer token, 8 concurrent requests was 865 ms (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "865",
   "unit": "ms",
   "metric": "time to first answer token",
   "concurrency": 8,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median client-side time to the first answer token with the reasoning block disabled, while 8 closed-loop streams run concurrently on the unit (each request measured from its own client).",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit. Closed-loop concurrency 8 — not an arrival-rate/SLO capacity claim. Context split across server slots (8192 per slot).",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-v2-c8-nothink-a",
    "run-20260803-v2-c8-nothink-a-r2",
    "run-20260803-v2-c8-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c8-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c8-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-v2-c8-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-nothink.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.interactive2.c8.ttfa-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.interactive2.c8.ttfa-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.interactive2.c8.ttfa-nothink.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first answer token with the reasoning block disabled, while 8 closed-loop streams run concurrently on the unit (each request measured from its own client).\" Claim strix.qwen36.interactive2.c8.ttfa-nothink (865 ms), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36.interactive2.c8.ttfa-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.longctx.c1.control-success",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — unanswerable-control honesty: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured unanswerable-control honesty was 100.0 % of requests (share of all control requests). The measurement was taken under the frozen long-context-v1@2026-08-03 workload; evidence level lab_unit_replicated, 6 valid runs across 2 physical units. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "unanswerable-control honesty",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 2,
   "statement": "Share of unanswerable-control requests where the model admitted the answer was absent from the document instead of fabricating one, with a distractor fact present.",
   "scope": "long-context-v1@2026-08-03",
   "aggregation": "share of all control requests",
   "limitations": "Six runs across two commercially identical units; the three 2026-08-05 runs re-measured the same cells under the corrected decline gate (errata 2026-08-05). Grounding check plus marker matching — a model could still decline in wording the marker list does not contain; the gate is declared brittle in the methodology. Synthetic controls with one distractor per document — not a general hallucination rate.",
   "evidence_level": "lab_unit_replicated",
   "status": "active",
   "run_ids": [
    "run-20260803-lc-nothink-a",
    "run-20260803-lc-nothink-a-r2",
    "run-20260803-lc-nothink-a-r3",
    "run-20260805-lc-nothink-b",
    "run-20260805-lc-nothink-b-r2",
    "run-20260805-lc-nothink-b-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r3",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260805-lc-nothink-b",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260805-lc-nothink-b-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260805-lc-nothink-b-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/control-success.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.longctx.c1.control-success/",
   "json": "https://agmind.ai/claims/strix.qwen36.longctx.c1.control-success.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.longctx.c1.control-success.md",
   "cite": "AGmind Systems Lab. \"Share of unanswerable-control requests where the model admitted the answer was absent from the document instead of fabricating one, with a distractor fact present.\" Claim strix.qwen36.longctx.c1.control-success (100.0 % of requests), evidence level lab_unit_replicated, scope long-context-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.longctx.c1.control-success/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.longctx.c1.needle-success",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — needle retrieval success: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured needle retrieval success was 100.0 % of requests (share of all needle requests). The measurement was taken under the frozen long-context-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "needle retrieval success",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Share of requests where the model retrieved a synthetic fact embedded at mid-document, across a 2k-32k-token context ladder in EN and RU.",
   "scope": "long-context-v1@2026-08-03",
   "aggregation": "share of all needle requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled. Synthetic needle retrieval at a fixed ~50% depth — not comprehension; tested to 32k tokens, no claim beyond.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-lc-nothink-a",
    "run-20260803-lc-nothink-a-r2",
    "run-20260803-lc-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/needle-success.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.longctx.c1.needle-success/",
   "json": "https://agmind.ai/claims/strix.qwen36.longctx.c1.needle-success.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.longctx.c1.needle-success.md",
   "cite": "AGmind Systems Lab. \"Share of requests where the model retrieved a synthetic fact embedded at mid-document, across a 2k-32k-token context ladder in EN and RU.\" Claim strix.qwen36.longctx.c1.needle-success (100.0 % of requests), evidence level lab_repeated, scope long-context-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.longctx.c1.needle-success/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.longctx.c1.ttft-2k-en",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first token at a 2k-token document: 1907 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first token at a 2k-token document was 1907 ms (median over valid 2k-band requests). The measurement was taken under the frozen long-context-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "1907",
   "unit": "ms",
   "metric": "time to first token at a 2k-token document",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median client-side time to the first token when the prompt carries a 2k-token English document.",
   "scope": "long-context-v1@2026-08-03",
   "aggregation": "median over valid 2k-band requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled. Synthetic needle retrieval at a fixed ~50% depth — not comprehension; tested to 32k tokens, no claim beyond.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-lc-nothink-a",
    "run-20260803-lc-nothink-a-r2",
    "run-20260803-lc-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-ctx2k-en.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-2k-en/",
   "json": "https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-2k-en.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-2k-en.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first token when the prompt carries a 2k-token English document.\" Claim strix.qwen36.longctx.c1.ttft-2k-en (1907 ms), evidence level lab_repeated, scope long-context-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-2k-en/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.longctx.c1.ttft-32k-en",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — time to first token at a 32k-token document: 33965 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first token at a 32k-token document was 33965 ms (median over valid 32k-band requests). The measurement was taken under the frozen long-context-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "33965",
   "unit": "ms",
   "metric": "time to first token at a 32k-token document",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median client-side time to the first token when the prompt carries a 32k-token English document.",
   "scope": "long-context-v1@2026-08-03",
   "aggregation": "median over valid 32k-band requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled. Synthetic needle retrieval at a fixed ~50% depth — not comprehension; tested to 32k tokens, no claim beyond.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-lc-nothink-a",
    "run-20260803-lc-nothink-a-r2",
    "run-20260803-lc-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-lc-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttft-ctx32k-en.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-32k-en/",
   "json": "https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-32k-en.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-32k-en.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first token when the prompt carries a 32k-token English document.\" Claim strix.qwen36.longctx.c1.ttft-32k-en (33965 ms), evidence level lab_repeated, scope long-context-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.longctx.c1.ttft-32k-en/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.structured.c1.e2e-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — end-to-end time to a complete answer: 844 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured end-to-end time to a complete answer was 844 ms (median over valid requests). The measurement was taken under the frozen structured-agent-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "844",
   "unit": "ms",
   "metric": "end-to-end time to a complete answer",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median end-to-end time to a complete strict-JSON answer with reasoning disabled.",
   "scope": "structured-agent-v1@2026-08-03",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit. Deterministic short tasks with closed label sets — no tool execution, no sandbox. Concurrency 1 only.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-sa-nothink-a",
    "run-20260803-sa-nothink-a-r2",
    "run-20260803-sa-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/e2e-thinking-vs-not.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-nothink.md",
   "cite": "AGmind Systems Lab. \"Median end-to-end time to a complete strict-JSON answer with reasoning disabled.\" Claim strix.qwen36.structured.c1.e2e-nothink (844 ms), evidence level lab_repeated, scope structured-agent-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.structured.c1.e2e-think4k",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — end-to-end time with reasoning on: 14677 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured end-to-end time with reasoning on was 14677 ms (median over valid requests). The measurement was taken under the frozen structured-agent-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "14677",
   "unit": "ms",
   "metric": "end-to-end time with reasoning on",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median end-to-end time to a complete strict-JSON answer with reasoning enabled at a 4096-token budget.",
   "scope": "structured-agent-v1@2026-08-03",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit. Deterministic short tasks with closed label sets — no tool execution, no sandbox. Concurrency 1 only.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-sa-think4k-a",
    "run-20260803-sa-think4k-a-r2",
    "run-20260803-sa-think4k-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-think4k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-think4k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-think4k-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/e2e-thinking-vs-not.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-think4k/",
   "json": "https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-think4k.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-think4k.md",
   "cite": "AGmind Systems Lab. \"Median end-to-end time to a complete strict-JSON answer with reasoning enabled at a 4096-token budget.\" Claim strix.qwen36.structured.c1.e2e-think4k (14677 ms), evidence level lab_repeated, scope structured-agent-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.structured.c1.e2e-think4k/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.structured.c1.task-success-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — strict-JSON task success: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured strict-JSON task success was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen structured-agent-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "strict-JSON task success",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, reasoning disabled.",
   "scope": "structured-agent-v1@2026-08-03",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit. Deterministic short tasks with closed label sets — no tool execution, no sandbox. Concurrency 1 only.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-sa-nothink-a",
    "run-20260803-sa-nothink-a-r2",
    "run-20260803-sa-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/json-task-success.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-nothink.md",
   "cite": "AGmind Systems Lab. \"Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, reasoning disabled.\" Claim strix.qwen36.structured.c1.task-success-nothink (100.0 % of requests), evidence level lab_repeated, scope structured-agent-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36.structured.c1.task-success-think4k",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — strict-JSON task success with reasoning on: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured strict-JSON task success with reasoning on was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen structured-agent-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "strict-JSON task success with reasoning on",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, reasoning enabled at a 4096-token budget.",
   "scope": "structured-agent-v1@2026-08-03",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit. Deterministic short tasks with closed label sets — no tool execution, no sandbox. Concurrency 1 only.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-sa-think4k-a",
    "run-20260803-sa-think4k-a-r2",
    "run-20260803-sa-think4k-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-think4k-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-think4k-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-sa-think4k-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/json-task-success.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-think4k/",
   "json": "https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-think4k.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-think4k.md",
   "cite": "AGmind Systems Lab. \"Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, reasoning enabled at a 4096-token budget.\" Claim strix.qwen36.structured.c1.task-success-think4k (100.0 % of requests), evidence level lab_repeated, scope structured-agent-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36.structured.c1.task-success-think4k/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36q4.rocm.c1.itl-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp ROCm) — inter-token latency: 18.7 ms/token",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, ROCm backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured inter-token latency was 18.7 ms/token (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "18.7",
   "unit": "ms/token",
   "metric": "inter-token latency",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, ROCm backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median inter-token latency (client-side decode-speed proxy) for Q4_K_M on the ROCm backend.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-rocm-v2-nothink-a",
    "run-20260803-rocm-v2-nothink-a-r2",
    "run-20260803-rocm-v2-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-v2-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-v2-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-v2-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/itl-median.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.itl-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.itl-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.itl-nothink.md",
   "cite": "AGmind Systems Lab. \"Median inter-token latency (client-side decode-speed proxy) for Q4_K_M on the ROCm backend.\" Claim strix.qwen36q4.rocm.c1.itl-nothink (18.7 ms/token), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36q4.rocm.c1.itl-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36q4.rocm.c1.task-success",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp ROCm) — strict-JSON task success: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, ROCm backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured strict-JSON task success was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen structured-agent-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "strict-JSON task success",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, ROCm backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, ROCm backend.",
   "scope": "structured-agent-v1@2026-08-03",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-rocm-sa-nothink-a",
    "run-20260803-rocm-sa-nothink-a-r2",
    "run-20260803-rocm-sa-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-sa-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-sa-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-sa-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/json-task-success.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.task-success/",
   "json": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.task-success.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.task-success.md",
   "cite": "AGmind Systems Lab. \"Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, ROCm backend.\" Claim strix.qwen36q4.rocm.c1.task-success (100.0 % of requests), evidence level lab_repeated, scope structured-agent-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36q4.rocm.c1.task-success/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36q4.rocm.c1.ttfa-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp ROCm) — time to first answer token: 215 ms",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, ROCm backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured time to first answer token was 215 ms (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "215",
   "unit": "ms",
   "metric": "time to first answer token",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, ROCm backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median client-side time to the first answer token with the reasoning block disabled, ROCm backend.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-rocm-v2-nothink-a",
    "run-20260803-rocm-v2-nothink-a-r2",
    "run-20260803-rocm-v2-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-v2-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-v2-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-rocm-v2-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/ttfa-nothink.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.ttfa-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.ttfa-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36q4.rocm.c1.ttfa-nothink.md",
   "cite": "AGmind Systems Lab. \"Median client-side time to the first answer token with the reasoning block disabled, ROCm backend.\" Claim strix.qwen36q4.rocm.c1.ttfa-nothink (215 ms), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36q4.rocm.c1.ttfa-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36q4.vulkan.c1.itl-nothink",
   "headline": "Qwen3.6-35B-A3B Q4_K_M on Ryzen AI Max+ 395 (llama.cpp Vulkan) — inter-token latency: 15.9 ms/token",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M), the measured inter-token latency was 15.9 ms/token (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "15.9",
   "unit": "ms/token",
   "metric": "inter-token latency",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q4_K_M)",
   "units_measured": 1,
   "statement": "Median inter-token latency (client-side decode-speed proxy) for Q4_K_M on the Vulkan backend.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260802-v2-nothink-a",
    "run-20260802-v2-nothink-a-r2",
    "run-20260802-v2-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260802-v2-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/itl-median.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36q4.vulkan.c1.itl-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36q4.vulkan.c1.itl-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36q4.vulkan.c1.itl-nothink.md",
   "cite": "AGmind Systems Lab. \"Median inter-token latency (client-side decode-speed proxy) for Q4_K_M on the Vulkan backend.\" Claim strix.qwen36q4.vulkan.c1.itl-nothink (15.9 ms/token), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36q4.vulkan.c1.itl-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36q8.vulkan.c1.itl-nothink",
   "headline": "Qwen3.6-35B-A3B Q8_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — inter-token latency: 18.7 ms/token",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q8_0), the measured inter-token latency was 18.7 ms/token (median over valid requests). The measurement was taken under the frozen interactive-assistant-v2@2026-08-02 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "18.7",
   "unit": "ms/token",
   "metric": "inter-token latency",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q8_0)",
   "units_measured": 1,
   "statement": "Median inter-token latency (client-side decode-speed proxy) for Q8_0 on the Vulkan backend.",
   "scope": "interactive-assistant-v2@2026-08-02",
   "aggregation": "median over valid requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-q8-v2-nothink-a",
    "run-20260803-q8-v2-nothink-a-r2",
    "run-20260803-q8-v2-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-q8-v2-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-q8-v2-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-q8-v2-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/itl-median.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.itl-nothink/",
   "json": "https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.itl-nothink.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.itl-nothink.md",
   "cite": "AGmind Systems Lab. \"Median inter-token latency (client-side decode-speed proxy) for Q8_0 on the Vulkan backend.\" Claim strix.qwen36q8.vulkan.c1.itl-nothink (18.7 ms/token), evidence level lab_repeated, scope interactive-assistant-v2@2026-08-02. https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.itl-nothink/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  },
  {
   "id": "strix.qwen36q8.vulkan.c1.task-success",
   "headline": "Qwen3.6-35B-A3B Q8_0 on Ryzen AI Max+ 395 (llama.cpp Vulkan) — strict-JSON task success: 100.0 % of requests",
   "answer": "On a Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified running llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14) with ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q8_0), the measured strict-JSON task success was 100.0 % of requests (share of all issued requests). The measurement was taken under the frozen structured-agent-v1@2026-08-03 workload; evidence level lab_repeated, 3 valid runs across 1 physical unit. The value is re-derived from the raw run records on every CI build.",
   "value": "100.0",
   "unit": "% of requests",
   "metric": "strict-JSON task success",
   "concurrency": 1,
   "system": "Beelink GTR9 Pro — AMD Ryzen AI Max+ 395, Radeon 8060S (gfx1151), 128 GB LPDDR5X-8000 unified",
   "runtime": "llama.cpp (server, Vulkan backend), b9049 (server_fingerprint b9049-2496f9c14)",
   "model": "ggml-org/Qwen3.6-35B-A3B-GGUF @ baec3ebee244 (Q8_0)",
   "units_measured": 1,
   "statement": "Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, Q8_0 quantization.",
   "scope": "structured-agent-v1@2026-08-03",
   "aggregation": "share of all issued requests",
   "limitations": "Three repeated runs on one unit, reasoning disabled.",
   "evidence_level": "lab_repeated",
   "status": "active",
   "run_ids": [
    "run-20260803-q8-sa-nothink-a",
    "run-20260803-q8-sa-nothink-a-r2",
    "run-20260803-q8-sa-nothink-a-r3"
   ],
   "runs": [
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-q8-sa-nothink-a",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-q8-sa-nothink-a-r2",
    "https://github.com/botAGI/agmind-lab/tree/main/runs/run-20260803-q8-sa-nothink-a-r3"
   ],
   "derivation_sql": "https://github.com/botAGI/agmind-lab/blob/main/catalog/claims/sql/json-task-success.sql",
   "permalink": "https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.task-success/",
   "json": "https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.task-success.json",
   "markdown": "https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.task-success.md",
   "cite": "AGmind Systems Lab. \"Share of strict-JSON automation requests whose output parsed and matched the ground truth per key, Q8_0 quantization.\" Claim strix.qwen36q8.vulkan.c1.task-success (100.0 % of requests), evidence level lab_repeated, scope structured-agent-v1@2026-08-03. https://agmind.ai/claims/strix.qwen36q8.vulkan.c1.task-success/",
   "license": "https://creativecommons.org/licenses/by/4.0/",
   "corrections": "https://agmind.ai/errata/"
  }
 ]
}