{
  "publisher": "The Fire Dev LLC",
  "published": "2026-10-06",
  "notice": "Historical observations from the publisher's own lab. Different models, prompts, runtimes and load are not directly comparable. These are not performance guarantees or current service status.",
  "records": [
    {"date":"2026-10-02","hardware":"Two NVIDIA GB10 machines, tensor parallel 2","model":"GLM-5.3-Flash uncensored, locally converted NVFP4","runtime":"vLLM knapcio TP2","workload":"idle-gated perf-loop prose","tokens_per_second":38.4,"source":"Publisher cluster inference operations benchmark record"},
    {"date":"2026-10-02","hardware":"Two NVIDIA GB10 machines, tensor parallel 2","model":"GLM-5.3-Flash uncensored, locally converted NVFP4","runtime":"vLLM knapcio TP2","workload":"idle-gated perf-loop code","tokens_per_second":68.3,"source":"Publisher cluster inference operations benchmark record"},
    {"date":"2026-10-02","hardware":"Two NVIDIA GB10 machines, tensor parallel 2","model":"GLM-5.3-Flash uncensored, locally converted NVFP4","runtime":"vLLM knapcio TP2","workload":"six concurrent streams, aggregate output","tokens_per_second":91,"source":"Publisher cluster inference operations benchmark record"},
    {"date":"2026-10-05","hardware":"Two NVIDIA GB10 machines, tensor parallel 2","model":"GLM-5.3-Flash uncensored","runtime":"vLLM","workload":"loaded cluster, temperature 0, thinking off, one benchmark stream alongside live traffic","tokens_per_second":6.2,"decode_first_token_ms":1689,"source":"Publisher infer-bench before table, local proxy route; 8/8 successful requests across four prompts"},
    {"date":"2026-10-05","hardware":"RTX 4080 16 GiB workstation","model":"Qwen3.5-2B Q4_K_M","runtime":"llama.cpp CUDA","workload":"about 100 words, 128 token cap, temperature 0, thinking off, concurrency 1","tokens_per_second":274.4,"decode_first_token_ms":15,"source":"Publisher infer-bench after table, direct fast-model route; 20/20 successful requests across four prompts"},
    {"date":"2026-10-05","hardware":"Intel Core Ultra 9 386H laptop iGPU, 30 GiB RAM","model":"Qwen3.5-2B Q4_K_M","runtime":"llama.cpp Vulkan","workload":"about 100 words, 128 token cap, temperature 0, thinking off, concurrency 1","tokens_per_second":24.6,"decode_first_token_ms":138,"source":"Publisher infer-bench before table, on-machine route; 12/12 successful requests across four prompts"},
    {"date":"2026-10-05","hardware":"Apple M4 Pro Mac mini, 24 GiB unified memory","model":"Qwen3.5-9B Q4_K_M","runtime":"llama.cpp b11081 Metal","workload":"about 100 words, 128 token cap, temperature 0, thinking off, concurrency 1; shared build load","tokens_per_second":8.9,"decode_first_token_ms":413,"source":"Publisher infer-bench before table, on-machine route; 12/12 successful requests across four prompts"},
    {"date":"2026-10-06","hardware":"Linux x64 workstation, 16 CPU threads, 124 GiB RAM; CPU model runtime","model":"Qwen3-4B-Instruct-2507 Q4_K_M","runtime":"llama.cpp b11430, context 65536","workload":"Write about 100 words explaining why a local model is private. Temperature 0, token cap 160; two completed runs","decode_tokens_per_second":[11.969932302310292,12.287438193098504],"prompt_tokens":23,"completion_tokens":[125,114],"source":"Publisher isolated Harness install; llama-server response timings"}
  ],
  "method_october_5": "Median of 2 to 5 runs per prompt after an unmeasured warmup. Decode rate = (completion tokens - 1) / (last token time - first token time). First-token latency is request to first streamed token. The cluster was serving other sessions; no idle benchmark was run on October 5. Successful-request totals cover short, decode, cold 1.6k and warm 1.6k prompts, not decode alone.",
  "cluster_identity_check": {"date":"2026-10-06","served_model":"glm-5.3-flash-uncensored","max_model_len":262144,"generation_benchmarked":false,"note":"Read-only metadata check. Engine had active traffic; historical idle benchmark is dated October 2."}
}
