{"rows":[{"id":"cmsnp1rko00jco001mvymwmsb","modelRevision":"main","promptTokens":0,"outputTokens":64,"contextLength":1024,"prefillTokens":null,"batchSize":112,"ttftMs":120,"tokSOut":6001.4,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:30.360Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode. DATA-PARALLEL across 4 cards; aggregate of 4 independent replicas. batch=112, concurrency=112, 100 input rows, 3 waves, max_tokens=64, ctx=1024, KV=q8, greedy; TTFT not comparable. Per-card median MODEL tok/s: d0:1495.3, d1:1504.4, d2:1500.1, d3:1501.6 (sum 6001.4); wall-clock aggregate 3889.4 tok/s (per-card wall d0:969.6, d1:975.7, d2:973.2, d3:970.9). MODEL tok/s is sum of per-card decode rates excluding inter-wave gap; wall includes all overhead. Do not compare DP aggregate to single-stream. engine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201 x4 quant: MQ4R = MagnumQuant 4-bit Redline (graded mixed-tier: MQ4 attn/router/shared + graded MQ4 routed experts, Redline retained-PM4 dispatch), 4.16 bpw effective (18700048128 bytes *8 / 35.95B params) model file: qwen3.6-35b-a3b.mq4r sha256 4685c140c46b1a6f prompt: uniform short-serve jsonl md5 3d22208dcf539818ff2e8341f53474ee, 64 max_new method: median of 3 waves, fresh process per wave, temperature 0, bit-exact outputs (sha ec47dcfb9b0c54e0) evidence: sealed case a3b-mq4r-dp4-b112-gfx1201-r9700-4x, corpus manifest manifest-sha12-8f3c2e1a4b9d not measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"hipfire-batch-generate --model ~/.hipfire/models/qwen3.6-35b-a3b.mq4r --batch 112 --max-seq 1024 --max-new 64 --devices 0,1,2,3 --data-parallel --kv q8 --fresh-process-per-wave","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.6-35B-A3B","displayName":"Qwen3.6-35B-A3B","family":"Qwen","params":36,"activeParams":3,"isMoE":true,"baseModel":null},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":4,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4R","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":1,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp232a00kao001r4wh9920","modelRevision":"main","promptTokens":0,"outputTokens":64,"contextLength":1024,"prefillTokens":null,"batchSize":100,"ttftMs":120,"tokSOut":5993.3,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:45.251Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode. DATA-PARALLEL across 4 cards; aggregate of 4 independent replicas. batch=100, concurrency=100, 100 input rows, 3 waves, max_tokens=64, ctx=1024, KV=q8, greedy; TTFT not comparable. Per-card median MODEL tok/s: d0:1493.8, d1:1500.3, d2:1499.3, d3:1499.9 (sum 5993.3); wall-clock aggregate 3885.8 tok/s (per-card wall d0:970.3, d1:973.6, d2:972.5, d3:969.4). MODEL tok/s is sum of per-card decode rates excluding inter-wave gap; wall includes all overhead. Do not compare DP aggregate to single-stream. engine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201 x4 quant: MQ4R = MagnumQuant 4-bit Redline (graded mixed-tier: MQ4 attn/router/shared + graded MQ4 routed experts, Redline retained-PM4 dispatch), 4.16 bpw effective (18700048128 bytes *8 / 35.95B params) model file: qwen3.6-35b-a3b.mq4r sha256 4685c140c46b1a6f prompt: uniform short-serve jsonl md5 3d22208dcf539818ff2e8341f53474ee, 64 max_new method: median of 3 waves, fresh process per wave, temperature 0, bit-exact outputs (sha ec47dcfb9b0c54e0) evidence: sealed case a3b-mq4r-dp4-b100-gfx1201-r9700-4x, corpus manifest manifest-sha12-8f3c2e1a4b9d not measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"hipfire-batch-generate --model ~/.hipfire/models/qwen3.6-35b-a3b.mq4r --batch 100 --max-seq 1024 --max-new 64 --devices 0,1,2,3 --data-parallel --kv q8 --fresh-process-per-wave","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.6-35B-A3B","displayName":"Qwen3.6-35B-A3B","family":"Qwen","params":36,"activeParams":3,"isMoE":true,"baseModel":null},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":4,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4R","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":2,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp1spu00jho001f7bpvkac","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":644.1,"tokSOut":4816.072,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:31.843Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+2e0c4d39f29f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: batched-prefill-coalesced build; pre-coalesce fixed-wave baseline B128 is 1523.1 tok/s (case lfm2.5-230m-gfx1201-b128r128-fixed).\nevidence: sealed case lfm2.5-230m-gfx1201-b128r128-batched-prefill-fixed, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 4096 --port 11543 --home-root /tmp/lmx-lfm230-b128-coalesce-home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm230-batched-prefill-coalesce-b128 --out /home/kaden/localmaxxing-benchmarks/validation/lfm230-local-b128r128-batched-prefill-coalesce.json --cli target/release/hipfire --daemon target/release/examples/daemon --device 0 --timeout 600 --thinking off","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+2e0c4d39f29f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":3,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp1tv300jko001nfbt14ek","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":580.5,"tokSOut":4347.308,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:33.327Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 1 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+48802a005625, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 1 runs, fresh process, temperature 0 (greedy)\ncaveats: R9700 230M opt-B128 single-card figure device d2 4347.3 tok/s; per-card d0/d1/d2/d3 = 4291/4310/4347/4309 tok/s (all similar).\nevidence: sealed case lfm25-230m-opt-b128-fixed-wave-20260809-gfx1201-r9700-d2, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 4096 --port 18822 --home-root /tmp/lmx_opt230_d2_home --log-dir /tmp/lmx_opt230_d2_logs --out /tmp/lmx_opt230_d2_b128.json --device 2","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+48802a005625","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":4,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp247500kfo001s4pg03pu","modelRevision":"main","promptTokens":0,"outputTokens":64,"contextLength":1024,"prefillTokens":null,"batchSize":64,"ttftMs":120,"tokSOut":4207.7,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:46.721Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode. DATA-PARALLEL across 4 cards; aggregate of 4 independent replicas. batch=64, concurrency=64, 100 input rows, 3 waves, max_tokens=64, ctx=1024, KV=q8, greedy; TTFT not comparable. Per-card median MODEL tok/s: d0:1046.6, d1:1053.0, d2:1052.2, d3:1055.9 (sum 4207.7); wall-clock aggregate 3058.6 tok/s (per-card wall d0:761.8, d1:766.0, d2:765.4, d3:765.4). MODEL tok/s is sum of per-card decode rates excluding inter-wave gap; wall includes all overhead. Do not compare DP aggregate to single-stream. engine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201 x4 quant: MQ4R = MagnumQuant 4-bit Redline (graded mixed-tier: MQ4 attn/router/shared + graded MQ4 routed experts, Redline retained-PM4 dispatch), 4.16 bpw effective (18700048128 bytes *8 / 35.95B params) model file: qwen3.6-35b-a3b.mq4r sha256 4685c140c46b1a6f prompt: uniform short-serve jsonl md5 3d22208dcf539818ff2e8341f53474ee, 64 max_new method: median of 3 waves, fresh process per wave, temperature 0, bit-exact outputs (sha ec47dcfb9b0c54e0) evidence: sealed case a3b-mq4r-dp4-b64-gfx1201-r9700-4x, corpus manifest manifest-sha12-8f3c2e1a4b9d not measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"hipfire-batch-generate --model ~/.hipfire/models/qwen3.6-35b-a3b.mq4r --batch 64 --max-seq 1024 --max-new 64 --devices 0,1,2,3 --data-parallel --kv q8 --fresh-process-per-wave","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.6-35B-A3B","displayName":"Qwen3.6-35B-A3B","family":"Qwen","params":36,"activeParams":3,"isMoE":true,"baseModel":null},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":4,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4R","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":5,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp1v0f00jno001z3qt1vjn","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":659.7,"tokSOut":3669.847,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:34.815Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+2e0c4d39f29f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: batched-prefill-coalesced build; pre-coalesce baseline B128 is 1237.7 tok/s (case lfm2.5-350m-gfx1201-b128r128-fixed).\nevidence: sealed case lfm2.5-350m-gfx1201-b128r128-batched-prefill-fixed, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 4096 --port 11541 --home-root /tmp/lmx-lfm350-b128-coalesce-home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm350-batched-prefill-coalesce-b128 --out /home/kaden/localmaxxing-benchmarks/validation/lfm350-local-b128r128-batched-prefill-coalesce.json --cli target/release/hipfire --daemon target/release/examples/daemon --device 0 --timeout 600 --thinking off","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+2e0c4d39f29f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":6,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp25c400kko001bfn1g0ch","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":604.6,"tokSOut":3283.31,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:48.196Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 1 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+48802a005625, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 1 runs, fresh process, temperature 0 (greedy)\ncaveats: R9700 350M opt-B128 single-card figure device d0 3283.3 tok/s; per-card spread d0/d1/d2/d3 = 3283/3189/3247/1609 tok/s (d3 outlier 1609). Do not use 4-card median without noting outlier.\nevidence: sealed case lfm25-350m-opt-b128-fixed-wave-20260809-gfx1201-r9700-d0, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 4096 --port 18830 --home-root /tmp/lmx_opt350_d0_home --log-dir /tmp/lmx_opt350_d0_logs --out /tmp/lmx_opt350_d0_b128.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+48802a005625","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":7,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp26gr00kno0012jshzkpm","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":256,"ttftMs":1260,"tokSOut":1876.917,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:49.660Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=256, concurrency=256, 256 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+2e0c4d39f29f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: batched-prefill-coalesced build; pre-coalesce baseline B256 is 1258.5 tok/s (case lfm2.5-350m-gfx1201-b256r256-fixed).\nevidence: sealed case lfm2.5-350m-gfx1201-b256r256-batched-prefill-fixed, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 256 --requests 256 --runs 3 --max-tokens 128 --max-seq 4096 --port 11542 --home-root /tmp/lmx-lfm350-b256-coalesce-home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm350-batched-prefill-coalesce-b256 --out /home/kaden/localmaxxing-benchmarks/validation/lfm350-local-b256r256-batched-prefill-coalesce.json --cli target/release/hipfire --daemon target/release/examples/daemon --device 0 --timeout 600 --thinking off","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+2e0c4d39f29f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":8,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp27le00kqo0015qt1pjdp","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":8,"ttftMs":80.25,"tokSOut":1704.611,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:51.122Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=8, concurrency=8, 16 requests/run, 3 runs, mode=continuous, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: FINAL continuous numbers shown; earlier continuous_refill sweep measured lower (B4R8 737.9 vs 1044.8, B8R16 1023.2 vs 1704.6).\nevidence: sealed case lfm2.5-230m-gfx1201-b8r16-FINAL-continuous, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 8 --requests 16 --runs 3 --max-tokens 128 --max-seq 4096 --port 11605 --home-root /home/kaden/lmx-final-runs/work/FINAL/lfm2.5-230m/continuous/home-b8r16 --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-230m/continuous/b8r16 --out /home/kaden/lmx-final-runs/lfm230-FINAL-b8r16-gfx1201-continuous.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":9,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp28q500kuo001070dwcev","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":256,"ttftMs":2537.9,"tokSOut":1597.704,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:52.589Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=256, concurrency=256, 256 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+439233d92959, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm25-230m-b256-fixed-wave-20260809-gfx1201-r9700, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 256 --requests 256 --runs 3 --max-tokens 128 --max-seq 2048 --port 18816 --home-root /tmp/lmx_cb_home6 --log-dir /tmp/lmx_cb_logs6 --out /tmp/lmx_cb230_b256.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+439233d92959","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":10,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp29vb00kxo00147qwfmi0","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":2067.8,"tokSOut":1585.588,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:54.071Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+439233d92959, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: pre-coalesce fixed-wave B128 1585.6 tok/s; opt batched-prefill-coalesced B128 per-card up to 4347 tok/s (case lfm25-230m-opt-b128...-d2).\nevidence: sealed case lfm25-230m-b128-fixed-wave-20260809-gfx1201-r9700, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 2048 --port 18812 --home-root /tmp/lmx_cb_home2 --log-dir /tmp/lmx_cb_logs2 --out /tmp/lmx_cb230_b128.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+439233d92959","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":11,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2b0g00l1o001ekfx6umo","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":2231.9,"tokSOut":1523.117,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:55.551Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b973024a973f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: pre-coalesce baseline (no batched prefill coalescing); coalesced build achieves 4816.1 tok/s (case lfm2.5-230m-gfx1201-b128r128-batched-prefill-fixed).\nevidence: sealed case lfm2.5-230m-gfx1201-b128r128-fixed, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 4096 --port 11530 --home-root /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b128r128/home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b128r128/logs --out /home/kaden/localmaxxing-benchmarks/validation/lfm230-local-b128r128.json --cli /home/kaden/hipfire-lmx-lfm/target/release/hipfire --daemon /home/kaden/hipfire-lmx-lfm/target/release/examples/daemon --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b973024a973f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":12,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp1w7300jqo001njoxttpa","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":8,"ttftMs":87.2,"tokSOut":1470.768,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:36.352Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=8, concurrency=8, 16 requests/run, 3 runs, mode=continuous_refill, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1100\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-350m-gfx1100-hip0-b8r16, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file /home/kaden/hipfire-lmx-final-26cf8e148/benchmarks/prompts/sweep/longform.txt --batch-size 8 --requests 16 --runs 3 --max-tokens 128 --max-seq 4096 --port 11554 --home-root /home/kaden/lmx-final-runs/homes/FINAL-gfx1100-350-batch-b8r16-home --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-350m/gfx1100-batch-b8r16 --out /home/kaden/lmx-final-runs/lfm2.5-350m-gfx1100-hip0-batch-b8r16.json --device 0 --timeout 600","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 7900 XTX","gpuCount":1,"vramGb":24,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 7900 XTX","hardwareGroupKey":"DISCRETE_GPU:rx 7900 xtx","rank":13,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2c5d00l5o0011vx4xq75","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":64,"ttftMs":1055.7,"tokSOut":1456.962,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:57.026Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=64, concurrency=64, 64 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+1bec7dc3d1b5, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-230m-gfx1201-b64r64-fixed, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 64 --requests 64 --runs 3 --max-tokens 128 --max-seq 4096 --port 11529 --home-root /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b64r64/home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b64r64/logs --out /home/kaden/localmaxxing-benchmarks/validation/lfm230-local-b64r64.json --cli /home/kaden/hipfire-lmx-lfm/target/release/hipfire --daemon /home/kaden/hipfire-lmx-lfm/target/release/examples/daemon --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+1bec7dc3d1b5","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":14,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2d9x00l9o001zb1nsuww","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":32,"ttftMs":407.2,"tokSOut":1376.722,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:58.485Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=32, concurrency=32, 64 requests/run, 3 runs, mode=continuous_refill, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+5f11453f892f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-230m-gfx1201-b32r64-continuous, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 32 --requests 64 --runs 3 --max-tokens 128 --max-seq 4096 --port 11525 --home-root /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b32r64/home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b32r64/logs --out /home/kaden/localmaxxing-benchmarks/validation/lfm230-local-b32r64.json --cli /home/kaden/hipfire-lmx-lfm/target/release/hipfire --daemon /home/kaden/hipfire-lmx-lfm/target/release/examples/daemon --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+5f11453f892f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":15,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2eef00lco001ryud4uiu","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":256,"ttftMs":2844.1,"tokSOut":1291.866,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:59.943Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=256, concurrency=256, 256 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+439233d92959, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm25-350m-b256-fixed-wave-20260809-gfx1201-r9700, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 256 --requests 256 --runs 3 --max-tokens 128 --max-seq 2048 --port 18815 --home-root /tmp/lmx_cb_home5 --log-dir /tmp/lmx_cb_logs5 --out /tmp/lmx_cb350_b256.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+439233d92959","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":16,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2fk000lfo001s09siyds","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":2246,"tokSOut":1278.939,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:01.440Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+439233d92959, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: pre-coalesce fixed-wave B128 1278.9 tok/s; opt batched-prefill-coalesced B128 per-card d0 3283.3 tok/s (spread includes d3 outlier 1609).\nevidence: sealed case lfm25-350m-b128-fixed-wave-20260809-gfx1201-r9700, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 2048 --port 18814 --home-root /tmp/lmx_cb_home4 --log-dir /tmp/lmx_cb_logs4 --out /tmp/lmx_cb350_b128.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+439233d92959","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":17,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2gp300ljo001b1fr238f","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":256,"ttftMs":4927.9,"tokSOut":1258.467,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:02.919Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=256, concurrency=256, 256 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+2ceddeb4f38d, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: pre-coalesce baseline; coalesced build achieves 1876.9 tok/s (case lfm2.5-350m-gfx1201-b256r256-batched-prefill-fixed).\nevidence: sealed case lfm2.5-350m-gfx1201-b256r256-fixed, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 256 --requests 256 --runs 3 --max-tokens 128 --max-seq 4096 --port 11533 --home-root /home/kaden/localmaxxing-benchmarks/raw/lfm350-local-b256r256/home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm350-local-b256r256/logs --out /home/kaden/localmaxxing-benchmarks/validation/lfm350-local-b256r256.json --cli /home/kaden/hipfire-lmx-lfm/target/release/hipfire --daemon /home/kaden/hipfire-lmx-lfm/target/release/examples/daemon --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+2ceddeb4f38d","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":18,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2hup00lmo001z7xdlo2y","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":16,"ttftMs":302.9,"tokSOut":1246.758,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:04.417Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=16, concurrency=16, 32 requests/run, 3 runs, mode=continuous_refill, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+5f11453f892f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-230m-gfx1201-b16r32-continuous, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 16 --requests 32 --runs 3 --max-tokens 128 --max-seq 4096 --port 11524 --home-root /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b16r32/home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm230-local-b16r32/logs --out /home/kaden/localmaxxing-benchmarks/validation/lfm230-local-b16r32.json --cli /home/kaden/hipfire-lmx-lfm/target/release/hipfire --daemon /home/kaden/hipfire-lmx-lfm/target/release/examples/daemon --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+5f11453f892f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":19,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2j0300lqo001fnw6dwfy","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":128,"ttftMs":2456.3,"tokSOut":1237.749,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:05.908Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=128, concurrency=128, 128 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+e734d6c8561d, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: pre-coalesce baseline; coalesced build achieves 3669.8 tok/s (case lfm2.5-350m-gfx1201-b128r128-batched-prefill-fixed).\nevidence: sealed case lfm2.5-350m-gfx1201-b128r128-fixed, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 128 --requests 128 --runs 3 --max-tokens 128 --max-seq 4096 --port 11532 --home-root /home/kaden/localmaxxing-benchmarks/raw/lfm350-local-b128r128/home --log-dir /home/kaden/localmaxxing-benchmarks/raw/lfm350-local-b128r128/logs --out /home/kaden/localmaxxing-benchmarks/validation/lfm350-local-b128r128.json --cli /home/kaden/hipfire-lmx-lfm/target/release/hipfire --daemon /home/kaden/hipfire-lmx-lfm/target/release/examples/daemon --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+e734d6c8561d","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":20,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp1xcx00jto001lp2i7emk","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":8,"ttftMs":114.4,"tokSOut":1181.636,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:37.857Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=8, concurrency=8, 16 requests/run, 3 runs, mode=continuous_refill, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1151\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-350m-gfx1151-b8r16-refill, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 8 --requests 16 --runs 3 --max-tokens 128 --max-seq 4096 --port 18164 --home-root /home/kaden/lmx-final-runs/homes/hipx-1-batch-b8r16 --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm350/gfx1151-continuous-b8r16 --out /home/kaden/lmx-final-runs/hipx-1-lfm350-b8r16-b8r16.json --device 1 --cli target/release/hipfire --daemon target/release/examples/daemon","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"UNIFIED","gpuName":null,"gpuCount":1,"vramGb":null,"chipVendor":"AMD","chipFamily":"Ryzen AI Max","chipVariant":"Ryzen AI Max 395+","unifiedMemoryGb":103,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Ryzen AI Max 395","hardwareGroupKey":"UNIFIED:ryzen ai max 395","rank":21,"reactionCounts":{},"myEmoji":null},{"id":"cmsnsxbrr02fbo0012bo2krxv","modelRevision":"main","promptTokens":38,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":64,"ttftMs":628.5,"tokSOut":1115.74,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T22:27:01.719Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode. batch=64, concurrency=64, 64 requests/run, 3 runs, continuous batching, max_tokens=128, KV=q8, greedy; TTFT measured under load.\nsingle R9700 (1 of 4 in the box; other cards ran unrelated jobs and were unused).\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201\nquant: MQ4R = MagnumQuant 4-bit Redline (MQ4 attention/router/shared + graded MQ4 routed experts; '.mq4r' also selects Redline retained-PM4 dispatch), 4.16 bpw effective.\nWORKLOAD (read before comparing): prompt merge_sort_thinking_off.txt md5 253c7ac50857fe6d0e10fb0d2c5e35c0, only 38 prompt tokens, 128 generated. The contextLength 4096 on this row is the max_seq ALLOCATION, not the prompt size. The Arc Pro B70 row on this page (1139.8 tok/s, concurrency 64) reports contextLength 16384, but that is likewise its --max-model-len; its own notes state diverse 512-token prompts. Its per-request prefill work is therefore ~13x ours, so this row is NOT like-for-like and should not be read as matching or beating it on equal workload.\nmethod: median of 3 runs; samples 1113.68 / 1115.74 / 1131.49 tok/s; per-lane 17.43 tok/s; median latency 7182.37 ms. All outputs byte-identical and coherent.\ncapacity: 64 lanes is the ceiling at max_seq 4096 on 32 GB; B96 and B128 both fail with 'hipMalloc: out of memory / continuous batch allocation failed'.\nevidence: 20260810T221534Z-a3b-mq4r-concurrency-ladder\nnot measured: peak VRAM.","engineFlags":{"commandSnippet":"python3 scripts/lmx_continuous_batch.py --model qwen3.6-35b-a3b.mq4r --prompt-file benchmarks/prompts/merge_sort_thinking_off.txt --batch-size 64 --requests 64 --runs 3 --max-tokens 128 --max-seq 4096 --device 1","tensorParallel":1,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.6-35B-A3B","displayName":"Qwen3.6-35B-A3B","family":"Qwen","params":36,"activeParams":3,"isMoE":true,"baseModel":null},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4R","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":22,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2k4r00lto001bsasg3ok","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":59.5,"tokSOut":1044.768,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:07.371Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=4, concurrency=4, 8 requests/run, 3 runs, mode=continuous, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: FINAL continuous numbers shown; earlier continuous_refill sweep measured lower (B4R8 737.9 vs 1044.8, B8R16 1023.2 vs 1704.6).\nevidence: sealed case lfm2.5-230m-gfx1201-b4r8-FINAL-continuous, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 8 --runs 3 --max-tokens 128 --max-seq 4096 --port 11604 --home-root /home/kaden/lmx-final-runs/work/FINAL/lfm2.5-230m/continuous/home-b4r8 --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-230m/continuous/b4r8 --out /home/kaden/lmx-final-runs/lfm230-FINAL-b4r8-gfx1201-continuous.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":23,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2l9o00lwo001th6itv48","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":63.4,"tokSOut":1016.965,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:08.844Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=4, concurrency=4, 4 requests/run, 3 runs, mode=continuous, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-230m-gfx1201-b4r4-FINAL-continuous, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 4 --runs 3 --max-tokens 128 --max-seq 4096 --port 11603 --home-root /home/kaden/lmx-final-runs/work/FINAL/lfm2.5-230m/continuous/home-b4r4 --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-230m/continuous/b4r4 --out /home/kaden/lmx-final-runs/lfm230-FINAL-b4r4-gfx1201-continuous.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":24,"reactionCounts":{},"myEmoji":null},{"id":"cmsnsxd8w02fgo001otjutgqo","modelRevision":"main","promptTokens":38,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":32,"ttftMs":519.15,"tokSOut":969.37,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T22:27:03.632Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode. batch=32, concurrency=32, 32 requests/run, 3 runs, continuous batching, max_tokens=128, KV=q8, greedy; TTFT measured under load.\nsingle R9700 (1 of 4 in the box; other cards ran unrelated jobs and were unused).\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201\nquant: MQ4R = MagnumQuant 4-bit Redline (MQ4 attention/router/shared + graded MQ4 routed experts; '.mq4r' also selects Redline retained-PM4 dispatch), 4.16 bpw effective.\nWORKLOAD (read before comparing): prompt merge_sort_thinking_off.txt md5 253c7ac50857fe6d0e10fb0d2c5e35c0, only 38 prompt tokens, 128 generated. The contextLength 4096 on this row is the max_seq ALLOCATION, not the prompt size. The Arc Pro B70 row on this page (1139.8 tok/s, concurrency 64) reports contextLength 16384, but that is likewise its --max-model-len; its own notes state diverse 512-token prompts. Its per-request prefill work is therefore ~13x ours, so this row is NOT like-for-like and should not be read as matching or beating it on equal workload.\nmethod: median of 3 runs; samples 968.61 / 969.37 / 971.82 tok/s; per-lane 30.29 tok/s; median latency 4171.65 ms. All outputs byte-identical and coherent.\ncapacity: 64 lanes is the ceiling at max_seq 4096 on 32 GB; B96 and B128 both fail with 'hipMalloc: out of memory / continuous batch allocation failed'.\nevidence: 20260810T221534Z-a3b-mq4r-concurrency-ladder\nnot measured: peak VRAM.","engineFlags":{"commandSnippet":"python3 scripts/lmx_continuous_batch.py --model qwen3.6-35b-a3b.mq4r --prompt-file benchmarks/prompts/merge_sort_thinking_off.txt --batch-size 32 --requests 32 --runs 3 --max-tokens 128 --max-seq 4096 --device 1","tensorParallel":1,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.6-35B-A3B","displayName":"Qwen3.6-35B-A3B","family":"Qwen","params":36,"activeParams":3,"isMoE":true,"baseModel":null},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4R","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":25,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2mfn00lzo001u3ft157u","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":71.35,"tokSOut":957.971,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:10.356Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=4, concurrency=4, 8 requests/run, 3 runs, mode=continuous_refill, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1151\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-230m-gfx1151-hip1-batch-B4R8-20260810T004005Z, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file /home/kaden/hipfire-lmx-final-26cf8e148/benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 8 --runs 3 --max-tokens 128 --max-seq 4096 --port 12004 --home-root /home/kaden/lmx-final-runs/homes/FINAL-hip1-batch-B4R8 --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-230m/gfx1151-batch/B4R8 --out /home/kaden/lmx-final-runs/lfm2.5-230m-gfx1151-hip1-batch-B4R8.json --device 1","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"UNIFIED","gpuName":null,"gpuCount":1,"vramGb":null,"chipVendor":"AMD","chipFamily":"Ryzen AI Max","chipVariant":"Ryzen AI Max 395+","unifiedMemoryGb":103,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Ryzen AI Max 395","hardwareGroupKey":"UNIFIED:ryzen ai max 395","rank":26,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2nkg00m2o0011pu80f03","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":72.75,"tokSOut":885.342,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:11.824Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=4, concurrency=4, 8 requests/run, 3 runs, mode=continuous_refill, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1100\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-350m-gfx1100-hip0-b4r8-honest, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file /home/kaden/hipfire-lmx-final-26cf8e148/benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 8 --runs 3 --max-tokens 128 --max-seq 4096 --port 11564 --home-root /home/kaden/lmx-final-runs/homes/FINAL-gfx1100-350-batch-b4r8-honest-home --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-350m/gfx1100-batch-b4r8-honest --out /home/kaden/lmx-final-runs/lfm2.5-350m-gfx1100-hip0-batch-b4r8-honest.json --device 0 --timeout 600","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 7900 XTX","gpuCount":1,"vramGb":24,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 7900 XTX","hardwareGroupKey":"DISCRETE_GPU:rx 7900 xtx","rank":27,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2opb00m5o001pk0qv7i3","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":80,"tokSOut":858.396,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:13.296Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=4, concurrency=4, 4 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1100\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-350m-gfx1100-hip0-b4r4-honest, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file /home/kaden/hipfire-lmx-final-26cf8e148/benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 4 --runs 3 --max-tokens 128 --max-seq 4096 --port 11563 --home-root /home/kaden/lmx-final-runs/homes/FINAL-gfx1100-350-batch-b4r4-honest-home --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-350m/gfx1100-batch-b4r4-honest --out /home/kaden/lmx-final-runs/lfm2.5-350m-gfx1100-hip0-batch-b4r4-honest.json --device 0 --timeout 600","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 7900 XTX","gpuCount":1,"vramGb":24,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 7900 XTX","hardwareGroupKey":"DISCRETE_GPU:rx 7900 xtx","rank":28,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2pua00m8o0011dkarrrm","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":84.1,"tokSOut":777.019,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:14.770Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=4, concurrency=4, 8 requests/run, 3 runs, mode=continuous_refill, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1151\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-350m-gfx1151-b4r8-refill, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 8 --runs 3 --max-tokens 128 --max-seq 4096 --port 18158 --home-root /home/kaden/lmx-final-runs/homes/hipx-1-batch-b4r8 --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm350/gfx1151-continuous-b4r8 --out /home/kaden/lmx-final-runs/hipx-1-lfm350-b4r8-b4r8.json --device 1 --cli target/release/hipfire --daemon target/release/examples/daemon","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"UNIFIED","gpuName":null,"gpuCount":1,"vramGb":null,"chipVendor":"AMD","chipFamily":"Ryzen AI Max","chipVariant":"Ryzen AI Max 395+","unifiedMemoryGb":103,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Ryzen AI Max 395","hardwareGroupKey":"UNIFIED:ryzen ai max 395","rank":29,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2qzh00mbo001e1rdy98b","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":84.2,"tokSOut":766.299,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:16.253Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=4, concurrency=4, 4 requests/run, 3 runs, mode=fixed_wave, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1151\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-350m-gfx1151-b4r4-fixed, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 4 --runs 3 --max-tokens 128 --max-seq 4096 --port 18154 --home-root /home/kaden/lmx-final-runs/homes/hipx-1-batch-b4r4-retry --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm350/gfx1151-continuous-b4r4-retry --out /home/kaden/lmx-final-runs/hipx-1-lfm350-b4r4-retry.json --device 1 --cli target/release/hipfire --daemon target/release/examples/daemon","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"UNIFIED","gpuName":null,"gpuCount":1,"vramGb":null,"chipVendor":"AMD","chipFamily":"Ryzen AI Max","chipVariant":"Ryzen AI Max 395+","unifiedMemoryGb":103,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Ryzen AI Max 395","hardwareGroupKey":"UNIFIED:ryzen ai max 395","rank":30,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp1yij00jwo001hu7bl10h","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":8,"ttftMs":240.75,"tokSOut":648.33,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:39.355Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=8, concurrency=8, 16 requests/run, 3 runs, mode=unrecorded, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1010\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: mode=unrecorded (harvest mode null; metadata.json on hipx contains no mode field; scripts/lmx_continuous_batch.py with batch/requests suggests continuous; not asserting).\nevidence: sealed case lfm230-gfx1010-B8R16, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file /home/kaden/hipfire-lmx-final-26cf8e148/benchmarks/prompts/sweep/longform.txt --batch-size 8 --requests 16 --runs 3 --max-tokens 128 --max-seq 4096 --port 12105 --home-root /home/kaden/lmx-final-runs/homes/batch-gfx1010-B8R16 --log-dir /home/kaden/lmx-final-runs/batch-logs/gfx1010/B8R16 --out /home/kaden/lmx-final-runs/batch-out/lfm230-gfx1010-B8R16.json --cli /home/kaden/hipfire-lmx-final-26cf8e148/target/release/hipfire --daemon /home/kaden/hipfire-lmx-final-26cf8e148/target/release/examples/daemon --device 2","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 5700 XT","gpuCount":1,"vramGb":8,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 5700 XT","hardwareGroupKey":"DISCRETE_GPU:rx 5700 xt","rank":31,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2s4l00mfo001ean545fb","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":77,"tokSOut":597.752,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:17.733Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 dispatch vs its own HIP control on same gfx1201 (hiptrx-r9700-x1). PM4 (auto) median 597.8 tok/s vs HIP median 493.9 tok/s, speedup 1.210x. Single-stream decode, batch 1, greedy, temperature 0. engine: hipfire 0.3.0+c1eab1f3290f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of campaign, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm230-mq4-redline-pm4-20260809-gfx1201-r9700-d3, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 77.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --expected-substring 'Deep learning' --device 3 --process-runs 3 --bench-runs 5 --daemon target/release/examples/daemon --cli target/release/hipfire --transport pm4 --context 128 --iterations 1000 --coherence-max-tokens 1024 --coherence-sampling greedy --work-dir /home/kaden/lmx-final-runs/work/3/lfm230/final-attempt1 --log-dir /home/kaden/lmx-final-runs/logs/3/lfm230/final-attempt1 --out /home/kaden/lmx-final-runs/lfm230-hiptrx-gfx1201-r9700-3-final-attempt1.json --env HOME=/home/kaden/lmx-final-runs/homes/hiptrx-3 --env HIPFIRE_HOME=/home/kaden/lmx-final-runs/homes/hiptrx-3/.hipfire","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+c1eab1f3290f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":32,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2t9n00mio001xsz4iszj","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":78,"tokSOut":597.189,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:19.212Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 dispatch vs its own HIP control on same gfx1201 (hiptrx-r9700-x1). PM4 (auto) median 597.2 tok/s vs HIP median 494.0 tok/s, speedup 1.209x. Single-stream decode, batch 1, greedy, temperature 0. engine: hipfire 0.3.0+c1eab1f3290f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of campaign, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm230-mq4-redline-pm4-20260809-gfx1201-r9700-d0, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 78.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --expected-substring 'Deep learning' --device 0 --process-runs 3 --bench-runs 5 --daemon /home/kaden/hipfire-lmx-final-26cf8e148/target/release/examples/daemon --cli /home/kaden/hipfire-lmx-final-26cf8e148/target/release/hipfire --transport pm4 --context 128 --iterations 1000 --coherence-max-tokens 1024 --coherence-sampling greedy --work-dir /home/kaden/lmx-final-runs/work/0/lfm230/attempt1 --log-dir /home/kaden/lmx-final-runs/logs/0/lfm230/attempt1 --env HOME=/home/kaden/lmx-final-runs/homes/hiptrx-0 --env HIPFIRE_HOME=/home/kaden/lmx-final-runs/homes/hiptrx-0/.hipfire --out /home/kaden/lmx-final-runs/lfm230-hiptrx-gfx1201-r9700-0-attempt1.json","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+c1eab1f3290f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":33,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2ue800mlo001oigh30re","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":76,"tokSOut":596.586,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:20.673Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 dispatch vs its own HIP control on same gfx1201 (hiptrx-r9700-x1). PM4 (auto) median 596.6 tok/s vs HIP median 492.2 tok/s, speedup 1.212x. Single-stream decode, batch 1, greedy, temperature 0. engine: hipfire 0.3.0+c1eab1f3290f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of campaign, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm230-mq4-redline-pm4-20260809-gfx1201-r9700-d1, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 76.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --expected-substring 'Deep learning' --device 1 --process-runs 3 --bench-runs 5 --daemon target/release/examples/daemon --cli target/release/hipfire --transport pm4 --context 128 --iterations 1000 --coherence-max-tokens 1024 --coherence-sampling greedy --work-dir /home/kaden/lmx-final-runs/work/1/lfm230/attempt1 --log-dir /home/kaden/lmx-final-runs/logs/1/lfm230/attempt1 --out /home/kaden/lmx-final-runs/lfm230-hiptrx-gfx1201-r9700-1-attempt1.json --env HOME=/home/kaden/lmx-final-runs/homes/hiptrx-1 --env HIPFIRE_HOME=/home/kaden/lmx-final-runs/homes/hiptrx-1/.hipfire","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+c1eab1f3290f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":34,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2viy00moo001fzhsnnnl","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":76,"tokSOut":594.099,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:22.139Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 dispatch vs its own HIP control on same gfx1201 (hiptrx-r9700-x1). PM4 (auto) median 594.1 tok/s vs HIP median 489.5 tok/s, speedup 1.214x. Single-stream decode, batch 1, greedy, temperature 0. engine: hipfire 0.3.0+c1eab1f3290f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of campaign, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm230-mq4-redline-pm4-20260809-gfx1201-r9700-d2, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 76.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --expected-substring 'Deep learning' --device 2 --process-runs 3 --bench-runs 5 --daemon /home/kaden/hipfire-lmx-final-26cf8e148/target/release/examples/daemon --cli /home/kaden/hipfire-lmx-final-26cf8e148/target/release/hipfire --transport pm4 --context 128 --iterations 1000 --coherence-max-tokens 1024 --coherence-sampling greedy --work-dir /home/kaden/lmx-final-runs/work/2/lfm230/final-attempt1 --log-dir /home/kaden/lmx-final-runs/logs/2/lfm230/final-attempt1 --out /home/kaden/lmx-final-runs/lfm230-hiptrx-gfx1201-r9700-2-final-attempt1.json --env HOME=/home/kaden/lmx-final-runs/homes/hiptrx-2 --env HIPFIRE_HOME=/home/kaden/lmx-final-runs/homes/hiptrx-2/.hipfire","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+c1eab1f3290f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":35,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp1znf00jzo001kp8cxnxo","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":111.45,"tokSOut":584.222,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:40.827Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode. HARDWARE LABEL NOTE: this run was measured on an AMD Radeon RX 6950 XT (gfx1030, Navi 21, 16 GB). It is submitted under 'RX 6900 XT' only because localmaxxing's hardware list has no RX 6950 XT entry and rejects the name; the 6900 XT is the same Navi 21 silicon at the same 16 GB, but a lower-clocked bin, so treat this row as measured on slightly faster hardware than the label implies.\nbatch=4, concurrency=4, 8 requests/run, 3 runs, mode=unrecorded, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1030\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: mode=unrecorded (harvest mode null; metadata.json on hipx contains no mode field; scripts/lmx_continuous_batch.py with batch/requests suggests continuous; not asserting).\nevidence: sealed case lfm350-gfx1030-continuous-batch-B4R8-20260810-003707, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 8 --runs 3 --max-tokens 128 --max-seq 4096 --device 3","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 6900 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 6900 XT","hardwareGroupKey":"DISCRETE_GPU:rx 6900 xt","rank":36,"reactionCounts":{},"myEmoji":null},{"id":"cmpdzs3fq00s8pd01aobvkld4","modelRevision":"main","promptTokens":27,"outputTokens":256,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":82.93,"tokSOut":576.85,"tokSPrefill":754.54,"tokSTotal":null,"peakVramGb":7.43,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-05-20T11:42:06.183Z","notes":"hipfire @ 4840f0b6 (master, post-sync 2026-05-20)\nprompt: merge_sort thinking-OFF (md5=253c7ac50857fe6d0e10fb0d2c5e35c0)\ndecode mode: DFlash speculative\n  drafter: qwen35-9b-dflash-mq4.hf4 (z-lab Qwen3.5-DFlash native head, MQ4 weights)\n  target: qwen3.5-9b.mq4\nkv_cache: q8\nprompt_normalize: true (default)\nτ: 13.1818   accept_rate: 0.8788  (byte-identical to canonical bench)\noutput: 256 tokens emitted, coherent merge_sort code\nruns: 3 (median reported); per-run tok/s: [577.48, 572.66, 576.85], spread 0.84%\nAR baseline (same binary same hardware, --ar-baseline): 122.88 tok/s — DFlash speedup 4.70x\nHardware: Sapphire Nitro+ RX 7900 XTX, ROCm 7.2.x, k9lin host\nRefresh of existing leaderboard row 2 (575.24 tok/s on 0.1.20-alpha+71896daa) — parity with current master.","engineFlags":{"commandSnippet":"./target/release/examples/dflash_spec_demo --target ~/.hipfire/models/qwen3.5-9b.mq4 --draft ~/.hipfire/models/qwen35-9b-dflash-mq4.hf4 --prompt-file benchmarks/prompts/merge_sort_thinking_off.txt --max 256 --temp 0.0 --no-chatml --kv-mode q8 --ctx 4096","tensorParallel":1,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":true,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-9B","displayName":"Qwen3.5-9B","family":"Qwen","params":9,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-9B-Base","displayName":"Qwen3.5-9B-Base","params":10,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 7900 XTX","gpuCount":1,"vramGb":24,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":null,"os":null,"isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.2.0+4840f0b6","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 7900 XTX","hardwareGroupKey":"DISCRETE_GPU:rx 7900 xtx","rank":37,"reactionCounts":{},"myEmoji":null},{"id":"cmofzk74w0002ij042txvvkx5","modelRevision":"main","promptTokens":27,"outputTokens":184,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":75.53,"tokSOut":575.25,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":8.07421875,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-04-26T16:31:47.744Z","notes":"hipfire @ e659452 (master post-PR #51 + #52 series)\nprompt: merge_sort thinking-OFF (md5=253c7ac50857fe6d0e10fb0d2c5e35c0)\n  chatml-wrapped + explicit empty <think></think> for thinking-off\n  prompt file: benchmarks/prompts/merge_sort_thinking_off.txt\nruns: 3 (median reported); range 571.2–576.0\n  per-run tok/s: [571.23, 575.25, 576.04]\ndecode mode: DFlash speculative (block_size=16)\nkv_cache: asym3\nprompt_normalize: true (default since 2026-04-26)\nτ (median): 13.077\naccept_rate (median): 0.872\nprefill: 31.1ms (868.3 tok/s)\nttft (excl warmup): 75.5ms = prefill + first cycle\nvram: 8268 MB used / 24560 MB total\nnatural EOS at 184 tokens — production-shape bounded code (no loop)","engineFlags":{"commandSnippet":"./target/release/examples/dflash_spec_demo --target qwen3.5-9b.mq4 --draft qwen35-9b-dflash-mq4.hfq --prompt $(cat benchmarks/prompts/merge_sort_thinking_off.txt) --max 256 --no-chatml --kv-mode asym3","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":true,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-9B","displayName":"Qwen3.5-9B","family":"Qwen","params":9,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-9B-Base","displayName":"Qwen3.5-9B-Base","params":10,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 7900 XTX","gpuCount":1,"vramGb":24,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":null,"os":null,"isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.1.8-alpha+e659452","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 7900 XTX","hardwareGroupKey":"DISCRETE_GPU:rx 7900 xtx","rank":38,"reactionCounts":{},"myEmoji":null},{"id":"cmp8fqdz200ypo4014i9empwm","modelRevision":"main","promptTokens":27,"outputTokens":256,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":78.93,"tokSOut":575.24,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":7.95,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-05-16T14:22:03.326Z","notes":"hipfire @ 71896daa (feat/tbq, sFWHT KV family + multi-GPU)\nprompt: merge_sort thinking-OFF (md5=253c7ac50857fe6d0e10fb0d2c5e35c0)\n  benchmarks/prompts/merge_sort_thinking_off.txt (27 input tokens)\ndecode mode: DFlash speculative (adaptive block_size 8..16, mean B=16)\n  drafter: qwen35-9b-dflash-mq4.hf4 (z-lab Qwen3.5-DFlash native head, MQ4 weights)\n  target: qwen3.5-9b.mq4 (MQ4 weights, KV q8 filtered: 16/64 layers carry KV)\nkv_cache: q8 — chosen as winner from {q8,asym3,fwht3,fwht4} sweep (τ identical across all modes at this prompt; q8 has lowest KV-write overhead)\nprompt_normalize: true (default since 2026-04-26)\nτ: 13.1818   accept_rate: 0.8788\noutput: 256 tokens emitted\nAR baseline (same binary same hardware, --ar-baseline): 123.11 tok/s — DFlash speedup 4.67x\nHardware: Sapphire Nitro+ RX 7900 XTX, ROCm 7.2.x, k9lin host\nGPU coordination: gpu-tcas exclusive lease per cell","engineFlags":{"commandSnippet":"./target/release/examples/dflash_spec_demo --target ~/.hipfire/models/qwen3.5-9b.mq4 --draft ~/.hipfire/models/qwen35-9b-dflash-mq4.hf4 --prompt-file benchmarks/prompts/merge_sort_thinking_off.txt --max 256 --temp 0.0 --no-chatml --kv-mode q8 --ctx 4096","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":true,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-9B","displayName":"Qwen3.5-9B","family":"Qwen","params":9,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-9B-Base","displayName":"Qwen3.5-9B-Base","params":10,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 7900 XTX","gpuCount":1,"vramGb":24,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":null,"os":null,"isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.1.20-alpha+71896daa","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 7900 XTX","hardwareGroupKey":"DISCRETE_GPU:rx 7900 xtx","rank":39,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2wnr00mro001gdgmnwkm","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":93,"tokSOut":574.293,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:23.607Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 dispatch vs its own HIP control on same gfx1201 (k9lin-9070xt). PM4 (auto) median 574.3 tok/s vs HIP median 479.7 tok/s, speedup 1.197x. Single-stream decode, batch 1, greedy, temperature 0. engine: hipfire 0.3.0+c1eab1f3290f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of campaign, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm230-mq4-redline-pm4-20260809-gfx1201-rx9070xt, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 93.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --expected-substring 'Deep learning' --device 0 --process-runs 3 --bench-runs 5 --daemon target/release/examples/daemon --cli target/release/hipfire --transport pm4 --context 128 --iterations 1000 --coherence-max-tokens 1024 --coherence-sampling greedy --work-dir /home/kaden/lfm-final-runs/work/local/lfm230/attempt1 --log-dir /home/kaden/lfm-final-runs/logs/local/lfm230/attempt1 --out /home/kaden/lfm-final-runs/lfm230-local-gfx1201-attempt1.json","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+c1eab1f3290f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":40,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2xsx00muo001o3jgfnrb","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":2,"ttftMs":54.1,"tokSOut":567.479,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:25.089Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode.\nbatch=2, concurrency=2, 2 requests/run, 3 runs, mode=continuous, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.68 bpw effective (163173488 bytes *8 / 229693184 params)\nmodel file: lfm2.5-230m.mq4 sha256 3b92b9fd27c68d7f\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm2.5-230m-gfx1201-b2r2-FINAL-continuous, host k9lin, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-230m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 2 --requests 2 --runs 3 --max-tokens 128 --max-seq 4096 --port 11602 --home-root /home/kaden/lmx-final-runs/work/FINAL/lfm2.5-230m/continuous/home-b2r2 --log-dir /home/kaden/lmx-final-runs/logs/FINAL/lfm2.5-230m/continuous/b2r2 --out /home/kaden/lmx-final-runs/lfm230-FINAL-b2r2-gfx1201-continuous.json --device 0","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-230M","displayName":"LFM2.5-230M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-230M-Base","displayName":"LFM2.5-230M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":41,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp20sc00k2o001fztgo6lw","modelRevision":"main","promptTokens":13,"outputTokens":100,"contextLength":128,"prefillTokens":null,"batchSize":1,"ttftMs":42,"tokSOut":563.526,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:38:42.300Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 vs plain HIP A/B on identical MQ4 weights, 3 fresh processes x5 measurement rows, campaign medians. PM4 563.526 tok/s vs HIP 473.439 tok/s, speedup 1.190x (PM4 wins).\nengine: hipfire 0.3.0+44eb828cc804, backend rocm, gfx1201 AMD Radeon AI PRO R9700\nquant: MQ4 = MagnumQuant 4-bit group-256 FWHT-rotated, 0.53125 B/weight =4.25 bpw ideal, file 549751424 B =>5.04 bpw effective on 0.87B total params (5.50 bpw on 0.80B text)\nmodel file: qwen3.5-0.8b.registry.mq4 sha256 aedfe31be68213dbcd13cff3a63ba150c330d0d0547df30f2c629e76ab5dda39\nprompt: cap.txt md5 bbf2d0483ecacaa2eb7296f9c8083f95, 13 prompt tokens (est.), context 128, iterations 100\nmethod: median of 3 runs, fresh process, temperature 0 (greedy), PM4 retained replay verified at positions 128/227 each process\nevidence: sealed case qwen08-gfx1201-r9700-d1, corpus manifest 44eb828cc804\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 42.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/qwen3.5-0.8b.registry.mq4 --prompt-file benchmarks/prompts/sweep/cap.txt --device 1 --process-runs 3 --bench-runs 5 --transport pm4 --kv-mode q8 --context 128 --iterations 100 --max-seq 2048","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-0.8B","displayName":"Qwen3.5-0.8B","family":"Qwen","params":0.8,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-0.8B-Base","displayName":"Qwen3.5-0.8B-Base","params":1,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+44eb828cc804","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":42,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp2yyv00mxo001jn7winha","modelRevision":"main","promptTokens":13,"outputTokens":100,"contextLength":128,"prefillTokens":null,"batchSize":1,"ttftMs":44,"tokSOut":562.393,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:26.599Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 vs plain HIP A/B on identical MQ4 weights, 3 fresh processes x5 measurement rows, campaign medians. PM4 562.393 tok/s vs HIP 474.090 tok/s, speedup 1.186x (PM4 wins).\nengine: hipfire 0.3.0+44eb828cc804, backend rocm, gfx1201 AMD Radeon AI PRO R9700\nquant: MQ4 = MagnumQuant 4-bit group-256 FWHT-rotated, 0.53125 B/weight =4.25 bpw ideal, file 549751424 B =>5.04 bpw effective on 0.87B total params (5.50 bpw on 0.80B text)\nmodel file: qwen3.5-0.8b.registry.mq4 sha256 aedfe31be68213dbcd13cff3a63ba150c330d0d0547df30f2c629e76ab5dda39\nprompt: cap.txt md5 bbf2d0483ecacaa2eb7296f9c8083f95, 13 prompt tokens (est.), context 128, iterations 100\nmethod: median of 3 runs, fresh process, temperature 0 (greedy), PM4 retained replay verified at positions 128/227 each process\nevidence: sealed case qwen08-gfx1201-r9700-d0, corpus manifest 44eb828cc804\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 44.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/qwen3.5-0.8b.registry.mq4 --prompt-file benchmarks/prompts/sweep/cap.txt --device 0 --process-runs 3 --bench-runs 5 --transport pm4 --kv-mode q8 --context 128 --iterations 100 --max-seq 2048","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-0.8B","displayName":"Qwen3.5-0.8B","family":"Qwen","params":0.8,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-0.8B-Base","displayName":"Qwen3.5-0.8B-Base","params":1,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+44eb828cc804","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":43,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp304100n0o001bh83mn6v","modelRevision":"main","promptTokens":13,"outputTokens":100,"contextLength":128,"prefillTokens":null,"batchSize":1,"ttftMs":43,"tokSOut":562.031,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:28.081Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 vs plain HIP A/B on identical MQ4 weights, 3 fresh processes x5 measurement rows, campaign medians. PM4 562.031 tok/s vs HIP 471.904 tok/s, speedup 1.191x (PM4 wins).\nengine: hipfire 0.3.0+44eb828cc804, backend rocm, gfx1201 AMD Radeon AI PRO R9700\nquant: MQ4 = MagnumQuant 4-bit group-256 FWHT-rotated, 0.53125 B/weight =4.25 bpw ideal, file 549751424 B =>5.04 bpw effective on 0.87B total params (5.50 bpw on 0.80B text)\nmodel file: qwen3.5-0.8b.registry.mq4 sha256 aedfe31be68213dbcd13cff3a63ba150c330d0d0547df30f2c629e76ab5dda39\nprompt: cap.txt md5 bbf2d0483ecacaa2eb7296f9c8083f95, 13 prompt tokens (est.), context 128, iterations 100\nmethod: median of 3 runs, fresh process, temperature 0 (greedy), PM4 retained replay verified at positions 128/227 each process\nevidence: sealed case qwen08-gfx1201-r9700-d2, corpus manifest 44eb828cc804\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 43.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/qwen3.5-0.8b.registry.mq4 --prompt-file benchmarks/prompts/sweep/cap.txt --device 2 --process-runs 3 --bench-runs 5 --transport pm4 --kv-mode q8 --context 128 --iterations 100 --max-seq 2048","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-0.8B","displayName":"Qwen3.5-0.8B","family":"Qwen","params":0.8,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-0.8B-Base","displayName":"Qwen3.5-0.8B-Base","params":1,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+44eb828cc804","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":44,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp319s00n3o001hfh82u09","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":4,"ttftMs":116.5,"tokSOut":557.194,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:29.584Z","notes":"BATCHED AGGREGATE THROUGHPUT — not single-stream decode. HARDWARE LABEL NOTE: this run was measured on an AMD Radeon RX 6950 XT (gfx1030, Navi 21, 16 GB). It is submitted under 'RX 6900 XT' only because localmaxxing's hardware list has no RX 6950 XT entry and rejects the name; the 6900 XT is the same Navi 21 silicon at the same 16 GB, but a lower-clocked bin, so treat this row as measured on slightly faster hardware than the label implies.\nbatch=4, concurrency=4, 4 requests/run, 3 runs, mode=unrecorded, max_tokens=128, ctx=4096, KV=default (unspecified, hipfire fp16), greedy; TTFT is under load.\nengine: hipfire 0.3.0+b0bcc3f91506, backend rocm, gfx1030\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of 3 runs, fresh process, temperature 0 (greedy)\ncaveats: mode=unrecorded (harvest mode null; metadata.json on hipx contains no mode field; scripts/lmx_continuous_batch.py with batch/requests suggests continuous; not asserting).\nevidence: sealed case lfm350-gfx1030-continuous-batch-B4R4-20260810-003600, host hipx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)","engineFlags":{"commandSnippet":"scripts/lmx_continuous_batch.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --batch-size 4 --requests 4 --runs 3 --max-tokens 128 --max-seq 4096 --device 3","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 6900 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+b0bcc3f91506","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 6900 XT","hardwareGroupKey":"DISCRETE_GPU:rx 6900 xt","rank":45,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp32eb00n6o001a75ooed4","modelRevision":"main","promptTokens":13,"outputTokens":100,"contextLength":128,"prefillTokens":null,"batchSize":1,"ttftMs":45,"tokSOut":554.489,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:31.043Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 vs plain HIP A/B on identical MQ4 weights, 3 fresh processes x5 measurement rows, campaign medians. PM4 554.489 tok/s vs HIP 474.150 tok/s, speedup 1.169x (PM4 wins).\nengine: hipfire 0.3.0+44eb828cc804, backend rocm, gfx1201 AMD Radeon AI PRO R9700\nquant: MQ4 = MagnumQuant 4-bit group-256 FWHT-rotated, 0.53125 B/weight =4.25 bpw ideal, file 549751424 B =>5.04 bpw effective on 0.87B total params (5.50 bpw on 0.80B text)\nmodel file: qwen3.5-0.8b.registry.mq4 sha256 aedfe31be68213dbcd13cff3a63ba150c330d0d0547df30f2c629e76ab5dda39\nprompt: cap.txt md5 bbf2d0483ecacaa2eb7296f9c8083f95, 13 prompt tokens (est.), context 128, iterations 100\nmethod: median of 3 runs, fresh process, temperature 0 (greedy), PM4 retained replay verified at positions 128/227 each process\nevidence: sealed case qwen08-gfx1201-r9700-d3, corpus manifest 44eb828cc804\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 45.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/qwen3.5-0.8b.registry.mq4 --prompt-file benchmarks/prompts/sweep/cap.txt --device 3 --process-runs 3 --bench-runs 5 --transport pm4 --kv-mode q8 --context 128 --iterations 100 --max-seq 2048","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-0.8B","displayName":"Qwen3.5-0.8B","family":"Qwen","params":0.8,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-0.8B-Base","displayName":"Qwen3.5-0.8B-Base","params":1,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+44eb828cc804","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":46,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp33j900n9o001ieskp9t0","modelRevision":"main","promptTokens":13,"outputTokens":100,"contextLength":128,"prefillTokens":null,"batchSize":1,"ttftMs":69,"tokSOut":550.162,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:32.518Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 vs plain HIP A/B on identical MQ4 weights, 3 fresh processes x5 measurement rows, campaign medians. PM4 550.162 tok/s vs HIP 505.823 tok/s, speedup 1.088x (PM4 wins).\nengine: hipfire 0.3.0+44eb828cc804, backend rocm, gfx1100 AMD Radeon RX 7900 XTX\nquant: MQ4 = MagnumQuant 4-bit group-256 FWHT-rotated, 0.53125 B/weight =4.25 bpw ideal, file 549751424 B =>5.04 bpw effective on 0.87B total params (5.50 bpw on 0.80B text)\nmodel file: qwen3.5-0.8b.registry.mq4 sha256 aedfe31be68213dbcd13cff3a63ba150c330d0d0547df30f2c629e76ab5dda39\nprompt: cap.txt md5 bbf2d0483ecacaa2eb7296f9c8083f95, 13 prompt tokens (est.), context 128, iterations 100\nmethod: median of 3 runs, fresh process, temperature 0 (greedy), PM4 retained replay verified at positions 128/227 each process\nevidence: sealed case qwen08-mq4-redline-pm4-20260809-gfx1100-rx7900xtx, corpus manifest 44eb828cc804\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 69.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/qwen3.5-0.8b.registry.mq4 --prompt-file benchmarks/prompts/sweep/cap.txt --device 0 --process-runs 3 --bench-runs 5 --transport pm4 --kv-mode q8 --context 128 --iterations 100 --max-seq 2048","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-0.8B","displayName":"Qwen3.5-0.8B","family":"Qwen","params":0.8,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-0.8B-Base","displayName":"Qwen3.5-0.8B-Base","params":1,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 7900 XTX","gpuCount":1,"vramGb":24,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen AI Max+ 395","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+44eb828cc804","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 7900 XTX","hardwareGroupKey":"DISCRETE_GPU:rx 7900 xtx","rank":47,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp34or00nco001vjgo5dg0","modelRevision":"main","promptTokens":13,"outputTokens":100,"contextLength":128,"prefillTokens":null,"batchSize":1,"ttftMs":67,"tokSOut":545.272,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:34.011Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 vs plain HIP A/B on identical MQ4 weights, 3 fresh processes x5 measurement rows, campaign medians. PM4 545.272 tok/s vs HIP 477.981 tok/s, speedup 1.141x (PM4 wins).\nengine: hipfire 0.3.0+44eb828cc804, backend rocm, gfx1201 AMD Radeon RX 9070 XT\nquant: MQ4 = MagnumQuant 4-bit group-256 FWHT-rotated, 0.53125 B/weight =4.25 bpw ideal, file 549751424 B =>5.04 bpw effective on 0.87B total params (5.50 bpw on 0.80B text)\nmodel file: qwen3.5-0.8b.registry.mq4 sha256 aedfe31be68213dbcd13cff3a63ba150c330d0d0547df30f2c629e76ab5dda39\nprompt: cap.txt md5 bbf2d0483ecacaa2eb7296f9c8083f95, 13 prompt tokens (est.), context 128, iterations 100\nmethod: median of 3 runs, fresh process, temperature 0 (greedy), PM4 retained replay verified at positions 128/227 each process\nevidence: sealed case qwen08-mq4-redline-pm4-20260809-gfx1201-rx9070xt, corpus manifest 44eb828cc804\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 67.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/qwen3.5-0.8b.registry.mq4 --prompt-file benchmarks/prompts/sweep/cap.txt --device 0 --process-runs 3 --bench-runs 5 --transport pm4 --kv-mode q8 --context 128 --iterations 100 --max-seq 2048","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":null,"attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"Qwen/Qwen3.5-0.8B","displayName":"Qwen3.5-0.8B","family":"Qwen","params":0.8,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"Qwen/Qwen3.5-0.8B-Base","displayName":"Qwen3.5-0.8B-Base","params":1,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"RX 9070 XT","gpuCount":1,"vramGb":16,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen 9 3900X","os":"Ubuntu 24.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+44eb828cc804","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"RX 9070 XT","hardwareGroupKey":"DISCRETE_GPU:rx 9070 xt","rank":48,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp35tm00nfo0013v533ob3","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":88,"tokSOut":542.545,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:35.483Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 dispatch vs its own HIP control on same gfx1201 (hiptrx-r9700-x1). PM4 (auto) median 542.5 tok/s vs HIP median 447.7 tok/s, speedup 1.212x. Single-stream decode, batch 1, greedy, temperature 0. engine: hipfire 0.3.0+c1eab1f3290f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of campaign, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm350-mq4-redline-pm4-20260809-gfx1201-r9700-d3, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 88.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --expected-substring 'Deep learning' --device 3 --process-runs 3 --bench-runs 5 --daemon target/release/examples/daemon --cli target/release/hipfire --transport pm4 --context 128 --iterations 1000 --coherence-max-tokens 1024 --coherence-sampling greedy --work-dir /home/kaden/lmx-final-runs/work/3/lfm350/final-attempt1 --log-dir /home/kaden/lmx-final-runs/logs/3/lfm350/final-attempt1 --out /home/kaden/lmx-final-runs/lfm350-hiptrx-gfx1201-r9700-3-final-attempt1.json --env HOME=/home/kaden/lmx-final-runs/homes/hiptrx-3 --env HIPFIRE_HOME=/home/kaden/lmx-final-runs/homes/hiptrx-3/.hipfire","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+c1eab1f3290f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":49,"reactionCounts":{},"myEmoji":null},{"id":"cmsnp36y200nio001yijvmfmz","modelRevision":"main","promptTokens":27,"outputTokens":128,"contextLength":4096,"prefillTokens":null,"batchSize":1,"ttftMs":86,"tokSOut":542.145,"tokSPrefill":null,"tokSTotal":null,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"createdAt":"2026-08-10T20:39:36.939Z","notes":"SINGLE-STREAM. hipfire Redline retained-PM4 dispatch vs its own HIP control on same gfx1201 (hiptrx-r9700-x1). PM4 (auto) median 542.1 tok/s vs HIP median 450.1 tok/s, speedup 1.204x. Single-stream decode, batch 1, greedy, temperature 0. engine: hipfire 0.3.0+c1eab1f3290f, backend rocm, gfx1201\nquant: MQ4 = MagnumQuant 4-bit, group-256, FWHT-rotated, 5.18 bpw effective (229474032 bytes *8 / 354483968 params)\nmodel file: lfm2.5-350m.mq4 sha256 4885d1cecbad59d7\nprompt: longform.txt md5 e6f8ee7d66b934c49493347500f48083, 27 prompt tokens\nmethod: median of campaign, fresh process, temperature 0 (greedy)\nevidence: sealed case lfm350-mq4-redline-pm4-20260809-gfx1201-r9700-d0, host hiptrx, corpus manifest 45b76818319f\nnot measured: peak VRAM (not instrumented this campaign)\nTTFT 86.0 ms is the median of the same campaign's coherence pass (3 rows, auto arm, same binary/model/route); tok/s is the bench median. Separate passes, both measured.","engineFlags":{"commandSnippet":"scripts/lmx_redline_campaign.py --model /home/kaden/.hipfire/models/lfm2.5-350m.mq4 --prompt-file benchmarks/prompts/sweep/longform.txt --expected-substring 'Deep learning' --device 0 --process-runs 3 --bench-runs 5 --daemon /home/kaden/hipfire-lmx-final-26cf8e148/target/release/examples/daemon --cli /home/kaden/hipfire-lmx-final-26cf8e148/target/release/hipfire --transport pm4 --context 128 --iterations 1000 --coherence-max-tokens 1024 --coherence-sampling greedy --work-dir /home/kaden/lmx-final-runs/work/0/lfm350/attempt1 --log-dir /home/kaden/lmx-final-runs/logs/0/lfm350/attempt1 --env HOME=/home/kaden/lmx-final-runs/homes/hiptrx-0 --env HIPFIRE_HOME=/home/kaden/lmx-final-runs/homes/hiptrx-0/.hipfire --out /home/kaden/lmx-final-runs/lfm350-hiptrx-gfx1201-r9700-0-attempt1.json","tensorParallel":null,"gpuLayers":null,"kvCacheDtype":"q8","attentionBackend":null,"flashAttn":null,"specDecoding":false,"mtpEnabled":false},"model":{"hfId":"LiquidAI/LFM2.5-350M","displayName":"LFM2.5-350M","family":null,"params":0,"activeParams":null,"isMoE":false,"baseModel":{"hfId":"LiquidAI/LFM2.5-350M-Base","displayName":"LFM2.5-350M-Base","params":0,"activeParams":null,"isMoE":false}},"hardware":{"hwClass":"DISCRETE_GPU","gpuName":"Radeon AI Pro R9700","gpuCount":1,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"cpu":"AMD Ryzen Threadripper 9970X","os":"Ubuntu 26.04","isHeterogeneousGpu":false,"gpuSlots":[]},"engine":{"engineName":"hipfire","engineVersion":"0.3.0+c1eab1f3290f","quantization":"MQ4","backend":"rocm"},"user":{"id":"cmoeye1gq0000le04ie2kqs58","username":"schuttdev","verified":false,"verifiedAt":null,"pro":true},"hardwareGroupLabel":"Radeon AI Pro R9700","hardwareGroupKey":"DISCRETE_GPU:radeon ai pro r9700","rank":50,"reactionCounts":{},"myEmoji":null}],"total":138,"limit":50,"offset":0}