[{"data":1,"prerenderedAt":3040},["ShallowReactive",2],{"blog-post-en-\u002Fblog\u002Fpaiton-qwen38-mxfp4-dflash2-r9700":3,"blog-posts-sidebar-en":2561},{"id":4,"title":5,"body":6,"categories":2544,"date":2548,"description":2549,"extension":2550,"heading":2551,"image":2552,"meta":2553,"navigation":709,"originalUrl":2554,"path":2555,"seo":2556,"slug":2557,"stem":2558,"updated":2559,"__hash__":2560},"blog\u002Fblog\u002Fpaiton-qwen38-mxfp4-dflash2-r9700.md","Qwen3.8: 400.7 tok\u002Fs on R9700 | Paiton",{"type":7,"value":8,"toc":2504},"minimark",[9,37,63,74,79,96,99,116,170,175,181,260,264,267,339,346,350,353,423,439,458,465,475,486,490,503,512,516,519,523,538,596,600,607,610,624,637,641,647,652,657,660,685,692,696,701,704,718,723,729,733,753,763,769,782,785,789,809,816,821,907,919,925,929,932,949,955,960,963,967,987,993,998,1011,1014,1019,1023,1036,1042,1047,1067,1073,1078,1081,1085,1098,1175,1186,1202,1213,1219,1224,1230,1234,1251,1271,1288,1291,1294,1298,1317,1331,1337,1342,1347,1351,1466,1475,1479,1488,1503,1561,1571,1591,1608,1612,1644,1654,1663,1667,1690,1694,1697,1707,1729,1733,1742,1749,1764,1775,1778,1803,1807,1843,1847,1854,1866,1879,1883,1892,1928,1934,1938,1941,1945,1948,1951,1965,1996,2004,2008,2028,2031,2049,2053,2059,2062,2065,2073,2077,2084,2104,2110,2114,2123,2127,2500],[10,11,12,16,17,20,21,24,25],"p",{},[13,14,15],"strong",{},"400.7 aggregate output tokens per second on one Radeon AI PRO R9700."," The 18 September release moves to ROCm 10 and vLLM 0.29.0, with a 65K throughput preset and configurable long-conversation profiles. Weighted decode is ",[13,18,19],{},"146.9 tok\u002Fs","; the JSON category reaches ",[13,22,23],{},"215.9 tok\u002Fs",".",[26,27,28],"sup",{},[29,30,36],"a",{"href":31,"ariaDescribedBy":32,"dataFootnoteRef":34,"id":35},"#user-content-fn-11",[33],"footnote-label","","user-content-fnref-11","1",[10,38,39,42,43,47,48,52,53,57,58,62],{},[13,40,41],{},"Updated 19 September 2026."," Start with the ",[29,44,46],{"href":45},"#current-release-4007-toks-on-one-r9700","current benchmark",", ",[29,49,51],{"href":50},"#long-conversations-prefix-caching-you-can-try","200K\u002F220K conversations and prefix caching",", or ",[29,54,56],{"href":55},"#run-the-rocm-10-release","current launch instructions",". The ",[29,59,61],{"href":60},"#original-16-september-comparison","original 57% comparison"," and the 17 September experiments remain below as historical evidence.",[10,64,65,68,69,73],{},[13,66,67],{},"A new release, not a new matched percentage claim."," This run uses ",[70,71,72],"code",{},"unsloth\u002FQwen3.8-27B-NVFP4"," through the MXFP4 runtime path, rather than the earlier AMD checkpoint. The runtime, sampling and benchmark workload also changed. We do not divide 400.7 by an older result to claim a controlled speedup over Radiance or the previous Paiton release.",[75,76,78],"h2",{"id":77},"current-release-4007-toks-on-one-r9700","Current release: 400.7 tok\u002Fs on one R9700",[10,80,81,82,85,86,89,90],{},"These results come from ",[13,83,84],{},"one complete BetterBench 0.6.0 validation run: 290 successful requests including warmups",", on one 32 GB R9700 at a ",[13,87,88],{},"300 W power limit",". The 65,536-token preset was tested with thinking disabled and APC off. Sampling was temperature 0.7, top-p 0.95, top-k 20 and seed 42. A power limit is not a measured electricity-use result.",[26,91,92],{},[29,93,36],{"href":31,"ariaDescribedBy":94,"dataFootnoteRef":34,"id":95},[33],"user-content-fnref-11-2",[97,98],"rocm-throughput-figure",{},[10,100,101],{},[102,103,104,105,111,112,24],"em",{},"65K release preset; 48 requests at each concurrency level. Aggregate output divided by total wall time, not per-user decode speed. ",[29,106,110],{"href":107,"rel":108},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002F8f56157c05eb6a53f6cdab00a115e47b42fed2d1\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FREADME.md#current-benchmark-results",[109],"nofollow","Published tables"," · ",[29,113,115],{"href":114},"\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002Fupdate-2026-09-19\u002Fresults.json","Chart data",[117,118,119,133],"table",{},[120,121,122],"thead",{},[123,124,125,129],"tr",{},[126,127,128],"th",{},"Concurrent requests",[126,130,132],{"align":131},"right","Aggregate output tok\u002Fs",[134,135,136,144,152,160],"tbody",{},[123,137,138,141],{},[139,140,36],"td",{},[139,142,143],{"align":131},"115.0",[123,145,146,149],{},[139,147,148],{},"2",[139,150,151],{"align":131},"203.2",[123,153,154,157],{},[139,155,156],{},"4",[139,158,159],{"align":131},"296.5",[123,161,162,165],{},[139,163,164],{},"8",[139,166,167],{"align":131},[13,168,169],{},"400.7",[171,172,174],"h3",{"id":173},"decode-across-real-task-categories","Decode across real task categories",[10,176,177,178,180],{},"The weighted decode score is ",[13,179,19],{},". The category medians below use five scored runs after one warmup per category. The JSON result is a category result, not the overall speed a user should expect.",[117,182,183,193],{},[120,184,185],{},[123,186,187,190],{},[126,188,189],{},"Category",[126,191,192],{"align":131},"Median decode tok\u002Fs",[134,194,195,203,211,219,229,237,245,253],{},[123,196,197,200],{},[139,198,199],{},"Chat",[139,201,202],{"align":131},"123.7",[123,204,205,208],{},[139,206,207],{},"Code",[139,209,210],{"align":131},"169.2",[123,212,213,216],{},[139,214,215],{},"File editing",[139,217,218],{"align":131},"185.0",[123,220,221,224],{},[139,222,223],{},"JSON",[139,225,226],{"align":131},[13,227,228],{},"215.9",[123,230,231,234],{},[139,232,233],{},"Math",[139,235,236],{"align":131},"182.7",[123,238,239,242],{},[139,240,241],{},"Prose",[139,243,244],{"align":131},"74.9",[123,246,247,250],{},[139,248,249],{},"Reasoning",[139,251,252],{"align":131},"112.5",[123,254,255,258],{},[139,256,257],{},"Summarization",[139,259,143],{"align":131},[171,261,263],{"id":262},"prefill-as-the-prompt-grows","Prefill as the prompt grows",[10,265,266],{},"BetterBench divides actual prompt tokens by time to first token. Each depth has eight scored runs after two warmups. Requested depth labels are not the actual tokenized prompt lengths.",[117,268,269,282],{},[120,270,271],{},[123,272,273,276,279],{},[126,274,275],{},"Requested depth",[126,277,278],{"align":131},"Median actual prompt tokens",[126,280,281],{"align":131},"Median prefill tok\u002Fs",[134,283,284,295,306,317,328],{},[123,285,286,289,292],{},[139,287,288],{},"2,000",[139,290,291],{"align":131},"1,516.5",[139,293,294],{"align":131},"3,503.6",[123,296,297,300,303],{},[139,298,299],{},"8,000",[139,301,302],{"align":131},"5,894.5",[139,304,305],{"align":131},"3,530.4",[123,307,308,311,314],{},[139,309,310],{},"16,000",[139,312,313],{"align":131},"11,802.0",[139,315,316],{"align":131},"3,536.9",[123,318,319,322,325],{},[139,320,321],{},"32,000",[139,323,324],{"align":131},"23,549.5",[139,326,327],{"align":131},"3,393.0",[123,329,330,333,336],{},[139,331,332],{},"64,000",[139,334,335],{"align":131},"47,016.5",[139,337,338],{"align":131},"3,113.0",[10,340,341,342,345],{},"These are measurements of the ",[13,343,344],{},"65K configuration",", not 200K throughput. “65K” means 65,536 total context tokens, the same count previously described as “64K”; the name alone does not represent a larger context window.",[75,347,349],{"id":348},"long-conversations-prefix-caching-you-can-try","Long conversations: prefix caching you can try",[10,351,352],{},"The new public launchers make context a startup setting, not a compiled image limit. The context budget includes the prompt, chat\u002Ftool formatting and generated output. VRAM and the checkpoint's supported range still constrain deployment; eight scheduled requests do not mean eight full 65K conversations fit at once.",[117,354,355,371],{},[120,356,357],{},[123,358,359,362,365,368],{},[126,360,361],{},"Current startup choice",[126,363,364],{"align":131},"Total context",[126,366,367],{"align":131},"Maximum scheduled requests",[126,369,370],{},"APC",[134,372,373,386,398,411],{},[123,374,375,378,381,383],{},[139,376,377],{},"65K release preset",[139,379,380],{"align":131},"65,536",[139,382,164],{"align":131},[139,384,385],{},"Off",[123,387,388,391,394,396],{},[139,389,390],{},"200K release preset",[139,392,393],{"align":131},"200,000",[139,395,36],{"align":131},[139,397,385],{},[123,399,400,403,406,408],{},[139,401,402],{},"Chat profile",[139,404,405],{"align":131},"200,000; tested override to 220,000",[139,407,36],{"align":131},[139,409,410],{},"On, experimental configuration",[123,412,413,416,419,421],{},[139,414,415],{},"Desktop profile",[139,417,418],{"align":131},"32,768",[139,420,36],{"align":131},[139,422,385],{},[10,424,425,428,429,432,433],{},[13,426,427],{},"The chat profile is now publicly runnable."," It enables prefix caching for an unchanged conversation history, uses an ",[13,430,431],{},"8 GiB KV pool",", 1,024-token prefill chunks and thinking disabled. Use a dedicated R9700: spare VRAM is tight. This is a different configuration from the throughput benchmark, not APC enabled by default in the release presets.",[26,434,435],{},[29,436,36],{"href":31,"ariaDescribedBy":437,"dataFootnoteRef":34,"id":438},[33],"user-content-fnref-11-3",[10,440,441,442,445,446,449,450,453,454,457],{},"In one 220K-context retrieval probe, a ",[13,443,444],{},"215,005-token prompt"," completed in ",[13,447,448],{},"130.00 seconds cold and 1.77 seconds on an identical repeat",", reusing ",[13,451,452],{},"213,840 cached tokens",". Both answers contained nine tokens at temperature zero. These are ",[13,455,456],{},"complete response times from a functional probe",", not TTFT, a general latency guarantee or faster decode. Changed-prefix retrieval, two longer answers, a subsequent XML tool call and a fresh short request also passed. This is not broad long-conversation quality validation.",[10,459,460,461,464],{},"The separate 200K image check passed a 198,989-token prompt followed by generation and ordinary\u002Fstreaming XML tool calls. The largest configured serving limit tested is ",[13,462,463],{},"220,000",", not the checkpoint's 262,144-token architectural ceiling.",[10,466,467,468,471,472,24],{},"Cache hits require an unchanged prefix still resident in memory. They avoid repeated prompt work, not generation of each new token. Disk caches used at startup are separate. The chat profile reports ",[70,469,470],{},"usage.prompt_tokens_details.cached_tokens","; streaming clients also need ",[70,473,474],{},"\"stream_options\":{\"include_usage\":true}",[10,476,477,478,481,482,485],{},"For a GPU shared with desktop applications, ",[70,479,480],{},"--profile desktop"," starts smaller, with a 2 GiB cache and one scheduled request. It is not a guarantee against running out of memory. Lowering context alone does not shrink a fixed KV allocation. These launchers are ",[13,483,484],{},"text-only","; the separate vision smoke test is not a released 200K\u002F220K multimodal profile.",[75,487,489],{"id":488},"run-the-rocm-10-release","Run the ROCm 10 release",[10,491,492,493,498,499,502],{},"Use Linux x86-64, Python 3, Docker, the Hugging Face CLI and one 32 GB Radeon AI PRO R9700 with working AMD GPU access. Get the ",[29,494,497],{"href":495,"rel":496},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002F8f56157c05eb6a53f6cdab00a115e47b42fed2d1\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FREADME.md#run-the-current-release",[109],"public repository"," and run the commands from its root. For reproducibility, this article follows commit ",[70,500,501],{},"8f56157c05eb6a53f6cdab00a115e47b42fed2d1",". The launchers pin the public images by immutable digest. Do not reuse the old AMD checkpoint or historical automatic downloader for this release.",[10,504,505,506,511],{},"Select the GPU through the launcher's documented ROCm visibility environment variables. They are forwarded unchanged; inspect existing masks rather than combining indices blindly. See the ",[29,507,510],{"href":508,"rel":509},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002F8f56157c05eb6a53f6cdab00a115e47b42fed2d1\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FREADME.md#gpu-context-and-memory-controls",[109],"GPU and memory controls",". Run only one profile on the card at a time.",[171,513,515],{"id":514},"start-the-200k-chat-profile","Start the 200K chat profile",[10,517,518],{},"The commands download the exact target and draft snapshots and start the server in the foreground. The first image pull is approximately 9.6 GB, excluding weights. Startup loads\u002Fconverts weights and compiles runtime components, so wait for readiness before measuring performance. Subsequent starts reuse the persistent cache; no private compiler checkout is needed.",[520,521],"qwen-launch",{"profile":522},"rocm10-chat",[10,524,525,526,531,532,537],{},"The API uses ",[13,527,528],{},[70,529,530],{},"http:\u002F\u002F127.0.0.1:18982\u002Fv1"," and model name ",[13,533,534],{},[70,535,536],{},"Qwen3.8",", not the historical port and model name below:",[539,540,544],"pre",{"className":541,"code":542,"language":543,"meta":34,"style":34},"language-bash shiki shiki-themes github-light github-dark","curl --fail http:\u002F\u002F127.0.0.1:18982\u002Fhealth\ncurl --fail http:\u002F\u002F127.0.0.1:18982\u002Fv1\u002Fchat\u002Fcompletions \\\n  -H 'Content-Type: application\u002Fjson' \\\n  -d '{\"model\":\"Qwen3.8\",\"messages\":[{\"role\":\"user\",\"content\":\"Write a Python function that removes duplicate items while preserving order.\"}],\"temperature\":0.7,\"top_p\":0.95,\"max_tokens\":256,\"stream\":true,\"chat_template_kwargs\":{\"enable_thinking\":false}}'\n","bash",[70,545,546,563,576,587],{"__ignoreMap":34},[547,548,551,555,559],"span",{"class":549,"line":550},"line",1,[547,552,554],{"class":553},"sScJk","curl",[547,556,558],{"class":557},"sj4cs"," --fail",[547,560,562],{"class":561},"sZZnC"," http:\u002F\u002F127.0.0.1:18982\u002Fhealth\n",[547,564,566,568,570,573],{"class":549,"line":565},2,[547,567,554],{"class":553},[547,569,558],{"class":557},[547,571,572],{"class":561}," http:\u002F\u002F127.0.0.1:18982\u002Fv1\u002Fchat\u002Fcompletions",[547,574,575],{"class":557}," \\\n",[547,577,579,582,585],{"class":549,"line":578},3,[547,580,581],{"class":557},"  -H",[547,583,584],{"class":561}," 'Content-Type: application\u002Fjson'",[547,586,575],{"class":557},[547,588,590,593],{"class":549,"line":589},4,[547,591,592],{"class":557},"  -d",[547,594,595],{"class":561}," '{\"model\":\"Qwen3.8\",\"messages\":[{\"role\":\"user\",\"content\":\"Write a Python function that removes duplicate items while preserving order.\"}],\"temperature\":0.7,\"top_p\":0.95,\"max_tokens\":256,\"stream\":true,\"chat_template_kwargs\":{\"enable_thinking\":false}}'\n",[171,597,599],{"id":598},"select-the-65k-benchmark-preset","Select the 65K benchmark preset",[10,601,602,603,606],{},"First stop the other server with ",[70,604,605],{},"docker stop paiton-qwen38-200k",". Keep the same target and draft directories, then use:",[520,608],{"profile":609},"rocm10-65k",[10,611,612,615,616,619,620,623],{},[13,613,614],{},"Set thinking explicitly in client requests to match the benchmark."," The release preset's model template enables thinking if omitted; the chat profile and reported benchmarks disable it. Use ",[70,617,618],{},"--thinking off"," for a server default or ",[70,621,622],{},"\"chat_template_kwargs\":{\"enable_thinking\":false}"," per request.",[10,625,626,627,630,631,636],{},"For 220K conversations, stop the running server and use ",[70,628,629],{},"bash models\u002FQwen3.8-MXFP4-DFlash2\u002Frun-rocm10-200k.sh --profile chat --context 220000",". Use the complete chat profile rather than assuming an isolated APC flag preserves memory requirements. The ",[29,632,635],{"href":633,"rel":634},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002F8f56157c05eb6a53f6cdab00a115e47b42fed2d1\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FREADME.md#200k-and-220k-with-prefix-caching",[109],"public guide"," documents the tested settings and limitations.",[75,638,640],{"id":639},"original-16-september-comparison","Original 16 September comparison",[10,642,643,646],{},[13,644,645],{},"Historical scope:"," everything in the original comparison below uses the earlier AMD checkpoint, vLLM 0.28 and matched 8K configuration. Its 57% gain remains valid for those tests; it is not a percentage claim for the ROCm 10 release.",[10,648,649],{},[13,650,651],{},"One Radeon AI PRO R9700. The same Qwen3.8-27B MXFP4 checkpoint. The same 5 GiB cache allocation. Paiton delivers 57% more aggregate throughput than our matched Radiance + DFlash2 baseline at eight concurrent requests and reduces median time to first token from 6.59 seconds to 195 milliseconds.",[10,653,654],{},[102,655,656],{},"Throughput and latency headlines use the original matched 188-request comparison at eight concurrent requests and an 8,192-token context limit. Both accelerated engines use the same target and DFlash2 snapshots. GPU artwork is illustrative.",[10,658,659],{},"This is a substantial step forward for serving a 27B model on one workstation GPU. Not just faster generation when one person is waiting. More requests progressing together, almost three times the estimated cache-token capacity, and dramatically less waiting under load.",[10,661,662,663,666,667,670,671,674,675,678],{},"Paiton reaches ",[13,664,665],{},"314.5 aggregate output tokens per second",", against ",[13,668,669],{},"200.3 tok\u002Fs"," for the matched accelerated baseline. Weighted serial decode improves by ",[13,672,673],{},"22%",", and prefill is faster at every tested prompt depth. The execution runs through a plugin on the official vLLM 0.28 ROCm runtime. ",[13,676,677],{},"The installed vLLM library stays unchanged.",[26,679,680],{},[29,681,148],{"href":682,"ariaDescribedBy":683,"dataFootnoteRef":34,"id":684},"#user-content-fn-1",[33],"user-content-fnref-1",[10,686,687,688,691],{},"The important combination is performance ",[13,689,690],{},"and"," deployment: a much stronger local serving result without moving to a separate inference-engine distribution.",[75,693,695],{"id":694},"see-it-in-action","See it in action",[10,697,698],{},[102,699,700],{},"Historical recording from the original release, not a demonstration of the new ROCm 10 benchmark or 220K chat profile.",[10,702,703],{},"Watch Qwen3.8 generate responses on one Radeon AI PRO R9700, with Paiton + DFlash2 running through regular vLLM. The recording shows streamed output alongside live generation figures and GPU activity.",[10,705,706],{},[707,708],"video",{"controls":709,"playsInline":709,"preload":710,"width":711,"height":712,"ariaLabel":713,"ariaDescribedBy":714,"poster":716,"src":717},true,"none",1354,1080,"Screen recording of Qwen3.8 generating responses with Paiton on one Radeon AI PRO R9700",[715],"paiton-live-demo-caption","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002Flive-demo-poster.webp","\u002Fasset\u002Fvideos\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002Fpaiton-qwen38-r9700-demo.mp4",[10,719,720],{"id":715},[102,721,722],{},"70-second screen recording with one active request at a time. The live figures describe the requests shown, not the eight-request aggregate benchmark below.",[10,724,725,728],{},[29,726,727],{"href":717},"Open the recording (MP4, 10.2 MB)",". Use the player's fullscreen control for a closer look.",[75,730,732],{"id":731},"a-strong-baseline-a-bigger-result","A strong baseline. A bigger result.",[10,734,735,736,741,742,24,745],{},"The investigation began with the impressive work around ",[29,737,740],{"href":738,"rel":739},"https:\u002F\u002Fgithub.com\u002Fmagiccodingman\u002Fvllm-radiance",[109],"vLLM-Radiance",". Seeing what the team achieved on AMD hardware gave us ideas and encouraged us to push our own R9700 implementation further. Its public documentation describes support for the same AMD Quark MXFP4 model and DFlash2 acceleration; its published performance results use ",[13,743,744],{},"two R9700s",[26,746,747],{},[29,748,752],{"href":749,"ariaDescribedBy":750,"dataFootnoteRef":34,"id":751},"#user-content-fn-2",[33],"user-content-fnref-2","3",[10,754,755],{},[13,756,757,758,24],{},"Radiance inspired the investigation. Paiton's native HIP integration includes techniques adapted from Radiance and libr4d, credited in the ",[29,759,762],{"href":760,"rel":761},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FTHIRD_PARTY_NOTICES.md",[109],"public third-party notices",[10,764,765,766],{},"We wanted to answer a different question: ",[13,767,768],{},"how much could we deliver from this model on one card, while keeping the regular vLLM deployment path?",[10,770,771,772,775,776],{},"We ran Radiance + DFlash2 and Paiton on vLLM + DFlash2 ourselves on ",[13,773,774],{},"one R9700",", using the same checkpoint, target and draft snapshots, FP8 KV cache, 5 GiB cache allocation and 8,192-token context ceiling. Both accelerated profiles used seven speculative tokens, greedy sampling and disabled prefix caching.",[26,777,778],{},[29,779,148],{"href":682,"ariaDescribedBy":780,"dataFootnoteRef":34,"id":781},[33],"user-content-fnref-1-2",[10,783,784],{},"Our percentage gains compare those matched single-GPU runs, not a one-card result against somebody else's two-card screenshot.",[75,786,788],{"id":787},"_57-more-throughput-with-a-growing-lead-under-load","57% more throughput, with a growing lead under load",[10,790,791,792,795,796,802],{},"The original 57% result comes from the ",[13,793,794],{},"full 188-request BetterBench preset",", not the shorter, 128-output-token-cap comparison. We retained the preset's original longer output budgets, ten measured passes per task category, 24 requests at each tested concurrency level and repeated prefill measurements. Both engines completed the full request set.",[26,797,798],{},[29,799,148],{"href":682,"ariaDescribedBy":800,"dataFootnoteRef":34,"id":801},[33],"user-content-fnref-1-3",[26,803,804],{},[29,805,156],{"href":806,"ariaDescribedBy":807,"dataFootnoteRef":34,"id":808},"#user-content-fn-4",[33],"user-content-fnref-4",[10,810,811],{},[812,813],"img",{"alt":814,"src":815},"Paiton reaches 314.5 aggregate output tokens per second versus 200.3 for Radiance at eight concurrent requests, leading at every tested concurrency.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002F01-full-throughput.webp",[10,817,818],{},[102,819,820],{},"Full 188-request workload, runs 495 \u002F 494. Aggregate throughput includes prefill and queueing. Higher is better.",[117,822,823,838],{},[120,824,825],{},[123,826,827,829,832,835],{},[126,828,128],{},[126,830,831],{"align":131},"Radiance + DFlash2",[126,833,834],{"align":131},"Paiton on vLLM + DFlash2",[126,836,837],{"align":131},"Paiton uplift",[134,839,840,857,874,891],{},[123,841,842,844,847,852],{},[139,843,36],{},[139,845,846],{"align":131},"76.9 tok\u002Fs",[139,848,849],{"align":131},[13,850,851],{},"89.6 tok\u002Fs",[139,853,854],{"align":131},[13,855,856],{},"16.5%",[123,858,859,861,864,869],{},[139,860,148],{},[139,862,863],{"align":131},"142.7 tok\u002Fs",[139,865,866],{"align":131},[13,867,868],{},"165.7 tok\u002Fs",[139,870,871],{"align":131},[13,872,873],{},"16.1%",[123,875,876,878,881,886],{},[139,877,156],{},[139,879,880],{"align":131},"187.1 tok\u002Fs",[139,882,883],{"align":131},[13,884,885],{},"254.8 tok\u002Fs",[139,887,888],{"align":131},[13,889,890],{},"36.2%",[123,892,893,895,897,902],{},[139,894,164],{},[139,896,669],{"align":131},[139,898,899],{"align":131},[13,900,901],{},"314.5 tok\u002Fs",[139,903,904],{"align":131},[13,905,906],{},"57.0%",[10,908,909,912,913],{},[13,910,911],{},"Paiton leads at every tested concurrency level."," The full task-category results also show a lead across code, reasoning, prose, JSON, file editing, summarization, math and chat. The improvement survives the longer workload instead of depending on a single favorable prompt.",[26,914,915],{},[29,916,148],{"href":682,"ariaDescribedBy":917,"dataFootnoteRef":34,"id":918},[33],"user-content-fnref-1-4",[10,920,921,922],{},"These are aggregate serving rates across the active workload. ",[13,923,924],{},"314.5 tok\u002Fs does not mean each of eight users receives 314.5 tok\u002Fs.",[75,926,928],{"id":927},"from-a-659-second-wait-to-a-195-millisecond-first-token","From a 6.59-second wait to a 195-millisecond first token",[10,930,931],{},"The throughput gain is large. The change in responsiveness under load is larger.",[10,933,934,935,938,939,942,943],{},"At four concurrent requests, median time to first token falls from ",[13,936,937],{},"1,627 ms to 180 ms",". At eight, it falls from ",[13,940,941],{},"6,586 ms to 195 ms: 97% lower",". These timings include queueing, so they describe the wait a client experiences before a response begins.",[26,944,945],{},[29,946,148],{"href":682,"ariaDescribedBy":947,"dataFootnoteRef":34,"id":948},[33],"user-content-fnref-1-5",[10,950,951],{},[812,952],{"alt":953,"src":954},"At eight concurrent requests, median time to first token falls from 6,586 milliseconds with Radiance to 195 milliseconds with Paiton.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002F02-time-to-first-token.webp",[10,956,957],{},[102,958,959],{},"Full 188-request workload. Lower is better. Radiance retains a 10–11 ms TTFT advantage at one and two concurrent requests; Paiton still leads throughput at both levels.",[10,961,962],{},"For a shared coding assistant or local agent endpoint, throughput alone is not enough. An endpoint that produces more tokens but leaves requests waiting to start can still feel slow. Here, the higher concurrency result comes with a substantially shorter wait for that first token.",[75,964,966],{"id":965},"the-same-5-gib-holds-almost-three-times-the-token-capacity","The same 5 GiB holds almost three times the token capacity",[10,968,969,970,973,974,977,978,24,981],{},"With ",[13,971,972],{},"exactly 5 GiB reserved on each engine",", the runtimes report ",[13,975,976],{},"25,746 cache-token slots for Radiance and 74,430 for Paiton",". That is ",[13,979,980],{},"2.89× estimated token capacity inside the same allocation",[26,982,983],{},[29,984,148],{"href":682,"ariaDescribedBy":985,"dataFootnoteRef":34,"id":986},[33],"user-content-fnref-1-6",[10,988,989],{},[812,990],{"alt":991,"src":992},"A 5 GiB cache allocation reports 25,746 token slots for Radiance and 74,430 for Paiton: 2.89 times the estimated capacity.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002F03-cache-capacity.webp",[10,994,995],{},[102,996,997],{},"Runtime-reported shared-cache capacity estimates, not physical VRAM capacity. The per-request context ceiling was 8,192 tokens in this original benchmark configuration.",[10,999,1000,1001,1004,1005],{},"The run logs capture up to ",[13,1002,1003],{},"three active requests for Radiance versus eight for Paiton"," in this matched configuration. Those are observations from these runs, not universal concurrency limits for either engine.",[26,1006,1007],{},[29,1008,148],{"href":682,"ariaDescribedBy":1009,"dataFootnoteRef":34,"id":1010},[33],"user-content-fnref-1-7",[10,1012,1013],{},"The extra capacity helps explain the stronger concurrent experience: more requests can make progress inside the same budget. The capacity figures and active-request observations align with the throughput and latency result, although they do not isolate cache handling as the sole cause of the improvement.",[10,1015,1016],{},[13,1017,1018],{},"More usable serving capacity, not more VRAM or a larger advertised context window.",[75,1020,1022],{"id":1021},"faster-decode-faster-prefill-across-the-workload","Faster decode. Faster prefill. Across the workload.",[10,1024,1025,1026,1029,1030],{},"The gains are not confined to admitting more requests. In the full workload, weighted serial decode rises from ",[13,1027,1028],{},"86.0 to 104.9 output tok\u002Fs: 22% higher",". This measures generation after the first token and is separate from the aggregate complete-workload throughput above.",[26,1031,1032],{},[29,1033,148],{"href":682,"ariaDescribedBy":1034,"dataFootnoteRef":34,"id":1035},[33],"user-content-fnref-1-8",[10,1037,1038],{},[812,1039],{"alt":1040,"src":1041},"Full-workload weighted serial decode rises from 86.0 to 104.9 output tokens per second, a 22 percent increase.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002F04-serial-decode.webp",[10,1043,1044],{},[102,1045,1046],{},"Weighted serial decode, full workload. Same target and draft snapshots on the same R9700.",[10,1048,1049,1050,1053,1054,1057,1058,24,1061],{},"Prefill improves at every tested prompt depth as well. At approximately ",[13,1051,1052],{},"1,556 \u002F 3,024 \u002F 5,226 input tokens",", Radiance measures ",[13,1055,1056],{},"2,828 \u002F 3,003 \u002F 2,926 input tok\u002Fs",". Paiton reaches ",[13,1059,1060],{},"3,182 \u002F 3,521 \u002F 3,367 input tok\u002Fs",[26,1062,1063],{},[29,1064,148],{"href":682,"ariaDescribedBy":1065,"dataFootnoteRef":34,"id":1066},[33],"user-content-fnref-1-9",[10,1068,1069],{},[812,1070],{"alt":1071,"src":1072},"Paiton delivers 3,182, 3,521 and 3,367 input tokens per second at three prompt depths, ahead of Radiance at each.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002F05-full-prefill.webp",[10,1074,1075],{},[102,1076,1077],{},"Full-workload prefill results: 12.5%, 17.2% and 15.1% higher, calculated from the displayed summary values. Actual prompt tokens divided by HTTP time to first token, not isolated kernel throughput.",[10,1079,1080],{},"Together, these results show improvement at several points a user notices: getting the prompt processed, starting the answer under load and generating the rest of it.",[75,1082,1084],{"id":1083},"against-stock-vllm-the-whole-engine-gap-is-striking","Against stock vLLM, the whole-engine gap is striking",[10,1086,1087,1088,1091,1092],{},"We also ran a separate ",[13,1089,1090],{},"54-request matrix with generation capped at 128 tokens",", comparing stock vLLM O2, Radiance + DFlash2 and Paiton on vLLM + DFlash2. The stock profile used the best O2 settings we tested.",[26,1093,1094],{},[29,1095,148],{"href":682,"ariaDescribedBy":1096,"dataFootnoteRef":34,"id":1097},[33],"user-content-fnref-1-10",[117,1099,1100,1113],{},[120,1101,1102],{},[123,1103,1104,1106,1109,1111],{},[126,1105,128],{},[126,1107,1108],{"align":131},"Stock vLLM O2",[126,1110,831],{"align":131},[126,1112,834],{"align":131},[134,1114,1115,1130,1145,1160],{},[123,1116,1117,1119,1122,1125],{},[139,1118,36],{},[139,1120,1121],{"align":131},"4.5 tok\u002Fs",[139,1123,1124],{"align":131},"78.0 tok\u002Fs",[139,1126,1127],{"align":131},[13,1128,1129],{},"90.0 tok\u002Fs",[123,1131,1132,1134,1137,1140],{},[139,1133,148],{},[139,1135,1136],{"align":131},"8.8 tok\u002Fs",[139,1138,1139],{"align":131},"151.2 tok\u002Fs",[139,1141,1142],{"align":131},[13,1143,1144],{},"160.5 tok\u002Fs",[123,1146,1147,1149,1152,1155],{},[139,1148,156],{},[139,1150,1151],{"align":131},"17.4 tok\u002Fs",[139,1153,1154],{"align":131},"189.2 tok\u002Fs",[139,1156,1157],{"align":131},[13,1158,1159],{},"231.5 tok\u002Fs",[123,1161,1162,1164,1167,1170],{},[139,1163,164],{},[139,1165,1166],{"align":131},"33.7 tok\u002Fs",[139,1168,1169],{"align":131},"175.6 tok\u002Fs",[139,1171,1172],{"align":131},[13,1173,1174],{},"328.5 tok\u002Fs",[10,1176,1177,1180],{},[102,1178,1179],{},"Separate 54-request matrix, runs 403 \u002F 493 \u002F 492. Values from the published benchmark summary. This table is not the source of the 57% headline.",[26,1181,1182],{},[29,1183,148],{"href":682,"ariaDescribedBy":1184,"dataFootnoteRef":34,"id":1185},[33],"user-content-fnref-1-11",[10,1187,1188,1189,1192,1193,24,1196],{},"At eight concurrent requests, aggregate throughput rises from ",[13,1190,1191],{},"33.7 tok\u002Fs on stock to 328.5 tok\u002Fs with Paiton",", nearly tenfold. Weighted serial decode in this shorter comparison rises from ",[13,1194,1195],{},"4.5 to 113.5 tok\u002Fs",[26,1197,1198],{},[29,1199,148],{"href":682,"ariaDescribedBy":1200,"dataFootnoteRef":34,"id":1201},[33],"user-content-fnref-1-12",[10,1203,1204,1205,1208,1209,1212],{},"The scope matters: stock uses checkpoint-native ",[13,1206,1207],{},"W4A4 emulation",", while the accelerated configurations use ",[13,1210,1211],{},"W4A8 execution plus DFlash2",". These are complete engine configurations on the same weights, not identical activation arithmetic or a claim that replacing one kernel delivers the entire gain. Nor do these figures describe every stock-vLLM model or quantization.",[10,1214,1215],{},[812,1216],{"alt":1217,"src":1218},"The separate 54-request prefill comparison shows Paiton ahead of stock vLLM O2 and Radiance at all three tested prompt depths.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002F07-three-engine-prefill.webp",[10,1220,1221],{},[102,1222,1223],{},"Prefill in the separate capped matrix, with actual tokenized prompt lengths. Keep these values separate from the full-workload prefill series above.",[10,1225,1226,1227],{},"That is why the main headline uses the harder comparison: ",[13,1228,1229],{},"Paiton against an already accelerated Radiance + DFlash2 baseline, confirmed on the longer workload.",[75,1231,1233],{"id":1232},"regular-vllm-the-installed-library-stays-unchanged","Regular vLLM. The installed library stays unchanged.",[10,1235,1236,1237,1243],{},"The deployment result is deliberate. Paiton integrates through vLLM's extension mechanisms and supplies native HIP runtime artifacts for the optimized execution. vLLM's plugin system is designed to support extensions without modifying its codebase.",[26,1238,1239],{},[29,1240,148],{"href":682,"ariaDescribedBy":1241,"dataFootnoteRef":34,"id":1242},[33],"user-content-fnref-1-13",[26,1244,1245],{},[29,1246,1250],{"href":1247,"ariaDescribedBy":1248,"dataFootnoteRef":34,"id":1249},"#user-content-fn-3",[33],"user-content-fnref-3","5",[10,1252,1253,1254,1257,1263],{},"Paiton's native HIP kernels and integrated DFlash2 support provide the optimized execution on the official vLLM ROCm runtime. The release supplies the runtime artifacts needed for the tested deployment, including the DFlash2 integration, so users can deploy the supported profile without a separate DFlash package. ",[13,1255,1256],{},"Our Paiton compiler remains proprietary.",[26,1258,1259],{},[29,1260,148],{"href":682,"ariaDescribedBy":1261,"dataFootnoteRef":34,"id":1262},[33],"user-content-fnref-1-14",[26,1264,1265],{},[29,1266,1270],{"href":1267,"ariaDescribedBy":1268,"dataFootnoteRef":34,"id":1269},"#user-content-fn-6",[33],"user-content-fnref-6","6",[10,1272,1273,1274,1277,1278,1281,1282],{},"For the original v1.0.0 release, we checked the ordinary ",[70,1275,1276],{},"vllm.entrypoints.openai.api_server"," entry point independently of the benchmark harness. It passed streaming chat, eight concurrent requests, a request at the configured 8K context boundary and a fresh request afterward. ",[13,1279,1280],{},"All 2,893 installed vLLM files matched the official base image."," Validation applies to the pinned tested runtime and supported profile, not every vLLM feature or future version.",[26,1283,1284],{},[29,1285,148],{"href":682,"ariaDescribedBy":1286,"dataFootnoteRef":34,"id":1287},[33],"user-content-fnref-1-15",[10,1289,1290],{},"Standard vLLM still carries its normal framework dependencies; Paiton's native libraries load independently of those frameworks. This is not a claim that the entire serving stack has become framework-free.",[10,1292,1293],{},"Completion and API checks are useful reliability checks. They are not an independent model-accuracy evaluation or a guarantee of identical generated text across execution profiles.",[75,1295,1297],{"id":1296},"more-output-from-the-same-active-hour","More output from the same active hour",[10,1299,1300,1301,1303,1304,1307,1308,1303,1310,1313,1314,24],{},"The full-workload concurrency-eight rates translate into a useful capacity illustration. Sustaining ",[13,1302,669],{}," would produce about ",[13,1305,1306],{},"721,000 output tokens per active hour",". Sustaining ",[13,1309,901],{},[13,1311,1312],{},"1.13 million",", roughly ",[13,1315,1316],{},"411,000 additional output tokens from the same hour",[10,1318,1319,1320,1323,1324,1327,1328,24],{},"Equivalently, one million output tokens would take ",[13,1321,1322],{},"1.387 active hours"," at the baseline rate or ",[13,1325,1326],{},"0.883 hours"," at Paiton's rate: ",[13,1329,1330],{},"36.3% less active time",[10,1332,1333],{},[812,1334],{"alt":1335,"src":1336},"At sustained measured rates, modeled active time per million output tokens falls from 1.387 hours to 0.883 hours: 36.3 percent less.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002F08-active-time-per-million.webp",[10,1338,1339],{},[102,1340,1341],{},"Arithmetic illustration using the full-workload concurrency-eight rates. Assumes those rates are sustained; this is not an hour-long measurement, energy measurement or financial-cost claim.",[10,1343,1344],{},[13,1345,1346],{},"The hardware does not change. How much useful work it can deliver does.",[75,1348,1350],{"id":1349},"original-benchmark-configuration","Original benchmark configuration",[117,1352,1353,1363],{},[120,1354,1355],{},[123,1356,1357,1360],{},[126,1358,1359],{},"Component",[126,1361,1362],{},"Matched profile",[134,1364,1365,1376,1386,1394,1402,1410,1418,1426,1434,1442,1450,1458],{},[123,1366,1367,1370],{},[139,1368,1369],{},"GPU",[139,1371,1372,1373],{},"One Radeon AI PRO R9700, ",[70,1374,1375],{},"gfx1201",[123,1377,1378,1381],{},[139,1379,1380],{},"Target checkpoint",[139,1382,1383],{},[70,1384,1385],{},"amd\u002FQwen3.8-27B-Quark-AWQ-MXFP4",[123,1387,1388,1391],{},[139,1389,1390],{},"Accelerated profiles",[139,1392,1393],{},"Radiance + DFlash2; Paiton on official vLLM + DFlash2",[123,1395,1396,1399],{},[139,1397,1398],{},"Paiton serving base",[139,1400,1401],{},"Official vLLM 0.28 ROCm runtime",[123,1403,1404,1407],{},[139,1405,1406],{},"Cache",[139,1408,1409],{},"FP8 KV; exactly 5 GiB reserved per engine",[123,1411,1412,1415],{},[139,1413,1414],{},"Context ceiling",[139,1416,1417],{},"8,192 tokens per request",[123,1419,1420,1423],{},[139,1421,1422],{},"Concurrent requests tested",[139,1424,1425],{},"1, 2, 4 and 8",[123,1427,1428,1431],{},[139,1429,1430],{},"Speculation",[139,1432,1433],{},"Same target and draft snapshots; seven speculative tokens",[123,1435,1436,1439],{},[139,1437,1438],{},"Sampling",[139,1440,1441],{},"Greedy",[123,1443,1444,1447],{},[139,1445,1446],{},"Prefix caching",[139,1448,1449],{},"Disabled",[123,1451,1452,1455],{},[139,1453,1454],{},"Headline workload",[139,1456,1457],{},"Full 188-request preset with original output budgets",[123,1459,1460,1463],{},[139,1461,1462],{},"Supporting stock comparison",[139,1464,1465],{},"Separate 54-request matrix, output cap 128",[10,1467,1468,1469],{},"These are results for the tested model, runtime and workload. They do not establish a guarantee for other GPUs, larger contexts, different drafters or untested concurrency levels.",[26,1470,1471],{},[29,1472,148],{"href":682,"ariaDescribedBy":1473,"dataFootnoteRef":34,"id":1474},[33],"user-content-fnref-1-16",[75,1476,1478],{"id":1477},"longer-conversations-two-released-profiles","Longer conversations: two released profiles",[10,1480,1481],{},[102,1482,1483,1484,1487],{},"17 September archive: the v1.1.0 profiles below are not the current ROCm 10 images. Use the ",[29,1485,1486],{"href":55},"current launch section"," for the latest release.",[10,1489,1490,1491,1494,1495],{},"The original result answers a serving-performance question at 8K. A practical coding endpoint also needs space for documents, conversation history and tool results. The ",[13,1492,1493],{},"17 September v1.1.0 images"," extend the available serving profiles without changing the pinned model weights or existing native libraries.",[26,1496,1497],{},[29,1498,1502],{"href":1499,"ariaDescribedBy":1500,"dataFootnoteRef":34,"id":1501},"#user-content-fn-8",[33],"user-content-fnref-8","7",[117,1504,1505,1520],{},[120,1506,1507],{},[123,1508,1509,1512,1514,1517],{},[126,1510,1511],{},"Profile",[126,1513,364],{"align":131},[126,1515,1516],{"align":131},"FP8 cache budget",[126,1518,1519],{},"Active-sequence limit",[134,1521,1522,1542],{},[123,1523,1524,1529,1534,1539],{},[139,1525,1526],{},[13,1527,1528],{},"64K v1.1.0",[139,1530,1531],{"align":131},[13,1532,1533],{},"65,536 tokens",[139,1535,1536],{"align":131},[13,1537,1538],{},"5 GiB",[139,1540,1541],{},"Up to 8, subject to available cache",[123,1543,1544,1549,1554,1559],{},[139,1545,1546],{},[13,1547,1548],{},"200K v1.1.0",[139,1550,1551],{"align":131},[13,1552,1553],{},"200,000 tokens",[139,1555,1556],{"align":131},[13,1557,1558],{},"8 GiB",[139,1560,36],{},[10,1562,1563,1566,1567,1570],{},[13,1564,1565],{},"Context is the whole request budget:"," the tokenized prompt, chat and tool formatting, and generated output. Reserve space for the answer. Eight scheduled sequences does ",[13,1568,1569],{},"not"," mean eight full-length 64K conversations fit simultaneously. Longer requests consume more cache and can queue or require recomputation.",[10,1572,1573,1574,1577,1578,1584],{},"These are practical serving profiles, not the model's architectural maximum. The target and drafter declare 262,144 positions, but the entire serving stack must fit alongside weights, cache and working memory. The 200K profile has ",[13,1575,1576],{},"tight VRAM headroom on the 32 GB R9700"," and serves one active request; others wait. Keep its packaged 4,096-token prefill chunk and concurrency settings.",[26,1579,1580],{},[29,1581,1502],{"href":1499,"ariaDescribedBy":1582,"dataFootnoteRef":34,"id":1583},[33],"user-content-fnref-8-2",[26,1585,1586],{},[29,1587,164],{"href":1588,"ariaDescribedBy":1589,"dataFootnoteRef":34,"id":1590},"#user-content-fn-9",[33],"user-content-fnref-9",[10,1592,1593,1594,1597,1598,1601,1602],{},"A final packaged-image check retrieved three markers from a synthetic ",[13,1595,1596],{},"195,999-token prompt",", producing a 57-token answer. That is a bounded functional check of the 200K profile, ",[13,1599,1600],{},"not broad long-context reasoning or coding validation",". It is separate from the throughput benchmarks. Likewise, two submitted 62,983-token prompts on the 64K image passed with queueing; that does not demonstrate simultaneous residency of two full contexts.",[26,1603,1604],{},[29,1605,164],{"href":1588,"ariaDescribedBy":1606,"dataFootnoteRef":34,"id":1607},[33],"user-content-fnref-9-2",[75,1609,1611],{"id":1610},"tool-calls-that-reach-the-client","Tool calls that reach the client",[10,1613,1614,1615,1618,1619,1625,1626,24,1632,1638],{},"The original parser did not match Qwen's XML tool format. Tool syntax could appear as ordinary response text instead of becoming the ",[13,1616,1617],{},"structured API call"," that a client needs to run a tool. The new images correct that server-side parsing with ",[13,1620,1621,1624],{},[70,1622,1623],{},"qwen3_xml"," tool parsing"," and ",[13,1627,1628,1631],{},[70,1629,1630],{},"qwen3"," reasoning parsing",[26,1633,1634],{},[29,1635,1502],{"href":1499,"ariaDescribedBy":1636,"dataFootnoteRef":34,"id":1637},[33],"user-content-fnref-8-3",[26,1639,1640],{},[29,1641,164],{"href":1588,"ariaDescribedBy":1642,"dataFootnoteRef":34,"id":1643},[33],"user-content-fnref-9-3",[10,1645,1646,1649,1650,1653],{},[13,1647,1648],{},"In these 17 September v1.1.0 images, thinking is disabled by default."," Applications can enable it per request with ",[70,1651,1652],{},"\"chat_template_kwargs\": {\"enable_thinking\": true}",". This changes request behavior, not the model weights or native-library bytes.",[10,1655,1656,1657,1662],{},"Both corrected images passed the published streaming\u002Fnon-streaming API tool probes. Separately, the corrected 64K configuration passed an actual OpenCode 1.18.31 file-read\u002Ffile-write check. These are bounded API and client checks, not a claim of general autonomous coding ability. The ",[29,1658,1661],{"href":1659,"rel":1660},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-agentic-64k\u002FREADME.md#separate-long-context-and-tool-checks",[109],"tool-test evidence"," records what was exercised and what was not.",[75,1664,1666],{"id":1665},"the-64k-image-in-a-short-prompt-benchmark","The 64K image in a short-prompt benchmark",[10,1668,1669,1670,1673,1674,47,1677,1680,1681,24,1684],{},"With the corrected defaults, we ran the ",[13,1671,1672],{},"52-request BetterBench quick workload"," on one R9700. At eight concurrent requests, the 64K image delivered ",[13,1675,1676],{},"304.10 aggregate output tok\u002Fs",[13,1678,1679],{},"67.81 median generation tok\u002Fs per request",", and ",[13,1682,1683],{},"350.84 ms median client time to first token",[26,1685,1686],{},[29,1687,164],{"href":1588,"ariaDescribedBy":1688,"dataFootnoteRef":34,"id":1689},[33],"user-content-fnref-9-4",[1691,1692],"benchmark-figure",{"kind":1693},"quick64k",[10,1695,1696],{},"These three measures answer different questions. Per-request generation measures output after the first streamed update; aggregate throughput measures completed output across the concurrency phase, including prefill. Client TTFT measures the wait for the first response, including HTTP and server waiting, but not the client's concurrency semaphore.",[10,1698,1699,1702,1703,1706],{},[13,1700,1701],{},"These are short-prompt results on a 64K-capable image, not generation with 64K prompts."," The concurrency prompts were 69–116 tokens. There were 10 discarded warmups, a 128-token output cap, and no request errors or preemptions. ",[13,1704,1705],{},"37 of 52 measured outputs reached the cap",", so successful requests and these speeds do not establish completed-task quality.",[10,1708,1709,1710,1713,1714,47,1719,1625,1724,24],{},"The new image defaults to thinking disabled; the earlier release used the checkpoint's thinking default. That changes generated content, lengths and speculative acceptance. ",[13,1711,1712],{},"This is not a controlled speedup comparison with the earlier release",", and it does not replace the original 57% result. See the ",[29,1715,1718],{"href":1716,"rel":1717},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-agentic-64k\u002FREADME.md",[109],"full report and methodology",[29,1720,1723],{"href":1721,"rel":1722},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-agentic-64k\u002Fsummary.json",[109],"summary",[29,1725,1728],{"href":1726,"rel":1727},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-agentic-64k\u002Fdata\u002Fresults.json",[109],"per-run results",[75,1730,1732],{"id":1731},"growing-conversations-stop-reprocessing-the-same-history","Growing conversations: stop reprocessing the same history",[10,1734,1735],{},[102,1736,1737,1738,1741],{},"17 September investigation. These measurements and availability statements describe the earlier v1.1.0 images and experimental adapter. The new public chat configuration is documented ",[29,1739,1740],{"href":50},"above","; do not apply the older 200K cache restriction to that separately tested profile.",[10,1743,1744,1745,1748],{},"A coding agent often resends the conversation so far: instructions, earlier answers, file excerpts and tool results, plus one new turn. ",[13,1746,1747],{},"Automatic prefix caching (APC) lets the server reuse an unchanged cached beginning instead of processing that history again."," This can matter more to the wait before an answer than decode speed alone.",[10,1750,1751,1752,1755,1756],{},"Thank you to the ",[13,1753,1754],{},"community contributor who identified this gap and demonstrated the workaround",". Our earlier fresh-prompt benchmarks did not measure the cost of repeatedly processing a growing history. APC is vLLM's existing reuse mechanism; the new investigation measures its integration with Paiton and the tradeoffs.",[26,1757,1758],{},[29,1759,1763],{"href":1760,"ariaDescribedBy":1761,"dataFootnoteRef":34,"id":1762},"#user-content-fn-10",[33],"user-content-fnref-10","9",[10,1765,1766,1767,1770,1771,1774],{},"At approximately 40K prompt tokens, the publicly reproducible ",[13,1768,1769],{},"stock-GDN APC"," path reduced cold-to-repeat response-start latency from ",[13,1772,1773],{},"15.58 to 1.13 seconds",". The compact APC-off path continued to process the repeated history. An experimental native-prefill APC candidate also shortened the wait, including on growing follow-ups.",[1691,1776],{"kind":1777},"apc",[10,1779,1780,1781,1784,1785,1788,1789,1792,1793,1796,1797],{},"The separate approximately 150K experiment used ",[13,1782,1783],{},"an 8 GiB cache, a 160,000-token configured context and one active request",". Its experimental native-prefill adapter reduced cold-to-repeat TTFT from ",[13,1786,1787],{},"88.46 to 2.04 seconds",", independently reproduced with a second distinct prefix at ",[13,1790,1791],{},"88.61 to 2.03 seconds",". Those are ",[13,1794,1795],{},"43–44× shorter response-start waits on cache hits",", not faster decode, a competitor comparison or a default benefit of the published image. The largest tested prompt was 150,645 tokens, so this does not qualify full 160K or 200K operation.",[26,1798,1799],{},[29,1800,1763],{"href":1760,"ariaDescribedBy":1801,"dataFootnoteRef":34,"id":1802},[33],"user-content-fnref-10-2",[171,1804,1806],{"id":1805},"what-was-available-on-17-september","What was available on 17 September",[1808,1809,1810,1817,1828,1837],"ul",{},[1811,1812,1813,1816],"li",{},[13,1814,1815],{},"The 17 September v1.1.0 images default to APC off."," The compact native path does not yet support prefix reuse.",[1811,1818,1819,1827],{},[13,1820,1821,1822,24],{},"Stock-GDN APC can be tried through the ",[29,1823,1826],{"href":1824,"rel":1825},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-prefix-caching\u002FREPRODUCE.md",[109],"published profile override and reproduction instructions"," That recipe uses the pinned 64K image, a 5 GiB cache and one active request.",[1811,1829,1830,1836],{},[13,1831,1832,1835],{},[70,1833,1834],{},"PAITON_PREFIX_CACHING=1"," is not distributed in these images."," Setting that convenience flag alone does not enable APC; use the documented override.",[1811,1838,1839,1842],{},[13,1840,1841],{},"The native-prefill APC adapter is experimental and unreleased."," Its measured results cannot be reproduced from the published image and profile alone.",[171,1844,1846],{"id":1845},"a-workload-choice-with-real-tradeoffs","A workload choice, with real tradeoffs",[10,1848,1849,1850,1853],{},"The current stock-state fallback has ",[13,1851,1852],{},"lower observed decode performance and usable cache capacity"," than the compact APC-off path. Reusing a long history can still save substantial waiting, but fresh prompts, changed early tokens, cache eviction and server restarts require processing again. The 40K comparisons used the same 5 GiB budget; the 150K panel is a separate 8 GiB experiment, not an equal-cache comparison against the compact path.",[10,1855,1856,1859,1860],{},[13,1857,1858],{},"Do not apply stock-state APC to the released 200K profile with its existing 8 GiB cache."," That budget is insufficient at a 200K context limit in the pinned sizing calculation. This was a sizing rejection, not an observed GPU out-of-memory event, and a larger allocation remains untested.",[26,1861,1862],{},[29,1863,1763],{"href":1760,"ariaDescribedBy":1864,"dataFootnoteRef":34,"id":1865},[33],"user-content-fnref-10-3",[10,1867,1868,1869,1874,1875,1878],{},"The published retrieval, cached-tool and growing-history checks are useful evidence, not broad model-quality or robustness certification. The ",[29,1870,1873],{"href":1871,"rel":1872},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-prefix-caching\u002FREADME.md",[109],"APC report"," retains the control arms, cache-miss checks, measured tradeoffs and limitations. The practical lesson is to measure ",[13,1876,1877],{},"the conversation",", not only how quickly the next answer decodes.",[75,1880,1882],{"id":1881},"run-it-on-your-r9700","Run it on your R9700",[10,1884,1885],{},[102,1886,1887,1888,1891],{},"Historical launch commands for reproducing the 16\u002F17 September results only. New deployments should use the ",[29,1889,1890],{"href":55},"ROCm 10 instructions"," above, with the new checkpoint, launchers and API port.",[10,1893,1894,1895,47,1900,1625,1905,1910,1911,24,1914,1922],{},"Use Linux x86-64, Docker and one Radeon AI PRO R9700 with working AMD GPU access. Start with the ",[29,1896,1899],{"href":1897,"rel":1898},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FLAUNCH-agentic-v1.1.0.md",[109],"published launch guide",[29,1901,1904],{"href":1902,"rel":1903},"https:\u002F\u002Fgithub.com\u002Fusers\u002FEliovp\u002Fpackages\u002Fcontainer\u002Fpackage\u002Fpaiton-vllm-plugin",[109],"GHCR package",[29,1906,1909],{"href":1907,"rel":1908},"https:\u002F\u002Fhuggingface.co\u002FEliovpAI\u002FQwen3.8-27B-Quark-AWQ-MXFP4-DFlash2-Paiton-RDNA4",[109],"updated Hugging Face companion",". The companion retains the unchanged v1.0.0 native overlay; the new serving profiles are the v1.1.0 ",[13,1912,1913],{},"GHCR images",[26,1915,1916],{},[29,1917,1921],{"href":1918,"ariaDescribedBy":1919,"dataFootnoteRef":34,"id":1920},"#user-content-fn-5",[33],"user-content-fnref-5","10",[26,1923,1924],{},[29,1925,1502],{"href":1499,"ariaDescribedBy":1926,"dataFootnoteRef":34,"id":1927},[33],"user-content-fnref-8-4",[10,1929,1930,1933],{},[13,1931,1932],{},"Run only one profile on the GPU at a time."," Both new commands share the persistent model-cache volume and bind the API to localhost. First startup downloads and verifies the pinned target and drafter. Wait for startup to finish before using the endpoint; stop the selected container before switching profiles. No compiler checkout or build is required.",[171,1935,1937],{"id":1936},"historical-64k-v110-profile","Historical 64K v1.1.0 profile",[520,1939],{"profile":1940},"64k",[171,1942,1944],{"id":1943},"historical-200k-v110-profile","Historical 200K v1.1.0 profile",[10,1946,1947],{},"One active request, 8 GiB cache, tight VRAM headroom. Keep the packaged concurrency and prefill settings.",[520,1949],{"profile":1950},"200k",[10,1952,1953,1954,1957,1958,531,1961,1964],{},"Check readiness with ",[70,1955,1956],{},"curl --fail http:\u002F\u002F127.0.0.1:8000\u002Fhealth",". Use base URL ",[70,1959,1960],{},"http:\u002F\u002F127.0.0.1:8000\u002Fv1",[70,1962,1963],{},"Qwen3.8-27B-Quark-AWQ-MXFP4",". For example, enable thinking explicitly for one request:",[539,1966,1968],{"className":541,"code":1967,"language":543,"meta":34,"style":34},"curl --fail http:\u002F\u002F127.0.0.1:8000\u002Fv1\u002Fchat\u002Fcompletions \\\n  -H 'Content-Type: application\u002Fjson' \\\n  -d '{\"model\":\"Qwen3.8-27B-Quark-AWQ-MXFP4\",\"messages\":[{\"role\":\"user\",\"content\":\"Explain the tradeoffs of prefix caching.\"}],\"temperature\":0,\"max_tokens\":256,\"stream\":true,\"chat_template_kwargs\":{\"enable_thinking\":true}}'\n",[70,1969,1970,1981,1989],{"__ignoreMap":34},[547,1971,1972,1974,1976,1979],{"class":549,"line":550},[547,1973,554],{"class":553},[547,1975,558],{"class":557},[547,1977,1978],{"class":561}," http:\u002F\u002F127.0.0.1:8000\u002Fv1\u002Fchat\u002Fcompletions",[547,1980,575],{"class":557},[547,1982,1983,1985,1987],{"class":549,"line":565},[547,1984,581],{"class":557},[547,1986,584],{"class":561},[547,1988,575],{"class":557},[547,1990,1991,1993],{"class":549,"line":578},[547,1992,592],{"class":557},[547,1994,1995],{"class":561}," '{\"model\":\"Qwen3.8-27B-Quark-AWQ-MXFP4\",\"messages\":[{\"role\":\"user\",\"content\":\"Explain the tradeoffs of prefix caching.\"}],\"temperature\":0,\"max_tokens\":256,\"stream\":true,\"chat_template_kwargs\":{\"enable_thinking\":true}}'\n",[10,1997,1998,1999,2003],{},"For APC, use the separate ",[29,2000,2002],{"href":1824,"rel":2001},[109],"stock-GDN profile override",", not an unsupported environment flag on these default commands.",[171,2005,2007],{"id":2006},"reproduce-the-original-8k-benchmark","Reproduce the original 8K benchmark",[10,2009,2010,2011,2014,2015,24,2020],{},"The ",[13,2012,2013],{},"v1.0.0 image remains available unchanged",", along with its ",[29,2016,2019],{"href":2017,"rel":2018},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Freleases\u002Ftag\u002Fqwen38-mxfp4-dflash2-rdna4-v1.0.0",[109],"runtime release notes",[26,2021,2022],{},[29,2023,2027],{"href":2024,"ariaDescribedBy":2025,"dataFootnoteRef":34,"id":2026},"#user-content-fn-7",[33],"user-content-fnref-7","11",[520,2029],{"profile":2030},"original",[10,2032,2033,2034,1625,2039,2044,2045,2048],{},"Use the ",[29,2035,2038],{"href":2036,"rel":2037},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Ftree\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2#run-the-v100-benchmark-release",[109],"original model guide",[29,2040,2043],{"href":2041,"rel":2042},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FBENCHMARKS.md",[109],"benchmark settings"," to reproduce the historical comparison. Its 5 GiB cache, 8,192-token limit, target and draft snapshots, thinking behavior, speculative settings and APC-off configuration are part of the result. The repository's historical ",[70,2046,2047],{},"serve.py"," still selects its original release lock, not either new profile.",[75,2050,2052],{"id":2051},"same-hardware-a-much-stronger-serving-result","Same hardware. A much stronger serving result.",[10,2054,2055,2058],{},[13,2056,2057],{},"57% more aggregate throughput. 97% lower median time to first token at eight concurrent requests. 2.89× estimated cache-token capacity."," Those remain the results of the original matched 188-request, 8K comparison on one Radeon AI PRO R9700 through regular vLLM.",[10,2060,2061],{},"Those historical steps led to the current ROCm 10 release: 400.7 aggregate tok\u002Fs in its separate 65K test and a publicly runnable experimental chat configuration for longer histories. Different checkpoints and workloads remain separate; neither replaces the matched scope of the original 57% comparison.",[10,2063,2064],{},"For local developers, that means a stronger shared endpoint from a single workstation GPU. For teams running AMD inference at scale, it is another demonstration of why execution efficiency matters alongside hardware capacity. The R9700 numbers are not a prediction of gains on other AMD platforms.",[10,2066,2067,2068,2072],{},"Running an AMD inference workload that should be delivering more? Talk to us about ",[29,2069,2071],{"href":2070},"\u002Fproducts\u002Fpaiton","Paiton",". Bring the model, workload and current baseline.",[171,2074,2076],{"id":2075},"credit-where-it-belongs","Credit where it belongs",[10,2078,1751,2079,2083],{},[29,2080,2082],{"href":738,"rel":2081},[109],"Radiance team"," for their excellent work on AMD inference and for providing the inspiration and a strong comparison point for this investigation.",[10,2085,2086,2087,47,2092,2097,2098,2103],{},"We also thank ",[29,2088,2091],{"href":2089,"rel":2090},"https:\u002F\u002Fgithub.com\u002Fvllm-project\u002Fvllm",[109],"vLLM",[29,2093,2096],{"href":2094,"rel":2095},"https:\u002F\u002Fcodeberg.org\u002FStillDeadcode\u002Flibr4d",[109],"StillDeadcode\u002Flibr4d",", the Qwen and DFlash2 teams, AMD's Quark checkpoint team and ",[29,2099,2102],{"href":2100,"rel":2101},"https:\u002F\u002Fgithub.com\u002FGGZ14\u002FBetterBench",[109],"BetterBench"," for the foundations, models and measurement tools that support this work.",[10,2105,2010,2106,2109],{},[29,2107,762],{"href":760,"rel":2108},[109]," identify the adapted Radiance\u002Flibr4d techniques and applicable component terms. This article describes published serving behavior and measurements only. Proprietary compiler documentation and implementation details remain private.",[171,2111,2113],{"id":2112},"continue-the-community-discussion","Continue the community discussion",[10,2115,2116,2117,2122],{},"Questions, deployment experiences or a workload we should measure next? Join the ",[29,2118,2121],{"href":2119,"rel":2120},"https:\u002F\u002Fwww.reddit.com\u002Fr\u002FROCm\u002Fcomments\u002F1whv60r\u002Fqwen38_27b_on_one_r9700_two_paitonvllm_benchmark\u002F",[109],"r\u002FROCm discussion about these Qwen3.8 results",". The thread discusses two separate comparisons; each retains its own methodology and is not interchangeable with the 17 September quick run or APC experiments.",[171,2124,2126],{"id":2125},"sources-and-benchmark-references","Sources and benchmark references",[2128,2129,2132,2137],"section",{"className":2130,"dataFootnotes":34},[2131],"footnotes",[75,2133,2136],{"className":2134,"id":33},[2135],"sr-only","Footnotes",[2138,2139,2140,2176,2300,2313,2325,2339,2349,2388,2435,2475,2487],"ol",{},[1811,2141,2143,47,2148,2153,2154,2161,2162,2161,2169],{"id":2142},"user-content-fn-11",[29,2144,2147],{"href":2145,"rel":2146},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002F8f56157c05eb6a53f6cdab00a115e47b42fed2d1\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002FREADME.md",[109],"Current release, benchmark tables and long-context checks",[29,2149,2152],{"href":2150,"rel":2151},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002F8f56157c05eb6a53f6cdab00a115e47b42fed2d1\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Flaunch-rocm10.py",[109],"public launcher and image pins",". Release dated 18 September; reviewed 19 September 2026. Charts are reproduced from the published rounded tables, not an independently rerun benchmark. ",[29,2155,2160],{"href":2156,"ariaLabel":2157,"className":2158,"dataFootnoteBackref":34},"#user-content-fnref-11","Back to reference 1",[2159],"data-footnote-backref","↩"," ",[29,2163,2160,2167],{"href":2164,"ariaLabel":2165,"className":2166,"dataFootnoteBackref":34},"#user-content-fnref-11-2","Back to reference 1-2",[2159],[26,2168,148],{},[29,2170,2160,2174],{"href":2171,"ariaLabel":2172,"className":2173,"dataFootnoteBackref":34},"#user-content-fnref-11-3","Back to reference 1-3",[2159],[26,2175,752],{},[1811,2177,2179,2180,2184,2185,2161,2190,2161,2197,2161,2204,2161,2211,2161,2218,2161,2225,2161,2232,2161,2239,2161,2246,2161,2253,2161,2260,2161,2268,2161,2276,2161,2284,2161,2292],{"id":2178},"user-content-fn-1","ElioVP's supplied benchmark summary and charts; accompanying ",[29,2181,2183],{"href":2041,"rel":2182},[109],"Paiton benchmark evidence",". Full-workload runs 495 \u002F 494; capped comparison runs 403 \u002F 493 \u002F 492. The original 16 September figures come from the supplied summary and charts rather than a raw request log. The new 17 September figures use the public data linked below. ",[29,2186,2160],{"href":2187,"ariaLabel":2188,"className":2189,"dataFootnoteBackref":34},"#user-content-fnref-1","Back to reference 2",[2159],[29,2191,2160,2195],{"href":2192,"ariaLabel":2193,"className":2194,"dataFootnoteBackref":34},"#user-content-fnref-1-2","Back to reference 2-2",[2159],[26,2196,148],{},[29,2198,2160,2202],{"href":2199,"ariaLabel":2200,"className":2201,"dataFootnoteBackref":34},"#user-content-fnref-1-3","Back to reference 2-3",[2159],[26,2203,752],{},[29,2205,2160,2209],{"href":2206,"ariaLabel":2207,"className":2208,"dataFootnoteBackref":34},"#user-content-fnref-1-4","Back to reference 2-4",[2159],[26,2210,156],{},[29,2212,2160,2216],{"href":2213,"ariaLabel":2214,"className":2215,"dataFootnoteBackref":34},"#user-content-fnref-1-5","Back to reference 2-5",[2159],[26,2217,1250],{},[29,2219,2160,2223],{"href":2220,"ariaLabel":2221,"className":2222,"dataFootnoteBackref":34},"#user-content-fnref-1-6","Back to reference 2-6",[2159],[26,2224,1270],{},[29,2226,2160,2230],{"href":2227,"ariaLabel":2228,"className":2229,"dataFootnoteBackref":34},"#user-content-fnref-1-7","Back to reference 2-7",[2159],[26,2231,1502],{},[29,2233,2160,2237],{"href":2234,"ariaLabel":2235,"className":2236,"dataFootnoteBackref":34},"#user-content-fnref-1-8","Back to reference 2-8",[2159],[26,2238,164],{},[29,2240,2160,2244],{"href":2241,"ariaLabel":2242,"className":2243,"dataFootnoteBackref":34},"#user-content-fnref-1-9","Back to reference 2-9",[2159],[26,2245,1763],{},[29,2247,2160,2251],{"href":2248,"ariaLabel":2249,"className":2250,"dataFootnoteBackref":34},"#user-content-fnref-1-10","Back to reference 2-10",[2159],[26,2252,1921],{},[29,2254,2160,2258],{"href":2255,"ariaLabel":2256,"className":2257,"dataFootnoteBackref":34},"#user-content-fnref-1-11","Back to reference 2-11",[2159],[26,2259,2027],{},[29,2261,2160,2265],{"href":2262,"ariaLabel":2263,"className":2264,"dataFootnoteBackref":34},"#user-content-fnref-1-12","Back to reference 2-12",[2159],[26,2266,2267],{},"12",[29,2269,2160,2273],{"href":2270,"ariaLabel":2271,"className":2272,"dataFootnoteBackref":34},"#user-content-fnref-1-13","Back to reference 2-13",[2159],[26,2274,2275],{},"13",[29,2277,2160,2281],{"href":2278,"ariaLabel":2279,"className":2280,"dataFootnoteBackref":34},"#user-content-fnref-1-14","Back to reference 2-14",[2159],[26,2282,2283],{},"14",[29,2285,2160,2289],{"href":2286,"ariaLabel":2287,"className":2288,"dataFootnoteBackref":34},"#user-content-fnref-1-15","Back to reference 2-15",[2159],[26,2290,2291],{},"15",[29,2293,2160,2297],{"href":2294,"ariaLabel":2295,"className":2296,"dataFootnoteBackref":34},"#user-content-fnref-1-16","Back to reference 2-16",[2159],[26,2298,2299],{},"16",[1811,2301,2303,2307,2308],{"id":2302},"user-content-fn-2",[29,2304,2306],{"href":738,"rel":2305},[109],"vLLM-Radiance project documentation",", including the same AMD Quark MXFP4 target, DFlash2 profile and explicitly dual-R9700 published measurements. Those external measurements are context, not the denominator of our headline. ",[29,2309,2160],{"href":2310,"ariaLabel":2311,"className":2312,"dataFootnoteBackref":34},"#user-content-fnref-2","Back to reference 3",[2159],[1811,2314,2316,2319,2320],{"id":2315},"user-content-fn-4",[29,2317,2102],{"href":2100,"rel":2318},[109],". The request counts, selected workload settings and results above come from our supplied run summary. ",[29,2321,2160],{"href":2322,"ariaLabel":2323,"className":2324,"dataFootnoteBackref":34},"#user-content-fnref-4","Back to reference 4",[2159],[1811,2326,2328,2333,2334],{"id":2327},"user-content-fn-3",[29,2329,2332],{"href":2330,"rel":2331},"https:\u002F\u002Fdocs.vllm.ai\u002Fen\u002Flatest\u002Fdesign\u002Fplugin_system\u002F",[109],"Official vLLM plugin-system documentation",". ",[29,2335,2160],{"href":2336,"ariaLabel":2337,"className":2338,"dataFootnoteBackref":34},"#user-content-fnref-3","Back to reference 5",[2159],[1811,2340,2342,2333,2344],{"id":2341},"user-content-fn-6",[29,2343,2071],{"href":2070},[29,2345,2160],{"href":2346,"ariaLabel":2347,"className":2348,"dataFootnoteBackref":34},"#user-content-fnref-6","Back to reference 6",[2159],[1811,2350,2352,1625,2356,2361,2362,2161,2367,2161,2374,2161,2381],{"id":2351},"user-content-fn-8",[29,2353,2355],{"href":1897,"rel":2354},[109],"Published 64K\u002F200K v1.1.0 launch instructions",[29,2357,2360],{"href":2358,"rel":2359},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fruntime.agentic-v1.1.0.lock.json",[109],"immutable image lock",". Reviewed 17 September 2026. ",[29,2363,2160],{"href":2364,"ariaLabel":2365,"className":2366,"dataFootnoteBackref":34},"#user-content-fnref-8","Back to reference 7",[2159],[29,2368,2160,2372],{"href":2369,"ariaLabel":2370,"className":2371,"dataFootnoteBackref":34},"#user-content-fnref-8-2","Back to reference 7-2",[2159],[26,2373,148],{},[29,2375,2160,2379],{"href":2376,"ariaLabel":2377,"className":2378,"dataFootnoteBackref":34},"#user-content-fnref-8-3","Back to reference 7-3",[2159],[26,2380,752],{},[29,2382,2160,2386],{"href":2383,"ariaLabel":2384,"className":2385,"dataFootnoteBackref":34},"#user-content-fnref-8-4","Back to reference 7-4",[2159],[26,2387,156],{},[1811,2389,2391,2395,2396,47,2399,1625,2403,2408,2409,2161,2414,2161,2421,2161,2428],{"id":2390},"user-content-fn-9",[29,2392,2394],{"href":1716,"rel":2393},[109],"64K image quick benchmark and tool-support report",", its ",[29,2397,1723],{"href":1721,"rel":2398},[109],[29,2400,2402],{"href":1726,"rel":2401},[109],"results",[29,2404,2407],{"href":2405,"rel":2406},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-agentic-64k\u002Fassets\u002Fconcurrency-64k-quick.svg",[109],"source SVG",". New figures are redrawn from the published data in ElioVP's visual style. ",[29,2410,2160],{"href":2411,"ariaLabel":2412,"className":2413,"dataFootnoteBackref":34},"#user-content-fnref-9","Back to reference 8",[2159],[29,2415,2160,2419],{"href":2416,"ariaLabel":2417,"className":2418,"dataFootnoteBackref":34},"#user-content-fnref-9-2","Back to reference 8-2",[2159],[26,2420,148],{},[29,2422,2160,2426],{"href":2423,"ariaLabel":2424,"className":2425,"dataFootnoteBackref":34},"#user-content-fnref-9-3","Back to reference 8-3",[2159],[26,2427,752],{},[29,2429,2160,2433],{"href":2430,"ariaLabel":2431,"className":2432,"dataFootnoteBackref":34},"#user-content-fnref-9-4","Back to reference 8-4",[2159],[26,2434,156],{},[1811,2436,2438,47,2442,47,2447,1625,2451,2455,2456,2161,2461,2161,2468],{"id":2437},"user-content-fn-10",[29,2439,2441],{"href":1871,"rel":2440},[109],"APC investigation",[29,2443,2446],{"href":2444,"rel":2445},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-prefix-caching\u002Fassets\u002Fplot-data.json",[109],"published plot data",[29,2448,2407],{"href":2449,"rel":2450},"https:\u002F\u002Fgithub.com\u002FEliovp-BV\u002Fpaiton-vllm-plugin\u002Fblob\u002Fmain\u002Fmodels\u002FQwen3.8-MXFP4-DFlash2\u002Fbenchmarks\u002F2026-09-17-prefix-caching\u002Fassets\u002Fprefix-reuse-latency.svg",[109],[29,2452,2454],{"href":1824,"rel":2453},[109],"working stock-APC reproduction instructions",". Native-prefill APC results require an unreleased adapter. ",[29,2457,2160],{"href":2458,"ariaLabel":2459,"className":2460,"dataFootnoteBackref":34},"#user-content-fnref-10","Back to reference 9",[2159],[29,2462,2160,2466],{"href":2463,"ariaLabel":2464,"className":2465,"dataFootnoteBackref":34},"#user-content-fnref-10-2","Back to reference 9-2",[2159],[26,2467,148],{},[29,2469,2160,2473],{"href":2470,"ariaLabel":2471,"className":2472,"dataFootnoteBackref":34},"#user-content-fnref-10-3","Back to reference 9-3",[2159],[26,2474,752],{},[1811,2476,2478,2333,2482],{"id":2477},"user-content-fn-5",[29,2479,2481],{"href":1907,"rel":2480},[109],"Paiton release companion on Hugging Face",[29,2483,2160],{"href":2484,"ariaLabel":2485,"className":2486,"dataFootnoteBackref":34},"#user-content-fnref-5","Back to reference 10",[2159],[1811,2488,2490,2494,2495],{"id":2489},"user-content-fn-7",[29,2491,2493],{"href":2017,"rel":2492},[109],"Paiton Qwen3.8 MXFP4 + DFlash2 runtime release",", including the runtime package and release notes. ",[29,2496,2160],{"href":2497,"ariaLabel":2498,"className":2499,"dataFootnoteBackref":34},"#user-content-fnref-7","Back to reference 11",[2159],[2501,2502,2503],"style",{},"html pre.shiki code .sScJk, html code.shiki .sScJk{--shiki-default:#6F42C1;--shiki-dark:#B392F0}html pre.shiki code .sj4cs, html code.shiki .sj4cs{--shiki-default:#005CC5;--shiki-dark:#79B8FF}html pre.shiki code .sZZnC, html code.shiki .sZZnC{--shiki-default:#032F62;--shiki-dark:#9ECBFF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"title":34,"searchDepth":565,"depth":565,"links":2505},[2506,2510,2511,2515,2516,2517,2518,2519,2520,2521,2522,2523,2524,2525,2526,2527,2528,2529,2533,2538,2543],{"id":77,"depth":565,"text":78,"children":2507},[2508,2509],{"id":173,"depth":578,"text":174},{"id":262,"depth":578,"text":263},{"id":348,"depth":565,"text":349},{"id":488,"depth":565,"text":489,"children":2512},[2513,2514],{"id":514,"depth":578,"text":515},{"id":598,"depth":578,"text":599},{"id":639,"depth":565,"text":640},{"id":694,"depth":565,"text":695},{"id":731,"depth":565,"text":732},{"id":787,"depth":565,"text":788},{"id":927,"depth":565,"text":928},{"id":965,"depth":565,"text":966},{"id":1021,"depth":565,"text":1022},{"id":1083,"depth":565,"text":1084},{"id":1232,"depth":565,"text":1233},{"id":1296,"depth":565,"text":1297},{"id":1349,"depth":565,"text":1350},{"id":1477,"depth":565,"text":1478},{"id":1610,"depth":565,"text":1611},{"id":1665,"depth":565,"text":1666},{"id":1731,"depth":565,"text":1732,"children":2530},[2531,2532],{"id":1805,"depth":578,"text":1806},{"id":1845,"depth":578,"text":1846},{"id":1881,"depth":565,"text":1882,"children":2534},[2535,2536,2537],{"id":1936,"depth":578,"text":1937},{"id":1943,"depth":578,"text":1944},{"id":2006,"depth":578,"text":2007},{"id":2051,"depth":565,"text":2052,"children":2539},[2540,2541,2542],{"id":2075,"depth":578,"text":2076},{"id":2112,"depth":578,"text":2113},{"id":2125,"depth":578,"text":2126},{"id":33,"depth":565,"text":2136},[2071,2545,2091,536,2546,2547],"AMD Radeon","Inference Optimization","DFlash2","2026-09-16T07:30:00Z","Qwen3.8 on one R9700: 400.7 aggregate tok\u002Fs with ROCm 10 and vLLM 0.29, plus public 200K\u002F220K chat profiles. Benchmarks, limits and launch commands.","md","400.7 aggregate tok\u002Fs. One Radeon. Regular vLLM.","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-qwen38-mxfp4\u002Fupdate-2026-09-19\u002Fhero.webp",{},"https:\u002F\u002Feliovp.com\u002Fblog\u002Fpaiton-qwen38-mxfp4-dflash2-r9700","\u002Fblog\u002Fpaiton-qwen38-mxfp4-dflash2-r9700",{"title":5,"description":2549},"paiton-qwen38-mxfp4-dflash2-r9700","blog\u002Fpaiton-qwen38-mxfp4-dflash2-r9700","2026-09-19T00:00:00Z","ch3zyRh0h51LlTV00hnX_Q3BcnKLWKaq2jbH8jlRwWc",[2562,2564,2575,2587,2597,2612,2621,2636,2667,2679,2701,2719,2738,2756,2774,2791,2803,2819,2834,2846,2855,2863,2878,2890,2901,2912,2922,2935,2945,2958,2969,2979,2990,2999,3011,3022,3031],{"path":2555,"title":5,"description":2549,"date":2548,"slug":2557,"image":2552,"originalUrl":2554,"categories":2563},[2071,2545,2091,536,2546,2547],{"path":2565,"title":2566,"description":2567,"date":2568,"slug":2569,"image":2570,"originalUrl":2571,"categories":2572},"\u002Fblog\u002Fpaiton-qwen38-neo-gguf-vllm-r9700","Qwen3.8 GGUF in vLLM: Faster Responses on One Radeon","Run the original NEO CODER MAX GGUF in vLLM with Paiton on an R9700. Explore measured latency gains, image input and local deployment.","2026-09-14T07:30:00Z","paiton-qwen38-neo-gguf-vllm-r9700","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-neo-gguf\u002F00-hero-neo-gguf-r9700.webp","https:\u002F\u002Feliovp.com\u002Fblog\u002Fpaiton-qwen38-neo-gguf-vllm-r9700",[2071,2545,2573,2574,2091],"Local AI","GGUF",{"path":2576,"title":2577,"description":2578,"date":2579,"slug":2580,"image":2581,"originalUrl":2582,"categories":2583},"\u002Fblog\u002Fpaiton-minimax-h3-radeon-ai-pro-r9700","MiniMax H3 on Radeon: 15-Second Video With Native Sound","Paiton generates a 15-second MiniMax H3 video with stereo audio on one Radeon AI PRO R9700 in 5m 33s, with 16.7% lower latency than matched stock.","2026-09-09T07:30:00Z","paiton-minimax-h3-radeon-ai-pro-r9700","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-minimax-h3\u002F00-featured-minimax-h3-r9700.webp","https:\u002F\u002Feliovp.com\u002Fblog\u002Fpaiton-minimax-h3-radeon-ai-pro-r9700",[2071,2545,2573,2584,2585,2586],"Video Generation","MiniMax H3","ComfyUI",{"path":2588,"title":2589,"description":2590,"date":2591,"slug":2592,"image":2593,"originalUrl":2588,"categories":2594},"\u002Fblog\u002Fpaiton-flux2-klein-radeon-ai-pro-r9700","Local FLUX.2 klein on Radeon AI PRO R9700: Faster Image Generation with Less VRAM","Paiton generates 1024×1024 FLUX.2 klein images in 1.054 seconds on an R9700: 16.2% lower warm latency and 33.4% lower peak Torch allocation.","2026-09-07T09:00:00","paiton-flux2-klein-radeon-ai-pro-r9700","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-flux2-klein\u002Ffox-paiton.webp",[2071,2545,2573,2595,2596,2586],"Image Generation","FLUX",{"path":2598,"title":2599,"description":2600,"date":2601,"slug":2602,"image":2603,"originalUrl":2604,"categories":2605},"\u002Fblog\u002Fpaiton-ornith15-radeon-ai-pro-r9700","Ornith 1.5 at 44.6 tok\u002Fs on One Radeon AI PRO R9700","Paiton serves Ornith 1.5 35B A3B at 44.63 output tok\u002Fs on one Radeon AI PRO R9700, 27% faster than tuned stock vLLM, with 21.3% lower modeled cost.","2026-09-05T09:00:00","paiton-ornith15-radeon-ai-pro-r9700","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-ornith15\u002F00-featured-ornith15-r9700.webp","https:\u002F\u002Feliovp.com\u002Fblog\u002Fpaiton-ornith15-radeon-ai-pro-r9700",[2071,2606,2545,2607,2608,2609,2546,2610,2091,2611],"Artificial Intelligence","AI Inference","GPU Performance","Inference Latency","Large Language Models","Cost Efficiency",{"path":2613,"title":2614,"description":2615,"date":2616,"slug":2617,"image":2618,"originalUrl":2619,"categories":2620},"\u002Fblog\u002Fpaiton-qwen38-radeon-ai-pro-r9700","Paiton: 21% More Qwen3.8 Throughput on Radeon AI PRO R9700","Paiton serves AMD’s Qwen3.8 27B at 39.77 output tokens\u002Fs on one Radeon AI PRO R9700, delivering 21% more throughput and 17.4% lower modeled cost.","2026-09-04T09:00:00","paiton-qwen38-radeon-ai-pro-r9700","\u002Fasset\u002Fimages\u002Fblog\u002Fpaiton-r9700\u002F00-featured-paiton-r9700.webp","https:\u002F\u002Feliovp.com\u002Fblog\u002Fpaiton-qwen38-radeon-ai-pro-r9700",[2071,2606,2545,2607,2608,2609,2546,2610,2091,2611],{"path":2622,"title":2623,"description":2624,"date":2625,"slug":2626,"image":2627,"originalUrl":2628,"categories":2629},"\u002Fblog\u002Fai-data-center-power-requirements-gpu-per-megawatt","AI Data Center Power Requirements: The GPU\u002FMW Illusion","Why do AI data-center proposals quote different GPU capacities? See how PUE, peak loads, storage, networking and cooling determine deployable compute.","2026-07-27T23:52:00","ai-data-center-power-requirements-gpu-per-megawatt","\u002Fasset\u002Fimages\u002Fblog\u002Fai-data-center-power-requirements-gpu-per-megawatt\u002Fgpu-per-megawatt-illusion.webp",null,[2630,2631,2632,2633,2634,2635],"All","AI Infrastructure","Data Centers","ModFlex","HPC","AMD Helios",{"path":2637,"title":2638,"description":2639,"date":2640,"slug":2641,"image":2642,"originalUrl":2643,"categories":2644},"\u002Fblog\u002Fpaiton-returns-to-its-diffusion-roots-optimizing-wan2-2-t2v-a14b-on-amd-mi355x","Wan2.2 Video Generation: Paiton on AMD MI355X","Compare Wan2.2-T2V-A14B video generation on AMD MI355X with Paiton and NVIDIA B200 using Diffusers, and explore our diffusion optimization approach.","2026-06-10T14:04:04","paiton-returns-to-its-diffusion-roots-optimizing-wan2-2-t2v-a14b-on-amd-mi355x","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fpaitonwan2.webp","https:\u002F\u002Feliovp.com\u002Fpaiton-returns-to-its-diffusion-roots-optimizing-wan2-2-t2v-a14b-on-amd-mi355x\u002F",[2630,2606,2071,2645,2646,2647,2648,2649,2650,2651,2652,2653,1369,2654,2655,2656,2657,2658,2659,2660,2071,2661,2662,2663,2664,2665,2666],"14B","AMD","B200","Benchmarks","Blackwell","Compute","Diffusion","Eliovp","Generative AI","Hardware","Inference","Instinct","MI355x","NVidia","On-Premise","Optimization","Sovereign AI","T2V","Text-to-Video","Tuning","Video-Generation","Wan2.2",{"path":2668,"title":2669,"description":2670,"date":2671,"slug":2672,"image":2673,"originalUrl":2674,"categories":2675},"\u002Fblog\u002Ffrom-the-attic-to-the-front-page-eliovp-recognized-as-a-pioneer-in-chip-optimization-data-center-infrastructure","ElioVP in De Tijd: Chip Optimization and Data Centers","Read about De Tijd's coverage of ElioVP, from its origins in chip optimization to its work on modular data centers and high-density cooling.","2026-02-10T20:48:12","from-the-attic-to-the-front-page-eliovp-recognized-as-a-pioneer-in-chip-optimization-data-center-infrastructure","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fphysicalnewspaper.webp","https:\u002F\u002Feliovp.com\u002Ffrom-the-attic-to-the-front-page-eliovp-recognized-as-a-pioneer-in-chip-optimization-data-center-infrastructure\u002F",[2630,2606,2676,2677,2646,2678,2676,2658],"Modular DC","Uncategorized","De Tijd",{"path":2680,"title":2681,"description":2682,"date":2683,"slug":2684,"image":2685,"originalUrl":2686,"categories":2687},"\u002Fblog\u002Fprivacy-is-geen-it-probleem-meer-het-is-een-strategische-prioriteit","AI Privacy: A Strategic Priority for Benelux Businesses","Explore generative AI privacy risks, trust, data retention and governance, and why Benelux businesses need a strategic approach to secure AI.","2026-01-29T13:51:11","privacy-is-geen-it-probleem-meer-het-is-een-strategische-prioriteit","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fheaderimage.webp","https:\u002F\u002Feliovp.com\u002Fprivacy-is-geen-it-probleem-meer-het-is-een-strategische-prioriteit\u002F",[2630,2606,2688,2677,2689,2690,2691,2692,2693,2694,2695,2696,2652,2697,2653,2698,2699,2700],"Trending","AI Act","Anthropomorphism","AVG","Benelux","ChatGPT","Cybersecurity","Data Governance","Data Security","GDPR","Microsoft Copilot","Privacy","Shadow AI",{"path":2702,"title":2703,"description":2704,"date":2705,"slug":2706,"image":2707,"originalUrl":2708,"categories":2709},"\u002Fblog\u002Fitsme-bij-ons-is-het-its-not-me-en-dit-is-waarom","Why We Do Not Use itsme: Privacy and Data Sovereignty","Why ElioVP does not use itsme: our assessment of identity metadata, cloud dependence, data sovereignty and authentication risks.","2025-11-27T09:32:14","itsme-bij-ons-is-het-its-not-me-en-dit-is-waarom","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Ffrontimage.webp","https:\u002F\u002Feliovp.com\u002Fitsme-bij-ons-is-het-its-not-me-en-dit-is-waarom\u002F",[2630,2710,2711,2712,2694,2713,2714,2715,2697,2716,2717,2718,2699],"AWS","Belgian Mobile ID","Cloud Act","Data Sovereignty","Digital Identity","eIDAS","itsme","Liberty Global","MyGov.be",{"path":2720,"title":2721,"description":2722,"date":2723,"slug":2724,"image":2725,"originalUrl":2726,"categories":2727},"\u002Fblog\u002Ffield-report-the-reality-of-building-agentic-ai-in-2025","Field Report. The Reality of Building Agentic AI in 2025","Lessons from building on-premise AI agents in 2025 cover workflow design, observability, model training, hallucinations and GPU memory limits.","2025-11-25T14:03:39","field-report-the-reality-of-building-agentic-ai-in-2025","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Ffieldreport.webp","https:\u002F\u002Feliovp.com\u002Ffield-report-the-reality-of-building-agentic-ai-in-2025\u002F",[2630,2606,2728,2688,2729,2730,2731,2732,2733,2734,2735,2736,2661,2737],"Solutions","Agentic AI","AI Engineering","AI Strategy","Autonomous Agents","Enterprise AI","Local LLM","Model Fine-Tuning","On-Premise AI","VRAM Optimization",{"path":2739,"title":2740,"description":2741,"date":2742,"slug":2743,"image":2744,"originalUrl":2745,"categories":2746},"\u002Fblog\u002Fthe-synthetic-unicorn-bubble","The Synthetic Unicorn Bubble","An analysis of AI neocloud investment risks, examining circular financing, infrastructure claims, contract terms and due diligence.","2025-11-24T19:22:28","the-synthetic-unicorn-bubble","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fsyntheticunicorn.webp","https:\u002F\u002Feliovp.com\u002Fthe-synthetic-unicorn-bubble\u002F",[2630,2606,2688,2631,2747,2748,2749,2750,2751,2752,2753,2754,2755],"AI Neocloud","Circular Financing","GPU Cloud","Investment Risks","Startup Valuation","Synthetic Bubble","Tech Analysis","Vaporware","Venture Capital",{"path":2757,"title":2758,"description":2759,"date":2760,"slug":2761,"image":2762,"originalUrl":2763,"categories":2764},"\u002Fblog\u002Fbuilding-the-engine-for-the-ai-race-the-4-month-path-to-nvidia-gb300-nvl72-power","NVIDIA GB300 NVL72: A Four-Month Modular Data Center Plan","Explore a modular data center design for NVIDIA GB300 NVL72, covering redundant power, hybrid cooling and a four-month deployment plan.","2025-11-20T14:10:19","building-the-engine-for-the-ai-race-the-4-month-path-to-nvidia-gb300-nvl72-power","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fsuperpodmodflexfrontimage.webp","https:\u002F\u002Feliovp.com\u002Fbuilding-the-engine-for-the-ai-race-the-4-month-path-to-nvidia-gb300-nvl72-power\u002F",[2630,2676,2677,2765,2631,2766,2767,2768,2769,2770,2771,2772,2773],"150kW Rack","DLC","High Density","Liquid Cooling","Modular Data Center","NVIDIA Blackwell Ultra","NVIDIA GB300","NVL72","Rapid Deployment",{"path":2775,"title":2776,"description":2777,"date":2778,"slug":2779,"image":2780,"originalUrl":2781,"categories":2782},"\u002Fblog\u002Fwhy-cuda-translation-wont-unlock-amds-real-potential","Why “CUDA” Translation Won’t Unlock AMD’s Real Potential","Why CUDA compatibility is not the same as AMD performance: explore ROCm, HIP, kernel tuning and the case for hardware-specific optimization.","2025-11-12T14:48:37","why-cuda-translation-wont-unlock-amds-real-potential","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fchatgpt-image-nov-11-2025-09_16_10-pm-1.webp","https:\u002F\u002Feliovp.com\u002Fwhy-cuda-translation-wont-unlock-amds-real-potential\u002F",[2630,2606,2071,2677,2783,2606,2784,2785,2786,2787,2788,2789,2071,2790],"AMD MI300X","CUDA Translation","FP8","GPU Optimization","High Performance Computing","HIP","Kernel Tuning","ROCm",{"path":2792,"title":2793,"description":2794,"date":2795,"slug":2796,"image":2797,"originalUrl":2798,"categories":2799},"\u002Fblog\u002Fpaiton-the-simplest-way-to-supercharge-ai-inference","Paiton: The Simplest Way to Supercharge AI Inference","Learn how Paiton integrates with existing inference stacks, with AMD MI300X benchmark results and performance-per-dollar comparisons.","2025-11-11T10:31:22","paiton-the-simplest-way-to-supercharge-ai-inference","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fpaiton-powaaah.webp","https:\u002F\u002Feliovp.com\u002Fpaiton-the-simplest-way-to-supercharge-ai-inference\u002F",[2630,2606,2071,2607,2800,2783,2611,2801,2546,2789,2071,2802,2091],"AMD Instinct","High Throughput","SGLang",{"path":2804,"title":2805,"description":2806,"date":2807,"slug":2808,"image":2809,"originalUrl":2810,"categories":2811},"\u002Fblog\u002Fstop-overpaying-paiton-mi300x-moe-beats-h200-b200-on-1m-tokens","Paiton MoE Benchmarks: MI300X vs H200 and B200","Compare Qwen3-30B-A3B MoE inference with Paiton on MI300X against H200 and B200, including throughput and cost per million tokens.","2025-09-26T13:36:18","stop-overpaying-paiton-mi300x-moe-beats-h200-b200-on-1m-tokens","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fhulkvshulkpaitonwins.webp","https:\u002F\u002Feliovp.com\u002Fstop-overpaying-paiton-mi300x-moe-beats-h200-b200-on-1m-tokens\u002F",[2630,2606,2071,2812,2783,2813,2546,2814,2815,2816,2817,2071,2818],"AI Benchmarks","Cost per Token","Mixture of Experts","MoE","NVIDIA B200","NVIDIA H200","Qwen3",{"path":2820,"title":2821,"description":2822,"date":2823,"slug":2824,"image":2825,"originalUrl":2826,"categories":2827},"\u002Fblog\u002Fagentic-ai-but-make-it-local-from-inbox-to-insight-to-action-en","Local Agentic AI: From Inbox to Action","Local-first AI agents turn email, documents and images into tickets, reports and actions, using models tailored to your data and systems.","2025-09-16T13:09:00","agentic-ai-but-make-it-local-from-inbox-to-insight-to-action-en","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Ffrontfotoblog.webp","https:\u002F\u002Feliovp.com\u002Fagentic-ai-but-make-it-local-from-inbox-to-insight-to-action-en\u002F",[2630,2606,2728,2677,2729,2828,2829,2830,2831,2734,2736,2661,2832,2833],"Damage Detection","Document Processing","Email Automation","Invoice Extraction","Ticket Automation","Workflow Automation",{"path":2835,"title":2836,"description":2837,"date":2838,"slug":2839,"image":2840,"originalUrl":2841,"categories":2842},"\u002Fblog\u002Fmi300x-fp8-data-parallel-benchmarks-8-64-gpus-h200-left-behind-b200-within-reach","MI300X FP8 Benchmarks: GPU Partitioning with Paiton","Explore Llama 3.1 8B FP8 benchmarks on partitioned MI300X GPUs with Paiton, comparing throughput and latency against NVIDIA H200 and B200.","2025-07-31T13:32:57","mi300x-fp8-data-parallel-benchmarks-8-64-gpus-h200-left-behind-b200-within-reach","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fimage-2-1.webp","https:\u002F\u002Feliovp.com\u002Fmi300x-fp8-data%e2%80%91parallel-benchmarks-8-64-gpus-h200-left-behind-b200-within-reach\u002F",[2630,2606,2071,2677,2843,2646,2647,2844,2845,2657,2658,2071,2091],"AI","H200","MI300X",{"path":2847,"title":2848,"description":2849,"date":2850,"slug":2851,"image":2852,"originalUrl":2853,"categories":2854},"\u002Fblog\u002Fapplicable-ai-for-businesses","Applicable AI for Businesses","Explore ElioVP's approach to local AI for business workflows, including custom model training and automated damage detection for logistics.","2025-07-09T21:35:30","applicable-ai-for-businesses","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fscherm_afbeelding-2025-07-09-om-23.30.17.webp","https:\u002F\u002Feliovp.com\u002Fapplicable-ai-for-businesses\u002F",[2630,2606,2728,2729,2828,2829,2830,2831,2734,2736,2661,2832,2833],{"path":2856,"title":2857,"description":2858,"date":2859,"slug":2860,"image":34,"originalUrl":2861,"categories":2862},"\u002Fblog\u002Fintroducing-paitons-free-evaluation-models","Introducing Paiton’s Free Evaluation Models","Test Paiton with free evaluation models for AMD GPUs. Compare text, vision and image generation performance using your own workloads.","2025-07-07T11:26:13","introducing-paitons-free-evaluation-models","https:\u002F\u002Feliovp.com\u002Fintroducing-paitons-free-evaluation-models\u002F",[2630,2606,2071],{"path":2864,"title":2865,"description":2866,"date":2867,"slug":2868,"image":2869,"originalUrl":2870,"categories":2871},"\u002Fblog\u002Fpaiton-dramatically-faster-startup-and-performance-for-llama-3-1-405b","Llama 3.1 405B: Faster Startup with Paiton on MI300X","See Paiton benchmarks for Llama 3.1 405B on eight AMD MI300X GPUs, covering model startup, tensor parallelism, throughput and latency.","2025-06-12T20:15:23","paiton-dramatically-faster-startup-and-performance-for-llama-3-1-405b","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fservingscreenshot.webp","https:\u002F\u002Feliovp.com\u002Fpaiton-dramatically-faster-startup-and-performance-for-llama-3-1-405b\u002F",[2630,2606,2071,2677,2607,2783,2872,2785,2873,2874,2875,2071,2876,2877],"Cold Start","Graph Compilation","Llama 3.1 405B","LLM Optimization","Startup Latency","Tensor Parallelism",{"path":2879,"title":2880,"description":2881,"date":2882,"slug":2883,"image":2884,"originalUrl":2885,"categories":2886},"\u002Fblog\u002Fpaiton-fp8-beats-nvidias-h200-on-amds-mi300x","Paiton FP8 Beats NVIDIA’s H200 on AMD’s MI300X","Compare Paiton on AMD MI300X with NVIDIA H200 for Llama 3.1 70B FP8, including throughput, first-token delay and latency across batch sizes.","2025-06-08T19:12:40","paiton-fp8-beats-nvidias-h200-on-amds-mi300x","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fblognewfp8.webp","https:\u002F\u002Feliovp.com\u002Fpaiton-fp8-beats-nvidias-h200-on-amds-mi300x\u002F",[2630,2606,2071,2677,2783,2887,2733,2653,2608,2609,2610,2874,2888,2889],"Cold Start Optimization","Model Serving","vLLM Optimization",{"path":2891,"title":2892,"description":2893,"date":2894,"slug":2895,"image":2896,"originalUrl":2897,"categories":2898},"\u002Fblog\u002Fmi300x-vs-h200-vs-rx-7900-xtx-vs-tenstorrent-n300s-with-vllm","MI300X vs H200 vs RX 7900 XTX vs Tenstorrent n300s with vLLM","Compare MI300X, H200, RX 7900 XTX and Tenstorrent n300s on Llama 3 8B with vLLM, including throughput, modeled token costs and hardware limits.","2025-05-09T14:03:58","mi300x-vs-h200-vs-rx-7900-xtx-vs-tenstorrent-n300s-with-vllm","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fcomparisontenstor.webp","https:\u002F\u002Feliovp.com\u002Fmi300x-vs-h200-vs-rx-7900-xtx-vs-tenstorrent-n300s-with-vllm\u002F",[2630,2606,2071,2728,2677,2646,2845,2658,2899,2900],"RX7900XTX","tenstorrent",{"path":2902,"title":2903,"description":2904,"date":2905,"slug":2906,"image":2907,"originalUrl":2908,"categories":2909},"\u002Fblog\u002Fclusterpl-empowering-gpu-cluster-investors-with-real-world-financial-insights","ClusterP&L: Financial Modeling for GPU Clusters","Explore how ClusterP&L models GPU cluster costs, profitability and investment scenarios, with ROI metrics, risk simulations and exportable reports.","2025-05-03T10:52:22","clusterpl-empowering-gpu-cluster-investors-with-real-world-financial-insights","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fcomparisonscenarios.webp","https:\u002F\u002Feliovp.com\u002Fclusterpl-empowering-gpu-cluster-investors-with-real-world-financial-insights\u002F",[2630,2606,2676,2728,2647,2844,2910,2658,2911],"MI325x","pnl calculator",{"path":2913,"title":2914,"description":2915,"date":2916,"slug":2917,"image":2918,"originalUrl":2919,"categories":2920},"\u002Fblog\u002Fcranking-out-faster-tokens-for-fewer-dollars-amd-mi300x-vs-nvidia-h200","AMD MI300X vs. NVIDIA H200: Qwen3-32B with Paiton","Compare Qwen3-32B benchmarks on Paiton-optimized AMD MI300X and NVIDIA H200, covering throughput, latency and hardware costs.","2025-05-02T21:10:30","cranking-out-faster-tokens-for-fewer-dollars-amd-mi300x-vs-nvidia-h200","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002F3ac59a73-2466-4422-b7e5-ef2e4a8ca58e.webp","https:\u002F\u002Feliovp.com\u002Fcranking-out-faster-tokens-for-fewer-dollars-amd-mi300x-vs-nvidia-h200\u002F",[2630,2606,2071,2843,2646,2844,2921,2658,2071,2091],"MI300",{"path":2923,"title":2924,"description":2925,"date":2926,"slug":2927,"image":2928,"originalUrl":2929,"categories":2930},"\u002Fblog\u002Fpower-meets-precision-high-density-modular-data-center-for-nvidia-nvl-deployments-1-2-mw","Modular Data Centers for NVIDIA NVL: 1 to 2 MW","Explore modular data center designs for NVIDIA NVL systems, covering power capacity, liquid cooling, redundancy and deployment planning.","2025-05-02T14:09:59","power-meets-precision-high-density-modular-data-center-for-nvidia-nvl-deployments-1-2-mw","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Feliovp_critical-1mw-pod_rev-2_transparent.webp","https:\u002F\u002Feliovp.com\u002Fpower-meets-precision-high-density-modular-data-center-for-nvidia-nvl-deployments-1-2-mw\u002F",[2630,2676,2931,2631,2767,2634,2768,2769,2932,2933,2772,2934],"1-2MW Data Center","NVIDIA Blackwell","NVIDIA NVL","Precision Cooling",{"path":2936,"title":2937,"description":2938,"date":2939,"slug":2940,"image":2941,"originalUrl":2942,"categories":2943},"\u002Fblog\u002Fexamining-ai-agents-in-the-medical-field-ai-that-speaks-dicom","Examining AI agents in the medical field: AI that speaks DICOM","Explore a local AI agent demo for DICOM workflows, from patient and study retrieval to a comparison of vision models using anonymized medical images.","2025-04-11T14:45:48","examining-ai-agents-in-the-medical-field-ai-that-speaks-dicom","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fhealthcareblog-1.webp","https:\u002F\u002Feliovp.com\u002Fexamining-ai-agents-in-the-medical-field-ai-that-speaks-dicom\u002F",[2630,2606,2728,2677,2843,2646,2944],"Healthcare",{"path":2946,"title":2947,"description":2948,"date":2949,"slug":2950,"image":2951,"originalUrl":2952,"categories":2953},"\u002Fblog\u002Feliovp-bv-your-trusted-partner-for-supply-chain-resilience-amidst-new-u-s-tariffs","U.S. Tariffs and AI Supply Chain Resilience: April 2025","Read ElioVP's April 2025 perspective on U.S. tariffs and supply chain resilience for AI servers, HPC systems and modular data centers.","2025-04-04T10:01:27","eliovp-bv-your-trusted-partner-for-supply-chain-resilience-amidst-new-u-s-tariffs","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Ftariffsshipping.webp","https:\u002F\u002Feliovp.com\u002Feliovp-bv-your-trusted-partner-for-supply-chain-resilience-amidst-new-u-s-tariffs\u002F",[2630,2688,2843,2646,2954,2955,2956,2957],"import","Taiwan","Tariffs","Trump",{"path":2959,"title":2960,"description":2961,"date":2962,"slug":2963,"image":2964,"originalUrl":2965,"categories":2966},"\u002Fblog\u002Fwhy-ai-agents-are-the-future","Why AI Agents Are the Future","Explore AI agents for ERP, CRM, finance and customer support, with practical use cases and a path from workflow assessment to pilot and deployment.","2025-03-23T22:06:59","why-ai-agents-are-the-future","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Ferp2.jpeg","https:\u002F\u002Feliovp.com\u002Fwhy-ai-agents-are-the-future\u002F",[2630,2606,2728,2843,2967,2968],"AI Agents","ERP",{"path":2970,"title":2971,"description":2972,"date":2973,"slug":2974,"image":2975,"originalUrl":2976,"categories":2977},"\u002Fblog\u002Fthe-rise-of-open-source-ai-model-optimization","The Rise of Open-Source AI Model Optimization","Explore open-source AI optimization trends, from quantization and mixture-of-experts models to hardware-aware tuning, RAG and edge deployment.","2025-03-22T20:59:23","the-rise-of-open-source-ai-model-optimization","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Friseofopensource.jpeg","https:\u002F\u002Feliovp.com\u002Fthe-rise-of-open-source-ai-model-optimization\u002F",[2630,2606,2688,2978,2646,1369,2658],"AI news",{"path":2980,"title":2981,"description":2982,"date":2983,"slug":2984,"image":2985,"originalUrl":2986,"categories":2987},"\u002Fblog\u002Fintroducing-our-benchmarking-tool-powered-by-dstack","Introducing Our Benchmarking Tool: Powered by dstack","Explore our dstack-powered tool for reproducible vLLM benchmarks, automated parameter sweeps and performance reports across local and cloud GPUs.","2025-03-20T14:21:59","introducing-our-benchmarking-tool-powered-by-dstack","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fbenchmarktool.jpeg","https:\u002F\u002Feliovp.com\u002Fintroducing-our-benchmarking-tool-powered-by-dstack\u002F",[2630,2606,2071,2843,2646,2988,2989,2845,2071],"benchmark","LLM",{"path":2991,"title":2992,"description":2993,"date":2994,"slug":2995,"image":2996,"originalUrl":2997,"categories":2998},"\u002Fblog\u002Foptimizing-qwq-32b-by-qwen-amd-mi300x-vs-nvidia-h200","Optimizing QwQ-32B (by Qwen): AMD MI300X vs. NVIDIA H200","Compare QwQ-32B throughput and latency on AMD MI300X with Paiton and NVIDIA H200, from small batches to higher concurrency.","2025-03-19T21:41:44","optimizing-qwq-32b-by-qwen-amd-mi300x-vs-nvidia-h200","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fpaiton4.jpeg","https:\u002F\u002Feliovp.com\u002Foptimizing-qwq-32b-by-qwen-amd-mi300x-vs-nvidia-h200\u002F",[2630,2606,2071],{"path":3000,"title":3001,"description":3002,"date":3003,"slug":3004,"image":3005,"originalUrl":3006,"categories":3007},"\u002Fblog\u002Feliovp-featured-on-amd-tech-talk-podcast","Eliovp Featured on AMD “Tech Talk” Podcast","Listen to Elio Van Puyvelde and Jim Greene on AMD's Tech Talk podcast, discussing ElioVP's origins and its AI hardware and software services.","2025-03-19T07:53:39","eliovp-featured-on-amd-tech-talk-podcast","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Ftechtalkjimgreene.jpeg","https:\u002F\u002Feliovp.com\u002Feliovp-featured-on-amd-tech-talk-podcast\u002F",[2630,2646,3008,3009,3010],"Jim Greene","Podcast","Tech Talk",{"path":3012,"title":3013,"description":3014,"date":3015,"slug":3016,"image":3017,"originalUrl":3018,"categories":3019},"\u002Fblog\u002Ffurther-optimizing-amd-powered-inference-with-paiton","Further Optimizing AMD-Powered Inference with Paiton","Explore Paiton's DeepSeek R1 Distill Llama 8B benchmarks on AMD MI300X, focusing on throughput and first-token latency at smaller batch sizes.","2025-03-13T06:18:30","further-optimizing-amd-powered-inference-with-paiton","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fpaitonpost3.webp","https:\u002F\u002Feliovp.com\u002Ffurther-optimizing-amd-powered-inference-with-paiton\u002F",[2630,2606,2071,2646,3020,3021,2844,2845,2910,2071,2091],"Deepseek","H100",{"path":3023,"title":3024,"description":3025,"date":3026,"slug":3027,"image":3028,"originalUrl":3029,"categories":3030},"\u002Fblog\u002Fa-first-look-at-paiton-in-action-deepseek-r1-distill-llama-3-1-8b","Paiton Benchmarks: DeepSeek R1 Distill Llama 3.1 8B","Compare stock and Paiton-optimized DeepSeek R1 Distill Llama 3.1 8B on AMD MI300X, with throughput and latency benchmarks across batch sizes.","2025-01-31T09:11:02","a-first-look-at-paiton-in-action-deepseek-r1-distill-llama-3-1-8b","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fpaitonpost2.webp","https:\u002F\u002Feliovp.com\u002Fa-first-look-at-paiton-in-action-deepseek-r1-distill-llama-3-1-8b\u002F",[2630,2606,2071,2646,3020,3021,2844,2845,2910,2071,2091],{"path":3032,"title":3033,"description":3034,"date":3035,"slug":3036,"image":3037,"originalUrl":3038,"categories":3039},"\u002Fblog\u002Fai-model-optimization-with-paiton","AI Model Optimization with Paiton","Learn how Paiton uses model compilation, custom kernels and kernel fusion to optimize AI inference on AMD GPUs.","2025-01-30T19:53:25","ai-model-optimization-with-paiton","\u002Fasset\u002Fimages\u002Fblog\u002Fimported\u002Fpaitonpost1.webp","https:\u002F\u002Feliovp.com\u002Fai-model-optimization-with-paiton\u002F",[2630,2606,2071,2646,3021,2844,2845,2910,2071,2091],1789853166451]