[{"data":1,"prerenderedAt":820},["ShallowReactive",2],{"blog-\u002Fget-inspired\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark":3,"relation-people":603},{"id":4,"title":5,"audioLink":6,"authors":7,"body":9,"description":592,"draft":593,"episodeNumber":6,"extension":594,"mainImage":595,"meta":596,"navigation":597,"path":598,"publishedAt":599,"seo":600,"stem":601,"transcriptPdf":6,"__hash__":602},"blog\u002Fget-inspired\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark.md","50+ Tokens\u002Fsec on a Desktop: Running LLMs on the NVIDIA DGX Spark",null,[8],"luc-bocahut",{"type":10,"value":11,"toc":589},"minimark",[12,40,45,53,56,78,85,89,92,124,131,138,142,148,159,164,171,174,186,189,193,200,206,209,214,222,225,229,234,240,244,251,254,276,283,289,293,296,302,333,339,343,347,352,355,360,367,371,374,377,380,383,386,389,392,395,398,401,404,407,410,413,416,427,435,439,449,455,464,470,479,485,487,490,494,505,515,518,520,523,530,534,543,552,567,576,585],[13,14,15,19,20,23,24,27,28,31,32,35,36,39],"p",{},[16,17,18],"strong",{},"TL;DR:"," At ",[16,21,22],{},"XRPL Commons",", we’ve been experimenting with ",[16,25,26],{},"local AI infrastructure"," for development workflows: coding assistants, document drafting, agent systems, and internal tooling. We got a ",[16,29,30],{},"30B-parameter LLM running at 51–54 tokens\u002Fsec"," on the ",[16,33,34],{},"NVIDIA DGX Spark"," by combining ",[16,37,38],{},"Mixture-of-Experts models, FP8 quantization, and a community Docker image that fixes Blackwell compatibility issues."," Full setup below, brought to you by our CTO.",[41,42,44],"h5",{"id":43},"why-we-wanted-local-llms","Why We Wanted Local LLMs",[13,46,47,48,23,50,52],{},"At ",[16,49,22],{},[16,51,26],{}," for development workflows: coding assistants, document drafting, agent systems, and internal tooling.",[13,54,55],{},"Our requirements were simple:",[57,58,59,66,72],"ul",{},[60,61,62,65],"li",{},[16,63,64],{},"Fast enough"," for interactive use",[60,67,68,71],{},[16,69,70],{},"Private"," enough to run on-premise",[60,73,74,77],{},[16,75,76],{},"Replicable"," across multiple machines",[13,79,80,81,84],{},"The ",[16,82,83],{},"DGX Spark"," looked like an interesting candidate. But achieving good performance requires understanding its real constraint.",[41,86,88],{"id":87},"the-hardware","The Hardware",[13,90,91],{},"The DGX Spark packs significant compute into a desktop system:",[57,93,94,99,104,109,114,119],{},[60,95,96],{},[16,97,98],{},"NVIDIA GB10 Blackwell GPU (SM 12.1)",[60,100,101],{},[16,102,103],{},"128 GB unified LPDDR5X memory",[60,105,106],{},[16,107,108],{},"273 GB\u002Fs memory bandwidth",[60,110,111],{},[16,112,113],{},"ARM Grace CPU (20 cores: 10 Cortex-X925 + 10 Cortex-A725)",[60,115,116],{},[16,117,118],{},"4 TB NVMe M.2 (~3.7 TB usable)",[60,120,121],{},[16,122,123],{},"DGX OS (Ubuntu 24.04)",[13,125,126,127,130],{},"The standout feature is the ",[16,128,129],{},"128 GB unified memory",", which allows very large models to run locally.",[13,132,133,134,137],{},"But the critical limitation is ",[16,135,136],{},"memory bandwidth",".",[41,139,141],{"id":140},"the-bandwidth-wall","The Bandwidth Wall",[13,143,144,145,137],{},"LLM inference is primarily ",[16,146,147],{},"memory-bandwidth bound",[13,149,150,151,154,155,158],{},"During autoregressive decoding, each token requires reading the ",[16,152,153],{},"active model weights"," from memory. With ",[16,156,157],{},"273 GB\u002Fs bandwidth",", the limits become clear:",[13,160,161],{},[16,162,163],{},"Model",[13,165,166],{},[167,168],"img",{"alt":169,"src":170},"","\u002Fimages\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark\u002F1.webp",[13,172,173],{},"Our first runs matched this almost exactly:",[57,175,176,181],{},[60,177,178],{},[16,179,180],{},"Qwen3-32B (bf16): 3.7 tok\u002Fs",[60,182,183],{},[16,184,185],{},"Qwen3-8B (bf16): 13.1 tok\u002Fs",[13,187,188],{},"Large models fit comfortably in memory, but generate tokens slowly.",[41,190,192],{"id":191},"the-moe-breakthrough","The MoE Breakthrough",[13,194,195,196,199],{},"The solution is ",[16,197,198],{},"Mixture-of-Experts (MoE)"," models.",[13,201,202,203,137],{},"Instead of activating the entire network, MoE models route each token through a ",[16,204,205],{},"subset of experts",[13,207,208],{},"Example:",[13,210,211],{},[16,212,213],{},"Qwen3-30B-A3B",[57,215,216,219],{},[60,217,218],{},"~30.5B total parameters",[60,220,221],{},"~3.3B active parameters per token",[13,223,224],{},"This dramatically changes the bandwidth math:",[13,226,227],{},[16,228,163],{},[13,230,231],{},[167,232],{"alt":169,"src":233},"\u002Fimages\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark\u002F2.webp",[13,235,236,237,137],{},"In practice, routing overhead, KV-cache reads, and software stack inefficiencies reduce throughput. Real systems typically achieve ",[16,238,239],{},"50–70% of theoretical bandwidth limits",[41,241,243],{"id":242},"the-blackwell-software-problem","The Blackwell Software Problem",[13,245,246,247,250],{},"The DGX Spark’s ",[16,248,249],{},"Blackwell GPU (SM 12.1)"," is new enough that much of the software stack is still catching up.",[13,252,253],{},"Issues we encountered:",[57,255,256,261,266,271],{},[60,257,258],{},[16,259,260],{},"FlashAttention 2 crashes",[60,262,263],{},[16,264,265],{},"vLLM MoE kernels missing SM 12.1",[60,267,268],{},[16,269,270],{},"PyTorch officially supports only SM 12.0",[60,272,273],{},[16,274,275],{},"CUDA graphs disabled in standard builds",[13,277,278,279,282],{},"We initially built ",[16,280,281],{},"vLLM from source",", patching build scripts and dependencies.",[13,284,285,286,137],{},"It worked, but required --enforce-eager mode (no CUDA graphs), which capped throughput at about ",[16,287,288],{},"30 tok\u002Fs",[41,290,292],{"id":291},"the-avarok-docker-image","The Avarok Docker Image",[13,294,295],{},"A community project solved most of these issues.",[13,297,80,298,301],{},[16,299,300],{},"Avarok dgx-vllm Docker image"," includes:",[57,303,304,310,315,321,327],{},[60,305,306,307],{},"Patched ",[16,308,309],{},"vLLM v0.16.0rc2",[60,311,312],{},[16,313,314],{},"SM 12.1 Blackwell support",[60,316,317,318],{},"Custom ",[16,319,320],{},"CUTLASS kernels",[60,322,323,324],{},"Software fallback for missing ",[16,325,326],{},"NVFP4 instructions",[60,328,329,330],{},"Working ",[16,331,332],{},"FlashAttention and CUDA graphs",[13,334,335,336,137],{},"Instead of hours compiling from source, deployment becomes a ",[16,337,338],{},"single Docker command",[41,340,342],{"id":341},"results","Results",[13,344,345],{},[16,346,163],{},[13,348,349],{},[167,350],{"alt":169,"src":351},"\u002Fimages\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark\u002F3.webp",[13,353,354],{},"The winning combination:",[13,356,357],{},[16,358,359],{},"MoE architecture + FP8 quantization + CUDA graphs via Avarok Docker",[13,361,362,363,366],{},"The FP8 model uses about ",[16,364,365],{},"110 GB of GPU memory",", leaving little headroom but delivering excellent throughput.",[41,368,370],{"id":369},"final-setup","Final Setup",[13,372,373],{},"Deployment is straightforward:",[13,375,376],{},"shell",[13,378,379],{},"docker pull avarok\u002Fdgx-vllm-nvfp4-kernel:v22",[13,381,382],{},"docker run -d \\",[13,384,385],{},"--name vllm \\",[13,387,388],{},"--gpus all \\",[13,390,391],{},"--shm-size=16g \\",[13,393,394],{},"--restart unless-stopped \\",[13,396,397],{},"-p 8000:8888 \\",[13,399,400],{},"-v \u002Fhome\u002F$USER\u002F.cache\u002Fhuggingface:\u002Froot\u002F.cache\u002Fhuggingface \\",[13,402,403],{},"-e MODEL=Qwen\u002FQwen3-30B-A3B-Instruct-2507-FP8 \\",[13,405,406],{},"-e PORT=8888 \\",[13,408,409],{},"-e GPU_MEMORY_UTIL=0.85 \\",[13,411,412],{},"-e MAX_MODEL_LEN=32768 \\",[13,414,415],{},"avarok\u002Fdgx-vllm-nvfp4-kernel:v22 serve",[13,417,418,419,422,423,426],{},"First startup takes ",[16,420,421],{},"10–20 minutes"," (model download and CUDA graph capture). After that, the container auto-starts and exposes an ",[16,424,425],{},"OpenAI-compatible API",":",[13,428,429],{},[430,431,432],"a",{"href":432,"rel":433},"http:\u002F\u002Flocalhost:8000\u002Fv1",[434],"nofollow",[41,436,438],{"id":437},"lessons-learned","Lessons Learned",[13,440,441,442,445,446,137],{},"**1. Understand the bottleneck ",[443,444],"br",{},"\n**On DGX Spark, ",[16,447,448],{},"memory bandwidth determines performance",[13,450,451,452,454],{},"**2. MoE models are ideal for bandwidth-limited systems ",[443,453],{},"\n**They dramatically reduce active weights per token.",[13,456,457,458,460,461,137],{},"**3. FP8 quantization is a free win ",[443,459],{},"\n**Throughput nearly doubled from ",[16,462,463],{},"30 → 51 tok\u002Fs",[13,465,466,467,469],{},"**4. Avoid building from source if possible ",[443,468],{},"\n**Community builds often include critical patches ahead of official releases.",[13,471,472,473,475,476,137],{},"**5. Stop competing inference servers first ",[443,474],{},"\n**One Spark OOM-killed during installation because ",[16,477,478],{},"Ollama was using ~100 GB of memory",[13,480,481,482,484],{},"**6. Kernel updates can break NVIDIA drivers ",[443,483],{},"\n**If nvidia-smi fails after reboot:",[13,486,376],{},[13,488,489],{},"sudo apt install linux-modules-nvidia-580-open-$(uname -r)",[41,491,493],{"id":492},"next-experiments","Next Experiments",[13,495,496,497,500,501,504],{},"One promising direction is experimenting with ",[16,498,499],{},"reasoning-optimized models",". A recent example is ",[16,502,503],{},"Qwen3.5-27B-Claude-4.6-Opus-Reasoning-Distilled"," (“Qwopus”), which distills structured reasoning from Claude 4.6 Opus into a Qwen3.5 base model.",[13,506,507,508,511,512,137],{},"While it is a ",[16,509,510],{},"dense model"," and therefore likely slower than our current MoE setup on the DGX Spark, it may offer ",[16,513,514],{},"stronger step-by-step reasoning for coding, math, and agent workflows",[13,516,517],{},"Testing it is simple—swap the model in the same container:",[13,519,376],{},[13,521,522],{},"-e MODEL=Jackrong\u002FQwen3.5-27B-Claude-4.6-Opus-Reasoning-Distilled",[13,524,525,526,529],{},"We plan to benchmark ",[16,527,528],{},"reasoning quality vs throughput"," alongside the current MoE setup.",[41,531,533],{"id":532},"references","References",[13,535,536,537,539],{},"NVIDIA DGX Spark ",[443,538],{},[430,540,541],{"href":541,"rel":542},"https:\u002F\u002Fwww.nvidia.com\u002Fen-us\u002Fdata-center\u002Fdgx-spark\u002F",[434],[13,544,545,546,548],{},"Avarok dgx-vllm Docker project ",[443,547],{},[430,549,550],{"href":550,"rel":551},"https:\u002F\u002Fgithub.com\u002Favarok-ai\u002Fdgx-vllm",[434],[13,553,554,555,557,561,562,566],{},"vLLM documentation ",[443,556],{},[430,558,559],{"href":559,"rel":560},"https:\u002F\u002Fdocs.vllm.ai",[434],"(",[430,563,564],{"href":564,"rel":565},"https:\u002F\u002Fdocs.vllm.ai\u002F",[434],")",[13,568,569,570,572],{},"Qwen3-30B-A3B FP8 model ",[443,571],{},[430,573,574],{"href":574,"rel":575},"https:\u002F\u002Fhuggingface.co\u002FQwen\u002FQwen3-30B-A3B-Instruct-2507-FP8",[434],[13,577,578,579,581],{},"Qwen3.5-27B-Claude-4.6-Opus-Reasoning-Distilled ",[443,580],{},[430,582,583],{"href":583,"rel":584},"https:\u002F\u002Fhuggingface.co\u002FJackrong\u002FQwen3.5-27B-Claude-4.6-Opus-Reasoning-Distilled",[434],[41,586,588],{"id":587},"let-us-know-what-youre-building","Let us know what you’re building.",{"title":169,"searchDepth":590,"depth":590,"links":591},2,[],"Local AI doesn’t have to mean slow. We got a 30B LLM running at 51–54 tok\u002Fs on a DGX Spark using MoE models, FP8 quantization, and a patched Blackwell stack. See how we made fast, private on‑prem AI actually work.",false,"md","\u002Fimages\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark\u002Fcover.webp",{},true,"\u002Fget-inspired\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark","2026-03-19",{"title":5,"description":592},"get-inspired\u002Fblog\u002F50-tokens-sec-on-a-desktop-running-llms-on-the-nvidia-dgx-spark","xY5al2OKBSm6gY4k8n14K_2JoAtTlzaV7RwcQYHNk7Q",[604,607,610,613,616,619,622,625,628,631,634,637,640,643,646,649,652,655,658,661,664,667,670,673,676,679,682,685,688,691,694,697,700,703,706,709,712,715,718,721,724,727,730,733,736,739,742,745,748,751,754,757,760,763,766,769,772,775,778,781,784,787,790,793,796,799,802,805,808,811,814,817],{"stem":605,"name":606},"people\u002Fadria-carrera-mas","Adrià Carrera Mas",{"stem":608,"name":609},"people\u002Falex-mavigok","Alex Mavigok",{"stem":611,"name":612},"people\u002Falexandre-duarte","Alexandre Duarte",{"stem":614,"name":615},"people\u002Famaury-mongreville","Amaury Mongreville",{"stem":617,"name":618},"people\u002Fartur-kirjakulov-phd","Artur Kirjakulov, PhD",{"stem":620,"name":621},"people\u002Fasheesh-birla","Asheesh Birla",{"stem":623,"name":624},"people\u002Faurelien-burget","Aurélien Burget",{"stem":626,"name":627},"people\u002Fcassie-hirsh","Cassie Hirsh",{"stem":629,"name":630},"people\u002Fcharley-huebner","Charley Huebner",{"stem":632,"name":633},"people\u002Fchris-dangerfield","Chris Dangerfield",{"stem":635,"name":636},"people\u002Fclay-tantor","Clay Graham",{"stem":638,"name":639},"people\u002Fcyrille-bourdeaux","Cyrille Bourdeaux",{"stem":641,"name":642},"people\u002Fdan-wallace","Dan Wallace",{"stem":644,"name":645},"people\u002Fdaniel-arroche","Daniel Arroche",{"stem":647,"name":648},"people\u002Fdarius-tumas","Darius Tumas",{"stem":650,"name":651},"people\u002Fdavid-bchiri","David Bchiri",{"stem":653,"name":654},"people\u002Fdavid-servais","David Servais",{"stem":656,"name":657},"people\u002Fdeath-ranger","Death Ranger",{"stem":659,"name":660},"people\u002Fdenis-angell","Denis Angell",{"stem":662,"name":663},"people\u002Fdr-nicolas-figay","Dr. Nicolas Figay",{"stem":665,"name":666},"people\u002Felie-aben-moha","Elie Aben Moha",{"stem":668,"name":669},"people\u002Felisa-bailly","Elisa Bailly",{"stem":671,"name":672},"people\u002Ffabiola-bernard","Fabiola Bernard",{"stem":674,"name":675},"people\u002Ffanny-brakchi","Fanny Brakchi",{"stem":677,"name":678},"people\u002Fferran-prat-tio","Ferran Prat Tió",{"stem":680,"name":681},"people\u002Ffig-squid-router","Fig",{"stem":683,"name":684},"people\u002Fflorence-tison","Florence Tison",{"stem":686,"name":687},"people\u002Fflorent-uzio","Florent Uzio",{"stem":689,"name":690},"people\u002Fflorian-alonso","Florian Alonso",{"stem":692,"name":693},"people\u002Ffrederic-martin","Frédéric Martin",{"stem":695,"name":696},"people\u002Fguillaume-evrat","Guillaume Evrat",{"stem":698,"name":699},"people\u002Fguy-levi-bochi","Guy Lévi-Bochi",{"stem":701,"name":702},"people\u002Fimad-el-aouny","Imad EL Aouny",{"stem":704,"name":705},"people\u002Fkrippenreiter","Krippenreiter",{"stem":707,"name":708},"people\u002Flara-pagnier","Lara Pagnier",{"stem":710,"name":711},"people\u002Flauren-berta","Lauren Berta",{"stem":713,"name":714},"people\u002Fluc-bocahut","Luc Bocahut",{"stem":716,"name":717},"people\u002Fmanthan-dave","Manthan Dave",{"stem":719,"name":720},"people\u002Fmarc-hayot","Marc Hayot",{"stem":722,"name":723},"people\u002Fmarco-brondani","Marco Brondani",{"stem":725,"name":726},"people\u002Fmarco-neri","Marco Neri",{"stem":728,"name":729},"people\u002Fmargaux-frisque","Margaux Frisque",{"stem":731,"name":732},"people\u002Fmartino-bettucci","Martino Bettucci",{"stem":734,"name":735},"people\u002Fmathilde-morineaux","Mathilde Morineaux",{"stem":737,"name":738},"people\u002Fmathis-sergent","Mathis Sergent",{"stem":740,"name":741},"people\u002Fmatt-mankins","Matt Mankins",{"stem":743,"name":744},"people\u002Fmawuena-tendar","Mawuena Tendar",{"stem":746,"name":747},"people\u002Fmayukha-vadari","Mayukha Vadari",{"stem":749,"name":750},"people\u002Fmelanie-damour","Melanie Damour",{"stem":752,"name":753},"people\u002Fmichael-burow","Michaël Burow",{"stem":755,"name":756},"people\u002Fneti-poland","Neti Poland",{"stem":758,"name":759},"people\u002Fnick-daze","Nick Dazé",{"stem":761,"name":762},"people\u002Fodelia-torteman","Odelia Torteman",{"stem":764,"name":765},"people\u002Fparisa-ghodous","Parisa Ghodous",{"stem":767,"name":768},"people\u002Fpaul-pagnier","Paul Pagnier",{"stem":770,"name":771},"people\u002Fpeter-rosberg","Peter Rosberg",{"stem":773,"name":774},"people\u002Frachel","Rachel Cervantes",{"stem":776,"name":777},"people\u002Frichard-deane","Richard Deane",{"stem":779,"name":780},"people\u002Frobert-kiuru","Robert Kiuru",{"stem":782,"name":783},"people\u002Fromain-thepaut","Romain Thepaut",{"stem":785,"name":786},"people\u002Fshen-morincome","Shen Morincome",{"stem":788,"name":789},"people\u002Fshota-natenadze","Shota Natenadze",{"stem":791,"name":792},"people\u002Fsimon","Simon Luling",{"stem":794,"name":795},"people\u002Fsolene-daviaud","Solène Daviaud",{"stem":797,"name":798},"people\u002Fthomas-cadorel","Thomas Cadorel",{"stem":800,"name":801},"people\u002Fthomas-colin","Thomas Colin",{"stem":803,"name":804},"people\u002Fthomas-hussenet","Thomas Hussenet",{"stem":806,"name":807},"people\u002Ftom-kiddle","Tom Kiddle",{"stem":809,"name":810},"people\u002Fvalentin-gonnot","Valentin Gonnot",{"stem":812,"name":813},"people\u002Fvera-radeva-hadjiev","Vera Radeva Hadjiev",{"stem":815,"name":816},"people\u002Fvincent-diallo","Vincent Diallo",{"stem":818,"name":819},"people\u002Fzsofi-borsi","Zsofi Borsi",1788427147278]