MODELARCH
Per-model architecture: layers, KV heads, head dim, params.
const MODELARCH={
"DeepSeek-V4-Flash (304B MoE, MLA+DSA)":{layers:43, kvh:0, hd:0, params:3.04e+11, act:1.3e+10, mla:true},
"DeepSeek-V4-Pro (1.6T MoE, MLA+DSA)":{layers:61, kvh:0, hd:0, params:1.6e+12, act:4.9e+10, mla:true},
"Xiaomi MiMo-V2.5 (310B MoE, SWA)":{layers:48, kvLayers:9, kvh:4, hd:192, params:3.1e+11, act:1.5e+10, mla:false},
"MiniMax-M3 (428B MoE, MSA)":{layers:60, kvh:4, hd:128, params:4.28e+11, act:2.3e+10, mla:false},
"Nemotron-3-Ultra-550B-A55B (hybrid Mamba)":{layers:60, kvLayers:12, kvh:2, hd:128, params:5.5e+11, act:5.5e+10, mla:false},
"Nemotron-3-Super-120B-A12B (hybrid Mamba)":{layers:88, kvLayers:8, kvh:2, hd:128, params:1.2e+11, act:1.2e+10, mla:false},
"Gemma-4-31B (dense, SWA)":{layers:60, kvLayers:10, kvh:16, hd:256, params:3.1e+10, mla:false},
"Gemma-4-26B-A4B (MoE, SWA)":{layers:30, kvLayers:5, kvh:8, hd:256, params:2.6e+10, act:4e+09, mla:false},
"Ling-3.0-flash (124B MoE, KDA+MLA)":{layers:42, kvLayers:7, kvh:0, hd:0, params:1.24e+11, act:5.1e+09, mla:true},
"Qwen2.5-7B-Instruct-1M": {layers:28,kvh:4, hd:128,params:7.6e9, mla:false},
"DeepSeek-R1-Distill-Llama-70B-FP8": {layers:80,kvh:8, hd:128,params:70.6e9, mla:false},
"Llama-4-Scout-17B-16E-FP8": {layers:48,kvh:8, hd:128,params:109e9, act:17e9, mla:false},
"Llama-3.1-8B-Instruct": {layers:32,kvh:8, hd:128,params:8.03e9, mla:false},
"Llama-3.3-70B-Instruct": {layers:80,kvh:8, hd:128,params:70.6e9, mla:false},
"Qwen2.5-72B-Instruct": {layers:80,kvh:8, hd:128,params:72.7e9, mla:false},
"Qwen2.5-Coder-32B-Instruct": {layers:64,kvh:8, hd:128,params:32.8e9, mla:false},
"Mixtral-8x7B-Instruct-v0.1": {layers:32,kvh:8, hd:128,params:46.7e9, act:12.9e9, mla:false},
"Gemma-3-27B-IT": {layers:62,kvLayers:10,kvh:16,hd:128,params:27.4e9, mla:false}, /* 5 local : 1 global interleave, window 1024: SWA layers hold no long-context KV */
"DeepSeek-R1 (V3, MLA)": {layers:61,kvh:0, hd:0, params:671e9, act:37e9, mla:true},
/* MoE weights are counted at TOTAL parameters: every expert must be resident. */
"Llama-3.1-405B-Instruct": {layers:126,kvh:8, hd:128,params:405e9, mla:false},
"Llama-3.2-3B-Instruct": {layers:28, kvh:8, hd:128,params:3.2e9, mla:false},
"Llama-4-Maverick-17B-128E": {layers:48, kvh:8, hd:128,params:400e9, act:17e9, mla:false},
"Qwen3-8B": {layers:36, kvh:8, hd:128,params:8.2e9, mla:false},
"Qwen3-32B": {layers:64, kvh:8, hd:128,params:32.8e9, mla:false},
"Qwen3-30B-A3B": {layers:48, kvh:4, hd:128,params:30.5e9, act:3.3e9, mla:false},
"Qwen3-235B-A22B": {layers:94, kvh:4, hd:128,params:235e9, act:22e9, mla:false},
"GLM-4-9B-Chat": {layers:40, kvh:2, hd:128,params:9.4e9, mla:false},
"GLM-4.5-Air (106B MoE)": {layers:46, kvh:8, hd:128,params:106e9, act:12e9, mla:false},
"GLM-4.5 (355B MoE)": {layers:92, kvh:8, hd:128,params:355e9, act:32e9, mla:false},
"Kimi-K2-Instruct (1T MoE, MLA)": {layers:61, kvh:0, hd:0, params:1000e9, act:32e9, mla:true},
"gpt-oss-20b": {layers:24, kvLayers:12, kvh:8, hd:64, params:21e9, act:3.6e9, mla:false}, /* alternating full/sliding(128) */
"gpt-oss-120b": {layers:36, kvLayers:18, kvh:8, hd:64, params:117e9, act:5.1e9, mla:false}, /* alternating full/sliding(128) */
"Mistral-Small-3.1-24B": {layers:40, kvh:8, hd:128,params:24e9, mla:false},
"Mistral-Large-2 (123B)": {layers:88, kvh:8, hd:128,params:123e9, mla:false},
"Phi-4 (14B)": {layers:40, kvh:10,hd:128,params:14.7e9, mla:false},
"Command-R+ (104B)": {layers:64, kvh:8, hd:128,params:104e9, mla:false},
/* GLM-5.2: 744B MoE (~40B active), MLA with DeepSeek Sparse Attention + IndexShare.
DSA cuts attention FLOPs, not the resident cache, so the cache is modelled as MLA. */
"GLM-5.2 (744B MoE, MLA+DSA)": {layers:78, kvh:0, hd:0, params:744e9, act:40e9, mla:true},
/* Kimi K3: 2.8T MoE (104B active), 93 layers = 69 Kimi Delta Attention (linear,
constant state) + 24 Gated MLA. Only the 24 MLA layers cache per token. */
"Kimi-K3 (2.8T MoE, KDA+MLA)": {layers:93, kvLayers:24, kvh:0, hd:0, params:2800e9, act:104e9, mla:true},
/* Qwen3.5 interleaves Gated DeltaNet (Mamba-family, constant recurrent state) with
softmax attention 3:1, so only every fourth layer grows a KV cache.
0.8B is confirmed at 24 layers / 6 attention; the flagship layer count is an estimate. */
"Qwen3.5-0.8B (GDN hybrid)": {layers:24, kvLayers:6, kvh:4, hd:128, params:0.8e9, mla:false},
"Qwen3.5-397B-A17B (GDN hybrid)": {layers:80, kvLayers:20, kvh:8, hd:128, params:397e9, act:17e9, mla:false}
}QUANTB
Bytes per weight at each supported quantization.
const QUANTB={fp16:2,fp8:1,int8:1,fp4:0.5,int4:0.5}kvPerTok
KV cache bytes per token. GQA, MLA and hybrid stacks.
function kvPerTok(a,kvq){
const L=a.kvLayers!=null?a.kvLayers:a.layers;
return a.mla?576*L*KVQB[kvq]:2*L*a.kvh*a.hd*KVQB[kvq];
}kvState
The whole memory state: budget, batch, context ceiling, what fits.
function kvState(){
const hw=(BOM==="hw"), hwp=HWP[$("hw-preset").value]||HWP.b200su;
const custom=(!hw)&&$("provider").value==="Custom";
let inst=null;
try{ if(!hw&&!custom) inst=CATALOG[$("provider").value].instances[$("itype").value]; }catch(_){}
const gpus=hw?hwp.gpus:custom?(parseInt($("gpus_per_server").value)||8):((inst&&inst.gpus)||8);
const gpuName=hw?hwp.gpu:((inst&&inst.gpu)||"H100 80GB");
const memGB=gpuMemGB(gpuName);
const prof=PROFILES[$("profile").value]||PROFILES["qwen7b-100k"];
const arch=MODELARCH[prof.model]||MODELARCH["Qwen2.5-7B-Instruct-1M"];
const wq=elVal("m-wq","fp8"), kq=elVal("m-kq","fp8");
const util=Math.min(.98,Math.max(.5,(parseFloat(elVal("m-util",90))||90)/100));
const ctx=Math.max(1024,parseInt(elVal("m-ctx",prof.ctx))||prof.ctx);
const block=Math.max(1,parseInt(elVal("m-block",16))||16);
/* TP may exceed `gpus` (cross-node tensor parallel over the fabric); PP
splits the model's layers into stages. A model replica occupies tp*pp
GPUs. Mirrors server.py kv_state so the page and the PDF agree. */
const nServers=Math.max(1,parseInt($("nservers").value)||1);
const totalGpus=Math.max(1,nServers*gpus);
const tp=Math.max(1,Math.min(totalGpus,parseInt(elVal("m-tp",gpus))||gpus));
const pp=Math.max(1,parseInt(elVal("m-pp",1))||1);
const gpusPerReplica=tp*pp;
const replicas=Math.max(1,Math.floor(totalGpus/gpusPerReplica)); /* full weight copies in the fleet */
const prefix=elVal("m-prefix","1")!=="0";
const GBb=1073741824;
const per=kvPerTok(arch,kq);
/* vLLM allocates whole blocks, so a sequence rounds up to the next block boundary */
const ctxPad=Math.ceil(ctx/block)*block;
const perSeq=per*ctxPad;
/* Linear-attention layers cost nothing per token but carry a constant state per
SEQUENCE, so it scales with batch. Estimated, not measured. */
const linLayers=Math.max(0,arch.layers-(arch.kvLayers!=null?arch.kvLayers:arch.layers));
const statePerSeq=linLayers*131072; /* ~128 KB per linear layer per sequence */
/* Per-NODE budget when a replica fits in a node (tp*pp<=gpus): the node holds
floor(gpus/gpusPerReplica) full weight copies. This is byte-identical to the
pre-TP/PP formula, so every tile that reads K.onGpu / K.budget is unchanged
for the common case. Cross-node TP (tp>gpus) has no per-node boundary, so
there the replica's own budget is the unit. */
let budget, onGpu, inferra, replicasPerNode;
const crossNode = gpusPerReplica>gpus;
if(!crossNode){
replicasPerNode=Math.max(1,Math.floor(gpus/gpusPerReplica));
budget=Math.max(0,gpus*memGB*GBb*util-replicasPerNode*arch.params*(QUANTB[wq]||1));
onGpu=perSeq+statePerSeq>0?Math.floor(budget/(perSeq+statePerSeq)):0;
inferra=per>0?Math.floor(budget/(per*Math.min(ctx,INF_RESIDENT)+statePerSeq)):0;
} else {
replicasPerNode=0; /* a replica spans >1 node; no whole replica lives in one node */
budget=Math.max(0,gpusPerReplica*memGB*GBb*util-arch.params*(QUANTB[wq]||1));
onGpu=perSeq+statePerSeq>0?Math.floor(budget/(perSeq+statePerSeq)):0; /* per cross-node replica */
inferra=per>0?Math.floor(budget/(per*Math.min(ctx,INF_RESIDENT)+statePerSeq)):0;
}
const want=parseInt(elVal("m-batch",0))||0;
/* 8 was unrealistically small for a serving fleet and quietly understated
decode, partly masking the prefill overstatement above. */
const batch=want>0?want:Math.max(1,Math.min(64,onGpu||1));
return {gpus,memGB,gpuName,arch,model:prof.model,wq,kq,util,ctx,per,perSeq,budget,
block,tp,pp,gpusPerReplica,replicas,replicasPerNode,totalGpus,nServers,prefix,ctxPad,statePerSeq,linLayers,
weights:replicas*arch.params*(QUANTB[wq]||1),
/* every expert must be RESIDENT, but a decode step only READS the active ones */
activeWeights:replicas*(arch.act||arch.params)*(QUANTB[wq]||1),
moe:!!arch.act,
onGpu:Math.max(0,onGpu),onGpuFleet:Math.max(0,onGpu*nServers),
inferra:Math.max(0,inferra),inferraFleet:Math.max(0,inferra*nServers),
maxCtxOnGpu:per>0?Math.floor(budget/per):0,batch,
maxModelCtx:MODELMAXCTX[prof.model]||prof.ctx,
fitsOnGpu:batch<=onGpu, fitsInferra:batch<=inferra,
inferraOnly:batch>onGpu&&batch<=inferra,
impossible:batch>inferra,
crossNodeTp:crossNode,
batchOk:m=>batch<=(m==="inferra"?inferra:onGpu)};
}GPUBW
Published HBM bandwidth per GPU family.
const GPUBW={B300:8.0e12,B200:8.0e12,B100:8.0e12,H200:4.8e12,H100:3.35e12,"A100 40":1.555e12,A100:2.0e12,A10G:0.6e12,A10:0.6e12,V100:0.9e12,L40S:0.864e12,L4:0.3e12,
/* GDDR7 desktop parts are bandwidth-rich for their price; GB10 is the
opposite -- huge capacity, 273 GB/s, so decode is bandwidth-bound long
before it is capacity-bound. That asymmetry is the interesting part. */
"RTX PRO 6000":1.792e12,"RTX 5090":1.792e12,"RTX 4090":1.008e12,GB10:0.273e12}baselineTps
Decode throughput as a bandwidth roofline.
function baselineTps(K){
const bw=gpuBW(K.gpuName);
/* A decode step streams the ACTIVE experts plus each sequence's cache, not the
whole parameter count. Charging total weights understated MoE by up to 27x. */
/* ONE REPLICA'S weights, not the fleet's. kvState() defines activeWeights as
replicas x active params -- a fleet total -- while this function returns a
PER-GPU rate, and replicas scale with the node count. So the denominator grew
linearly with fleet size and per-GPU throughput fell as 1/n, exactly
cancelling the GPU count: 22 tok/s/GPU at 2 nodes, 11 at 4, 6 at 8, 1 at 32,
with fleet capacity pinned near 350 tok/s at every size. Adding hardware
bought nothing but cost, which is why profit fell as capex rose and why the
optimizer kept choosing tiny fleets.
A decode step streams the active experts of the replica serving it, plus that
sequence's cache. The other replicas are decoding their own batches on their
own GPUs, in parallel. */
const perReplicaActive=(K.replicas>0)?K.activeWeights/K.replicas:K.activeWeights;
const denom=perReplicaActive+K.batch*K.per*K.ctx;
if(denom<=0) return 1;
const raw=K.batch*bw*decodeEff(K.gpuName)/denom;
return Math.max(1,Math.round(raw*tpEfficiency(K.tp,K.gpus)*ppEfficiency(K.pp,K.batch)*specMult()*ctxSpreadEff()));
}prefillSecs
Prefill time. The quadratic fit and its coefficients.
function prefillSecs(n,kvTok,K){
/* measured single-user latency, exactly as published */
const scale=(kvTok||PREFILL_KVREF)/PREFILL_KVREF;
return (PREFILL_A*n+PREFILL_BB*n*n)*scale;
}mult
Whole-turn capacity multiplier. Amdahl over prefill only.
const mult=(k<=1||turn.inf<=0)?1:Math.max(1,turn.base/turn.inf);
/* A GPU does not decode all the time. Every turn spends tPre on prefill before
any token is emitted, and the roofline above gives PEAK decode throughput. So
scale it by the share of the turn actually spent decoding. Leaving this out
overstated fleet capacity by roughly 4x on long-context workloads, and the
error stayed invisible while only output tokens were billed. */
const dutyBase=turn.base>0?Math.max(.004,turn.tDec/turn.base):1;
const totalGpus=n*gpusPer;
const perGpuBase=Math.max(parseFloat($("p-tps").value)||prof.tpsGpu,1);
const monthHours=hours/24*HOURS_PER_MONTH;
/* PREFILL IS A CEILING, NOT A DEDUCTION.
The duty cycle already charges prefill for the time it takes, but time is
not the only thing prefill can run out of: a fleet cannot decode more
output than it can feed with prompt tokens. The disaggregated path has
always applied this -- min(Dcap, F/ppo) in disaggApply -- and the single
fleet never did, so it billed 1.66x more prompt tokens than it could
physically prefill, on a page where input carries ~86% of revenue.
Apply the same bound here, from the same roofline.
The two designs are bounded DIFFERENTLY, and that asymmetry is the product
argument rather than a thumb on the scale: a cache-less fleet must prefill
every prompt in full, while a warm cache re-reads most of it, so Inferra's
prompt ceiling is higher by exactly the reuse factor the rest of the model
already uses. Where the baseline becomes prefill-limited and Inferra does
not, the gap between them widens -- which is what serving long context on
the same hardware actually looks like. */
const _pfPool=(function(){
try{
if(!hw) return 0; /* cloud instances: no node roofline to price */
const arch2=MODELARCH[prof.model]||{};
const r=disaggNodeRates($("hw-preset").value,true,arch2.act||arch2.params||7e9);
return (r>0)?r*n:0;
}catch(_){ return 0; }
}WLREUSE
Per-workload cache reuse share.
const WLREUSE={code:.92,claw:.95,chat:.78,docs:.97,mm:.55,bench:.30}wlReuse
Blended reuse share across the workload mix.
function wlReuse(){
/* A measured shared-prefix share from the customer's own logs beats anything we
can infer from a workload mix, so it wins outright when supplied. */
const meas=parseFloat(elVal("cal-hit",""));
if(isFinite(meas)&&meas>0) return Math.min(.99,meas/100);
const c=wlCounts(); let num=0,den=0;
wlKeys().forEach(k=>{ const r=WLREUSE[k]!=null?WLREUSE[k]:0.8; num+=r*c[k]; den+=c[k]; });
return den>0?num/den:.8;
}modeReuse
How much of that reuse each mode can actually retain.
function modeReuse(m){
let a=wlReuse();
try{
const K=kvState();
if(!K.prefix) return 0; /* nothing is shared between requests */
/* prefix sharing only matches on block boundaries, so coarse blocks lose the tail */
if(K.block>16) a*=Math.max(0.6,1-(K.block-16)/512);
}catch(_){}
if(m==="off") return 0; /* nothing below HBM survives the next batch */
if(m==="fs") return a*hbmRetention(); /* bounded by leftover HBM, not a constant */
return a*.98; /* the NVMe tier holds the whole history */
}__profitFixes
Ranked, costed suggestions when a configuration loses money.
window.__profitFixes=function(s,inf){
const out=[], tok=inf.tokFleet/1e6;
/* 1. price: the exact list price that clears cost with a 15% margin */
/* Input tokens already cover part of the cost, so the output price only has to
close the remainder. Ignoring that would tell people to overprice. */
const need=Math.max(0.01,outPriceFloor(inf.cpm*1.15,inf));
if(need>s.price){
const m=MKT[$("profile").value];
const beats=(m&&m.ok)?m.prices.filter(v=>v>need).length:null;
out.push({v:(need-s.price)*tok,gain:"+"+fmt$((need-s.price)*tok)+" / mo",
text:`raise your list price from ${fmt$2(s.price)} to <b>${fmt$2(need)}</b> per 1M`
+(beats!=null?`, still cheaper than ${beats} of ${m.n} listings for this model.`:".")});
}
/* (licence-term pricing tip removed: licence pricing is never shown on-site) */
/* 3. fleet: idle nodes are pure cost */
if(s.util<0.55&&s.n>4){
const want=Math.max(2,Math.ceil(s.n*Math.max(s.util,0.05)/0.7));
if(want<s.n){
const save=(s.n-want)*s.hourly*730;
out.push({v:save,gain:"+"+fmt$(save)+" / mo",
text:`you are serving at ${Math.round(s.util*100)}% utilization. Run
<b>${want} nodes</b> instead of ${s.n}, or sell the idle capacity.`});
}
}
/* 4. denser weights free HBM for more tenants per node */
try{
const K=kvState(), cap=gpuCap(K.gpuName);
if(K.wq!=="fp4"&&cap.w.indexOf("fp4")>=0)
out.push({v:1,gain:"more tenants / node",
text:`this GPU supports <b>FP4</b>. Halving the weight footprint frees HBM for a larger
continuous batch on the same hardware.`});
else if(K.kq!=="fp8")
out.push({v:1,gain:"more tenants / node",
text:`drop the KV cache to <b>FP8</b> to double the sessions this node can hold.`});
}catch(_){}
/* 5. power */
const kwh=parseFloat(($("hw-kwh")||{}).value)||0.08;
if(kwh>0.06) out.push({v:0.5,gain:"lower $/node/hr",
text:`electricity is set to ${fmt$2(kwh)}/kWh. Colo at a cheaper rate, or a better PUE,
drops the amortized cost of every node.`});
return out.sort((a,b)=>b.v-a.v).slice(0,4);
}