{
  "id": "example-moe-8gpu",
  "status": "example",
  "title": "Mid-size MoE decode on an 8-GPU node",
  "published": "2026-10-06",
  "summary": "Format example. Three configurations of the same node, model and SLO: the production image as deployed, the vendor's latest recommended image, and a tuned decode path.",
  "hardware": {
    "vendor": "Example",
    "accelerator": "Accelerator A (example)",
    "count": 8,
    "rated_bw_tbs_per_gpu": 5.0,
    "driver": "example-driver 1.2.3",
    "host": "2-socket server, 8 accelerators"
  },
  "software": {
    "engine": "vLLM",
    "engine_version": "0.x (example)",
    "kernels": "vendor kernel library (example)"
  },
  "model": {
    "name": "Example-MoE-30B-A3B",
    "shape": "MoE",
    "quantization": "FP8 weights, FP8 KV cache",
    "parallelism": "8 replicas × TP1"
  },
  "slo": {
    "ttft_p99_ms": 2000,
    "tpot_p99_ms": 50
  },
  "workload": {
    "trace": "chat-mix-v1",
    "input_tokens": "median 1,024 · p90 4,096",
    "output_tokens": "median 256 · p90 1,024",
    "arrival": "Poisson, rate swept to SLO boundary"
  },
  "power": {
    "source": "BMC via Redfish",
    "sampling_hz": 1,
    "window": "600 s steady state"
  },
  "accuracy": {
    "suite": "gsm8k-500",
    "threshold_pts": 0.5
  },
  "runs": 5,
  "variance_band_pct": 3,
  "fleet": {
    "cards": 64
  },
  "configs": [
    {
      "key": "baseline",
      "label": "Production",
      "mbu_pct": 31,
      "node_power_kw": 6.1,
      "goodput_tok_s": 2400,
      "ttft_p99_ms": 1710,
      "tpot_p99_ms": 47,
      "accuracy": { "score": 88.4, "delta_pts": 0 }
    },
    {
      "key": "vendor",
      "label": "Vendor latest",
      "mbu_pct": 44,
      "node_power_kw": 6.6,
      "goodput_tok_s": 3300,
      "ttft_p99_ms": 1650,
      "tpot_p99_ms": 46,
      "accuracy": { "score": 88.3, "delta_pts": -0.1 }
    },
    {
      "key": "tuned",
      "label": "Tuned",
      "mbu_pct": 63,
      "node_power_kw": 7.0,
      "goodput_tok_s": 4500,
      "ttft_p99_ms": 1590,
      "tpot_p99_ms": 48,
      "accuracy": { "score": 88.2, "delta_pts": -0.2 }
    }
  ],
  "not_covered": [
    "Context lengths above 8,192 input tokens",
    "Quantization formats other than FP8 weights with FP8 KV cache",
    "Concurrency beyond the SLO boundary found in this run",
    "Multi-node deployments and prefill/decode disaggregation",
    "Workload traces other than chat-mix-v1"
  ],
  "reproduce": "docker run --rm --network host -v $PWD:/work tokenwatt/bench run \\\n  --endpoint http://localhost:8000/v1 \\\n  --spec /work/example-moe-8gpu.spec.yaml --runs 5 --label <config>\n\ntokenwatt report production vendor-latest tuned"
}
