{
  "id": "h100-qwen3-30b-a3b-moe-kernel-tuning",
  "status": "published",
  "title": "Tuned vs default fused-MoE kernels: Qwen3-30B-A3B FP8 on 1\u00d7 H100 SXM",
  "published": "2026-10-08",
  "summary": "Lambda Cloud 1\u00d7 H100 SXM, 2026-10-08: vLLM 0.31.0 with its default fused-MoE Triton configs, then with configs tuned for this GPU and model (2.5 hours of kernel tuning), then default again to bound drift. Quick mode. The tuned kernels made no measurable difference to SLO goodput.",
  "hardware": {
    "vendor": "NVIDIA",
    "accelerator": "NVIDIA H100 80GB HBM3",
    "count": 1,
    "rated_bw_tbs_per_gpu": 3.35,
    "driver": "580.105.08",
    "host": "Lambda Cloud instance"
  },
  "software": {
    "engine": "vLLM",
    "engine_version": "0.31.0",
    "kernels": "vllm/vllm-openai:latest"
  },
  "model": {
    "name": "Qwen3-30B-A3B-FP8",
    "shape": "MoE",
    "quantization": "FP8 weights, BF16 KV cache",
    "parallelism": "1 replica \u00d7 TP1"
  },
  "slo": {
    "ttft_p99_ms": 2000.0,
    "tpot_p99_ms": 50.0
  },
  "workload": {
    "trace": "chat-mix-v1",
    "input_tokens": "median 1,024 \u00b7 p90 4,096",
    "output_tokens": "median 256 \u00b7 p90 1,024",
    "arrival": "Poisson, rate swept to SLO boundary"
  },
  "power": {
    "source": "GPU board power sum (NVIDIA)",
    "kind": "board",
    "sampling_hz": 1,
    "window": "120 s steady state"
  },
  "accuracy": {
    "suite": "gsm8k-5shot",
    "threshold_pts": 1.0
  },
  "runs": 2,
  "variance_band_pct": 0.7,
  "fleet": {
    "cards": 64
  },
  "configs": [
    {
      "key": "baseline",
      "label": "default kernels",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 56.3,
      "node_power_kw": 0.623,
      "goodput_tok_s": 4515.0,
      "ttft_p99_ms": 338.0,
      "tpot_p99_ms": 44.2,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 180
      },
      "accuracy": {
        "score": 96.0,
        "delta_pts": 0.0,
        "spread_pts": 0.0
      }
    },
    {
      "key": "tuned",
      "label": "tuned kernels",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 57.9,
      "node_power_kw": 0.641,
      "goodput_tok_s": 4522.8,
      "ttft_p99_ms": 333.0,
      "tpot_p99_ms": 42.3,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 186
      },
      "accuracy": {
        "score": 96.0,
        "delta_pts": 0.0,
        "spread_pts": 0.0
      }
    },
    {
      "key": "other",
      "label": "default kernels (repeat)",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 57.0,
      "node_power_kw": 0.631,
      "goodput_tok_s": 4516.4,
      "ttft_p99_ms": 333.0,
      "tpot_p99_ms": 43.0,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 183
      },
      "accuracy": {
        "score": 96.0,
        "delta_pts": 0.0,
        "spread_pts": 0.0
      }
    }
  ],
  "not_covered": [
    "Whole-node power (no BMC access on this host)",
    "Context lengths beyond the chat-mix-v1 trace",
    "Quick mode: short windows, few repeats \u2014 validation only, not for publishing"
  ],
  "reproduce": "tokenwatt run --spec lambda-qwen3-30b-a3b-fp8-tp1-dp1.spec.yaml --label 'moe-default' --role baseline --out runs/moe-default.json\ntokenwatt run --spec lambda-qwen3-30b-a3b-fp8-tp1-dp1.spec.yaml --label 'moe-tuned' --role tuned --out runs/moe-tuned.json\ntokenwatt run --spec lambda-qwen3-30b-a3b-fp8-tp1-dp1.spec.yaml --label 'moe-default-2' --role other --out runs/moe-default-2.json\n\ntokenwatt report runs/moe-default.json runs/moe-tuned.json runs/moe-default-2.json --title 'Tuned vs default fused-MoE kernels: Qwen3-30B-A3B FP8 on 1\u00d7 H100 SXM'",
  "spec": {
    "name": "lambda-qwen3-30b-a3b-fp8-tp1-dp1",
    "hardware": {
      "adapter": "nvidia",
      "rated_bw_tbs_per_gpu": null,
      "gpus": null,
      "host": "Lambda Cloud instance"
    },
    "engine": {
      "endpoint": "http://127.0.0.1:8000/v1",
      "name": "vLLM",
      "version": "0.31.0",
      "kernels": "vllm/vllm-openai:latest",
      "api": "completions",
      "model": "Qwen/Qwen3-30B-A3B-FP8",
      "ignore_eos": true,
      "api_key_env": null,
      "extra_body": {}
    },
    "model": {
      "name": "Qwen3-30B-A3B-FP8",
      "config": "/home/ubuntu/tokenwatt-bench/models/Qwen__Qwen3-30B-A3B-FP8/config.json",
      "weight_dtype": "fp8",
      "kv_dtype": "bf16",
      "quantization": "",
      "replicas": 1,
      "tensor_parallel": 1,
      "vocab_size": 151936,
      "geometry": {}
    },
    "slo": {
      "ttft_p99_ms": 2000.0,
      "tpot_p99_ms": 50.0
    },
    "workload": {
      "trace": "chat-mix-v1",
      "input_tokens": null,
      "output_tokens": null,
      "seed": 0
    },
    "measurement": {
      "warmup_s": 45,
      "window_s": 120,
      "search_window_s": 60,
      "repeats": 2,
      "request_timeout_s": 300,
      "max_concurrency": 8192
    },
    "sweep": {
      "rates": null,
      "start_rps": 8.0,
      "growth": 1.25,
      "max_rps": 2000,
      "bisect_steps": 3,
      "backoff": 0.9,
      "max_backoffs": 4
    },
    "accuracy": {
      "suite": "/home/ubuntu/tokenwatt-bench/suites/gsm8k-200.jsonl",
      "name": "gsm8k-5shot",
      "scorer": "numeric",
      "max_drop_pts": 1.0,
      "seeds": 1,
      "max_tokens": 384,
      "stop": [
        "\n\nQuestion:"
      ],
      "concurrency": 32,
      "limit": 50
    },
    "power": {
      "source": "board",
      "sampling_hz": 1,
      "url": null,
      "username": null,
      "password_env": null,
      "verify_tls": false,
      "chassis": null,
      "host": null,
      "command": null,
      "field": null
    },
    "not_covered": [
      "Whole-node power (no BMC access on this host)",
      "Context lengths beyond the chat-mix-v1 trace",
      "Quick mode: short windows, few repeats \u2014 validation only, not for publishing"
    ]
  }
}