{
  "id": "lab-h100-qwen3-30b-a3b-kv-cache",
  "status": "draft",
  "title": "Qwen3-30B-A3B FP8 on 1\u00d7 H100: BF16 vs FP8 KV cache",
  "published": "2026-10-06",
  "summary": "Validation run on a Lambda Cloud 1\u00d7 H100 SXM5 instance with vLLM 0.31.0. Quick mode: 120 s windows, 2 repeats, 50 GSM8K items with one seed. FP8 KV cache raised SLO goodput and tokens per joule but failed the accuracy gate on this small suite.",
  "hardware": {
    "vendor": "NVIDIA",
    "accelerator": "NVIDIA H100 80GB HBM3",
    "count": 1,
    "rated_bw_tbs_per_gpu": 3.35,
    "driver": "580.105.08",
    "host": "Lambda Cloud instance"
  },
  "software": {
    "engine": "vLLM",
    "engine_version": "0.31.0",
    "kernels": "vllm/vllm-openai:latest"
  },
  "model": {
    "name": "Qwen3-30B-A3B-FP8",
    "shape": "MoE",
    "quantization": "FP8 weights; KV cache varies by configuration",
    "parallelism": "1 replica \u00d7 TP1"
  },
  "slo": {
    "ttft_p99_ms": 2000.0,
    "tpot_p99_ms": 50.0
  },
  "workload": {
    "trace": "chat-mix-v1",
    "input_tokens": "median 1,024 \u00b7 p90 4,096",
    "output_tokens": "median 256 \u00b7 p90 1,024",
    "arrival": "Poisson, rate swept to SLO boundary"
  },
  "power": {
    "source": "GPU board power sum (NVIDIA)",
    "kind": "board",
    "sampling_hz": 1,
    "window": "120 s steady state"
  },
  "accuracy": {
    "suite": "gsm8k-5shot",
    "threshold_pts": 1.0
  },
  "runs": 2,
  "variance_band_pct": 1.0,
  "fleet": {
    "cards": 64
  },
  "configs": [
    {
      "key": "baseline",
      "label": "vllm-default",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 56.6,
      "node_power_kw": 0.637,
      "goodput_tok_s": 4572.3,
      "ttft_p99_ms": 340.0,
      "tpot_p99_ms": 44.6,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 199
      },
      "accuracy": {
        "score": 96.0,
        "delta_pts": 0.0,
        "spread_pts": 0.0
      }
    },
    {
      "key": "tuned",
      "label": "vllm-fp8kv",
      "engine": "vLLM 0.31.0 \u00b7 KV FP8",
      "mbu_pct": 45.3,
      "node_power_kw": 0.616,
      "goodput_tok_s": 5056.9,
      "ttft_p99_ms": 352.0,
      "tpot_p99_ms": 43.7,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 133
      },
      "accuracy": {
        "score": 88.0,
        "delta_pts": -8.0,
        "spread_pts": 0.0
      }
    }
  ],
  "not_covered": [
    "Whole-node power (cloud VM: no BMC access)",
    "Context lengths beyond the chat-mix-v1 trace",
    "Quick mode: short windows, few repeats \u2014 validation only, not for publishing",
    "Whole-node power: energy figures are GPU board power only (no CPUs, memory, NICs, fans or PSU losses)"
  ],
  "reproduce": "tokenwatt run --spec lambda-qwen3-30b-a3b-fp8-tp1-dp1.spec.yaml --label 'vllm-default' --role baseline --out runs/vllm-default.json\ntokenwatt run --spec lambda-qwen3-30b-a3b-fp8-tp1-dp1.spec.yaml --label 'vllm-fp8kv' --role tuned --out runs/vllm-fp8kv.json\n\ntokenwatt report runs/vllm-default.json runs/vllm-fp8kv.json --title 'Qwen3-30B-A3B FP8 on 1\u00d7 H100: BF16 vs FP8 KV cache'",
  "spec": {
    "name": "lambda-qwen3-30b-a3b-fp8-tp1-dp1",
    "hardware": {
      "adapter": "nvidia",
      "rated_bw_tbs_per_gpu": null,
      "gpus": null,
      "host": "Lambda Cloud instance"
    },
    "engine": {
      "endpoint": "http://127.0.0.1:8000/v1",
      "name": "vLLM",
      "version": "0.31.0",
      "kernels": "vllm/vllm-openai:latest",
      "api": "completions",
      "model": "Qwen/Qwen3-30B-A3B-FP8",
      "ignore_eos": true,
      "api_key_env": null
    },
    "model": {
      "name": "Qwen3-30B-A3B-FP8",
      "config": "/home/ubuntu/tokenwatt-bench/models/Qwen__Qwen3-30B-A3B-FP8/config.json",
      "weight_dtype": "fp8",
      "kv_dtype": "bf16",
      "quantization": "",
      "replicas": 1,
      "tensor_parallel": 1,
      "vocab_size": 151936,
      "geometry": {}
    },
    "slo": {
      "ttft_p99_ms": 2000.0,
      "tpot_p99_ms": 50.0
    },
    "workload": {
      "trace": "chat-mix-v1",
      "input_tokens": null,
      "output_tokens": null,
      "seed": 0
    },
    "measurement": {
      "warmup_s": 45,
      "window_s": 120,
      "search_window_s": 60,
      "repeats": 2,
      "request_timeout_s": 300,
      "max_concurrency": 8192
    },
    "sweep": {
      "rates": null,
      "start_rps": 2,
      "growth": 2,
      "max_rps": 2000,
      "bisect_steps": 3,
      "backoff": 0.9,
      "max_backoffs": 4
    },
    "accuracy": {
      "suite": "/home/ubuntu/tokenwatt-bench/suites/gsm8k-200.jsonl",
      "name": "gsm8k-5shot",
      "scorer": "numeric",
      "max_drop_pts": 1.0,
      "seeds": 1,
      "max_tokens": 384,
      "stop": [
        "\n\nQuestion:"
      ],
      "concurrency": 32,
      "limit": 50
    },
    "power": {
      "source": "board",
      "sampling_hz": 1,
      "url": null,
      "username": null,
      "password_env": null,
      "verify_tls": false,
      "chassis": null,
      "host": null,
      "command": null,
      "field": null
    },
    "not_covered": [
      "Whole-node power (cloud VM: no BMC access)",
      "Context lengths beyond the chat-mix-v1 trace",
      "Quick mode: short windows, few repeats \u2014 validation only, not for publishing"
    ]
  }
}