{
  "id": "h200-llama-3.3-70b-prefill-chunk-tuning",
  "status": "published",
  "title": "Llama-3.3-70B FP8 on 1\u00d7 H200: prefill chunk size (max-num-batched-tokens)",
  "published": "2026-10-09",
  "summary": "Quick-mode comparison of vLLM's default max_num_batched_tokens (8192) against 2048 and 1024 on one RunPod Secure Cloud H200 (US-CO-1), 2026-10-09. Same node, model and workload for all three. The 15% gain at 1024 is about the size of quick-mode run-to-run variation; treat it as a lead. MBU recomputed with FP8 weight bytes (the original run files mis-detected compressed-tensors FP8).",
  "hardware": {
    "vendor": "NVIDIA",
    "accelerator": "NVIDIA H200",
    "count": 1,
    "rated_bw_tbs_per_gpu": 4.8,
    "driver": "580.178.04",
    "host": "RunPod secure cloud pod, "
  },
  "software": {
    "engine": "vLLM",
    "engine_version": "0.31.0",
    "kernels": "vllm/vllm-openai:v0.31.0"
  },
  "model": {
    "name": "Llama-3.3-70B-Instruct-FP8-dynamic",
    "shape": "Dense",
    "quantization": "FP8 weights (compressed-tensors), BF16 KV cache",
    "parallelism": "1 replica \u00d7 TP1"
  },
  "slo": {
    "ttft_p99_ms": 2000.0,
    "tpot_p99_ms": 50.0
  },
  "workload": {
    "trace": "chat-mix-v1",
    "input_tokens": "median 1,024 \u00b7 p90 4,096",
    "output_tokens": "median 256 \u00b7 p90 1,024",
    "arrival": "Poisson, rate swept to SLO boundary"
  },
  "power": {
    "source": "GPU board power sum (NVIDIA)",
    "kind": "board",
    "sampling_hz": 1,
    "window": "120 s steady state"
  },
  "accuracy": {
    "suite": "gsm8k-5shot",
    "threshold_pts": 1.0
  },
  "runs": 2,
  "variance_band_pct": 8.3,
  "fleet": {
    "cards": 64
  },
  "configs": [
    {
      "key": "other",
      "label": "default (8,192 tokens/step)",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 61.5,
      "node_power_kw": 0.639,
      "goodput_tok_s": 233.7,
      "ttft_p99_ms": 1226.0,
      "tpot_p99_ms": 37.1,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 72
      },
      "accuracy": {
        "score": 94.0,
        "delta_pts": 0.0,
        "spread_pts": 0.0
      }
    },
    {
      "key": "other",
      "label": "2,048 tokens/step",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 60.6,
      "node_power_kw": 0.643,
      "goodput_tok_s": 249.8,
      "ttft_p99_ms": 1279.0,
      "tpot_p99_ms": 38.3,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 76
      },
      "accuracy": {
        "score": 92.0,
        "delta_pts": -2.0,
        "spread_pts": 0.0
      }
    },
    {
      "key": "other",
      "label": "1,024 tokens/step",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 60.8,
      "node_power_kw": 0.642,
      "goodput_tok_s": 268.9,
      "ttft_p99_ms": 1316.0,
      "tpot_p99_ms": 36.4,
      "slo_pass_count": 2,
      "repeats": 2,
      "throttle_counts": {
        "sw_power_cap": 83
      },
      "accuracy": {
        "score": 92.0,
        "delta_pts": -2.0,
        "spread_pts": 0.0
      }
    }
  ],
  "not_covered": [
    "Whole-node power (no BMC access on this host)",
    "Context lengths beyond the chat-mix-v1 trace",
    "Quick mode: short windows, few repeats \u2014 validation only, not for publishing"
  ],
  "reproduce": "tokenwatt run --spec lambda-llama-3.3-70b-instruct-fp8-dynamic-tp1-dp1.spec.yaml --label 'llama70b-default' --role other --out runs/llama70b-default.json\ntokenwatt run --spec lambda-llama-3.3-70b-instruct-fp8-dynamic-tp1-dp1.spec.yaml --label 'llama70b-mnbt2048' --role other --out runs/llama70b-mnbt2048.json\ntokenwatt run --spec lambda-llama-3.3-70b-instruct-fp8-dynamic-tp1-dp1.spec.yaml --label 'llama70b-mnbt1024' --role other --out runs/llama70b-mnbt1024.json\n\ntokenwatt report runs/llama70b-default.json runs/llama70b-mnbt2048.json runs/llama70b-mnbt1024.json --title 'Llama-3.3-70B FP8 on 1\u00d7 H200: prefill chunk size (max-num-batched-tokens)'",
  "spec": {
    "name": "lambda-llama-3.3-70b-instruct-fp8-dynamic-tp1-dp1",
    "hardware": {
      "adapter": "nvidia",
      "rated_bw_tbs_per_gpu": null,
      "gpus": null,
      "host": "RunPod secure cloud pod, "
    },
    "engine": {
      "endpoint": "http://127.0.0.1:8000/v1",
      "name": "vLLM",
      "version": "0.31.0",
      "kernels": "vllm/vllm-openai:v0.31.0",
      "api": "completions",
      "model": "RedHatAI/Llama-3.3-70B-Instruct-FP8-dynamic",
      "ignore_eos": true,
      "api_key_env": null,
      "extra_body": {}
    },
    "model": {
      "name": "Llama-3.3-70B-Instruct-FP8-dynamic",
      "config": "/root/tokenwatt-bench/models/RedHatAI__Llama-3.3-70B-Instruct-FP8-dynamic/config.json",
      "weight_dtype": "fp8",
      "kv_dtype": "bf16",
      "quantization": "",
      "replicas": 1,
      "tensor_parallel": 1,
      "vocab_size": 128256,
      "geometry": {}
    },
    "slo": {
      "ttft_p99_ms": 2000.0,
      "tpot_p99_ms": 50.0
    },
    "workload": {
      "trace": "chat-mix-v1",
      "input_tokens": null,
      "output_tokens": null,
      "seed": 0
    },
    "measurement": {
      "warmup_s": 45,
      "window_s": 120,
      "search_window_s": 60,
      "repeats": 2,
      "request_timeout_s": 300,
      "max_concurrency": 8192
    },
    "sweep": {
      "rates": null,
      "start_rps": 0.5,
      "growth": 1.5,
      "max_rps": 2000,
      "bisect_steps": 3,
      "backoff": 0.9,
      "max_backoffs": 4
    },
    "accuracy": {
      "suite": "/root/tokenwatt-bench/suites/gsm8k-200.jsonl",
      "name": "gsm8k-5shot",
      "scorer": "numeric",
      "max_drop_pts": 1.0,
      "seeds": 1,
      "max_tokens": 384,
      "stop": [
        "\n\nQuestion:"
      ],
      "concurrency": 32,
      "limit": 50
    },
    "power": {
      "source": "board",
      "sampling_hz": 1,
      "url": null,
      "username": null,
      "password_env": null,
      "verify_tls": false,
      "chassis": null,
      "host": null,
      "command": null,
      "field": null
    },
    "not_covered": [
      "Whole-node power (no BMC access on this host)",
      "Context lengths beyond the chat-mix-v1 trace",
      "Quick mode: short windows, few repeats \u2014 validation only, not for publishing"
    ]
  }
}