{
  "id": "h100-qwen3-30b-a3b-fp8-vllm-0-31",
  "status": "published",
  "title": "Qwen3-30B-A3B FP8 on 1\u00d7 H100 SXM with vLLM 0.31.0",
  "published": "2026-10-06",
  "summary": "Full-mode measurement on a Lambda Cloud 1\u00d7 H100 SXM5 instance: 10-minute windows, 5 repeats at the SLO boundary, GSM8K 5-shot on 200 items \u00d7 3 seeds. Default vLLM configuration with a BF16 KV cache. Energy is GPU board power.",
  "hardware": {
    "vendor": "NVIDIA",
    "accelerator": "NVIDIA H100 80GB HBM3",
    "count": 1,
    "rated_bw_tbs_per_gpu": 3.35,
    "driver": "580.105.08",
    "host": "Lambda Cloud instance"
  },
  "software": {
    "engine": "vLLM",
    "engine_version": "0.31.0",
    "kernels": "vllm/vllm-openai:latest"
  },
  "model": {
    "name": "Qwen3-30B-A3B-FP8",
    "shape": "MoE",
    "quantization": "FP8 weights, BF16 KV cache",
    "parallelism": "1 replica \u00d7 TP1"
  },
  "slo": {
    "ttft_p99_ms": 2000.0,
    "tpot_p99_ms": 50.0
  },
  "workload": {
    "trace": "chat-mix-v1",
    "input_tokens": "median 1,024 \u00b7 p90 4,096",
    "output_tokens": "median 256 \u00b7 p90 1,024",
    "arrival": "Poisson, rate swept to SLO boundary"
  },
  "power": {
    "source": "GPU board power sum (NVIDIA)",
    "kind": "board",
    "sampling_hz": 1,
    "window": "600 s steady state"
  },
  "accuracy": {
    "suite": "gsm8k-5shot",
    "threshold_pts": 1.0
  },
  "runs": 5,
  "variance_band_pct": 2.2,
  "fleet": {
    "cards": 64
  },
  "configs": [
    {
      "key": "baseline",
      "label": "vllm-default",
      "engine": "vLLM 0.31.0 \u00b7 KV BF16",
      "mbu_pct": 58.1,
      "node_power_kw": 0.601,
      "goodput_tok_s": 3995.9,
      "ttft_p99_ms": 327.0,
      "tpot_p99_ms": 37.3,
      "slo_pass_count": 5,
      "repeats": 5,
      "throttle_counts": {
        "sw_power_cap": 1870
      },
      "accuracy": {
        "score": 94.17,
        "delta_pts": 0.0,
        "spread_pts": 0.5
      }
    }
  ],
  "not_covered": [
    "Whole-node power (no BMC access on this host)",
    "Context lengths beyond the chat-mix-v1 trace"
  ],
  "reproduce": "tokenwatt run --spec lambda-qwen3-30b-a3b-fp8-tp1-dp1.spec.yaml --label 'vllm-default' --role baseline --out runs/vllm-default.json\n\ntokenwatt report runs/vllm-default.json --title 'Qwen3-30B-A3B FP8 on 1\u00d7 H100 SXM with vLLM 0.31.0'",
  "spec": {
    "name": "lambda-qwen3-30b-a3b-fp8-tp1-dp1",
    "hardware": {
      "adapter": "nvidia",
      "rated_bw_tbs_per_gpu": null,
      "gpus": null,
      "host": "Lambda Cloud instance"
    },
    "engine": {
      "endpoint": "http://127.0.0.1:8000/v1",
      "name": "vLLM",
      "version": "0.31.0",
      "kernels": "vllm/vllm-openai:latest",
      "api": "completions",
      "model": "Qwen/Qwen3-30B-A3B-FP8",
      "ignore_eos": true,
      "api_key_env": null
    },
    "model": {
      "name": "Qwen3-30B-A3B-FP8",
      "config": "/home/ubuntu/tokenwatt-bench/models/Qwen__Qwen3-30B-A3B-FP8/config.json",
      "weight_dtype": "fp8",
      "kv_dtype": "bf16",
      "quantization": "",
      "replicas": 1,
      "tensor_parallel": 1,
      "vocab_size": 151936,
      "geometry": {}
    },
    "slo": {
      "ttft_p99_ms": 2000.0,
      "tpot_p99_ms": 50.0
    },
    "workload": {
      "trace": "chat-mix-v1",
      "input_tokens": null,
      "output_tokens": null,
      "seed": 0
    },
    "measurement": {
      "warmup_s": 120,
      "window_s": 600,
      "search_window_s": 120.0,
      "repeats": 5,
      "request_timeout_s": 600,
      "max_concurrency": 8192
    },
    "sweep": {
      "rates": null,
      "start_rps": 9.0,
      "growth": 1.25,
      "max_rps": 2000,
      "bisect_steps": 4,
      "backoff": 0.9,
      "max_backoffs": 4
    },
    "accuracy": {
      "suite": "/home/ubuntu/tokenwatt-bench/suites/gsm8k-200.jsonl",
      "name": "gsm8k-5shot",
      "scorer": "numeric",
      "max_drop_pts": 1.0,
      "seeds": 3,
      "max_tokens": 384,
      "stop": [
        "\n\nQuestion:"
      ],
      "concurrency": 32,
      "limit": 200
    },
    "power": {
      "source": "board",
      "sampling_hz": 1,
      "url": null,
      "username": null,
      "password_env": null,
      "verify_tls": false,
      "chassis": null,
      "host": null,
      "command": null,
      "field": null
    },
    "not_covered": [
      "Whole-node power (no BMC access on this host)",
      "Context lengths beyond the chat-mix-v1 trace"
    ]
  }
}