# Generated from run-config.json. Contains no credentials or cluster endpoints.
schema_version: 1
source: "run-config.json"
tested_configuration:
  package: "engine-configuration"
  endpoint_picker:
    image: "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0"
    replicas: 1
  model_service:
    image: "registry.redhat.io/rhaii/vllm-cuda-rhel9@sha256:ad06abf3bb5235ebb5b2df84cd1b9fd09e823f0ff2eebfc82bb4590275ccfe0b"
    model: "openai/gpt-oss-20b"
    gpu: "1 NVIDIA H100"
    prefix_cache: "disabled"
    gpu_memory_utilization: 0.9
    max_model_length: 32768
  traffic:
    arrival_mode: "closed_loop"
    input_tokens: "varies by sweep"
    output_tokens: "varies by sweep"
  measurement:
    direct_metric_interval_seconds: 0.5
    prometheus_interval_seconds: 5
  runner:
    source: "pipeline/benchmark.py"
    sha256: "3811ec26c46bf3a26fa643698ec54bf569bb4bc99c3ea22ca18f805cb077b8e0"
    traffic_driver: "native closed loop"
    launcher: "pipeline/run-in-cluster.sh"
  execution:
    arrival_mode: "closed_loop"
    prompt_pool_size: 24
    warmup_seconds: 30
    warmup_concurrency: 2
    steady_state_trim_seconds: 30
    traffic_seed: 42
    direct_metric_interval_seconds: 0.5
    prometheus_interval_seconds: 5
    cache_mode: "off"
  sweeps:
    - name: "capacity curve"
      closed_loop_concurrency:
        minimum: 8
        maximum: 160
      seconds_per_point: 180
    - name: "max-num-seqs"
      fixed_concurrency: 192
      values:
        - 64
        - 96
        - 128
        - 160
        - 192
      selected: 128
      selected_and_boundary_repeats: 3
    - name: "max-num-batched-tokens"
      fixed_max_num_seqs: 128
      values:
        - 4096
        - 8192
        - 16384
      selected: 8192
      selected_repeats: 3
