# Generated from run-config.json. Contains no credentials or cluster endpoints.
schema_version: 1
source: "run-config.json"
tested_configuration:
  package: "utilization-detectors"
  endpoint_picker:
    image: "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0"
    replicas: 1
  model_service:
    image: "registry.redhat.io/rhaii/vllm-cuda-rhel9@sha256:ad06abf3bb5235ebb5b2df84cd1b9fd09e823f0ff2eebfc82bb4590275ccfe0b"
    model: "openai/gpt-oss-20b"
    gpu: "1 NVIDIA H100"
    prefix_cache: "disabled"
    gpu_memory_utilization: 0.9
    max_model_length: 32768
  traffic:
    arrival_mode: "closed_loop"
    input_tokens: "varies by sweep"
    output_tokens: "varies by sweep"
  measurement:
    direct_metric_interval_seconds: 0.5
    prometheus_interval_seconds: 5
  runner:
    source: "pipeline/benchmark.py"
    sha256: "3811ec26c46bf3a26fa643698ec54bf569bb4bc99c3ea22ca18f805cb077b8e0"
    traffic_driver: "native closed loop"
    launcher: "pipeline/run-in-cluster.sh"
  execution:
    arrival_mode: "closed_loop"
    prompt_pool_size: 24
    warmup_seconds: 30
    warmup_concurrency: 2
    steady_state_trim_seconds: 30
    traffic_seed: 42
    direct_metric_interval_seconds: 0.5
    prometheus_interval_seconds: 5
    cache_mode: "off"
  sweeps:
    - name: "queue-depth threshold"
      scenario_file: "queue-depth-scenario.json"
      values:
        - 1
        - 2
        - 4
        - 5
        - 8
      matched_repeat_values:
        - 5
        - 8
      repeats: 3
    - name: "KV-cache threshold"
      scenario_file: "kv-threshold-scenario.json"
      values:
        - 0.5
        - 0.6
        - 0.7
        - 0.75
        - 0.8
        - 0.9
        - 1.0
      matched_repeat_values:
        - "flow-control-off"
        - 0.75
        - 0.8
      repeats: 3
