{
  "schema": "stable-upstream-selected-workload-shapes-benchmark-v1",
  "date": "2026-08-09",
  "topology": {
    "endpoint_picker_replicas": 1,
    "model_replicas": 1,
    "gpus_per_model_replica": 1
  },
  "hardware": {
    "gpu_per_model_replica": "NVIDIA H100"
  },
  "model_service": {
    "model": "GPT-OSS 20B",
    "runtime": "Red Hat AI Inference Server vLLM",
    "tensor_parallel_size": 1,
    "max_model_len": 32768,
    "max_num_batched_tokens": 8192,
    "max_num_seqs": 128,
    "gpu_memory_utilization": 0.9,
    "prefix_cache": "disabled"
  },
  "endpoint_picker": {
    "version": "llm-d Endpoint Picker v0.9.0",
    "flow_control_gate": "enabled",
    "detector": "request-concurrency",
    "max_concurrency": 128,
    "headroom": 0.1,
    "picker": "random",
    "priority_bands": [
      100,
      50,
      0,
      -10
    ],
    "fairness": "round-robin within each priority band"
  },
  "traffic": {
    "production_scenarios": [
      {
        "name": "chat-short-output",
        "duration_s": 180,
        "analysis_windows": [
          {
            "name": "baseline",
            "start_s": 20,
            "end_s": 55
          },
          {
            "name": "surge",
            "start_s": 80,
            "end_s": 135
          },
          {
            "name": "recovery",
            "start_s": 155,
            "end_s": 175
          }
        ],
        "tenants": [
          {
            "fairness_id": "chat-short-output",
            "priority": 100,
            "input_tokens": 1024,
            "output_tokens": 128,
            "phases": [
              {
                "start_s": 0,
                "duration_s": 60,
                "rate_pattern": "noisy_sinusoidal",
                "rate_center": 12,
                "rate_amplitude": 2.5,
                "period_s": 37,
                "phase_offset": 0.0,
                "rate_noise": 0.6,
                "rate_spike_probability": 0.02
              },
              {
                "start_s": 60,
                "duration_s": 90,
                "rate_pattern": "noisy_sinusoidal",
                "rate_center": 24,
                "rate_amplitude": 5,
                "period_s": 31,
                "phase_offset": 0.0,
                "rate_noise": 1.2,
                "rate_spike_probability": 0.02,
                "ramp_s": 10
              },
              {
                "start_s": 150,
                "duration_s": 30,
                "rate_pattern": "noisy_sinusoidal",
                "rate_center": 8,
                "rate_amplitude": 2,
                "period_s": 23,
                "phase_offset": 0.0,
                "rate_noise": 0.4
              }
            ]
          }
        ]
      },
      {
        "name": "agentic-longer-output",
        "duration_s": 180,
        "analysis_windows": [
          {
            "name": "baseline",
            "start_s": 20,
            "end_s": 55
          },
          {
            "name": "surge",
            "start_s": 80,
            "end_s": 135
          },
          {
            "name": "recovery",
            "start_s": 155,
            "end_s": 175
          }
        ],
        "tenants": [
          {
            "fairness_id": "agentic-longer-output",
            "priority": 100,
            "input_tokens": 4096,
            "output_tokens": 512,
            "phases": [
              {
                "start_s": 0,
                "duration_s": 60,
                "rate_pattern": "noisy_sinusoidal",
                "rate_center": 2,
                "rate_amplitude": 0.5,
                "period_s": 37,
                "phase_offset": 0.2,
                "rate_noise": 0.1,
                "rate_spike_probability": 0.02
              },
              {
                "start_s": 60,
                "duration_s": 90,
                "rate_pattern": "noisy_sinusoidal",
                "rate_center": 7.5,
                "rate_amplitude": 1.4,
                "period_s": 31,
                "phase_offset": 0.2,
                "rate_noise": 0.375,
                "rate_spike_probability": 0.02,
                "ramp_s": 10
              },
              {
                "start_s": 150,
                "duration_s": 30,
                "rate_pattern": "noisy_sinusoidal",
                "rate_center": 2,
                "rate_amplitude": 0.5,
                "period_s": 23,
                "phase_offset": 0.2,
                "rate_noise": 0.1
              }
            ]
          }
        ]
      }
    ]
  },
  "execution": {
    "arrival": "open-loop Poisson replay with noisy sinusoidal phases",
    "duration_seconds": 180,
    "production_repeats_per_shape": 3,
    "load_generator": "GuideLLM",
    "cache_mode": "off"
  },
  "runner": {
    "scenario_source": "pipeline/benchmark.py",
    "sha256": "3811ec26c46bf3a26fa643698ec54bf569bb4bc99c3ea22ca18f805cb077b8e0",
    "traffic_driver": "GuideLLM",
    "guidellm_version": "0.7.0",
    "trace_compiler": "pipeline/guidellm_trace.py",
    "launcher": "pipeline/run_guidellm_scenario.py",
    "scenario_file": "scenarios.json"
  }
}
