{
  "source": "Own Lium test pod (RTX PRO 6000 Blackwell 96 GB), Surogate Rune 26B-A4B v3 with the production vLLM image and flags; NOT the production API, no customer traffic",
  "measured": "2026-10-07",
  "method": "Unique random state per request, input_tokens read from the engine usage field, 30 requests per cell, 0 errors. Deterministic production profile (batch-invariant, no chunked prefill), GPU running only Rune.",
  "published_limits": { "s1-fast": 4096, "s1-pro": 32000, "s1-vision": 32000 },
  "single_request_latency_ms_p50": {
    "4k (4,103-4,164 tokens)": 191,
    "8k (8,148-8,262 tokens)": 396,
    "16k (16,295-16,433 tokens)": 924,
    "32k (31,865-32,047 tokens)": 2299
  },
  "note": "Concurrent long requests queue (latency is about the number of concurrent requests times the single-request time). The gateway therefore admits one long-input request at a time and answers further ones with HTTP 429 and Retry-After: 3."
}
