"""A tiny in-process OpenAI-compatible stub for the load-test module to hit through
httpx.MockTransport — no real socket, no new dependency (aiohttp) pulled in just for a test
fixture. It starts fresh in every test that asks for `fake_server`.

Its simulated latency scales with how many requests it is CURRENTLY serving concurrently
(`in_flight`), so a real run of run_load_test against it at increasing concurrency levels
produces a monotonically increasing p50 — the qualitative shape a real vLLM server under GPU
contention has, and the thing test_load_run.py's one test asserts.
"""

from __future__ import annotations

import asyncio

import httpx
import pytest


class FakeVLLMServer:
    """`base_latency_s` is what a single, uncontended request costs; `per_inflight_s` is
    added for every OTHER request the server is serving at the moment this one starts."""

    def __init__(self, base_latency_s: float = 0.01, per_inflight_s: float = 0.02) -> None:
        self.base_latency_s = base_latency_s
        self.per_inflight_s = per_inflight_s
        self.in_flight = 0
        self.max_in_flight_seen = 0
        self.calls = 0

    async def handle(self, request: httpx.Request) -> httpx.Response:
        self.calls += 1
        self.in_flight += 1
        self.max_in_flight_seen = max(self.max_in_flight_seen, self.in_flight)
        try:
            delay = self.base_latency_s + self.per_inflight_s * (self.in_flight - 1)
            await asyncio.sleep(delay)
            return httpx.Response(200, json={"choices": [{"text": "ok"}]})
        finally:
            self.in_flight -= 1

    def transport(self) -> httpx.MockTransport:
        return httpx.MockTransport(self.handle)


@pytest.fixture
def fake_server() -> FakeVLLMServer:
    return FakeVLLMServer()
