"""py-07 — AsyncLLMClient. STARTER: the signatures and docstrings are the contract; fill in the
bodies.

The client composes the two limits every real provider imposes: a TokenBucket for requests
per second and an asyncio.Semaphore for connections in flight. It retries a TransientError
exactly once, after `backoff` seconds, and it never swallows cancellation.
"""

from __future__ import annotations

import asyncio
from collections.abc import Awaitable, Callable, Sequence

from .bucket import TokenBucket
from .transport import MockTransport

SleepFn = Callable[[float], Awaitable[None]]


class AsyncLLMClient:
    def __init__(
        self,
        transport: MockTransport,
        bucket: TokenBucket,
        max_in_flight: int,
        *,
        backoff: float = 0.5,
        sleep: SleepFn = asyncio.sleep,
    ) -> None:
        """`max_in_flight` bounds concurrent `transport.send` calls (asyncio.Semaphore);
        `backoff` is the wait before the single retry; `sleep` is what waits (injected so the
        tests can fake the clock). Raise ValueError if max_in_flight < 1."""
        ...

    async def ask(self, prompt: str) -> str:
        """One request: take a semaphore slot, then a bucket token, then `transport.send`.
        On TransientError: `await sleep(backoff)`, take a NEW token (a retry is a new
        request) and send once more; a second failure propagates. asyncio.CancelledError is
        never caught — a cancelled ask ends cancelled, with no retry."""
        ...

    async def ask_many(self, prompts: Sequence[str]) -> list[str]:
        """All prompts concurrently (within the two limits); the result list is in PROMPT
        order, whatever order the replies arrive in."""
        ...
