"""py-07 — TokenBucket. STARTER: the signature and docstrings are the contract; fill in the bodies.

A token bucket bounds how OFTEN requests may start: it holds up to `capacity` tokens, refills
at `rate` tokens per second, and a request takes `n` tokens before it may begin. A full bucket
lets a burst of `capacity` requests through at once; after that, requests are spaced 1/rate
apart. It says nothing about how many requests are in flight — that is the client's semaphore.

`clock` and `sleep` are injected so the tests can drive it on a fake clock (see
tests/conftest.py). Never call time.sleep here: the whole event loop would stop with you.
"""

from __future__ import annotations

import asyncio
import time
from collections.abc import Awaitable, Callable

SleepFn = Callable[[float], Awaitable[None]]


class TokenBucket:
    def __init__(
        self,
        rate: float,
        capacity: int,
        *,
        clock: Callable[[], float] = time.monotonic,
        sleep: SleepFn = asyncio.sleep,
    ) -> None:
        """Start FULL (`capacity` tokens). `rate` tokens are added per second, continuously,
        never beyond `capacity` — an hour idle does not earn an hour of burst.
        Raise ValueError if rate <= 0 or capacity < 1."""
        ...

    async def acquire(self, n: int = 1) -> None:
        """Wait until `n` tokens are available, then take them. Waiting is `await sleep(...)`
        on the injected sleep, never a blocking call. Concurrent callers are served in the
        order they asked (a Lock or a queue), and the bucket never over-admits: in any window
        of one second at most `capacity + rate` starts, and once drained at most `rate`."""
        ...
