"""RAGAS-style faithfulness / answer-relevance / context-precision, judged by a deterministic
`MockJudge` (a fixture lookup, never a model) so every score in the tests is exact.

`faithfulness`, `answer_relevance`, `context_precision` and `run` below are what you implement.
`MockJudge`, `EvalRow` and `_similarity` are plumbing: the fixture-backed stand-in for an LLM
judge, the row type, and the token-overlap similarity `answer_relevance` needs.
"""

from __future__ import annotations

import re
from dataclasses import dataclass, field

_TOKEN_RE = re.compile(r"[a-z0-9]+")


def _tokens(text: str) -> set[str]:
    return set(_TOKEN_RE.findall(text.lower()))


def _similarity(a: str, b: str) -> float:
    """Jaccard similarity of `a` and `b`'s lowercase word tokens; 1.0 when both have no tokens
    at all, 0.0 when exactly one does."""
    ta, tb = _tokens(a), _tokens(b)
    if not ta and not tb:
        return 1.0
    if not ta or not tb:
        return 0.0
    return len(ta & tb) / len(ta | tb)


@dataclass
class MockJudge:
    """A deterministic stand-in for an LLM judge: every call below is a lookup in a fixture
    dict, never a model call, so the metrics computed from it are exact numbers in the tests
    instead of "roughly right"."""

    statement_support: dict[tuple[str, str], bool] = field(default_factory=dict)
    generated_questions: dict[str, list[str]] = field(default_factory=dict)
    chunk_relevance: dict[tuple[str, str], bool] = field(default_factory=dict)

    def supports(self, statement: str, chunk: str) -> bool:
        """Would the judge say `chunk` backs up `statement`?"""
        return self.statement_support.get((statement, chunk), False)

    def questions_from(self, answer: str) -> list[str]:
        """Questions the judge reckons `answer` is answering -- what `answer_relevance` compares
        the row's real question against."""
        return self.generated_questions.get(answer, [])

    def relevant(self, question: str, chunk: str) -> bool:
        """Would the judge say `chunk` is relevant to answering `question`?"""
        return self.chunk_relevance.get((question, chunk), False)


@dataclass
class EvalRow:
    question: str
    answer: str
    statements: list[str]
    contexts: list[str]


def faithfulness(row: EvalRow, judge: MockJudge) -> float:
    """The fraction of `row.statements` that `judge.supports` against *some* chunk in
    `row.contexts` (a statement counts once it is supported by any single chunk, not all of
    them). 1.0 when `row.statements` is empty -- an answer that states nothing states nothing
    unfaithful."""
    ...


def answer_relevance(row: EvalRow, judge: MockJudge) -> float:
    """The average `_similarity(row.question, gq)` over `gq in judge.questions_from(row.answer)`
    -- the questions the judge reckons the answer is actually answering, each compared back to
    the question that was actually asked. 0.0 when the judge generated no questions at all (a
    non-answer is not relevant to anything)."""
    ...


def context_precision(row: EvalRow, judge: MockJudge) -> float:
    """RAGAS's average precision over `row.contexts` in rank order: walk the list 1-indexed as
    `k`, and for every `k` whose chunk is `judge.relevant(row.question, chunk)`, add precision@k
    (relevant chunks seen in the top `k` / `k`) to a running sum. Return that sum divided by the
    total number of relevant chunks in `row.contexts` -- so a relevant chunk ranked first scores
    higher than the same chunk ranked last, and a docset with no relevant chunk at all scores
    0.0 (never divide by zero)."""
    ...


def run(dataset: list[EvalRow], judge: MockJudge) -> str:
    """A GitHub-flavored markdown table: a header row `question | faithfulness |
    answer_relevance | context_precision`, a `| --- | --- | --- | --- |` separator, one row per
    `dataset` entry with each score formatted to 2 decimal places, and a final row whose question
    column reads `**mean**` holding each column's mean (also to 2 decimal places)."""
    ...
