import pytest

from ragas_mock import EvalRow, MockJudge, answer_relevance, context_precision, faithfulness, run

EVEREST_CONTEXT = "Mount Everest, the tallest mountain above sea level, has its summit at 8849m."


def test_faithfulness_scores_fraction_supported():
    judge = MockJudge(
        statement_support={
            ("Everest is the tallest mountain.", EVEREST_CONTEXT): True,
            ("Its summit is 8849 meters high.", EVEREST_CONTEXT): True,
        }
    )
    full = EvalRow(
        question="How tall is Everest?",
        answer="",
        statements=["Everest is the tallest mountain.", "Its summit is 8849 meters high."],
        contexts=[EVEREST_CONTEXT],
    )
    assert faithfulness(full, judge) == pytest.approx(1.0), (
        "both statements are in judge.statement_support as True against the one context; "
        f"expected 1.0, got {faithfulness(full, judge)}"
    )

    partial = EvalRow(
        question="How tall is Everest, and when was it first climbed?",
        answer="",
        statements=[
            "Everest is the tallest mountain.",
            "Its summit is 8849 meters high.",
            "Everest was first climbed in 1953.",
        ],
        contexts=[EVEREST_CONTEXT],
    )
    score = faithfulness(partial, judge)
    assert score == pytest.approx(2 / 3), (
        "the third statement has no entry in judge.statement_support, so judge.supports "
        f"defaults to False for it; 2 of 3 statements supported should give 2/3, got {score}"
    )


def test_faithfulness_empty_statements_scores_one():
    row = EvalRow(question="Q", answer="A", statements=[], contexts=[EVEREST_CONTEXT])
    score = faithfulness(row, MockJudge())
    assert score == pytest.approx(1.0), (
        f"an answer with zero statements has stated nothing unfaithful; expected 1.0, got {score}"
    )


def test_answer_relevance_close_paraphrase_scores_higher_than_unrelated_answer():
    question = "How tall is Mount Everest?"
    close_answer = "Mount Everest's summit reaches 8849 meters above sea level."
    far_answer = "Coral reefs support a quarter of all marine species."
    judge = MockJudge(
        generated_questions={
            close_answer: [
                "How tall is Mount Everest?",
                "What is the height of Mount Everest?",
            ],
            far_answer: ["What do coral reefs support?"],
        }
    )
    close_row = EvalRow(question=question, answer=close_answer, statements=[], contexts=[])
    far_row = EvalRow(question=question, answer=far_answer, statements=[], contexts=[])

    close_score = answer_relevance(close_row, judge)
    assert close_score == pytest.approx(2 / 3), (
        "one generated question is an exact match (similarity 1.0), the other shares only "
        f"'is', 'mount', 'everest' out of 9 unique tokens (similarity 1/3); average is 2/3, "
        f"got {close_score}"
    )

    far_score = answer_relevance(far_row, judge)
    assert far_score == pytest.approx(0.0), (
        f"the coral-reef answer's generated question shares no token with the Everest question; "
        f"expected 0.0, got {far_score}"
    )

    no_fixture_row = EvalRow(
        question=question, answer="an answer not in the fixture", statements=[], contexts=[]
    )
    no_fixture_score = answer_relevance(no_fixture_row, judge)
    assert no_fixture_score == pytest.approx(0.0), (
        "judge.questions_from returns [] for an answer with no fixture entry -- averaging over "
        f"zero questions must not divide by zero, and a non-answer is not relevant; "
        f"got {no_fixture_score}"
    )


def test_context_precision_relevant_chunk_ranked_first_beats_ranked_last():
    question = "Where do coral reefs form?"
    relevant_chunk = "Coral reefs form in warm, shallow ocean waters."
    irrelevant_chunk = "The stock market closed higher on Tuesday."
    judge = MockJudge(chunk_relevance={(question, relevant_chunk): True})

    ranked_first = EvalRow(
        question=question, answer="", statements=[], contexts=[relevant_chunk, irrelevant_chunk]
    )
    ranked_last = EvalRow(
        question=question, answer="", statements=[], contexts=[irrelevant_chunk, relevant_chunk]
    )

    first_score = context_precision(ranked_first, judge)
    last_score = context_precision(ranked_last, judge)
    assert first_score == pytest.approx(1.0), (
        f"the only relevant chunk is at rank 1, so precision@1 (1/1) is the whole score; "
        f"expected 1.0, got {first_score}"
    )
    assert last_score == pytest.approx(0.5), (
        "the same relevant chunk at rank 2 contributes precision@2 (1/2); a version that ignores "
        f"rank and just divides relevant-count by total-count would wrongly give 0.5 for both "
        f"orderings -- expected 0.5 here, got {last_score}"
    )


def test_context_precision_no_relevant_chunk_scores_zero():
    question = "Where do coral reefs form?"
    row = EvalRow(
        question=question,
        answer="",
        statements=[],
        contexts=["The stock market closed higher on Tuesday.", "Rain is expected on Friday."],
    )
    score = context_precision(row, MockJudge())
    assert score == pytest.approx(0.0), (
        f"no chunk is judge.relevant for this question; must return 0.0, not divide by the zero "
        f"relevant-chunk count, got {score}"
    )


def test_run_writes_markdown_table_with_mean_row():
    judge = MockJudge(
        statement_support={("S1", "C1"): True, ("S2a", "C2"): True},
        generated_questions={"A1": ["Q1"], "A2": ["X2"]},
        chunk_relevance={("Q1", "C1"): True},
    )
    dataset = [
        EvalRow(question="Q1", answer="A1", statements=["S1"], contexts=["C1"]),
        EvalRow(question="Q2", answer="A2", statements=["S2a", "S2b"], contexts=["C2"]),
    ]
    table = run(dataset, judge)

    assert "| Q1 | 1.00 | 1.00 | 1.00 |" in table, (
        f"row 1 is fully supported, its generated question matches exactly, and its only chunk "
        f"is relevant -- every score should be 1.00; got:\n{table}"
    )
    assert "| Q2 | 0.50 | 0.00 | 0.00 |" in table, (
        "row 2: 1 of 2 statements supported (0.50), the generated question 'X2' shares no token "
        f"with 'Q2' (0.00), and its chunk is not in chunk_relevance (0.00); got:\n{table}"
    )
    assert "| **mean** | 0.75 | 0.50 | 0.50 |" in table, (
        f"the mean row must average each column across both dataset rows; got:\n{table}"
    )
