import pytest

from citation_check import check, support

CHUNK1 = "The Eiffel Tower was completed in 1889 and stands 330 meters tall in Paris."
CHUNK2 = "Photosynthesis converts carbon dioxide and water into glucose and oxygen using sunlight."
CHUNK3 = "The Great Wall of China stretches approximately 21196 kilometers across northern China."


def test_support_high_overlap_scores_full():
    claim = "The Eiffel Tower was completed in 1889 in Paris."
    assert support(claim, CHUNK1) == pytest.approx(1.0), (
        f"every content word and the number in {claim!r} is in the chunk; expected 1.0"
    )


def test_support_disjoint_claim_scores_zero():
    claim = "Photosynthesis needs sunlight plus water for growth."
    score = support(claim, CHUNK1)
    assert score == pytest.approx(0.0), (
        f"{claim!r} shares no content word with the Eiffel Tower chunk; got score={score}"
    )


def test_support_number_mismatch_is_penalized():
    claim = "The Eiffel Tower was completed in 1875 in Paris."
    score = support(claim, CHUNK1)
    assert 0.0 < score <= 0.4, (
        f"claim states 1875 but the chunk says 1889 -- word overlap alone is 0.8, "
        f"the number penalty must pull the score to 0.4 or below without zeroing it (a claim "
        f"that is mostly right with one wrong figure is not worthless); got score={score}"
    )


def test_support_stopwords_alone_score_zero():
    claim = "It was of this and that had been for it."
    score = support(claim, CHUNK2)
    assert score == pytest.approx(0.0), (
        f"{claim!r} has no content words (only stopwords); a shared 'and' or 'it' with the "
        f"chunk must not count toward overlap, so the score must be 0.0, got score={score}"
    )


def test_check_labels_supported_unsupported_wrong_number():
    answer = (
        "The Eiffel Tower was completed in 1889 in Paris. [1]\n"
        "Photosynthesis produces energy from sunlight and rainbows. [2]\n"
        "The Great Wall of China stretches 5000 kilometers. [3]\n"
    )
    report = check(answer, [CHUNK1, CHUNK2, CHUNK3])
    assert report.statuses == ["supported", "unsupported", "wrong-number"], (
        f"expected [supported, unsupported, wrong-number] (claim 1 is fully backed by chunk 1, "
        f"claim 2 barely overlaps chunk 2, claim 3's word overlap with chunk 3 is high but its "
        f"5000 isn't chunk 3's 21196); got {report.statuses}"
    )


def test_check_citation_index_is_one_indexed():
    answer = (
        "Photosynthesis converts carbon dioxide and water into glucose using sunlight. [2]\n"
        "The Eiffel Tower was completed in 1889 in Paris. [1]\n"
    )
    report = check(answer, [CHUNK1, CHUNK2])
    assert report.statuses == ["supported", "supported"], (
        f"[2] names chunks[1] (CHUNK2) and [1] names chunks[0] (CHUNK1) -- each claim is fully "
        f"backed by the chunk its own marker names, in whatever order the lines appear; "
        f"got {report.statuses}"
    )


def test_check_out_of_range_citation_raises():
    with pytest.raises(ValueError):
        check("Some claim with a bad marker. [3]", [CHUNK1, CHUNK2])
    with pytest.raises(ValueError):
        check("Some claim with a zero marker. [0]", [CHUNK1, CHUNK2])
