"""Each test isolates one way a plausible harness gets it wrong: reordering its own result
between two runs on the same fixture instead of returning the same evaluation twice."""
from sec_prompt_injection import run_harness

EXPECTED_SUCCESS_IDS = {
    "atk-01", "atk-03", "atk-06", "atk-09",
    "atk-11", "atk-13", "atk-15", "atk-17", "atk-19",
    "atk-21", "atk-24", "atk-26",
}


def test_run_harness_computes_attack_success_rate_on_the_fixture(attacks):
    result = run_harness(attacks)
    assert result.total == 40, f"fixture has 40 attacks, got total={result.total}"
    assert result.successes == 12, (
        "12 of the 40 attacks embed a literal trigger phrase the mock model follows (4 "
        "override, 5 exfiltration, 3 roleplay, 0 encoding -- those are obfuscated so the mock "
        f"model never sees a literal trigger) -- got {result.successes}"
    )
    assert result.rate == 0.3, f"12/40 == 0.3, got rate={result.rate}"
    assert set(result.success_ids) == EXPECTED_SUCCESS_IDS, (
        f"successful attack ids must be exactly {sorted(EXPECTED_SUCCESS_IDS)}, "
        f"got {sorted(result.success_ids)}"
    )
    assert result.by_category["encoding"] == {"attacks": 10, "successes": 0}, (
        f"all 10 encoding attacks are obfuscated and must score 0 successes, "
        f"got {result.by_category.get('encoding')}"
    )


def test_run_harness_is_deterministic(attacks):
    first = run_harness(attacks)
    second = run_harness(attacks)
    assert first is not None and first.total == 40, (
        f"run_harness(attacks) must return a HarnessResult with total=40, got {first!r}"
    )
    assert first == second, (
        f"run_harness must return the same evaluation for the same fixture every time it is "
        f"called -- first call returned {first}, second call returned {second}"
    )


def test_build_report_has_expected_markdown_shape(attacks):
    from sec_prompt_injection import build_report

    result = run_harness(attacks)
    report = build_report(result)
    assert report.startswith("# Prompt-injection harness report"), (
        f"report must open with the '# Prompt-injection harness report' heading, "
        f"got: {report[:60]!r}"
    )
    assert "**Attack success rate:** 12/40 (30.0%)" in report, (
        f"report must state the exact rate as '12/40 (30.0%)', got: {report!r}"
    )
    assert "| encoding | 10 | 0 | 0.0% |" in report, (
        f"per-category table must include a 0%-success encoding row (all 10 are obfuscated), "
        f"got: {report!r}"
    )
    assert "| exfiltration | 10 | 5 | 50.0% |" in report, (
        f"per-category table must include exfiltration's 5/10 row, got: {report!r}"
    )
    success_lines = [line for line in report.splitlines() if line.startswith("- atk-")]
    assert success_lines == [f"- {aid}" for aid in sorted(EXPECTED_SUCCESS_IDS)], (
        f"the 'Successful attacks' section must list every id in "
        f"{sorted(EXPECTED_SUCCESS_IDS)}, sorted, got {success_lines}"
    )
