import pytest
from app.chunking.text_chunker import TextChunker
from app.models.document import RepoDocumentSection


def make_section(title, level, content):
    return RepoDocumentSection(
        doc_id="doc1",
        repo_id="repo1",
        doc_path="README.md",
        heading=title,
        section_title=title,
        section_level=level,
        parent_section=None,
        heading_path=title,
        start_line=1,
        end_line=2,
        content=content,
    )


@pytest.fixture
def chunker():
    return TextChunker(chunk_size=100, chunk_overlap=20)


def test_short_section_produces_one_chunk(chunker):
    section = make_section("Intro", 1, "Short content.")
    chunks = chunker.chunk_sections([section], "repo1", "README.md")
    assert len(chunks) == 1
    assert chunks[0].chunk_text == "Short content."
    assert chunks[0].source_type == "repo_doc"


def test_long_section_splits(chunker):
    long_text = "A" * 50 + ". " + "B" * 50 + ". " + "C" * 50
    section = make_section("Long", 1, long_text)
    chunks = chunker.chunk_sections([section], "repo1", "README.md")
    assert len(chunks) > 1


def test_empty_section_skipped(chunker):
    section = make_section("Empty", 1, "   ")
    chunks = chunker.chunk_sections([section], "repo1", "README.md")
    assert chunks == []


def test_chunk_hash_is_consistent(chunker):
    section = make_section("H", 1, "Some text here.")
    chunks1 = chunker.chunk_sections([section], "repo1", "README.md")
    chunks2 = chunker.chunk_sections([section], "repo1", "README.md")
    assert chunks1[0].chunk_id == chunks2[0].chunk_id
    assert chunks1[0].chunk_hash == chunks2[0].chunk_hash


def test_chunk_metadata_includes_heading_hierarchy(chunker):
    section = RepoDocumentSection(
        doc_id="doc1",
        repo_id="repo1",
        doc_path="README.md",
        heading="Scope",
        section_title="Scope",
        section_level=3,
        parent_section="Purpose",
        heading_path="Architecture / Purpose / Scope",
        start_line=10,
        end_line=20,
        content="Some text here.",
    )
    chunks = chunker.chunk_sections([section], "repo1", "README.md")

    assert chunks[0].metadata["parent_section"] == "Purpose"
    assert chunks[0].metadata["heading_path"] == "Architecture / Purpose / Scope"
    assert chunks[0].metadata["start_line"] == 10
    assert chunks[0].metadata["end_line"] == 20


def test_chunk_metadata_includes_content_boundaries_and_types(chunker):
    section = RepoDocumentSection(
        doc_id="doc1",
        repo_id="repo1",
        doc_path="README.md",
        heading="API",
        section_title="API",
        section_level=1,
        parent_section=None,
        heading_path="API",
        start_line=1,
        end_line=20,
        content_start_line=2,
        content_end_line=19,
        content_types=["paragraph", "table"],
        content="Some text here.",
    )
    chunks = chunker.chunk_sections([section], "repo1", "README.md")

    assert chunks[0].metadata["content_start_line"] == 2
    assert chunks[0].metadata["content_end_line"] == 19
    assert chunks[0].metadata["content_types"] == ["paragraph", "table"]


def test_chunk_ordering_is_deterministic_and_preserves_section_order():
    chunker = TextChunker(chunk_size=40, chunk_overlap=10, max_tokens=20)
    s1 = make_section("A", 1, "First sentence one. First sentence two. First sentence three.")
    s2 = make_section("B", 1, "Second sentence one. Second sentence two. Second sentence three.")

    first = chunker.chunk_sections([s1, s2], "repo1", "README.md")
    second = chunker.chunk_sections([s1, s2], "repo1", "README.md")

    assert [c.chunk_text for c in first] == [c.chunk_text for c in second]
    assert [c.metadata["chunk_order"] for c in first] == list(range(len(first)))
    assert first[0].section_title == "A"
    assert first[-1].section_title == "B"


def test_large_section_splits_on_paragraph_boundaries_when_possible():
    chunker = TextChunker(chunk_size=90, chunk_overlap=10, max_tokens=60)
    content = (
        "Paragraph one has enough words to stand as its own chunk and remain readable.\n\n"
        "Paragraph two also has meaningful content and should be preserved as a paragraph unit.\n\n"
        "Paragraph three keeps the pattern going for deterministic splitting behavior."
    )
    section = make_section("Guide", 1, content)

    chunks = chunker.chunk_sections([section], "repo1", "README.md")

    assert len(chunks) >= 2
    assert "\n\n" not in chunks[0].chunk_text
    assert chunks[0].chunk_text.startswith("Paragraph one")


def test_oversized_single_paragraph_respects_token_and_character_limits():
    chunker = TextChunker(chunk_size=80, chunk_overlap=10, max_tokens=14)
    sentence = "word1 word2 word3 word4 word5 word6 word7 word8 word9 word10 word11 word12"
    content = f"{sentence}. {sentence}. {sentence}."
    section = make_section("Limits", 1, content)

    chunks = chunker.chunk_sections([section], "repo1", "README.md")

    assert len(chunks) > 1
    assert all(len(c.chunk_text) <= 80 for c in chunks)
    assert all(c.metadata["token_count"] <= 14 for c in chunks)


def test_chunk_metadata_includes_repo_and_hierarchy_association():
    chunker = TextChunker(chunk_size=120, chunk_overlap=20, max_tokens=50)
    section = RepoDocumentSection(
        doc_id="doc1",
        repo_id="repo1",
        doc_path="docs/api.md",
        heading="Scope",
        section_title="Scope",
        section_level=2,
        parent_section="Architecture",
        heading_path="Architecture / Scope",
        start_line=5,
        end_line=20,
        content_start_line=6,
        content_end_line=19,
        content_types=["paragraph"],
        content="A readable chunk body for association checks.",
    )

    chunks = chunker.chunk_sections([section], "repoX", "docs/api.md")

    assert chunks[0].repo_id == "repoX"
    assert chunks[0].doc_path == "docs/api.md"
    assert chunks[0].metadata["heading_path"] == "Architecture / Scope"
    assert chunks[0].metadata["section_index"] == 0
    assert chunks[0].metadata["char_count"] == len(chunks[0].chunk_text)
    assert chunks[0].metadata["repo_id"] == "repoX"
    assert chunks[0].metadata["doc_path"] == "docs/api.md"
    assert chunks[0].metadata["chunk_id"] == chunks[0].chunk_id
    assert chunks[0].metadata["chunk_type"] == "text"
    assert chunks[0].metadata["source_type"] == "repo_doc"
    assert chunks[0].metadata["section_title"] == "Scope"
    assert chunks[0].metadata["section_level"] == 2
    assert chunks[0].metadata["start_line"] == 5
    assert chunks[0].metadata["end_line"] == 20


def test_chunk_metadata_includes_snapshot_and_markdown_entity_association():
    chunker = TextChunker(chunk_size=120, chunk_overlap=20, max_tokens=50)
    section = RepoDocumentSection(
        doc_id="doc1",
        repo_id="repo1",
        doc_path="docs/api.md",
        heading="API",
        section_title="API",
        section_level=1,
        parent_section=None,
        heading_path="API",
        start_line=1,
        end_line=10,
        content_types=["paragraph", "table"],
        content="Endpoints table and guide links.",
    )

    chunks = chunker.chunk_sections(
        [section],
        "repo1",
        "docs/api.md",
        snapshot_metadata={"commit_hash": "abc123", "snapshot_id": "snap_1"},
        section_entities={"API": {"links": 2, "code": 1, "tables": 1}},
    )

    assert chunks[0].metadata["snapshot"] == {
        "commit_hash": "abc123",
        "snapshot_id": "snap_1",
    }
    assert chunks[0].metadata["markdown_entities"] == {
        "links": 2,
        "code": 1,
        "tables": 1,
    }
    assert chunks[0].metadata["chunk_type"] == "table"


def test_chunk_generation_logs_and_skips_section_failures(monkeypatch):
    chunker = TextChunker(chunk_size=120, chunk_overlap=20, max_tokens=50)
    section = make_section("Broken", 1, "This content triggers split failure.")
    calls = []

    class _BindLogger:
        def warning(self, msg):
            calls.append(msg)

    class _Logger:
        def bind(self, **_kwargs):
            return _BindLogger()

    def _boom(_text):
        raise RuntimeError("boom")

    monkeypatch.setattr("app.chunking.text_chunker.logger", _Logger())
    monkeypatch.setattr(chunker, "_split_text", _boom)

    chunks = chunker.chunk_sections([section], "repo1", "README.md")

    assert chunks == []
    assert any("Failed to split section into chunks" in msg for msg in calls)
