import pytest
from app.chunking.text_chunker import TextChunker
from app.models.document import RepoDocumentSection


def make_section(title, level, content):
    return RepoDocumentSection(
        doc_id="doc1",
        repo_id="repo1",
        doc_path="README.md",
        heading=title,
        section_title=title,
        section_level=level,
        content=content,
    )


@pytest.fixture
def chunker():
    return TextChunker(chunk_size=100, chunk_overlap=20)


def test_short_section_produces_one_chunk(chunker):
    section = make_section("Intro", 1, "Short content.")
    chunks = chunker.chunk_sections([section], "repo1", "README.md")
    assert len(chunks) == 1
    assert chunks[0].chunk_text == "Short content."
    assert chunks[0].source_type == "repo_doc"


def test_long_section_splits(chunker):
    long_text = "A" * 50 + ". " + "B" * 50 + ". " + "C" * 50
    section = make_section("Long", 1, long_text)
    chunks = chunker.chunk_sections([section], "repo1", "README.md")
    assert len(chunks) > 1


def test_empty_section_skipped(chunker):
    section = make_section("Empty", 1, "   ")
    chunks = chunker.chunk_sections([section], "repo1", "README.md")
    assert chunks == []


def test_chunk_hash_is_consistent(chunker):
    section = make_section("H", 1, "Some text here.")
    chunks1 = chunker.chunk_sections([section], "repo1", "README.md")
    chunks2 = chunker.chunk_sections([section], "repo1", "README.md")
    assert chunks1[0].chunk_hash == chunks2[0].chunk_hash
