"""
Extract Q&A pairs from raw Slack history and write faq/topics/<slug>.md files.

Reads tools/.slack_raw.json (produced by fetch_slack.py) and regenerates:
  faq/topics/<channel-slug>.md   — Q&A pairs for that channel
  faq/INDEX.md                   — one line per topic for the agent to scan

A message is treated as a question when it ends with '?' or starts with a
common interrogative word and the thread has at least one reply.
"""

import json
import logging
import re
from pathlib import Path

logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
log = logging.getLogger("extract_qa")

RAW_INPUT = Path(__file__).parent / ".slack_raw.json"
FAQ_DIR = Path(__file__).parent.parent / "faq"
TOPICS_DIR = FAQ_DIR / "topics"
INDEX_FILE = FAQ_DIR / "INDEX.md"

_QUESTION_RE = re.compile(
    r'\?[\s.!]*$'
    r'|^(how|what|why|when|where|who|which|can|does|is|are|will|should|do)\b',
    re.IGNORECASE | re.MULTILINE,
)


def _is_question(text: str) -> bool:
    return bool(_QUESTION_RE.search(text.strip()))


def _slugify(name: str) -> str:
    return re.sub(r'[^a-z0-9]+', '-', name.lower()).strip('-')


def _first_answer(replies: list[dict]) -> str | None:
    """Return the first non-empty reply text."""
    return next(
        (r.get("text", "").strip() for r in replies if r.get("text", "").strip()),
        None,
    )


def extract_qa(messages: list[dict]) -> list[dict]:
    pairs = []
    for msg in messages:
        text = (msg.get("text") or "").strip()
        if not text or not _is_question(text):
            continue
        replies = msg.get("replies") or []
        answer = _first_answer(replies)
        if answer:
            pairs.append({"question": text, "answer": answer})
    return pairs


def write_topic(slug: str, channel_name: str, pairs: list[dict]) -> str:
    """Write faq/topics/<slug>.md and return an INDEX line."""
    lines = [f"# {channel_name}\n"]
    for i, p in enumerate(pairs, 1):
        lines.append(f"## Q{i}: {p['question']}\n")
        lines.append(f"{p['answer']}\n")
    TOPICS_DIR.mkdir(parents=True, exist_ok=True)
    (TOPICS_DIR / f"{slug}.md").write_text("\n".join(lines), encoding="utf-8")
    summary = pairs[0]["question"][:80].rstrip("?")
    return f"{slug} — {summary}"


def run():
    if not RAW_INPUT.exists():
        log.error("Raw input not found: %s — run fetch_slack.py first", RAW_INPUT)
        raise SystemExit(1)

    raw = json.loads(RAW_INPUT.read_text(encoding="utf-8"))
    index_entries = []

    for channel_name, data in raw.items():
        pairs = extract_qa(data.get("messages", []))
        log.info("#%s: %d Q&A pairs extracted", channel_name, len(pairs))
        if not pairs:
            continue
        slug = _slugify(channel_name)
        index_entries.append(write_topic(slug, channel_name, pairs))

    index_lines = [
        "# FAQ Index",
        "# One line per topic: <slug> — <short summary>",
    ] + sorted(index_entries)
    INDEX_FILE.write_text("\n".join(index_lines) + "\n", encoding="utf-8")
    log.info("Written %s with %d topic(s)", INDEX_FILE, len(index_entries))


if __name__ == "__main__":
    run()
