#!/usr/bin/env python3
"""mail/index_mail - rebuild the brain with every markdown under var/mail.

Ladybug corrupts its WAL when brand-new leafs are bulk-inserted while the
FTS/VECTOR indexes already exist, so indexing ALWAYS runs as a fresh rebuild
(repo corpus + var/mail), matching the proven-safe `kb/index --rebuild` path.
Conversion and indexing stay separate: conversion can crash in native docling
and must not leave the brain DB mid-transaction.

  bin/mail/index_mail                 rebuild the index incl. all mail
  bin/mail/index_mail --dry-run       count without writing
  bin/mail/index_mail --limit N       cap messages included
  bin/mail/index_mail --since D       only messages dated >= D (YYYY-MM-DD)
"""
from __future__ import annotations

import argparse
import json
import sys
from pathlib import Path

ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(ROOT / "bin" / "tools"))

from kblib import DB_PATH, VAR, connect, ensure_indexes, init_schema, stats, upsert_leaf  # noqa: E402
from mdleaves import read_markdown, to_all, walk_markdown  # noqa: E402


def msg_date(md: Path) -> str:
    j = md.parent / "message.json"
    try:
        d = json.loads(j.read_text(encoding="utf-8"))
        return (d.get("receivedDate") or d.get("receivedAt") or "")[:10]
    except Exception:
        return ""


def mail_leafs(limit: int, since: str, repo: str = "ooMail") -> list[dict]:
    root = ROOT / "var" / "mail"
    mds = sorted(root.rglob("message.md"))
    if since:
        mds = [m for m in mds if msg_date(m) >= since]
    if limit:
        mds = mds[:limit]
    leafs: list[dict] = []
    for md in mds:
        files = [md] + sorted((md.parent / "attachments").glob("*.md"))
        for f in files:
            if not f.exists():
                continue
            for lf in to_all(read_markdown(f), f, repo=repo):
                lf["source"] = f"ooMail:{md.parent.name}:{f.name}"
                lf["how"] = "mail/import"
                leafs.append(lf)
    return leafs


def main(argv: list[str]) -> int:
    p = argparse.ArgumentParser(description="rebuild the brain incl. all mail")
    p.add_argument("--dry-run", action="store_true", help="count only, write nothing")
    p.add_argument("--limit", type=int, default=0, help="cap messages included")
    p.add_argument("--since", default="", help="only messages dated >= YYYY-MM-DD")
    p.add_argument("--json", action="store_true")
    a = p.parse_args(argv)

    mail = mail_leafs(a.limit, a.since)
    if a.dry_run:
        print(f"mail/index_mail: {len(mail)} mail leafs would be indexed")
        return 0

    # Fresh rebuild: delete DB, index repo corpus + mail, create indexes once
    # at the end. Never insert into an already-indexed DB (WAL corruption).
    VAR.mkdir(exist_ok=True)
    if DB_PATH.exists():
        DB_PATH.unlink()

    corpus = _load_corpus()
    leafs = corpus + mail

    db, conn = connect(DB_PATH, read_only=False)
    init_schema(conn)
    embed = _embedder()
    done, total = _index_leafs(conn, leafs, embed)
    ensure_indexes(conn)
    s = stats(conn)
    conn.close()
    db.close()

    result = {"indexed": done, "corpus_total": total, "mail_leafs": len(mail),
              **{k: v for k, v in s.items() if k in ("total", "by_root")}}
    print(json.dumps(result, indent=2) if a.json else
          f"mail/index_mail: indexed {done}/{total} leafs (mail={len(mail)}); db total {s['total']}")
    return 0


CORPUS_DEFAULTS = ["README.md", "PLAN.md", "AGENTS.md", "docs", "skills"]


def _load_corpus() -> list[dict]:
    files: list[Path] = []
    for entry in CORPUS_DEFAULTS:
        p = ROOT / entry
        if p.is_file():
            files.append(p)
        elif p.is_dir():
            files.extend(walk_markdown(p))
    leafs: list[dict] = []
    for path in files:
        try:
            leafs.extend(to_all(read_markdown(path), path, repo="eSlider/2dph"))
        except OSError as e:
            print(f"mail/index_mail: skip {path}: {e}", file=sys.stderr)
    return leafs


def _index_leafs(conn, leafs: list[dict], embed_fn) -> tuple[int, int]:
    count = 0
    for lf in leafs:
        query = f"{lf['heading']}\n\n{lf['text']}"
        emb = embed_fn(lf["text"]) if lf["text"] else None
        upsert_leaf(conn, text=query, root="info", confidence="confirmed",
                    source=lf["source"], source_rev="mail" if lf.get("how") == "mail/import" else "working-tree",
                    how=lf.get("how", "kb/index"), loc=lf["source"], type_=lf.get("type", "reference"),
                    embedding=emb)
        count += 1
    return count, len(leafs)


def _embedder():
    from model2vec import StaticModel
    model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
    return lambda text: model.encode([text])[0].astype(float).tolist()


if __name__ == "__main__":
    sys.exit(main(sys.argv[1:]))
