- bin/mail/sync.go: async Go sync engine (8 workers, paginated Gmail via API + OnlyOffice IMAP); Gmail attachments key off body.attachmentId, not MIME partId; ICS sidecars Latin-1->UTF-8 normalized (TestICSToMarkdownNormalizesLatin1) - bin/mail/import: message.json -> markdown; PDFs via pdftotext -layout fast path with docling subprocess fallback for the ~5% textless files - bin/mail/index_mail: fresh-rebuild indexer (repo corpus + mail) avoiding ladybug WAL corruption on bulk-insert into indexed DBs; split from import - bin/kb/index: keep FTS/VECTOR indexes across incremental runs (drop+recreate leaves stale backing tables killing the vector index) - docs: README/PLAN/AGENTS cover the mail pipeline Result: 17,835 messages -> 28,918 info leafs, FTS+HNSW healthy.
137 lines
4.8 KiB
Python
Executable File
137 lines
4.8 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""mail/index_mail - rebuild the brain with every markdown under var/mail.
|
|
|
|
Ladybug corrupts its WAL when brand-new leafs are bulk-inserted while the
|
|
FTS/VECTOR indexes already exist, so indexing ALWAYS runs as a fresh rebuild
|
|
(repo corpus + var/mail), matching the proven-safe `kb/index --rebuild` path.
|
|
Conversion and indexing stay separate: conversion can crash in native docling
|
|
and must not leave the brain DB mid-transaction.
|
|
|
|
bin/mail/index_mail rebuild the index incl. all mail
|
|
bin/mail/index_mail --dry-run count without writing
|
|
bin/mail/index_mail --limit N cap messages included
|
|
bin/mail/index_mail --since D only messages dated >= D (YYYY-MM-DD)
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
ROOT = Path(__file__).resolve().parents[2]
|
|
sys.path.insert(0, str(ROOT / "bin" / "tools"))
|
|
|
|
from kblib import DB_PATH, VAR, connect, create_fts_and_vector, init_schema, stats, upsert_leaf # noqa: E402
|
|
from mdleaves import read_markdown, to_all, walk_markdown # noqa: E402
|
|
|
|
|
|
def msg_date(md: Path) -> str:
|
|
j = md.parent / "message.json"
|
|
try:
|
|
d = json.loads(j.read_text(encoding="utf-8"))
|
|
return (d.get("receivedDate") or d.get("receivedAt") or "")[:10]
|
|
except Exception:
|
|
return ""
|
|
|
|
|
|
def mail_leafs(limit: int, since: str, repo: str = "ooMail") -> list[dict]:
|
|
root = ROOT / "var" / "mail"
|
|
mds = sorted(root.rglob("message.md"))
|
|
if since:
|
|
mds = [m for m in mds if msg_date(m) >= since]
|
|
if limit:
|
|
mds = mds[:limit]
|
|
leafs: list[dict] = []
|
|
for md in mds:
|
|
files = [md] + sorted((md.parent / "attachments").glob("*.md"))
|
|
for f in files:
|
|
if not f.exists():
|
|
continue
|
|
for lf in to_all(read_markdown(f), f, repo=repo):
|
|
lf["source"] = f"ooMail:{md.parent.name}:{f.name}"
|
|
lf["how"] = "mail/import"
|
|
leafs.append(lf)
|
|
return leafs
|
|
|
|
|
|
def main(argv: list[str]) -> int:
|
|
p = argparse.ArgumentParser(description="rebuild the brain incl. all mail")
|
|
p.add_argument("--dry-run", action="store_true", help="count only, write nothing")
|
|
p.add_argument("--limit", type=int, default=0, help="cap messages included")
|
|
p.add_argument("--since", default="", help="only messages dated >= YYYY-MM-DD")
|
|
p.add_argument("--json", action="store_true")
|
|
a = p.parse_args(argv)
|
|
|
|
mail = mail_leafs(a.limit, a.since)
|
|
if a.dry_run:
|
|
print(f"mail/index_mail: {len(mail)} mail leafs would be indexed")
|
|
return 0
|
|
|
|
# Fresh rebuild: delete DB, index repo corpus + mail, create indexes once
|
|
# at the end. Never insert into an already-indexed DB (WAL corruption).
|
|
VAR.mkdir(exist_ok=True)
|
|
if DB_PATH.exists():
|
|
DB_PATH.unlink()
|
|
|
|
corpus = _load_corpus()
|
|
leafs = corpus + mail
|
|
|
|
db, conn = connect(DB_PATH, read_only=False)
|
|
init_schema(conn)
|
|
embed = _embedder()
|
|
done, total = _index_leafs(conn, leafs, embed)
|
|
create_fts_and_vector(conn, force=True)
|
|
s = stats(conn)
|
|
conn.close()
|
|
db.close()
|
|
|
|
result = {"indexed": done, "corpus_total": total, "mail_leafs": len(mail),
|
|
**{k: v for k, v in s.items() if k in ("total", "by_root")}}
|
|
print(json.dumps(result, indent=2) if a.json else
|
|
f"mail/index_mail: indexed {done}/{total} leafs (mail={len(mail)}); db total {s['total']}")
|
|
return 0
|
|
|
|
|
|
CORPUS_DEFAULTS = ["README.md", "PLAN.md", "AGENTS.md", "docs", "skills"]
|
|
|
|
|
|
def _load_corpus() -> list[dict]:
|
|
files: list[Path] = []
|
|
for entry in CORPUS_DEFAULTS:
|
|
p = ROOT / entry
|
|
if p.is_file():
|
|
files.append(p)
|
|
elif p.is_dir():
|
|
files.extend(walk_markdown(p))
|
|
leafs: list[dict] = []
|
|
for path in files:
|
|
try:
|
|
leafs.extend(to_all(read_markdown(path), path, repo="eSlider/2dph"))
|
|
except OSError as e:
|
|
print(f"mail/index_mail: skip {path}: {e}", file=sys.stderr)
|
|
return leafs
|
|
|
|
|
|
def _index_leafs(conn, leafs: list[dict], embed_fn) -> tuple[int, int]:
|
|
count = 0
|
|
for lf in leafs:
|
|
query = f"{lf['heading']}\n\n{lf['text']}"
|
|
emb = embed_fn(lf["text"]) if lf["text"] else None
|
|
upsert_leaf(conn, text=query, root="info", confidence="confirmed",
|
|
source=lf["source"], source_rev="mail" if lf.get("how") == "mail/import" else "working-tree",
|
|
how=lf.get("how", "kb/index"), loc=lf["source"], type_=lf.get("type", "reference"),
|
|
embedding=emb)
|
|
count += 1
|
|
return count, len(leafs)
|
|
|
|
|
|
def _embedder():
|
|
from model2vec import StaticModel
|
|
model = StaticModel.from_pretrained("minishlab/potion-multilingual-128M")
|
|
return lambda text: model.encode([text])[0].astype(float).tolist()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main(sys.argv[1:]))
|