#!/usr/bin/env python3
"""Extract explicit withdrawal notices and their directly named dependencies."""
import argparse
import json
import re
from collections import defaultdict, deque
from pathlib import Path
from urllib.parse import unquote


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--source", required=True, type=Path)
    parser.add_argument("--output", required=True, type=Path)
    args = parser.parse_args()
    preprints = args.source / "preprints"
    notices = []
    for path in sorted(preprints.glob("*/README.md")):
        text = path.read_text()
        if "Withdrawal notice:" not in text:
            continue
        title = re.search(r"^# \[Withdrawal notice: (.*?)\]\(", text, re.M)
        date = re.search(r"Withdrawn on (.*?)\.", text)
        archived = re.findall(r"\[Pre-withdrawal PDF\]\((https://[^)]+)\)", text)
        notices.append({"id": path.parent.name, "title": title[1] if title else path.parent.name,
                        "notice_date": date[1] if date else None, "notice_path": str(path.relative_to(args.source)),
                        "archived_pdf_urls": archived, "statement_false": False,
                        "status": "proof_withdrawn", "text": text})
    ids = {n["id"] for n in notices}
    edges = []
    for notice in notices:
        if not re.search(r"proof relies on", notice["text"], re.I):
            continue
        for target in re.findall(r"https://github\.com/openai/math/blob/[^/]+/preprints/([^/]+)/", notice["text"]):
            target = unquote(target)
            if target != notice["id"] and target in ids:
                edge = {"dependent": notice["id"], "dependency": target,
                        "evidence": notice["notice_path"], "kind": "explicit_notice_dependency"}
                if edge not in edges:
                    edges.append(edge)
    dependents = defaultdict(list)
    for edge in edges:
        dependents[edge["dependency"]].append(edge["dependent"])
    impacts = {}
    for identifier in sorted(ids):
        found, queue = set(), deque(dependents[identifier])
        while queue:
            item = queue.popleft()
            if item not in found:
                found.add(item)
                queue.extend(dependents[item])
        impacts[identifier] = sorted(found)
    for notice in notices:
        del notice["text"]
    report = {"mode": "EXPLICIT_NOTICE_GRAPH_ONLY", "notices": notices, "edges": edges,
              "transitive_impact": impacts,
              "limits": ["The graph includes only dependencies explicitly named in withdrawal notices.",
                         "Withdrawn proofs do not establish that the mathematical statements are false.",
                         "The notice date is distinct from the section date in history.md.",
                         "Other mathematical and software dependencies require separate reviewed evidence."]}
    args.output.parent.mkdir(parents=True, exist_ok=True)
    with args.output.open("x") as stream:
        json.dump(report, stream, indent=2)
        stream.write("\n")
    print(json.dumps({"withdrawal_notices": len(notices), "explicit_dependency_edges": len(edges),
                      "notice_dates": sorted({n["notice_date"] for n in notices}),
                      "maximum_direct_or_transitive_impact": max((len(v) for v in impacts.values()), default=0)}))


if __name__ == "__main__":
    main()
