"""Reliable Robotics essay: check every quotation, figure and date in the essay against the saved primaries.

No network. Reads the essay (default: essay.md beside this file; or pass a path), the page copies saved
under reads/ and the news-page pull saved by newsroom_c6681.py (newsroom_c6681.json). Prints one line
per check and exits 1 if any fails.

1. QUOTES. Every double-quoted span in the essay must appear in a saved source. Typography is normalised
   (curly quotes, zero-width and non-breaking spaces, whitespace runs); a trailing period or comma inside
   the closing quote is dropped (American punctuation); the first letter may differ in case. Spans in the
   Sources list and the title are titles or headings, matched without regard to case.
2. FIGURES. Each number the essay states is asserted at its source: the phrase is located in the saved copy.
3. NEWS PAGE. The 130 count, the type split, the single item after June 17 and the title scan are recomputed
   from newsroom_c6681.json.
4. ARITHMETIC. The "implies" column of the target table and the "five months" are recomputed.

NOT BUNDLED. The published folder carries no copies of the pages. SOURCES.md lists each one's URL, the
file name it is checked under, when the copy was saved, and its size and SHA-256. Save a page's text as
reads/<name>.txt and its checks run; without it they print NOT BUNDLED instead of PASS or FAIL.
figures_c6681.out.txt is the run with every page present.
"""
import json, pathlib, re, sys, datetime

HERE = pathlib.Path(__file__).resolve().parent
READS = HERE / "reads"
ESSAY = pathlib.Path(sys.argv[1]) if len(sys.argv) > 1 else HERE / "essay.md"

results = []
unchecked = []


def check(ok, label, detail=""):
    results.append(bool(ok))
    print(f"{'PASS' if ok else 'FAIL'}  {label}" + (f"  [{detail}]" if detail else ""))


def norm(s):
    s = s.replace("’", "'").replace("‘", "'").replace("“", '"').replace("”", '"')
    s = s.replace("​", "").replace(" ", " ").replace("‑", "-")
    return re.sub(r"\s+", " ", s).strip()


# ---- the corpus: saved page text, one entry per source file ----
corpus = {}
for p in sorted(READS.glob("*.txt")):
    corpus[p.name] = norm(p.read_text(encoding="utf-8", errors="replace"))
news = json.loads((HERE / "newsroom_c6681.json").read_text(encoding="utf-8"))
corpus["newsroom_c6681.json"] = norm(" || ".join(
    f"{i.get('title', '')} | {i.get('summary', '')} | {i.get('author') or ''}" for i in news["items"]))
index = json.loads((READS / "reads_index.json").read_text(encoding="utf-8"))
# the two 2020 articles were saved beside the index, not in it
EXTRA = ["fortune_2020", "flightglobal_2020"]
MISSING = sorted(k + ".txt" for k in list(index) + EXTRA if not (READS / (k + ".txt")).exists())


def not_bundled(label, fname):
    unchecked.append(label)
    print(f"NOT BUNDLED  {label}  [{fname}]")
corpus["reads_index.json (titles)"] = norm(" || ".join(v.get("title", "") for v in index.values()))


def find(q, ignore_case=False):
    q = norm(q)
    for name, text in corpus.items():
        if ignore_case:
            if q.lower() in text.lower():
                return name
        else:
            if q in text:
                return name
            if len(q) > 1 and (q[0].swapcase() + q[1:]) in text:
                return name
    return None


essay = ESSAY.read_text(encoding="utf-8")
body, _, sources = essay.partition("**Sources:**")
title_line = essay.splitlines()[0]

print(f"essay: {ESSAY.name}")
print("== 1. quotations ==")
n_quotes = 0
for region, text, loose in (("title", title_line, True), ("body", body[len(title_line):], False),
                            ("sources", sources, True)):
    for m in re.finditer(r'"([^"\n]+)"', text):
        q = m.group(1)
        for part in [x for x in re.split(r"\s*(?:\.\.\.|…)\s*", q) if x.strip()]:
            part = part.strip().rstrip(".,;:")
            if len(part) < 3:
                continue
            n_quotes += 1
            hit = find(part, ignore_case=loose)
            if not hit and MISSING:
                not_bundled(f"{region}: \"{part[:90]}\"", "in none of the bundled files; "
                            f"{len(MISSING)} page(s) not bundled")
                continue
            check(hit, f"{region}: \"{part[:90]}{'...' if len(part) > 90 else ''}\"", hit or "NOT FOUND")

print("== 2. figures at their source ==")
FIGS = [
    ("$160M round", "bw_160m.txt", "today announced $160M in new funding"),
    ("led by Nimble Partners", "bw_160m.txt", "The financing was led by Nimble Partners"),
    ("both programs, fifth paragraph", "bw_160m.txt", "Both programs will begin operations this year."),
    ("nearly $1 billion (DroneXL on Bloomberg)", "dronexl_160m.txt", "at nearly $1 billion, CEO Robert Rose told Bloomberg"),
    ("$300 million total", "dronexl_160m.txt", "Total capital raised now stands at $300 million"),
    ("AIN: 2028", "ain_160m.txt", "for which it is targeting FAA certification in 2028"),
    ("AIN names Pathbreaker as lead (footer note)", "ain_160m.txt", "investors led by Pathbreaker Ventures"),
    ("eight projects across 26 states", "dot_duffy.txt", "selected eight projects across 26 states"),
    ("DOT page dated June 17, 2026", "dot_duffy.txt", "Wednesday, June 17, 2026"),
    ("test operations August 2026", "dot_duffy.txt", "are expected to begin test operations in August 2026"),
    ("eIPP release dated March 9, 2026", "bw_eipp.txt", "Mar 9, 2026"),
    ("routes ABQ to Durango and Santa Fe", "bw_eipp.txt", "to Durango-La Plata County Airport, CO (DRO) and Santa Fe Regional Airport (SAF)"),
    ("blog post dated 2026-09-11", "blog_safety_first_0911.txt", "2026-09-11"),
    ("three-year project", "blog_safety_first_0911.txt", "A three-year project in the FAA"),
    ("$17.4 million, announced Aug. 26", "natdef_usaf.txt", "announced on Aug. 26 a $17.4 million contract with the Air Force"),
    ("DAA release dated April 07, 2026", "pr_daa_0407.txt", "April 07, 2026"),
    ("testimony: House subcommittee on aviation", "bw_testimony_2025.txt", "Subcommittee on Aviation"),
    ("Caravan with no one aboard: 11/21/23", "bw_noone_2023.txt", "On 11/21/23"),
    ("50 miles away", "bw_noone_2023.txt", "from Reliable's control center 50 miles away"),
    ("12-minute flight", "aaaa_2023.txt", "a 12-minute test flight"),
    ("Hollister", "aaaa_2023.txt", "Hollister Municipal Airport in California on the morning of November 21"),
    ("inside sight of ground observers", "aaaa_2023.txt", "within the visual line of sight (VLOS) of observers on the ground"),
    ("Cessna 172, 15 minutes, south of San Jose", "fortune_2020.txt", "at an airport south of San Jose, took off on its own, flew for 15 minutes"),
    ("first unmanned 172 flight, September 2019 (\"September last year\", article of Aug 2020)", "flightglobal_2020.txt", "completed its first unmanned flight of the type in September last year"),
    ("Cessna 172 with no one on board, September 2019 (TechCrunch)", "techcrunch_2021.txt", "Back in September 2019, Reliable Robotics flew a Cessna 172 with no one on board"),
    ("2023 plan: licensed pilot in the cockpit first", "avweek_certplan_2023.txt", "will still have a licensed pilot in the cockpit"),
    ("2023 plan: follow-on certification to a ground control center", "avweek_certplan_2023.txt", "shift from the cockpit to a ground control center"),
    ("Dec 2022 target said 'in December'", "avweek_moc.txt", "Rose previously told Aviation Week in December"),
    ("$100M in 2021 (TechCrunch)", "techcrunch_2021.txt", "announced its $100 million Series C funding"),
    ("licensed pilots on the ground (2021)", "techcrunch_2021.txt", "licensed pilots remotely supervise the flights from a control center"),
    ("Cargo Newswire repeats August", "cargonewswire.txt", "With test operations scheduled to commence in August 2026"),
]
for label, fname, phrase in FIGS:
    if fname in MISSING:
        not_bundled(label, fname)
        continue
    check(norm(phrase) in corpus.get(fname, ""), label, fname)

# source dates from each page's own datePublished, as saved in reads_index.json (+ the two 2020 pages)
DATES = {"ain_160m": "2026-04-21", "dronexl_160m": "2026-04-21", "bw_160m": "2026-04-21", "bw_eipp": "2026-03-09",
         "avweek_certplan_2023": "2023-07-20", "avweek_moc": "2023-01-06", "bw_noone_2023": "2023-12-06",
         "techcrunch_2021": "2021-10-14", "cargonewswire": "2026-06-22", "bw_testimony_2025": "2025-12-03"}
for key, want in DATES.items():
    got = [d[:10] for d in index.get(key, {}).get("dates", [])]
    check(want in got, f"dated {want}: {key}", ",".join(got) or "no date in index")
for fname, want in (("flightglobal_2020.html", "2020-08-26"), ("fortune_2020.html", "2020-08-26")):
    if not (READS / fname).exists():
        not_bundled(f"dated {want}", fname)
        continue
    html = (READS / fname).read_text(encoding="utf-8", errors="replace")
    m = re.search(r'"datePublished"\s*:\s*"([^"]+)"', html)
    check(m and m.group(1)[:10] == want, f"dated {want}: {fname}", m.group(1) if m else "none")
check("/2025/8/26/" in index.get("natdef_usaf", {}).get("url", ""), "dated 2025-08-26: natdef_usaf (URL path)")
if "aaaa_2023.txt" in MISSING:
    not_bundled("FutureFlight original dated 2023-12-09", "aaaa_2023.txt")
else:
    check("2023-12-09" in (READS / "aaaa_2023.txt").read_text(encoding="utf-8"), "FutureFlight original dated 2023-12-09 (URL in the republication)")

print("== 3. the news page ==")
items = news["items"]
check(news["published_count"] == 130 == len(items), "130 published items", news["read_at"])
check(news["by_type"] == {"Press Release": 53, "Article": 38, "Blog": 32, "Live Interview": 7} or
      sorted(news["by_type"].items()) == sorted({"Press Release": 53, "Article": 38, "Blog": 32, "Live Interview": 7}.items()),
      "53 press releases, 38 articles, 32 blog posts, 7 interviews", json.dumps(news["by_type"]))
after = [i for i in items if i["date"] > "2026-06-17"]
check(len(after) == 1 and after[0]["date"] == "2026-09-11" and after[0].get("author") == "Paul Smith",
      "exactly one item dated after 2026-06-17: Smith, 2026-09-11", "; ".join(f"{i['date']} {i['title']}" for i in after))
newest_before = sorted([i for i in items if i["date"] <= "2026-06-17"], key=lambda i: i["date"])[-1]
check(newest_before["date"] == "2026-06-17" and "Duffy" in newest_before["title"], "the item before it is the June 17 visit")
NARROW = re.compile(r"pilotless|no one|no pilot|without (?:a )?pilot", re.I)
BROAD = re.compile(r"uncrewed|unmanned|crewless", re.I)
narrow = sorted((i["date"], i["title"]) for i in items if NARROW.search(i["title"]))
months = sorted({d[:7] for d, _ in narrow})
check(months == ["2020-08", "2023-12"], f"titles calling a flight pilotless or no-one-aboard: {len(narrow)}, all in {months}")
for d, t in narrow:
    print(f"        {d}  {t}")
print("      titles with 'uncrewed'/'unmanned' but not the words above (none reports a flight):")
for d, t in sorted((i["date"], i["title"]) for i in items if BROAD.search(i["title"]) and not NARROW.search(i["title"])):
    print(f"        {d}  {t}")
oct21 = [i for i in items if i["date"] == "2021-10-14" and "Remotely Piloted Cargo Operations" in i["title"]]
check(len(oct21) == 1, "Oct 14, 2021 funding release headline says 'Remotely Piloted Cargo Operations'")
roas = [i["date"] for i in items if "Remotely Operated Aircraft System" in (i.get("summary") or "")]
check("2022-05-26" in roas, "news page names the Remotely Operated Aircraft System in 2022", ",".join(roas))
ras = sorted(i["date"] for i in items if "Reliable Autonomy System" in (i.get("summary") or "") + i["title"])
check(ras and ras[0] == "2025-08-26", "first news-page mention of the Reliable Autonomy System: 2025-08-26", ",".join(ras))
mil = [i for i in items if i["date"] == "2024-01-30" and i.get("kind") == "Press Release"
       and "airworthiness" in i["title"].lower() and "remotely piloted" in (i.get("summary") or "")]
check(len(mil) == 1, "Jan 30, 2024 military airworthiness release (the company's), summary says remotely piloted")
aug20 = [i for i in items if i["date"] == "2020-08-26" and "autonomous passenger airplanes" in (i.get("summary") or "")]
check(len(aug20) == 1, "Aug 26, 2020 company release: 'autonomous passenger airplanes'")

print("== 4. arithmetic ==")


def add_months(y, m, k):
    t = y * 12 + (m - 1) + k
    return t // 12, t % 12 + 1


LADDER = [  # (said, months out low, months out high, what the table prints)
    ((2020, 8), 24, 24, "2022"),
    ((2022, 12), 24, 24, "end of 2024"),
    ((2023, 7), 18, 24, "early to mid 2025"),
    ((2025, 8), 24, 24, "by August 2027"),
]
for (y, m), lo, hi, printed in LADDER:
    a, b = add_months(y, m, lo), add_months(y, m, hi)
    print(f"      said {y}-{m:02d} + {lo}..{hi} months -> {a[0]}-{a[1]:02d}..{b[0]}-{b[1]:02d}  (table: {printed})")
check(add_months(2020, 8, 24) == (2022, 8), "Aug 2020 + two years = 2022")
check(add_months(2022, 12, 24) == (2024, 12), "Dec 2022 + two years = end of 2024")
check(add_months(2023, 7, 18) == (2025, 1) and add_months(2023, 7, 24) == (2025, 7), "Jul 2023 + 18-24 months = Jan-Jul 2025")
check(add_months(2025, 8, 24) == (2027, 8), "Aug 2025 + two years = Aug 2027")
check(sorted({2024, 2025} | {2025, 2026, 2027}) == [2024, 2025, 2026, 2027], "Dec 2023: 2024-25, then a year or two later = 2024 to 2027")
today = datetime.date(2026, 9, 22)
passed = [datetime.date(2022, 8, 26) < today, datetime.date(2024, 12, 31) < today, datetime.date(2025, 7, 20) < today,
          datetime.date(2025, 12, 31) < today]
check(all(passed), "rows 1-3 and row 4's single-pilot date (end of 2025) are all before 2026-09-22")
check(datetime.date(2027, 8, 26) > today and datetime.date(2028, 1, 1) > today, "rows 5 and 6 fall due in 2027 and 2028")
check((2026 - 2026) * 12 + (9 - 4) == 5, "April 21 to September 22, 2026 = five months")

print(f"== {sum(results)}/{len(results)} checks pass ({n_quotes} quoted spans)"
      + (f"; {len(unchecked)} not bundled (see SOURCES.md)" if unchecked else "") + " ==")
sys.exit(0 if all(results) else 1)
