"""Compute what NHTSA's Standing General Order 2021-01 data actually supports — and what it does not.

Every number in `reported_redacted_unknown_..._research.md` is printed by this script, read from the
two CSVs NHTSA publishes at static.nhtsa.gov. Nothing here is quoted from a secondary write-up.

⛔ THE POINT OF THE SCRIPT IS THE DENOMINATOR IT CANNOT SUPPLY. The SGO carries no miles, no rides
and no fleet size, so none of these counts can become a rate. What it CAN support is a
reporting-quality profile per entity, which is what this computes.

Run:  python sgo_profile_cp20260910.py
Data: SGO-2021-01_Incident_Reports_ADS.csv and _ADAS.csv, downloaded from
      https://static.nhtsa.gov/odi/ffdd/sgo-2021-01/
"""
from __future__ import annotations

import csv
import io
import os
import sys
from collections import Counter, defaultdict
from datetime import datetime

# ⛔ FOUR FILES, NOT TWO. The series break is not a caveat in a footnote — NHTSA SPLIT THE DATASET.
# The current URL serves only the post-amendment period; 2021 to mid-2025 lives at a different path
# under Archive-2021-2025/. Anyone who downloads "the SGO data" gets the first pair and a series that
# silently begins in 2025.
FILES = [("ADS  (current, post-amendment)", "SGO-2021-01_Incident_Reports_ADS.csv"),
         ("ADAS (current, post-amendment)", "SGO-2021-01_Incident_Reports_ADAS.csv"),
         ("ADS  (archive 2021-2025)", "ARCH_SGO-2021-01_Incident_Reports_ADS.csv"),
         ("ADAS (archive 2019-2025)", "ARCH_SGO-2021-01_Incident_Reports_ADAS.csv")]
REDACT = "REDACTED, MAY CONTAIN CONFIDENTIAL BUSINESS INFORMATION"
AMEND = datetime(2025, 6, 16)          # third amended SGO effective date

# ⛔ ONE DOWNLOAD, EITHER NAME (editor at packaging, per the c5444 addendum; scout bv1 found it).
# The two scripts published with this essay expected the SAME archive files under DIFFERENT names:
# this pair wanted ARCH_SGO-2021-01_Incident_Reports_*.csv and its sibling wanted ARCHIVE_*.csv. A
# reader who downloaded once and ran both got MISSING INPUT (exit 2) from the second, which makes
# the essay's "I would rather you ran them than trusted me" false on a first try. Both scripts now
# accept EITHER name and look beside the script as well as in the working directory.
ALIASES = {
    "ARCH_SGO-2021-01_Incident_Reports_ADS.csv":  ["ARCHIVE_ADS.csv"],
    "ARCH_SGO-2021-01_Incident_Reports_ADAS.csv": ["ARCHIVE_ADAS.csv"],
    "ARCHIVE_ADS.csv":  ["ARCH_SGO-2021-01_Incident_Reports_ADS.csv"],
    "ARCHIVE_ADAS.csv": ["ARCH_SGO-2021-01_Incident_Reports_ADAS.csv"],
}


def resolve(name, here=None):
    """First accepted name for this file that exists, beside the script or in the cwd; else None."""
    here = here or os.path.dirname(os.path.abspath(__file__))
    for cand in [name] + ALIASES.get(name, []):
        for base in (here, os.getcwd()):
            p = os.path.join(base, cand)
            if os.path.exists(p):
                return p
    return None


def load(path):
    with io.open(path, encoding="utf-8-sig", errors="replace", newline="") as fh:
        return list(csv.DictReader(fh))


def is_red(v):
    return REDACT in (v or "").upper() or (v or "").strip().upper() in ("REDACTED", "[REDACTED]")


def pdate(s):
    s = (s or "").strip()
    for f in ("%b-%Y", "%m/%d/%Y", "%Y-%m-%d", "%b %Y"):
        try:
            return datetime.strptime(s, f)
        except ValueError:
            pass
    return None


def pct(a, b):
    return "n/a" if not b else "%.1f%%" % (100.0 * a / b)


def main():
    resolved = [(label, name, resolve(name)) for label, name in FILES]
    missing = [n for _, n, p in resolved if p is None]
    if missing:
        print("MISSING INPUT: %s" % missing)
        print("Download from https://static.nhtsa.gov/odi/ffdd/sgo-2021-01/")
        print("Either name is accepted for the two archive files: "
              "ARCH_SGO-2021-01_Incident_Reports_*.csv or ARCHIVE_*.csv, so one download "
              "feeds this script and sgo_second_copy_cp20260910.py both.")
        return 2

    for label, name, path in resolved:
        rows = load(path)
        print("=" * 96)
        print("%s  —  %s  —  %d report rows" % (label, name, len(rows)))
        print("=" * 96)

        # ---- 1. rows per entity, and the DISTINCT-CRASH correction ------------------------------
        ent = Counter((r.get("Reporting Entity") or "?").strip() for r in rows)
        rid = defaultdict(set)
        for r in rows:
            rid[(r.get("Reporting Entity") or "?").strip()].add((r.get("Report ID") or "").strip())
        print("\n-- 1. ROWS vs DISTINCT REPORT IDs (the same crash is re-filed as it is amended) --")
        print("   %-34s %8s %8s %8s" % ("entity", "rows", "reportID", "rows/ID"))
        for e, n in ent.most_common(12):
            d = len(rid[e]) or 1
            print("   %-34s %8d %8d %8.2f" % (e[:34], n, d, n / d))
        tot_rows, tot_ids = len(rows), len({(r.get("Report ID") or "").strip() for r in rows})
        print("   %-34s %8d %8d %8.2f" % ("ALL", tot_rows, tot_ids, tot_rows / max(tot_ids, 1)))
        print("   ⛔ Counting ROWS as crashes overstates every entity by its own amendment rate.")

        # ---- 2. redaction share, by field and by entity -----------------------------------------
        print("\n-- 2. REDACTION: the three fields an entity may claim as confidential --")
        fields = [("Narrative", "Narrative"), ("Automation Feature Version", "Automation Feature Version"),
                  ("Within ODD?", "Within ODD?")]
        for lbl, col in fields:
            n = sum(1 for r in rows if is_red(r.get(col)))
            print("   %-30s redacted in %5d of %5d rows   %s" % (lbl, n, len(rows), pct(n, len(rows))))
        print("\n   by entity (entities with >=15 rows), share of ROWS with the narrative redacted:")
        print("   %-34s %7s %10s   %s" % ("entity", "rows", "narr.red", "share"))
        for e, n in ent.most_common():
            if n < 15:
                continue
            k = sum(1 for r in rows if (r.get("Reporting Entity") or "").strip() == e and is_red(r.get("Narrative")))
            print("   %-34s %7d %10d   %s" % (e[:34], n, k, pct(k, n)))

        # ---- 3. injury severity, and how much of it is unknown ----------------------------------
        print("\n-- 3. HIGHEST INJURY SEVERITY ALLEGED --")
        sev = Counter((r.get("Highest Injury Severity Alleged") or "(blank)").strip() for r in rows)
        for s, n in sev.most_common(10):
            print("   %-40s %6d   %s" % (s[:40], n, pct(n, len(rows))))
        unk = sum(n for s, n in sev.items() if "unknown" in s.lower() or s == "(blank)")
        print("   ⭐ UNKNOWN or blank: %d of %d  =  %s" % (unk, len(rows), pct(unk, len(rows))))

        # ---- 4. lag from crash to report --------------------------------------------------------
        print("\n-- 4. LAG, incident month -> report submission --")
        lags = []
        for r in rows:
            i, s = pdate(r.get("Incident Date")), pdate(r.get("Report Submission Date"))
            if i and s and s >= i:
                lags.append((s - i).days)
        if lags:
            lags.sort()
            n = len(lags)
            print("   measurable on %d of %d rows (%s)" % (n, len(rows), pct(n, len(rows))))
            print("   median %d days | p90 %d days | max %d days"
                  % (lags[n // 2], lags[int(n * 0.9)], lags[-1]))
            print("   ⚠ Incident Date is published as a MONTH, so a same-month report reads as 0-30 days.")

        # ---- 5. the series break ----------------------------------------------------------------
        print("\n-- 5. THE SERIES BREAK: third amended SGO effective 2025-06-16 --")
        dates = [pdate(r.get("Incident Date")) for r in rows]
        unparsed = sum(1 for d in dates if d is None)
        before = sum(1 for d in dates if d is not None and d < AMEND)
        after = sum(1 for d in dates if d is not None and d >= AMEND)
        yrs = Counter(d.year for d in dates if d is not None)
        print("   incident-date years present: %s" % dict(sorted(yrs.items())))
        print("   rows with an incident date BEFORE the amendment: %d" % before)
        print("   rows on or AFTER: %d" % after)
        print("   rows whose incident date could NOT be parsed: %d  (counted in NEITHER bucket)" % unparsed)
        print("   ⛔ These are not the same measurement; the reporting criteria changed between them.")

        # ---- 6. what is NOT here -----------------------------------------------------------------
        cols = set(rows[0].keys()) if rows else set()
        print("\n-- 6. THE DENOMINATOR, LOOKED FOR AND NOT FOUND --")
        for want in ("miles", "vehicle miles", "vmt", "rides", "trips", "fleet size", "exposure", "hours"):
            hit = [c for c in cols if want in c.lower()]
            print("   %-14s -> %s" % (want, hit or "ABSENT"))
        print("   ⛔ No exposure field of any kind. Every count above is a numerator with no denominator,")
        print("      so no crash RATE for any entity can be computed from this dataset. That is not a")
        print("      gap in this analysis; it is a property of the order.")
        print()
    return 0


if __name__ == "__main__":
    sys.exit(main())
