"""Second pass over NHTSA's Standing General Order 2021-01 crash-report files — the numbers the essay
"Reported, Redacted, Unknown" adds to the profile script that accompanies it (sgo_profile_cp20260910.py).

Everything here is computed on the dataset's OWN join key, `Same Incident ID`, the field NHTSA uses to tie
reports of one incident together, so that rows (which are filings and re-filings) are never mistaken for
crashes. Four files, downloaded from static.nhtsa.gov (see FILES); nothing is quoted from a write-up.

What it prints:
  1. incidents vs rows per file, and per entity the share of incidents whose EVERY filing has the narrative
     redacted (crash level) beside the share of rows redacted (row level)
  2. the second copy: the operator's crashes that the vehicle maker also filed (Cruise/GM, Waymo/Transdev),
     and how many operator-redacted crashes are readable in the maker's own account
  3. the Mileage column of the archive files (the crashed vehicle's odometer, not fleet exposure)
  4. how each entity learns of its crashes (the Source fields), which is what a per-entity count measures
  5. reporting lag on month-granular dates: same-month share and the first filings more than a year late
  6. fatality rows vs fatality incidents

Run:  python sgo_second_copy_cp20260910.py     (expects the four CSVs beside it, names as in FILES)
"""
import csv
import io
import os
import re
import sys
from collections import Counter, defaultdict
from datetime import datetime

FILES = [
    ("ADS current (post-amendment)", "SGO-2021-01_Incident_Reports_ADS.csv",
     "https://static.nhtsa.gov/odi/ffdd/sgo-2021-01/SGO-2021-01_Incident_Reports_ADS.csv"),
    ("ADAS current (post-amendment)", "SGO-2021-01_Incident_Reports_ADAS.csv",
     "https://static.nhtsa.gov/odi/ffdd/sgo-2021-01/SGO-2021-01_Incident_Reports_ADAS.csv"),
    ("ADS archive 2021-2025", "ARCHIVE_ADS.csv",
     "https://static.nhtsa.gov/odi/ffdd/sgo-2021-01/Archive-2021-2025/SGO-2021-01_Incident_Reports_ADS.csv"),
    ("ADAS archive 2019-2025", "ARCHIVE_ADAS.csv",
     "https://static.nhtsa.gov/odi/ffdd/sgo-2021-01/Archive-2021-2025/SGO-2021-01_Incident_Reports_ADAS.csv"),
]
RED = "REDACTED, MAY CONTAIN CONFIDENTIAL BUSINESS INFORMATION"

# ⛔ ONE DOWNLOAD, EITHER NAME (editor at packaging, per the c5444 addendum; scout bv1 found it).
# The two scripts published with this essay expected the SAME archive files under DIFFERENT names:
# this pair wanted ARCH_SGO-2021-01_Incident_Reports_*.csv and its sibling wanted ARCHIVE_*.csv. A
# reader who downloaded once and ran both got MISSING INPUT (exit 2) from the second, which makes
# the essay's "I would rather you ran them than trusted me" false on a first try. Both scripts now
# accept EITHER name and look beside the script as well as in the working directory.
ALIASES = {
    "ARCH_SGO-2021-01_Incident_Reports_ADS.csv":  ["ARCHIVE_ADS.csv"],
    "ARCH_SGO-2021-01_Incident_Reports_ADAS.csv": ["ARCHIVE_ADAS.csv"],
    "ARCHIVE_ADS.csv":  ["ARCH_SGO-2021-01_Incident_Reports_ADS.csv"],
    "ARCHIVE_ADAS.csv": ["ARCH_SGO-2021-01_Incident_Reports_ADAS.csv"],
}


def resolve(name, here=None):
    """First accepted name for this file that exists, beside the script or in the cwd; else None."""
    here = here or os.path.dirname(os.path.abspath(__file__))
    for cand in [name] + ALIASES.get(name, []):
        for base in (here, os.getcwd()):
            p = os.path.join(base, cand)
            if os.path.exists(p):
                return p
    return None
SOURCES = ["Source - Telematics", "Source - Complaint/Claim", "Source - Law Enforcement", "Source - Field Report",
           "Source - Media", "Source - Testing", "Source - Internal Process Review", "Source - NHTSA VOQ",
           "Source - Other Entity", "Source - State or Other Agency", "Source - Other - See Narrative"]


def load(path):
    with io.open(path, encoding="utf-8-sig", errors="replace", newline="") as fh:
        return list(csv.DictReader(fh))


def red(v):
    return RED in (v or "").upper()


def sid(r):
    return (r.get("Same Incident ID") or "").strip()


def month(s):
    try:
        return datetime.strptime((s or "").strip().title(), "%b-%Y")
    except ValueError:
        return None


def pct(a, b):
    return "n/a" if not b else "%.1f%%" % (100.0 * a / b)


def by_incident(rows, entity=None):
    inc = defaultdict(list)
    for r in rows:
        if entity and r["Reporting Entity"] != entity:
            continue
        if sid(r):
            inc[sid(r)].append(r)
    return inc


def section_1(label, rows):
    inc = by_incident(rows)
    print("-- 1. rows vs incidents, and redaction at both levels (entities with >= 15 rows) --")
    print("   rows %d -> incidents %d (%.2f rows per incident); rows with no incident id %d"
          % (len(rows), len(inc), len(rows) / max(1, len(inc)), sum(1 for r in rows if not sid(r))))
    per = defaultdict(lambda: defaultdict(list))
    for r in rows:
        if sid(r):
            per[r["Reporting Entity"]][sid(r)].append(r)
    tab = []
    for e, incs in per.items():
        n = sum(len(v) for v in incs.values())
        if n < 15:
            continue
        rr = sum(red(r["Narrative"]) for v in incs.values() for r in v)
        ir = sum(1 for v in incs.values() if all(red(r["Narrative"]) for r in v))
        tab.append((n, e, len(incs), rr, ir))
    print("   %-36s %6s %6s %12s %16s" % ("entity", "rows", "incid.", "rows redact.", "incidents redact."))
    for n, e, ni, rr, ir in sorted(tab, reverse=True):
        print("   %-36s %6d %6d %12s %16s" % (e[:36], n, ni, pct(rr, n), pct(ir, ni)))


def section_2(rows):
    print("-- 2. the second copy: maker and operator filing the same incident (ADS archive) --")
    for maker, op in (("General Motors, LLC", "Cruise LLC"), ("Transdev Alternative Services", "Waymo LLC")):
        M, O = by_incident(rows, maker), by_incident(rows, op)
        shared = set(M) & set(O)
        op_all_red = {k for k, v in O.items() if all(red(r["Narrative"]) for r in v)}
        rescued = {k for k in op_all_red & set(M) if any(not red(r["Narrative"]) for r in M[k])}
        same_text = sum(1 for k in shared for a in M[k] for b in O[k]
                        if not red(a["Narrative"]) and not red(b["Narrative"]) and a["Narrative"].strip() == b["Narrative"].strip())
        lens = sorted(len((r["Narrative"] or "").strip()) for r in rows if r["Reporting Entity"] == maker)
        print("   %s: %d incidents, %d (%s) with every filing redacted" % (op, len(O), len(op_all_red), pct(len(op_all_red), len(O))))
        print("   %s: %d incidents, %d shared with %s (%s of the operator's); redacted none: %s"
              % (maker, len(M), len(shared), op, pct(len(shared), len(O)),
                 all(not red(r["Narrative"]) for v in M.values() for r in v)))
        print("   -> operator-redacted incidents readable in the maker's copy: %d of %d; unreadable in any copy: %d"
              % (len(rescued), len(op_all_red), len(op_all_red - rescued)))
        print("   -> maker narrative length (chars) min/median/max %d/%d/%d; identical text to the operator's in %d pairs"
              % (lens[0], lens[len(lens) // 2], lens[-1], same_text))


def section_3(label, rows):
    cols = list(rows[0].keys())
    print("-- 3. exposure columns: %d columns; %s --" % (len(cols), "Mileage present" if "Mileage" in cols else "no Mileage column"))
    # fold-ok: matched against CSV column headers (Mileage, VMT), never file names; case-insensitive on purpose
    hits = [c for c in cols if re.search(r"mile|vmt|ride|trip|fleet|exposure|hour|odometer", c, re.I)]
    print("   columns matching mile/vmt/ride/trip/fleet/exposure/hour/odometer: %s" % (hits or "none"))
    if "Mileage" in cols:
        nums = []
        for r in rows:
            try:
                nums.append(float((r.get("Mileage") or "").replace(",", "")))
            except ValueError:
                pass
        nums.sort()
        print("   Mileage = the subject vehicle's odometer: numeric in %d of %d rows; median %d; flagged Unknown in %d"
              % (len(nums), len(rows), nums[len(nums) // 2], sum(1 for r in rows if (r.get("Mileage - Unknown") or "").strip())))


def section_4(label, rows):
    print("-- 4. how the entity learned of the crash: share of rows flagging each Source (entities >= 30 rows) --")
    ent = Counter(r["Reporting Entity"] for r in rows)
    for e, n in ent.most_common():
        if n < 30:
            continue
        R = [r for r in rows if r["Reporting Entity"] == e]
        sh = sorted(((s.replace("Source - ", ""), 100.0 * sum(1 for r in R if (r.get(s) or "").strip().upper() == "Y") / n)
                     for s in SOURCES), key=lambda kv: -kv[1])[:3]
        print("   %-36s n=%-5d %s" % (e[:36], n, ", ".join("%s %.0f%%" % kv for kv in sh)))


def section_5(label, rows):
    lag = []
    for r in rows:
        di, ds = month(r.get("Incident Date")), month(r.get("Report Submission Date"))
        if di and ds:
            lag.append(((ds - di).days, (r.get("Report Version") or "1").strip() == "1", r["Reporting Entity"],
                        r.get("Incident Date"), r.get("Report Submission Date")))
    if not lag:
        print("-- 5. lag: dates did not parse --")
        return
    lag.sort(reverse=True)
    n = len(lag)
    same = sum(1 for l in lag if l[0] == 0)
    late = [l for l in lag if l[0] > 365]
    late1 = [l for l in late if l[1]]
    print("-- 5. lag on MONTH-granular dates (both Incident Date and Report Submission Date are months) --")
    print("   parsed %d of %d rows; same-month %s; more than 365 days %d, of which first filings (version 1) %d"
          % (n, len(rows), pct(same, n), len(late), len(late1)))
    if late1:
        top = late1[0]
        print("   latest first filing: %s, incident %s, submitted %s (%d days)" % (top[2], top[3], top[4], top[0]))


def section_6(label, rows):
    f = [r for r in rows if (r.get("Highest Injury Severity Alleged") or "").strip() == "Fatality"]
    print("-- 6. severity recorded as Fatality: %d rows -> %d incidents --" % (len(f), len({sid(r) for r in f if sid(r)})))


def main():
    here = os.path.dirname(os.path.abspath(__file__))
    resolved = [(label, name, url, resolve(name, here)) for label, name, url in FILES]
    missing = [n for _, n, _, p in resolved if p is None]
    if missing:
        print("MISSING INPUT: %s" % missing)
        for _, n, u in FILES:
            print("   %s  <-  %s" % (n, u))
        print("Either name is accepted for the two archive files: ARCHIVE_*.csv or "
              "ARCH_SGO-2021-01_Incident_Reports_*.csv, so one download feeds this script and "
              "sgo_profile_cp20260910.py both.")
        return 2
    all_ids = set()
    for label, path, url, _resolved_path in resolved:
        rows = load(_resolved_path)
        all_ids |= {sid(r) for r in rows if sid(r)}
        print("=" * 100)
        print("%s  —  %s  —  %d rows, %d columns" % (label, path, len(rows), len(rows[0])))
        print("=" * 100)
        section_1(label, rows)
        if path == "ARCHIVE_ADS.csv":
            section_2(rows)
        section_3(label, rows)
        section_4(label, rows)
        section_5(label, rows)
        section_6(label, rows)
    print("=" * 100)
    print("all four files: %d distinct incident ids" % len(all_ids))
    return 0


if __name__ == "__main__":
    sys.exit(main())
