"""figures_c6733.py - check every quotation and figure in the c6733 essay against saved copies of its sources.

Essay: "The Forgetting Curve Everyone Cites Is Ebbinghaus's Own Column, Relabeled. The 90% Is Not in It."
Run:   python figures_c6733.py [essay.md]      (default: essay.md next to this script). No network.
       python figures_c6733.py --manifest      lists every saved source it reads: bytes, sha256, when saved, where.
Exit:  0 when every check passes, 1 otherwise.

What it checks
1. QUOTES. Every span inside double quotes in the essay body, and the blockquote, must be found in a saved source
   (normalized: tags stripped, entities decoded, curly quotes and dashes flattened, whitespace collapsed, case
   ignored), or be listed in LABELS with the reason it is not a quotation (a paraphrase of a retelling, a row label).
   The quotations the essay attributes to a named source are also checked against THAT source (ATTRIBUTED).
2. TABLES. Ebbinghaus's two tables are scanned images in the online edition, so they are transcribed below by eye
   (sources/ebb_table23.jpg, sources/ebb_table24.jpg). The transcription is checked for internal consistency
   (II + IV = 100; observed - calculated = the printed difference) and against an independent transcription:
   Murre & Dros (2015) Table 3, parsed from the article XML.
3. DERIVED FIGURES. Every figure the essay computes (column-IV equivalents, the MeltingSpot arithmetic, the
   exponential's values, the Murre & Dros savings from their Tables 1 and 2) is recomputed here.
4. REFIT. The fit statistics are read from refit_ebbinghaus_c6733.out.txt and two of them are recomputed from the
   printed parameters.
5. NUMBERS. Every number in the essay body must be one of the figures produced or declared here (NUMBERS_DECLARED
   carries the source of each count, date and constant).
"""
import hashlib
import html
import json
import math
import re
import sys
from pathlib import Path

HERE = Path(__file__).resolve().parent
SRC = HERE / "sources"
ARGS = [a for a in sys.argv[1:] if not a.startswith("--")]
ESSAY = Path(ARGS[0]) if ARGS else HERE / "essay.md"

# every saved copy this script (or the eye, for the two scans) reads, and where the original can be read in public
MANIFEST = [
    ("ebb_ch7.html", "https://psychclassics.yorku.ca/Ebbinghaus/memory7.htm"),
    ("ebb_ch7.txt", "text of ebb_ch7.html"),
    ("ebb_table23.jpg", "https://psychclassics.yorku.ca/Ebbinghaus/table23.jpg (read by eye)"),
    ("ebb_table24.jpg", "https://psychclassics.yorku.ca/Ebbinghaus/table24.jpg (read by eye)"),
    ("murre_dros_2015.xml", "https://doi.org/10.1371/journal.pone.0120644 (article XML)"),
    ("epmc_savings_pure.json", "https://doi.org/10.3758/s13423-022-02172-3 (Europe PMC record, abstract)"),
    ("epmc_bahrick1984.json", "https://doi.org/10.1037/0096-3445.113.1.1 (Europe PMC record, abstract)"),
    ("rubin_wenzel_1996.pdf", "https://doi.org/10.1037/0033-295X.103.4.734"),
    ("rubin_wenzel_1996.txt", "text of rubin_wenzel_1996.pdf"),
    ("meme_meltingspot_io_en_blog_forgetting_curve_corporate_training.html", "https://meltingspot.io/en/blog/forgetting-curve-corporate-training"),
    ("meme_www_talentcards_com_blog_ebbinghaus_forgetting_curve_.html", "https://www.talentlms.com/blog/ebbinghaus-forgetting-curve/ (the page's canonical address)"),
    ("meme_www_todaysclass_com_blog_fighting_the_forgetting_curve.html", "https://www.todaysclass.com/blog/fighting-the-forgetting-curve"),
    ("meme_trainmeuk_co_uk_resources_ebbinghaus_effect_employees_forget.html", "https://trainmeuk.co.uk/resources/ebbinghaus-effect-employees-forget-training-24-hours"),
    ("prior_mcdowall_idtips_20251109.html", "https://idtips.substack.com/p/sunday-l-and-d-myth-4-people-forget (saved with all but its JSON-LD scripts emptied)"),
    ("ours_we_wrote_25_reminders_20260426.html", "https://vibeagentmaking.com/blog/we-wrote-25-reminders-and-made-the-same-mistake-every-time/"),
]
if "--manifest" in sys.argv:
    from datetime import datetime, timezone
    print("%-72s %9s  %-20s  %-64s  %s" % ("saved copy (sources/)", "bytes", "saved (UTC)", "sha256", "public address"))
    for name, url in MANIFEST:
        f = SRC / name
        st = f.stat()
        when = datetime.fromtimestamp(st.st_mtime, timezone.utc).strftime("%Y-%m-%d %H:%M")
        print("%-72s %9d  %-20s  %s  %s" % (name, st.st_size, when, hashlib.sha256(f.read_bytes()).hexdigest(), url))
    sys.exit(0)

results = []


def check(ok, label):
    results.append((bool(ok), label))
    print(("ok    " if ok else "FAIL  ") + label)


# ---------------------------------------------------------------- source texts
def norm(s):
    s = html.unescape(s)
    # curly quotes, en/em/non-breaking hyphens, the ellipsis and the no-break space, written as code points
    for a, b in ((0x2018, "'"), (0x2019, "'"), (0x201C, '"'), (0x201D, '"'), (0x2013, "-"), (0x2014, "-"),
                 (0x2011, "-"), (0x2026, "..."), (0x00A0, " ")):
        s = s.replace(chr(a), b)
    return re.sub(r"\s+", " ", s).strip().lower()


def page_text(name):
    t = (SRC / name).read_text(encoding="utf-8", errors="replace")
    t = re.sub(r"(?is)<script.*?</script>|<style.*?</style>", " ", t)
    return norm(re.sub(r"(?s)<[^>]+>", " ", t))


def epmc_abstract(name):
    found = []

    def walk(o):
        if isinstance(o, dict):
            for k, v in o.items():
                if k == "abstractText" and isinstance(v, str):
                    found.append(v)
                walk(v)
        elif isinstance(o, list):
            for v in o:
                walk(v)

    walk(json.loads((SRC / name).read_text(encoding="utf-8")))
    return norm(re.sub(r"(?s)<[^>]+>", " ", " ".join(found)))


# the two column headings of the summary table, read off the scan (sources/ebb_table23.jpg)
TABLE23_HEADINGS = ("So much of the series learned was retained that in relearning a saving of Q% of the time of "
                    "original learning was made. The amount forgotten was thus equivalent to v% of the original in "
                    "terms of time of learning.")

SOURCES = {
    "ebbinghaus_ch7": norm((SRC / "ebb_ch7.txt").read_text(encoding="utf-8", errors="replace")),
    "ebbinghaus_table23_scan": norm(TABLE23_HEADINGS),
    "murre_dros_2015": page_text("murre_dros_2015.xml"),
    "murre_chessa_2023": epmc_abstract("epmc_savings_pure.json"),
    "bahrick_1984": epmc_abstract("epmc_bahrick1984.json"),
    "rubin_wenzel_1996": norm((SRC / "rubin_wenzel_1996.txt").read_text(encoding="utf-8", errors="replace")),
    "meltingspot": page_text("meme_meltingspot_io_en_blog_forgetting_curve_corporate_training.html"),
    "talentlms": page_text("meme_www_talentcards_com_blog_ebbinghaus_forgetting_curve_.html"),
    "todaysclass": page_text("meme_www_todaysclass_com_blog_fighting_the_forgetting_curve.html"),
    "trainmeuk": page_text("meme_trainmeuk_co_uk_resources_ebbinghaus_effect_employees_forget.html"),
    "mcdowall": page_text("prior_mcdowall_idtips_20251109.html"),
    "ours_april": page_text("ours_we_wrote_25_reminders_20260426.html"),
}

# ---------------------------------------------------------------- the essay
essay_raw = ESSAY.read_text(encoding="utf-8")
body = essay_raw.split("\n---\n")[0]              # everything above the footer rule
body_lines = [ln for ln in body.splitlines() if not ln.startswith("#")]
prose = "\n".join(ln for ln in body_lines if not ln.startswith("|") and not ln.startswith(">"))
blockquote = " ".join(ln[1:].strip() for ln in body_lines if ln.startswith(">"))

# spans the essay puts in quotation marks that are not quotations of a source
LABELS = {
    "half in an hour": "the retellings' claim, paraphrased (compare TalentLMS, Today's Class, our April essay)",
    "seventy percent in a day": "the retellings' claim, paraphrased (TrainMeUK headline, TalentLMS, Today's Class)",
    "1 hour": "the summary table's row label, read off the scan",
    "8.8 hours": "the summary table's row label, read off the scan",
    "forgotten": "one word, the column IV heading's own term",
    "humans": "one word quoted from the April passage (checked in the blockquote)",
    "newly learned information": "quoted from the April passage (checked in the blockquote)",
    "most of it within a week": "quoted from the April passage (checked in the blockquote)",
    "in terms of time of learning": "column IV heading, table header cell",
}

quoted = re.findall(r'"([^"]+)"', prose)
for q in quoted:
    nq = norm(q).strip(" ,.;:")
    # a quotation nested inside a quotation is written with single marks; the source may use double ones
    hits = [k for k, t in SOURCES.items() if nq in t or nq.replace("'", '"') in t]
    if hits:
        check(True, "quote in %s: \"%s\"" % (hits[0], q[:70]))
    elif nq in LABELS:
        check(True, "label (not a quotation: %s): \"%s\"" % (LABELS[nq], q))
    else:
        check(False, "quoted span found in no saved source and not a declared label: \"%s\"" % q)

# the blockquote is our own April passage, verbatim
check(norm(blockquote) in SOURCES["ours_april"], "blockquote = our April 26 passage, verbatim (%d chars)" % len(blockquote))
for w in ("humans", "newly learned information", "most of it within a week", "lower bound"):
    check(w in norm(blockquote), "April passage contains the phrase the essay corrects: %s" % w)

# quotations attributed to a named source must be in THAT source
ATTRIBUTED = {
    "meltingspot": ["roughly 56% is forgotten within one hour of learning, about 66% within one day, and around 75% within six days",
                    "cognitive science predicts that 90% of the material will be gone by Friday",
                    "If 90% of that training content is not retained past the first week, the effective return on investment is not just low. It is catastrophic.",
                    "roughly $940", "memory of new material follows an exponential decay curve",
                    "What the Ebbinghaus forgetting curve actually tells us",
                    "spend an average of $1,254 per employee per year on training"],
    "talentlms": ["found that within an hour of learning new information people tend to forget up to 50% of it. Within 24 hours, this can increase to 70%. By the end of the week, people tend to retain only about 25% of what they've learned.",
                  "studies have shown that, without reinforcement, people tend to forget up to 90% of what they've learned within a month"],
    "todaysclass": ["on average we forget 50% of new information within an hour, and about 70% within one day",
                    "Within 30 days, we forget up to 90% of what we learned",
                    "the percentage of information retained declined exponentially over time without reinforcement"],
    "trainmeuk": ["Why Employees Forget 70% of Training in 24 Hours (The Ebbinghaus Effect)", "It's a biology problem",
                  "80% within a month"],
    "mcdowall": ["you cannot legitimately convert savings scores into \"percentage forgotten.\"",
                 "Ebbinghaus's exponential decline curve"],
    "ebbinghaus_ch7": ["The investigations in question fell in the year 1879-80 and comprised 163 double tests",
                       "Each double test consisted in learning eight series of 13 syllables each (with the exception of 38 double tests",
                       "The learning was continued until two errorless recitations of the series in question were possible",
                       "This saving in work is each time the measure for the amount remembered at the end of the interval",
                       "for a certain individual, and for a series of 13 syllables",
                       "with merely approximate estimates, not involving exact calculation by the method of least squares",
                       "However, it is upheld by observations to be stated presently, so that I am in doubt about it",
                       "k = 1.84", "c = 1.25"],
    "murre_dros_2015": ["One subject spent 70 hours learning lists and relearning them",
                        "20 minutes, 1 hour, 9 hours, 1 day, 2 days, 6 days and 31 days",
                        "The second author, J. Dros, (22 years, male) was the only subject",
                        "who was 29 during his experiments in 1879",
                        "published only in German, without an English abstract",
                        "it has never been cited in international journals in English",
                        "an average increase in learning time of 2.67 s per day for a list",
                        "the corrected savings measure would be 0.137 for the 31 day interval instead of 0.0410",
                        "is still well below the values for the three others",
                        "University of Amsterdam", "the two subjects in an earlier German replication",
                        "we did not have the seven months available that Ebbinghaus invested in the experiment, but we had to accommodate our design to a 75 day period",
                        "Heller O , Mack W , Seitz J ( 1991 )", "Zeitschrift für Psychologie 199 : 3 - 18"],
    "murre_chessa_2023": ["prove mathematically that Ebbinghaus' savings measure is independent of initial encoding strength, learning time, and relearning times"],
    "bahrick_1984": ["Retention of Spanish learned in school was tested over a 50-year period for 733 individuals",
                     "Tests of reading comprehension, recall, and recognition vocabulary and grammar"],
    "rubin_wenzel_1996": ["A sample of 210 published data sets were assembled", "Each was fit to 105 different 2-parameter functions"],
}
for src, qs in ATTRIBUTED.items():
    for q in qs:
        check(norm(q) in SOURCES[src], "attributed to %s: \"%s\"" % (src, q[:70]))

# locators: which section of each retelling a quotation sits under, and that the chapter describes no lower bound
check(not re.search(r"lower bound|lower limit|minimum", SOURCES["ebbinghaus_ch7"]), "Chapter VII text has no 'lower bound' / 'lower limit' / 'minimum'")
ms = SOURCES["meltingspot"]
i56, i90, i1254 = ms.find("roughly 56% is forgotten"), ms.find("gone by friday"), ms.find("$1,254 per employee")
i_head1 = ms.rfind("what the ebbinghaus forgetting curve actually tells us", 0, i56)   # the heading over the 56/66/75
i_head2 = ms.find("the real financial cost of forgetting in l&d", i90)                # the next heading after the Friday line
check(0 <= i_head1 < i56 < i90 < i_head2 < i1254, "MeltingSpot: 56/66/75 and the Friday 90% sit in the first section; the $1,254 in the next")

# dates, from each page's own metadata or byline
RAW = {k: (SRC / f).read_text(encoding="utf-8", errors="replace") for k, f in (
    ("meltingspot", "meme_meltingspot_io_en_blog_forgetting_curve_corporate_training.html"),
    ("talentlms", "meme_www_talentcards_com_blog_ebbinghaus_forgetting_curve_.html"),
    ("todaysclass", "meme_www_todaysclass_com_blog_fighting_the_forgetting_curve.html"),
    ("trainmeuk", "meme_trainmeuk_co_uk_resources_ebbinghaus_effect_employees_forget.html"),
    ("mcdowall", "prior_mcdowall_idtips_20251109.html"),
    ("ours_april", "ours_we_wrote_25_reminders_20260426.html"))}
for k, pat, label in (
        ("meltingspot", r'"datePublished":"2026-06-08', "MeltingSpot published 2026-06-08"),
        ("talentlms", r'"datePublished":"2023-03-30', "TalentLMS published 2023-03-30"),
        ("talentlms", r'"dateModified":"2026-09-21', "TalentLMS modified 2026-09-21"),
        ("talentlms", r'og:site_name" content="TalentLMS Blog"', "the TalentLMS page names itself 'TalentLMS Blog'"),
        ("todaysclass", r"2023-05-05T", "Today's Class 2023-05-05"),
        ("todaysclass", r"2025-04-28T", "Today's Class modified 2025-04-28"),
        ("trainmeuk", r'"datePublished":"2025-12-03', "TrainMeUK published 2025-12-03"),
        ("mcdowall", r'"datePublished":"2025-11-09', "McDowall published 2025-11-09"),
        ("mcdowall", r'name="author" content="Tom McDowall"', "McDowall byline"),
        ("ours_april", r'"datePublished": "2026-04-26"', "our April essay published 2026-04-26")):
    check(re.search(pat, RAW[k]), "date/byline: " + label)

# ---------------------------------------------------------------- Ebbinghaus's tables (transcribed by eye from the scans)
# table23: after X hours | II saving Q% | III P.E.m | IV amount forgotten v%
T23 = [(0.33, 58.2, 1.0, 41.8), (1.0, 44.2, 1.0, 55.8), (8.8, 35.8, 1.0, 64.2), (24.0, 33.7, 1.2, 66.3),
       (48.0, 27.8, 1.4, 72.2), (6 * 24.0, 25.4, 1.3, 74.6), (31 * 24.0, 21.1, 0.8, 78.9)]
# table24: t (minutes) | b observed | b calculated | difference
T24 = [(20, 58.2, 57.0, +1.2), (64, 44.2, 46.7, -2.5), (526, 35.8, 34.5, +1.3), (1440, 33.7, 30.4, +3.3),
       (2 * 1440, 27.8, 28.1, -0.3), (6 * 1440, 25.4, 24.9, +0.5), (31 * 1440, 21.1, 21.2, -0.1)]
Q = [r[1] for r in T23]
IV = [r[3] for r in T23]
check(all(abs(q + v - 100.0) < 1e-9 for q, v in zip(Q, IV)), "table23: column II + column IV = 100 on every row")
check([r[1] for r in T24] == Q, "table24 observed column = table23 savings column")
check(all(abs(round(o - c, 1) - d) < 1e-9 for _, o, c, d in T24), "table24: observed - calculated = printed difference on every row")
worst = max(T24, key=lambda r: abs(r[3]))
check(worst[0] == 1440 and worst[3] == 3.3, "table24: the largest miss of the seven is at 1 day, +3.3")


def his(t, k=1.84, c=1.25):
    return 100 * k / (math.log10(t) ** c + k)


check(all(abs(round(his(t), 1) - c) <= 0.1 + 1e-9 for t, _, c, _ in T24), "his formula (k=1.84, c=1.25) reproduces his calculated column within 0.1")
check(abs(his(60) - 46.7) >= 0.5, "with a rounded 60 minutes his formula gives %.2f, not his printed 46.7 (the exact 64 is needed)" % his(60))

# Murre & Dros (2015) Tables 1, 2 and 3, parsed from the article XML
xml = (SRC / "murre_dros_2015.xml").read_text(encoding="utf-8", errors="replace")


def xml_table(tid):
    m = re.search(r'<table-wrap[^>]*id="[^"]*%s"[^>]*>(.*?)</table-wrap>' % tid, xml, re.S)
    rows = []
    for r in re.findall(r"<tr>(.*?)</tr>", m.group(1), re.S):
        cells = [re.sub(r"\s+", " ", html.unescape(re.sub(r"<[^>]+>", " ", c))).strip()
                 for c in re.findall(r"<t[dh][^>]*>(.*?)</t[dh]>", r, re.S)]
        rows.append(cells)
    return rows


INTERVALS = ["20 min", "1 hour", "9 hours", "1 day", "2 days", "6 days", "31 days"]
t3 = {r[0]: [float(x) for x in r[1:5]] for r in xml_table("t003") if r and r[0] in INTERVALS}
check(len(t3) == 7, "Murre & Dros Table 3 parsed: 7 intervals")
E, MACK, SEITZ, DROS = ([t3[i][j] for i in INTERVALS] for j in range(4))
check(all(abs(e - q / 100) < 1e-9 for e, q in zip(E, Q)), "M&D Table 3 Ebbinghaus column = the scan's column II / 100, to the digit")
t1 = {r[0]: r for r in xml_table("t001") if r and r[0] in INTERVALS}
dros_from_reps = [round((float(t1[i][2]) - float(t1[i][5])) / float(t1[i][2]), 3) for i in INTERVALS]
check(dros_from_reps == DROS, "M&D Table 3 Dros column = (learning - relearning repetitions) / learning, Table 1, all 7 intervals: %s" % dros_from_reps)
avg = [r for r in xml_table("t002") if r and r[0] == "Average"][0]
s1, s2, q31 = float(avg[-3]), float(avg[-2]), float(avg[-1])
check(abs((s1 - s2) / s1 - q31) < 0.001 and q31 == 0.090, "M&D Table 2 average at 31 days from seconds: (%g - %g) / %g = %.4f, printed %.3f" % (s1, s2, s1, (s1 - s2) / s1, q31))
ONE_DAY, NINE_H = INTERVALS.index("1 day"), INTERVALS.index("9 hours")
bump = [col[ONE_DAY] >= col[NINE_H] for col in (E, MACK, SEITZ, DROS)]
check(bump == [False, True, True, True], "one-day saving >= nine-hour saving in three of four columns (Mack, Seitz, Dros)")

# ---------------------------------------------------------------- derived figures the essay states
DERIVED = {}


def derive(label, value, shown):
    DERIVED[shown] = label
    check(("%s" % shown) == value, "%s = %s" % (label, value))


derive("MeltingSpot 56 = round(55.8)", str(round(55.8)), "56")
derive("MeltingSpot 66 = round(66.3)", str(round(66.3)), "66")
derive("MeltingSpot 75 = round(74.6)", str(round(74.6)), "75")
derive("$940 as a share of $1,254", "%.0f%%" % (100 * 940 / 1254), "75%")
derive("90% of $1,254", "${:,.0f}".format(0.9 * 1254), "$1,129")
derive("90 minus the table's maximum 78.9", "%.0f" % (90 - max(IV)), "11")
derive("column II drop, 20 minutes to one day", "%.1f" % (Q[0] - Q[3]), "24.5")
derive("column II drop, six days to 31 days", "%.1f" % (Q[5] - Q[6]), "4.3")
derive("largest column IV value", "%.1f%%" % max(IV), "78.9%")
derive("Dros six-day loss, column-IV terms, the largest of four at six days", "%.1f%%" % (100 * (1 - min(MACK[5], SEITZ[5], DROS[5], E[5]))), "83.2%")
derive("Mack 31 days, column-IV terms", "%.1f%%" % (100 * (1 - MACK[6])), "74.2%")
derive("Seitz 31 days, column-IV terms", "%.1f%%" % (100 * (1 - SEITZ[6])), "79.9%")
derive("Dros 31 days, column-IV terms", "%.1f%%" % (100 * (1 - DROS[6])), "95.9%")
derive("Dros 31 days drift-corrected (0.137), column-IV terms", "%.1f%%" % (100 * (1 - 0.137)), "86.3%")
derive("Dros 31 days from Table 2 seconds (0.090), column-IV terms", "%.1f%%" % (100 * (1 - q31)), "91.0%")
derive("relearning at 64 minutes as share of learning (column IV, rounded)", "%.0f%%" % IV[1], "56%")
derive("relearning at 31 days as share of learning (column IV, rounded)", "%.0f%%" % IV[6], "79%")
derive("saving when the second session takes 70% as long", "%d%%" % (100 - 70), "30%")
derive("saving when relearning takes 60% as long", "%d%%" % (100 - 60), "40%")

# ---------------------------------------------------------------- the refit
out = (HERE / "refit_ebbinghaus_c6733.out.txt").read_text(encoding="utf-8")
rows = {m.group(1).strip(): m.groups()[1:] for m in re.finditer(r"^(log, his k=1\.84 c=1\.25|log, re-fitted|power a\*t\^-d|exponential a\*e\^\(-t/s\))\s+([\d.]+)\s+([\d.]+)\s+([\d.]+)\s+(.*)$", out, re.M)}
check(len(rows) == 4, "refit output: four model rows")
check(rows["log, his k=1.84 c=1.25"][:3] == ("1.73", "24.5", "21.3"), "refit: his formula RMSE 1.73, 24.5 at 7 d, 21.3 at 30 d")
check(rows["power a*t^-d"][0] == "1.91", "refit: power law RMSE 1.91")
check(rows["exponential a*e^(-t/s)"][:3] == ("9.10", "32.4", "16.3"), "refit: exponential RMSE 9.10, 32.4 at 7 d, 16.3 at 30 d")
ma = re.search(r"a=([\d.]+) s=(\d+) min", rows["exponential a*e^(-t/s)"][3])
a_exp, s_exp = float(ma.group(1)), float(ma.group(2))
T_MIN = [r[0] for r in T24]
rmse_his = math.sqrt(sum((his(t) - q) ** 2 for t, q in zip(T_MIN, Q)) / 7)
rmse_exp = math.sqrt(sum((a_exp * math.exp(-t / s_exp) - q) ** 2 for t, q in zip(T_MIN, Q)) / 7)
check(abs(rmse_his - 1.73) < 0.005, "recomputed: his formula RMSE %.3f" % rmse_his)
check(abs(rmse_exp - 9.10) < 0.01, "recomputed from the printed a, s: exponential RMSE %.3f" % rmse_exp)
check(9.10 / 1.73 > 5, "exponential error / his formula's error = %.2f (more than five times)" % (9.10 / 1.73))
derive("exponential at t=0 (a), rounded", "%.0f%%" % a_exp, "40%")
derive("exponential at 20 minutes", "%.1f%%" % (a_exp * math.exp(-20 / s_exp)), "39.9%")
derive("exponential at one day", "%.1f%%" % (a_exp * math.exp(-1440 / s_exp)), "38.8%")

# ---------------------------------------------------------------- every number in the body is accounted for
NUMBERS_DECLARED = {
    "2026": "dates (MeltingSpot, TalentLMS modified, our April essay)", "2023": "TalentLMS and Today's Class dates",
    "2025": "TrainMeUK and McDowall dates", "1885": "the book's year", "1913": "the translation's year",
    "1879-80": "Ebbinghaus ch. VII, Section 28", "1879": "the experiment's first year", "1880": "the experiment's second year",
    "1879-80)": "table header", "2015": "Murre & Dros", "1991": "Heller, Mack & Seitz", "1996": "Rubin & Wenzel",
    "1984": "Bahrick", "8": "June 8 (MeltingSpot)", "21": "September 21 (TalentLMS modified)", "26": "April 26 (ours)",
    "163": "double tests, Section 28", "13": "syllables per series, Section 28", "38": "six-series tests, Section 28",
    "28": "Section 28", "29": "Section 29 / his age (Murre & Dros)", "22": "Dros's age (Murre & Dros)",
    "70": "hours (Murre & Dros abstract) / TrainMeUK headline / the 70% example", "75": "days (Murre & Dros)",
    "2.67": "seconds a day per list (Murre & Dros)", "0.137": "drift-corrected saving (Murre & Dros)",
    "0.0410": "Murre & Dros's own text", "0.090": "M&D Table 2 average, 31 days", "0.041": "M&D Table 3",
    "210": "Rubin & Wenzel", "105": "Rubin & Wenzel", "2": "'2-parameter' (Rubin & Wenzel)", "733": "Bahrick",
    "50": "Bahrick's 50 years / the retellings' 50%", "3-6": "Bahrick", "30": "Bahrick's 30 years / the 30% example / 'Within 30 days'",
    "24": "'Within 24 hours' (TalentLMS) / TrainMeUK headline / '24 hour data point'", "1": "row label '1 hour'",
    "1,254": "ATD figure quoted by MeltingSpot", "940": "MeltingSpot", "90%": "the retellings",
    "90": "the retellings", "50%": "the retellings", "70%": "the retellings", "25%": "TalentLMS", "80%": "TrainMeUK",
    "56%": "column IV at 64 minutes, rounded", "66": "MeltingSpot", "56": "MeltingSpot", "75%": "MeltingSpot / $940 share",
    "60%": "illustration", "40%": "illustration / exponential start", "64": "table24 t", "526": "table24 t",
    "8.8": "table23 row label", "20": "minutes, table24 t", "31": "days, table23/24", "6": "days", "9": "hours (M&D)",
    "1.73": "refit", "1.91": "refit", "9.10": "refit", "32.4%": "refit", "24.5%": "refit", "16.3%": "refit",
    "21.3%": "refit", "0.1": "footer check", "100": "column II + IV", "Q%": "heading", "v%": "heading",
    "3.3": "table24 difference at 1 day", "33.7": "table23 1 day", "1.84": "his k", "1.25": "his c",
    "140": "the April passage's '140 years' (quoted)", "66%": "MeltingSpot quote", "80": "'1879-80' written with an en-dash",
}
table_vals = {"%.1f%%" % x for x in Q + IV} | {"%.1f" % x for x in Q + IV} | {"%.3f" % x for x in E + MACK + SEITZ + DROS}
allowed = set(NUMBERS_DECLARED) | set(DERIVED) | table_vals
nums = re.findall(r"\$?\d[\d,]*(?:\.\d+)?(?:-\d+)?%?", body)
unaccounted = sorted({n.rstrip(",") for n in nums} - allowed - {n.rstrip(",").lstrip("$") for n in allowed})
unaccounted = [n for n in unaccounted if n not in allowed and n.lstrip("$") not in allowed]
check(not unaccounted, "every number in the essay body is a checked figure or a declared constant (%d tokens)%s"
      % (len(nums), "" if not unaccounted else "; UNACCOUNTED: %s" % unaccounted))

# ---------------------------------------------------------------- summary
bad = [lab for ok, lab in results if not ok]
print("\n%d checks, %d failed" % (len(results), len(bad)))
print("ALL CHECKS PASS" if not bad else "FAILED:\n  " + "\n  ".join(bad))
sys.exit(1 if bad else 0)
