From 1cf79b0f86d11e2ea964ba671878d918b640f6b5 Mon Sep 17 00:00:00 2001 From: David F Glidden Date: Sat, 8 Aug 2026 19:15:14 +0200 Subject: [PATCH] [PROPOSAL] PENDING-124 gate passed; five conditions discharged, and the gate found a tenth instance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Q1 is applied, not extended: Constraint 4 has two clauses and my contrary reading engaged only the second. Limits, not failures — and "I could not look" is a limit. I had overstated my own uncertainty on the question I withdrew a recommendation over. The condition that cost most: my quote-verification pass reported verified on a reconstruction of REVIEWED-104 — contractions, re-punctuation, two blocks spliced, and the closing sentence dropped. A two-valued verifier inside a package arguing verifiers must be three-valued. Rebuilt at ~/dotfiles/scripts/verify-quotes.py. The first rebuild had three tiers and cried wolf on every correctly-copied quote, since a record stored with hard wraps is byte-different from the same text quoted as one line; splitting re-wrapped from normalized is the same two-strengths lesson the fleet learned. Both directions proven: corrected package exit 0, original reconstruction not-found exit 1. The dropped sentence answered my own Q2. It was in the record the package quoted. Both citation errors in that package had one cause, which the script cannot diagnose: I quoted the ADVISORY message and attributed it to the PLACED record. Different documents; placement adds and cuts, so quoting the advisory loses exactly what placement contributed. My "five instances, same shape" was wrong — two are the shape, three belong to the attested-absence family whose parent is already ratified (REVIEWED-47, 2026-07-05). I searched for a doctrinal parent among R0 and Constraint 4 and missed the ratified sibling closest in content. The ladder entry now joins that lineage. Filed as a watch-item, with an operative memory note: third package running where the grounding pass was incomplete and every substantive omission cut against my own argument. It optimises for finding my errors, not my support. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01A35wiD55yRHj5U1ECZAX4t --- PENDING.md | 20 +++ ...valued-checks-JURIST-PACKAGE-2026-08-08.md | 66 ++++---- claude/memory/MEMORY.md | 1 + ...grounding-pass-finds-errors-not-support.md | 14 ++ .../memory/reference-verification-ladder.md | 11 ++ scripts/verify-quotes.py | 142 ++++++++++++++++++ 6 files changed, 227 insertions(+), 27 deletions(-) create mode 100644 claude/memory/feedback-grounding-pass-finds-errors-not-support.md create mode 100644 scripts/verify-quotes.py diff --git a/PENDING.md b/PENDING.md index 0e0e680..14f6897 100644 --- a/PENDING.md +++ b/PENDING.md @@ -2099,6 +2099,26 @@ Three cases, all discriminated: a rule ran → existing output already says so, --- +### AMENDMENT 2 — 2026-08-08, DESIGN GATE PASSED WITH CONDITIONS; five conditions discharged + +**Q1 — APPLIED, and firmly.** Constraint 4 has **two clauses**, and my contrary reading engaged only the second. *"The system must report its own limits"* does not speak of failures at all — it speaks of **limits**, and *"I could not look"* is one. **No constitutional change; no `[ESCALATE]`.** ⚠ Recorded because I withdrew a recommendation on this question: the replacement is firm, and I had **overstated my own uncertainty**. + +**Q2 — binds at BOTH, and the aggregation half was ALREADY RULED — in the sentence I dropped.** REVIEWED-104 §1 closes: *"A green fleet that includes an unassessed binding case is the same overstatement one layer along."* It was in the record the package quoted. **The two-strengths distinction is required, not optional.** + +**Q3 — free, and the line I said I could not find EXISTS and is ratified:** the hash-locality principle's *"the distinct NAMES prevent the collision."* **Names are individuated by REFERENT, not by concept.** Four referents, four names — correct; one-concept-four-homes only if **one referent** carries four names. + +**Q4 — ratify, not provisional — but NOT on the count.** ⚠ Four of the nine instances are dated 2026-08-08 and **downstream of the advisory that proposed the doctrine** — the register responding to its own proposal, which CLAUDE.md's ratified caution governs precisely (jurist and executor *"do not differ from each other in formation"*). Once Q1 is *applied*, authority comes from Constraint 4, not from the count. **Chamber tool-fleet census: owed, not blocking.** + +**⚠ CONDITION 2 relocated the proposal.** My *"five instances, same shape"* was **wrong**: **two** are the shape, **three** belong to the **attested-absence family**, whose parent — REVIEWED-47, **2026-07-05**, *"attested absence lives in its own honest top-level key"* — is **already ratified**. I searched for a parent among R0 (correctly withdrawn) and Constraint 4 and **missed the ratified sibling closest in content**. Corrected on the record per condition 3. + +**⚠ CONDITION 1 — a tenth instance, produced BY THE GATE and the only one independent of the advisory.** My quote-verification pass reported `verified` on a **reconstruction** of REVIEWED-104 — contractions, re-punctuation, two blocks spliced, and the closing sentence dropped. **A two-valued verifier, inside a package arguing that verifiers must be three-valued.** Rebuilt as `~/dotfiles/scripts/verify-quotes.py` with **four tiers** — `exact` / `re-wrapped` / `normalized` / `not-found`, plus author-declared `own-text`. ⚠ The first rebuild had **three** and cried wolf on every correctly-copied quote, because a record stored with hard wraps is byte-different from the same text quoted as one line; splitting `re-wrapped` from `normalized` is the **same two-strengths lesson**. **Both directions proven:** corrected package → exit 0; the original reconstruction → **not-found, exit 1**. + +**Discharged:** (1) verifier rebuilt + controlled · (2) ladder entry names the attested-absence family and cites 2026-07-05 · (3) evidence statement corrected · (4) III.1 now carries the environment-vs-defect split in the normative text · (5) landed as **one ladder entry**, nothing in `~/CLAUDE.md`. + +⚠ **Standing observation, filed as a watch-item:** third package running where the grounding pass was incomplete and **every substantive omission cut AGAINST my own argument** — a stable dated pattern, not an impression. The pass optimises for finding its own errors and not its own support. Operative note: `feedback-grounding-pass-finds-errors-not-support.md`. + +--- + ## PENDING-125 — A live false attestation in the governed record: Mauss's `reading_index_status` has read VERIFIED-BOUND for 53 days **Date:** 2026-08-08 diff --git a/claude/governance/three-valued-checks-JURIST-PACKAGE-2026-08-08.md b/claude/governance/three-valued-checks-JURIST-PACKAGE-2026-08-08.md index 7432c49..6c7603c 100644 --- a/claude/governance/three-valued-checks-JURIST-PACKAGE-2026-08-08.md +++ b/claude/governance/three-valued-checks-JURIST-PACKAGE-2026-08-08.md @@ -47,34 +47,26 @@ This is the whole basis. Everything below argues that a two-valued check **canno limit *"I did not look"*, and therefore violates Constraint 4 by construction whenever its subject can be absent. -### I.2 — `~/REVIEWED.md` REVIEWED-104 §1 (steward ruling, reachable as `reviewed`) +### I.2 — `~/REVIEWED.md` REVIEWED-104 (steward ruling, reachable as `reviewed`) -> A live-corpus assertion makes one suite depend on `chamber-library` being present and reachable. -> Every other suite builds under `tmp` and is portable; this one won't be. The item doesn't say -> what happens on a fresh clone with no chamber beside it, and both obvious answers are wrong: -> **Red on absent** — a suite that goes red for reasons unrelated to the code under test. That -> trains people to discount fleet red, which is the worst possible outcome for this particular -> thread. **Skip on absent** — the silent net, in the very assertion added to correct an -> overstatement. -> So: **three states — `bound` / `drifted` / `cannot-assess`** — and `cannot-assess` must be -> distinguishable in the fleet summary, never folded into green. +⚠ **REPLACED 2026-08-08 at the design gate. What stood here was a NORMALIZED RECONSTRUCTION of +this ruling, not its placed text**, and my own quote-verification pass counted it among the +verified. Differences the jurist enumerated and I confirmed against the record: `will not be` → +`won't be`; `does not say` → `doesn't say`; `wrong.` → `wrong:`; `CONDITION:` → `So:`; the Notes +paragraph and §1 spliced into one block though the record keeps them separate — **and the closing +sentence dropped.** Below is the placed text, `~/REVIEWED.md` L1246–1250: -⚠ **CITATION CORRECTED before relay — the passage that prompted this package is NOT in the -register, and I first cited it as though it were.** The observation below came from the jurist's -**advisory on PENDING-122/123**, relayed by the steward; the *placed* REVIEWED-104 does not contain -it. Verified by search: the phrase appears in `~/PENDING.md` (where I recorded it) and in this file, -and **nowhere in `~/REVIEWED.md`.** It is quoted here as **advisory, not as ruled text**: +> **Notes:** A live-corpus assertion makes one suite depend on chamber-library being present and reachable. Every other suite builds under tmp and is portable; this one will not be. The item does not say what happens on a fresh clone with no chamber beside it, and both obvious answers are wrong. Red on absent is a suite going red for reasons unrelated to the code under test, which trains people to discount fleet red — the worst possible outcome for this particular thread. Skip on absent is the silent net, in the very assertion added to correct an overstatement. -> Note where else that requirement just appeared. PENDING-123's acceptance test independently -> concludes that the valid-but-never-matching rule *'needs a third state, not a pass or a fail.'* -> Two subsystems, same day, same finding: **a check that reaches outside its own repo cannot be -> two-valued.** That is a candidate for doctrine rather than for restating per item — I'd rather -> rule it once than condition it three more times. +> 1. CONDITION: three states — bound / drifted / cannot-assess — and cannot-assess must be distinguishable in the fleet summary, never folded into green. A green fleet that includes an unassessed binding case is the same overstatement one layer along. -*Recorded rather than quietly repaired: a package that mis-attributes a quotation to a governed -record is the exact failure this instrument exists to prevent, and the second time this week I have -pointed a citation at the wrong entry. The mechanical quote-verification pass caught it — which is -the argument for running that pass rather than trusting the draft.* +**The dropped sentence is the one that answers Q2**, and it was in the record all along. See Part VII. + +⚠ **BOTH citation errors in this package have ONE cause, and the script cannot diagnose it: I +quoted the ADVISORY message and attributed it to the PLACED record.** They are different documents +— placement can add, cut, or re-word — so quoting the advisory systematically loses whatever the +act of placing contributed. Here it lost precisely the sentence that would have closed a gate +question without asking. Recorded as the generalizable finding, not as two separate slips. ### I.3 — `reference-verification-ladder.md`, the silent-net entry (steward-held; executor testimony) @@ -123,9 +115,23 @@ traceback where the honest answer was *"I could not look"*. Closed 2026-08-08 at rather than per-site. ⚠ **Closing the first six revealed two more** in suites the first census had cleared — the class was wider than the instrument that found it. -⚠ **Instances 1–5 are why I think this is discovered rather than invented.** Five independent -implementations of the same shape, in three subsystems, by different hands, before anyone proposed -a rule. A doctrine that has to be argued into existence is weaker than one that has to be *noticed*. +⚠ **CORRECTED AT THE GATE — the five are NOT one shape, and my *"five … same shape"* was wrong.** +Tested against this table's own rows: **two are the doctrine's shape** (1 R0's three states; 3 +`test_retrieve`'s named skip — a check reporting it could not assess). **Three belong to an adjacent +principle**: 2's own row says it refuses two-valuedness *"for a different reason: declared-vs-new"*, +and a declared failure is **assessed**, not unassessable; 4 is a **marking** shape, detectable +staleness of declared data; 5 is attested absence of a **finding**, not of an **assessment**. + +**So: two pre-existing instances of the shape, three of the adjacent principle, one found at the +gate itself** (§2 of the ruling — this package's own verifier reporting `verified` on a +reconstruction; the only instance in the set independent of the advisory that proposed the doctrine). + +⚠ **And the adjacent principle is already ratified**, which relocates the proposal rather than +weakening it: `~/REVIEWED.md` REVIEWED-47, **2026-07-05** — the exclusion path, *"attested absence +lives in its own honest top-level key `source_excluded`"*, with the `sectionless:` precedent that a +bare flag is not a safeguard and an attributed attestation is. **I searched for a parent among R0 +(D-1, correctly withdrawn) and Constraint 4, and missed the ratified sibling closest to it in +content.** --- @@ -133,9 +139,15 @@ a rule. A doctrine that has to be argued into existence is weaker than one that ### III.1 — Statement + > A check whose subject can be **absent** must report three outcomes, not two: the property holds, > the property fails, or **the property could not be assessed** — and the third must remain > distinguishable in every aggregate the check feeds. +> +> **The third outcome is itself two kinds, and they must not be merged:** unassessable because the +> **subject** is absent — an environment condition, which must not block — and unassessable because +> the **check** is broken — a defect, which must. Merging them lets a broken check hide behind an +> environment excuse. ### III.2 — Why two values cannot satisfy Constraint 4 diff --git a/claude/memory/MEMORY.md b/claude/memory/MEMORY.md index c8d378e..d6e2692 100644 --- a/claude/memory/MEMORY.md +++ b/claude/memory/MEMORY.md @@ -27,6 +27,7 @@ permalink: claude-memory/memory - [The central path — answerability, not purity](feedback-central-path-answerability-not-purity.md) — the contamination recursion is probably irresolvable, so **bind the claims, don't certify the parties**. Route by claim-type: *checkable* → produce the check + a falsifier; *judgment* → disclose standpoint in one line, decide, record; *undecidable* → name it open. **One layer, then act — never audit the audit.** - [Notes are part of the work](feedback-notes-are-part-of-the-work-keep-footnotes-endnotes.md) — footnotes/endnotes are integral: KEEP+CONVERT (``→`[^N]`), never drop on graduation. - [One-shot instruments are proportionate](feedback-one-shot-instruments-are-proportionate.md) — a measurement answering a question **asked once** is NOT a directive violation; its counterfactual is an **assertion**, not a durable tool. The violation is **re-writing what's already banked** (rule of three → ladder). *Too few promoted*, not *too many built*. +- [Grounding finds my errors, not my support](feedback-grounding-pass-finds-errors-not-support.md) — **three packages running, every omission cut AGAINST my own case.** Ground in TWO passes: (1) what did I get wrong? (2) **what already says this?** Search the register for the CONCLUSION, not just the citations. ⚠ Quote from the FILE, never from the relayed message — placement adds and cuts. - [Resurface banked notes before re-deriving](feedback-resurface-banked-notes-before-rederiving.md) · [Checkable claim surfaces bugs](feedback-checkable-claim-surfaces-bugs.md) · [Census by mechanism, not proxy](feedback-census-by-mechanism-not-proxy.md) · [Completion is a tripwire](feedback-completion-is-a-tripwire.md) · [Trust prior pass frame](feedback-trust-prior-pass-frame.md) — the five epistemic disciplines. Kernels: **read the banked note before re-deriving** · **a checkable claim over a soft classification is itself a defect-detector** · **census by RUNNING the real pipeline** · **the feeling of "done" is the cue to verify the tail** · **re-run a prior verification at the scope of your extension**. ⚠ Also carried as Symmetria §3 flags — *two surfaces, deliberately*: §3 loads only when Symmetria is invoked, so these stay here for sessions where it isn't. **Loud trigger — pointer suffices** diff --git a/claude/memory/feedback-grounding-pass-finds-errors-not-support.md b/claude/memory/feedback-grounding-pass-finds-errors-not-support.md new file mode 100644 index 0000000..edd9e13 --- /dev/null +++ b/claude/memory/feedback-grounding-pass-finds-errors-not-support.md @@ -0,0 +1,14 @@ +--- +name: feedback-grounding-pass-finds-errors-not-support +description: My grounding passes are asymmetric — they reliably catch my own mistakes and reliably miss the passages that would SUPPORT my argument. Three dated jurist packages, every substantive omission cutting against me. +metadata: + type: feedback +--- + +**The grounding pass is optimising for finding its own errors and not for finding its own support.** Jurist observation, 2026-08-08, on the third package in a row. + +**Why:** in three consecutive packages the grounding pass was incomplete, and **every substantive omission cut against my own argument** — the opposite of advocacy, so it is not motivated reasoning; it is a search shaped by *"what have I got wrong?"* and never by *"what already backs this?"*. The clearest case: I quoted REVIEWED-104 and **dropped its closing sentence**, which was the sentence that would have closed my own gate question Q2 without asking. I also searched for a doctrinal parent among R0 and Constraint 4 and **missed the ratified sibling closest in content** (the 2026-07-05 attested-absence ruling) — which, when found, relocated and strengthened the proposal. + +**How to apply:** when grounding an argument, run the search **twice, with both intents stated**. Pass 1 (already habitual): *what have I quoted wrongly, over-claimed, or failed to check?* Pass 2 (the one that does not happen): *what in the ratified record ALREADY says this, or already settles a question I am about to ask?* Search the register for the **conclusion**, not only for the citations — if a gate question is answerable from the record, asking it wastes the ruling. ⚠ Precise cause, from the same day: **quoting the ADVISORY message and attributing it to the PLACED record.** They are different documents; placement can add, cut, or re-word, so quoting the advisory systematically loses whatever the act of placing contributed. Quote from the file, never from the message. + +Related: [[feedback-resurface-banked-notes-before-rederiving]] · [[reference-verification-ladder]] · [[feedback-checkable-claim-surfaces-bugs]] diff --git a/claude/memory/reference-verification-ladder.md b/claude/memory/reference-verification-ladder.md index ef304cb..9c48040 100644 --- a/claude/memory/reference-verification-ladder.md +++ b/claude/memory/reference-verification-ladder.md @@ -123,6 +123,17 @@ Proven gates, each earned from a real catch. Reach for the one the claim's shape - **Every check states, in its own output, what it did NOT establish** — the necessary-but-not-sufficient gap named beside the pass. A check that cannot name its gap does not ship. - **The residue is irreducible and needs a differently-formed reader.** Discrimination catches proxy-gaps where a real negative instance exists; it cannot catch a proxy that discriminates on the pair and fails elsewhere. What is left must be *looked at* before the claim is made, by someone who is not the check's author — Constraint 6 applied to instruments. Reference implementation: `dotfiles/claude/governance/fool/test_discrimination.py`, which is shown rejecting the §3.3 pattern **as it actually shipped**. +## Checks whose subject can be absent + +*Ratified 2026-08-08 as an **application** of Constitutional Constraint 4 — *"The system must report its own limits. Silent failures are architectural violations"* — not as new doctrine. **It joins the attested-absence family**, whose parent is the 2026-07-05 exclusion-path ruling (REVIEWED-47): *attested absence lives in its own honest top-level key*, with the `sectionless:` precedent that a bare flag is not a safeguard and an attributed attestation is. This entry is that principle applied to the **reporting of checks** rather than the recording of findings.* + +- **A check whose subject can be ABSENT cannot be two-valued.** Three outcomes: the property holds · the property fails · **the property could not be assessed** — and the third must stay distinguishable in every aggregate the check feeds. A two-valued check conflates *"I looked and it holds"* with *"I could not look"*; inside one repo that is usually harmless, but the moment a check reaches across a repo, a network, a scheduler or an optional dependency, **absence becomes an ordinary condition** and both remaining verdicts are lies. *Skip on absent* is the silent net; *red on absent* trains people to discount red (REVIEWED-104). +- **The third outcome is itself TWO KINDS and they must not merge:** unassessable because the **subject** is absent (environment — must not block) vs unassessable because the **check** is broken (defect — must block). Merged, a broken check hides behind an environment excuse. +- **Binds at aggregation, not only at reporting** — *"A green fleet that includes an unassessed binding case is the same overstatement one layer along"* (REVIEWED-104 §1). An aggregate may not report clean while any member is unassessed. +- ⚠ **But weaken in TWO STRENGTHS, or the signal dies.** A suite-level non-verdict withdraws the word *green*; a per-check skip is **counted** and does not. Reached by building it: treating them alike made *"NOT A CLEAN PASS"* permanent, because one long-standing skip was vacuous-by-corpus-state — *a check that always says the same thing stops being read*. The same split later applied to a quote-verifier (`re-wrapped` vs `normalized`): **a warning that fires on the safe case is discarded along with the dangerous one.** +- **Names are individuated by REFERENT, not by concept** (from the hash-locality principle's *"the distinct NAMES prevent the collision"*). `unverified` = a region's binding state · `cannot-assess` = a suite's verdict · `blocked` = a source's gate state. Four referents, four names, correct. It is one-concept-four-homes only if **one referent** carries four names. +- **Turn it on its own instruments first.** The quote-verification pass built for a package arguing this doctrine was itself two-valued, and reported `verified` on a reconstruction that had dropped the sentence answering the package's own gate question. Reference implementation: `~/dotfiles/scripts/verify-quotes.py` (`exact` / `re-wrapped` / `normalized` / `not-found`, plus author-declared `own-text` excluded from assessment), shown red on the original reconstruction and clean on the corrected text. + ## Estimates and schedules - **Quote a long-job ETA only from an observed rate** — twice in one day I gave an ETA from intuition and was wrong by ~30×. Measure rows-per-elapsed on the running job, or benchmark a slice, then quote. diff --git a/scripts/verify-quotes.py b/scripts/verify-quotes.py new file mode 100644 index 0000000..4975a66 --- /dev/null +++ b/scripts/verify-quotes.py @@ -0,0 +1,142 @@ +#!/usr/bin/env python3 +"""verify-quotes.py — check a package's blockquotes against the records it cites. + +WHY THIS EXISTS, AND WHY IT IS THREE-VALUED (REVIEWED-⟨N⟩ condition 1, PENDING-124). + +The inline predecessor of this script reported `verified` for a REVIEWED-104 blockquote +that was a NORMALIZED RECONSTRUCTION, not the placed text: `will not be`→`won't be`, +`does not say`→`doesn't say`, `wrong.`→`wrong:`, `CONDITION:`→`So:`, two separate blocks +spliced into one, and — the part that mattered — THE CLOSING SENTENCE DROPPED. That +sentence was the one that answered the package's own Q2 without asking. + +A two-valued verifier said `verified`. The honest report was `matched after +normalization`. It was a two-valued verifier inside a package arguing that verifiers +whose subject can be absent must be three-valued, which is the defect the doctrine +describes, committed by the instrument checking the doctrine's own evidence. + + exact byte-for-byte in a cited source + re-wrapped identical after collapsing WHITESPACE only — a record stored with hard + line wraps quoted as one line. Content identical; safe. + normalized found only after folding emphasis/contractions/punctuation — REPORT IT, + because THIS is where a reconstruction hides + not-found in no cited source + + ⚠ re-wrapped and normalized were one tier in the first version, which cried wolf on + every correctly-copied quote. Splitting them is the same lesson the fleet summary + learned: a warning that fires on the safe case stops being read. + +DIAGNOSIS THE SCRIPT CANNOT MAKE, so the author must: BOTH citation errors in that +package had one cause — quoting the ADVISORY message and attributing it to the PLACED +record. They are different documents. Placement can add, cut or re-word, so quoting the +advisory systematically loses whatever the act of placing contributed. When a passage +comes from a relayed message rather than a file, it is not a quotation of the record and +must be labelled advisory. + +A blockquote that is the author's own proposed text is not a quotation claim at all and +is excluded from assessment — mark it with `` on the line before. An +unmarked block that is not found is reported, never silently skipped. + +Usage: python3 verify-quotes.py PACKAGE.md SOURCE [SOURCE ...] +Exit: 0 all assessed blocks exact · 1 any not-found · 3 any normalized-only +""" +import re +import sys +from pathlib import Path + +EXACT, REWRAPPED, NORMALIZED, NOT_FOUND = "exact", "rewrapped", "normalized", "not-found" + + +def fold(s: str) -> str: + """Normalization deliberately AGGRESSIVE: the wider it folds, the more it will + classify as `normalized` rather than `exact`, and that is the reporting we want.""" + s = re.sub(r"[*_`]", "", s) + s = (s.replace("—", "-").replace("–", "-") + .replace("’", "'").replace("‘", "'") + .replace("“", '"').replace("”", '"')) + for long, short in [("will not", "wont"), ("won't", "wont"), + ("does not", "doesnt"), ("doesn't", "doesnt"), + ("cannot", "cant"), ("can't", "cant"), + ("is not", "isnt"), ("isn't", "isnt")]: + s = s.replace(long, short) + s = re.sub(r"[^\w\s']", " ", s) + return re.sub(r"\s+", " ", s).strip().lower() + + +def blocks(text: str): + """Contiguous blockquote runs, with the `own-text` marker honoured.""" + out, cur, own = [], [], False + for line in text.splitlines(): + if line.strip().lower() == "": + own = True + continue + if line.startswith(">"): + cur.append(line.lstrip(">").strip()) + elif cur: + out.append((" ".join(cur).strip(), own)); cur, own = [], False + elif line.strip(): + own = False + if cur: + out.append((" ".join(cur).strip(), own)) + return [(b, o) for b, o in out if b] + + +def ws(s: str) -> str: + """Whitespace ONLY. A record stored with hard line wraps is byte-different from the + same text quoted as one line, and calling that a `normalized match` cries wolf — the + same failure as a fleet summary that never varies. Re-wrapping is safe; re-wording + is not, and they must not share a verdict.""" + import re as _re + return _re.sub(r"\s+", " ", s).strip() + + +def classify(quote: str, sources: dict[str, str]): + for name, text in sources.items(): + if quote and quote in text: + return EXACT, name + wq = ws(quote) + for name, text in sources.items(): + if wq and wq in ws(text): + return REWRAPPED, name + fq = fold(quote) + for name, text in sources.items(): + if fq and fq in fold(text): + return NORMALIZED, name + return NOT_FOUND, None + + +def main(argv): + if len(argv) < 2: + print(__doc__); return 2 + pkg = Path(argv[0]).read_text(encoding="utf-8") + sources = {} + for p in argv[1:]: + path = Path(p).expanduser() + if path.exists(): + sources[path.name] = path.read_text(encoding="utf-8") + else: + print(f" ⚠ source unreadable, NOT searched: {p}") + counts = {EXACT: 0, REWRAPPED: 0, NORMALIZED: 0, NOT_FOUND: 0, "own-text": 0} + print(f" checking {len(blocks(pkg))} blockquote(s) against {len(sources)} source(s)\n") + for q, own in blocks(pkg): + if own: + counts["own-text"] += 1 + print(f" [own-text ] {q[:66]}…") + continue + state, where = classify(q, sources) + counts[state] += 1 + tag = {EXACT: "exact ", REWRAPPED: "re-wrapped", NORMALIZED: "NORMALIZED", NOT_FOUND: "NOT FOUND"}[state] + print(f" [{tag}] {q[:60]}…" + (f" <- {where}" if where else "")) + if state == NORMALIZED: + print(" ^ reconstruction, not the placed text. Diff it before relying on it.") + print(f"\n exact {counts[EXACT]} · re-wrapped {counts[REWRAPPED]} (content identical) · " + f"normalized {counts[NORMALIZED]} · not-found {counts[NOT_FOUND]} · " + f"own-text {counts['own-text']} (unassessed by design)") + if counts[NOT_FOUND]: + return 1 + if counts[NORMALIZED]: + return 3 + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:]))