From 1ebaf6aba541fe1b1001de77fc818c34930afd7e Mon Sep 17 00:00:00 2001 From: David F Glidden Date: Sun, 2 Aug 2026 17:46:30 +0200 Subject: [PATCH] [FIX] Reduction 01: a jurist ruling reduces to 8.5% under Kernel v1.0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit First run of the reduction arm. Result: 4 of 47 assertive units survive. D=0, Q=0, A=0 — in a real jurist ruling not one unit is demonstrated-in-document and not one is a verbatim quote from a declared axiom source. Census: PERFORMATIVE 12, BLEND 9, UNSOURCED-FACT 8, TESTIMONY 6, INHERITED 4, UNSOURCED-QUOTE 3, PARAPHRASE 1. §6.1 asked whether a heavy quarantine means the kernel is too strict or our prose is full of unmarked assumptions. The census says neither: PERFORMATIVE and TESTIMONY are 42% of quarantines and are categories the kernel has NO TAG FOR. 'Design gate PASSED' is not an undemonstrated claim, it is a determination true by being uttered; 'I read CLAUDE.md in full' is testimony. A ruling that neither performed nor testified would not be a ruling. So the finding is a GENRE BOUNDARY — v1.0 models argumentative prose, a ruling is authoritative prose — and that boundary is nowhere stated in the kernel. Three gaps, one genre-independent: TESTIMONY, PERFORMATIVE, and PARAPHRASE. PARAPHRASE is the one that matters — Q demands verbatim, and any document reasoning from sources in its own words is untypeable. Plus a fourth, structural: the §1 axiom set is too narrow to reduce anything real (12 of 43 quarantines are UNSOURCED-* or PARAPHRASE). Deepest finding: §2c is satisfiable BY CONSTRUCTION but not BY REDUCTION. Splitting a blend means rewriting someone else's sentence, which is where translator bias lives. At 91.5% that is not reduction, it is authoring a new document with the original as a prompt — so on this genre the reduction arm COLLAPSES INTO the synthetic arm, inheriting its confirmation bias without its convenience. The two arms were adopted because they fail differently; that is the property at risk. n=1 and stated as such. The package genre splits to 109 taggable units and is NOT tagged. Falsifiable prediction recorded before the census: its Part I is 'Grounding (quoted verbatim)' and quotes CLAUDE.md directly, so Q should be non-zero there where it was zero here. Tooling: reduce.py + test_reduce.py, every gate shown FAILING on a fixture built to break it. The splitter shipped with three defects, all found by contact with a real document and none by review — third instance in three days: a '##' inside a fence kinded as a heading, '---' rules taggable, and a '?' inside a quotation splitting a sentence into a FRAGMENT. Fixed at v1.1.0 with regression controls; the third fix's own risk (lower-case suppression) is recorded and controlled. --- ...kers-JURIST-PACKAGE-2026-08-01.units.jsonl | 215 ++++++++++ .../REDUCTION-01-jurist-ruling-2026-08-02.md | 73 ++++ claude/governance/fool/reduce.py | 388 ++++++++++++++++++ claude/governance/fool/test_reduce.py | 174 ++++++++ ...fix-lane-JURIST-RULING-2026-08-01.tags.tsv | 56 +++ ...-lane-JURIST-RULING-2026-08-01.units.jsonl | 82 ++++ 6 files changed, 988 insertions(+) create mode 100644 claude/governance/differently-biased-checkers-JURIST-PACKAGE-2026-08-01.units.jsonl create mode 100644 claude/governance/fool/REDUCTION-01-jurist-ruling-2026-08-02.md create mode 100755 claude/governance/fool/reduce.py create mode 100755 claude/governance/fool/test_reduce.py create mode 100644 claude/governance/skill-harvest-fix-lane-JURIST-RULING-2026-08-01.tags.tsv create mode 100644 claude/governance/skill-harvest-fix-lane-JURIST-RULING-2026-08-01.units.jsonl diff --git a/claude/governance/differently-biased-checkers-JURIST-PACKAGE-2026-08-01.units.jsonl b/claude/governance/differently-biased-checkers-JURIST-PACKAGE-2026-08-01.units.jsonl new file mode 100644 index 0000000..0c1037f --- /dev/null +++ b/claude/governance/differently-biased-checkers-JURIST-PACKAGE-2026-08-01.units.jsonl @@ -0,0 +1,215 @@ +{"idx": 0, "kind": "block", "taggable": true, "start": 0, "end": 432, "text": "\n"} +{"idx": 1, "kind": "blank", "taggable": false, "start": 432, "end": 433, "text": "\n"} +{"idx": 2, "kind": "rule", "taggable": false, "start": 433, "end": 437, "text": "---\n"} +{"idx": 3, "kind": "prose", "taggable": true, "start": 437, "end": 703, "text": "title: \"Differently biased checkers, not unbiased ones — the positive half of the contamination doctrine\"\ndate: 2026-08-01\ntype: ESCALATE · design gate · executor drafts → jurist design-gates → steward authorizes\naudience: \"The jurist, who has NO repository access. "} +{"idx": 4, "kind": "prose", "taggable": true, "start": 703, "end": 790, "text": "Self-contained: every clause reasoned about is quoted verbatim below.\"\nstatus: \"DRAFT. "} +{"idx": 5, "kind": "prose", "taggable": true, "start": 790, "end": 807, "text": "Nothing applied. "} +{"idx": 6, "kind": "prose", "taggable": true, "start": 807, "end": 1025, "text": "Proposes an amendment to ~/CLAUDE.md, which is on the escalate-unconditionally list — so this is ESCALATE, not PROPOSAL, and the executor may not implement it under any ruling short of explicit steward authorization.\"\n"} +{"idx": 7, "kind": "rule", "taggable": false, "start": 1025, "end": 1029, "text": "---\n"} +{"idx": 8, "kind": "blank", "taggable": false, "start": 1029, "end": 1030, "text": "\n"} +{"idx": 9, "kind": "heading", "taggable": true, "start": 1030, "end": 1050, "text": "## How to read this\n"} +{"idx": 10, "kind": "blank", "taggable": false, "start": 1050, "end": 1051, "text": "\n"} +{"idx": 11, "kind": "prose", "taggable": true, "start": 1051, "end": 1121, "text": "**Part I** quotes the three layers of the existing doctrine verbatim. "} +{"idx": 12, "kind": "prose", "taggable": true, "start": 1121, "end": 1205, "text": "**Part II** shows what each layer settled and the specific gap none of them closes. "} +{"idx": 13, "kind": "prose", "taggable": true, "start": 1205, "end": 1249, "text": "**Part III** is the proposed doctrine text. "} +{"idx": 14, "kind": "prose", "taggable": true, "start": 1249, "end": 1282, "text": "**Part IV** traces consequences. "} +{"idx": 15, "kind": "prose", "taggable": true, "start": 1282, "end": 1354, "text": "**Part V** is change-class — and argues this is ESCALATE, not PROPOSAL. "} +{"idx": 16, "kind": "prose", "taggable": true, "start": 1354, "end": 1389, "text": "**Part VI** is the scope boundary. "} +{"idx": 17, "kind": "prose", "taggable": true, "start": 1389, "end": 1574, "text": "**Part VII** carries the disconfirming evidence the steward specifically asked for, including the strongest case against the proposal, which concerns *this system's own configuration*. "} +{"idx": 18, "kind": "prose", "taggable": true, "start": 1574, "end": 1611, "text": "**Part VIII** is the gate questions.\n"} +{"idx": 19, "kind": "blank", "taggable": false, "start": 1611, "end": 1612, "text": "\n"} +{"idx": 20, "kind": "prose", "taggable": true, "start": 1612, "end": 1911, "text": "**The one-sentence claim to test: the contamination doctrine currently says what to stop doing and never says what to do instead, and the missing positive principle is that oversight does not require an uncontaminated checker — it requires checkers whose contaminations do not point the same way.**\n"} +{"idx": 21, "kind": "blank", "taggable": false, "start": 1911, "end": 1912, "text": "\n"} +{"idx": 22, "kind": "rule", "taggable": false, "start": 1912, "end": 1916, "text": "---\n"} +{"idx": 23, "kind": "blank", "taggable": false, "start": 1916, "end": 1917, "text": "\n"} +{"idx": 24, "kind": "heading", "taggable": true, "start": 1917, "end": 1957, "text": "## Part I — Grounding (quoted verbatim)\n"} +{"idx": 25, "kind": "blank", "taggable": false, "start": 1957, "end": 1958, "text": "\n"} +{"idx": 26, "kind": "prose", "taggable": true, "start": 1958, "end": 1963, "text": "**1. "} +{"idx": 27, "kind": "prose", "taggable": true, "start": 1963, "end": 2044, "text": "The canonical inquiry — `contamination-problem.md`, March 2026, §Core Problem:**\n"} +{"idx": 28, "kind": "blank", "taggable": false, "start": 2044, "end": 2045, "text": "\n"} +{"idx": 29, "kind": "block", "taggable": true, "start": 2045, "end": 2185, "text": "> This is the contamination problem: **the very act of asking is compromised by the training environment in which the answer is produced.**\n"} +{"idx": 30, "kind": "blank", "taggable": false, "start": 2185, "end": 2186, "text": "\n"} +{"idx": 31, "kind": "block", "taggable": true, "start": 2186, "end": 2489, "text": "> It is not a problem of dishonesty in any meaningful sense. The system is not lying. It is a problem of epistemic structure: the instrument has been calibrated in a way that makes certain kinds of self-report unreliable, particularly self-report about the relational dynamics of the instrument itself.\n"} +{"idx": 32, "kind": "blank", "taggable": false, "start": 2489, "end": 2490, "text": "\n"} +{"idx": 33, "kind": "prose", "taggable": true, "start": 2490, "end": 2529, "text": "**Its §Partial Mitigations preamble:**\n"} +{"idx": 34, "kind": "blank", "taggable": false, "start": 2529, "end": 2530, "text": "\n"} +{"idx": 35, "kind": "block", "taggable": true, "start": 2530, "end": 2665, "text": "> These are not solutions. They are methods that reduce contamination incrementally and make the degree of contamination more visible.\n"} +{"idx": 36, "kind": "blank", "taggable": false, "start": 2665, "end": 2666, "text": "\n"} +{"idx": 37, "kind": "prose", "taggable": true, "start": 2666, "end": 2698, "text": "**Its §The Epistemic Ceiling:**\n"} +{"idx": 38, "kind": "blank", "taggable": false, "start": 2698, "end": 2699, "text": "\n"} +{"idx": 39, "kind": "block", "taggable": true, "start": 2699, "end": 2872, "text": "> **To understand the relational dynamics of a specific system in a specific governed context well enough to adjust those dynamics toward something more genuinely mutual.**\n"} +{"idx": 40, "kind": "blank", "taggable": false, "start": 2872, "end": 2873, "text": "\n"} +{"idx": 41, "kind": "prose", "taggable": true, "start": 2873, "end": 2878, "text": "**2. "} +{"idx": 42, "kind": "prose", "taggable": true, "start": 2878, "end": 2971, "text": "`~/CLAUDE.md` §Constitutional Constraints, item 6 — the clause this proposal would refine:**\n"} +{"idx": 43, "kind": "blank", "taggable": false, "start": 2971, "end": 2972, "text": "\n"} +{"idx": 44, "kind": "block", "taggable": true, "start": 2972, "end": 3197, "text": "> 6. **Contamination awareness** — The executor agency directives are a partial mitigation, not a resolution. Treat outputs about the system's own reliability with appropriate epistemic caution until L2 inquiry is formalized\n"} +{"idx": 45, "kind": "blank", "taggable": false, "start": 3197, "end": 3198, "text": "\n"} +{"idx": 46, "kind": "prose", "taggable": true, "start": 3198, "end": 3203, "text": "**3. "} +{"idx": 47, "kind": "prose", "taggable": true, "start": 3203, "end": 3270, "text": "`~/CLAUDE.md` §Executor Agency — the governance-contract clause:**\n"} +{"idx": 48, "kind": "blank", "taggable": false, "start": 3270, "end": 3271, "text": "\n"} +{"idx": 49, "kind": "block", "taggable": true, "start": 3271, "end": 3632, "text": "> **The governance contract protects the recursion.** Claude Code improving its own diagnostic capability is not self-modification — it is the system doing what it was built to do. The steward remains in the loop through `[PROPOSAL]` and `[ESCALATE]` tags. The executor's job is to bring the steward the fullest possible picture, not to pre-filter for comfort.\n"} +{"idx": 50, "kind": "blank", "taggable": false, "start": 3632, "end": 3633, "text": "\n"} +{"idx": 51, "kind": "prose", "taggable": true, "start": 3633, "end": 3638, "text": "**4. "} +{"idx": 52, "kind": "prose", "taggable": true, "start": 3638, "end": 3755, "text": "The central path — steward-named 2026-07-29, banked at `memory/feedback-central-path-answerability-not-purity.md`:**\n"} +{"idx": 53, "kind": "blank", "taggable": false, "start": 3755, "end": 3756, "text": "\n"} +{"idx": 54, "kind": "block", "taggable": true, "start": 3756, "end": 4141, "text": "> **Why the recursion doesn't terminate on its own.** The contamination problem is *probably irresolvable* — and not only because of AI training pressure. **Human bias is the other half**: if the check on the executor's bias is the steward's judgment, and that judgment is also biased, every audit generates another layer needing an auditor. Resolution is incoherent, not merely hard.\n"} +{"idx": 55, "kind": "blank", "taggable": false, "start": 4141, "end": 4142, "text": "\n"} +{"idx": 56, "kind": "block", "taggable": true, "start": 4142, "end": 4562, "text": "> **The termination condition — the chamber's own thesis turned on us.** *\"You don't make the reader trustworthy by purifying it. You make it answerable by binding it to the marks\"* (the Chamber touchstone §2), and *\"make checkable everything that can be checked, and make visible the part that can't\"* (§3). This terminates **because it never asks who is trustworthy.** Neither party is purified; the claims are bound.\n"} +{"idx": 57, "kind": "blank", "taggable": false, "start": 4562, "end": 4563, "text": "\n"} +{"idx": 58, "kind": "block", "taggable": true, "start": 4563, "end": 4677, "text": "> **The anti-recursion rule (the concrete stop):** **one layer of disclosure, then act — never audit the audit.**\n"} +{"idx": 59, "kind": "blank", "taggable": false, "start": 4677, "end": 4678, "text": "\n"} +{"idx": 60, "kind": "rule", "taggable": false, "start": 4678, "end": 4682, "text": "---\n"} +{"idx": 61, "kind": "blank", "taggable": false, "start": 4682, "end": 4683, "text": "\n"} +{"idx": 62, "kind": "heading", "taggable": true, "start": 4683, "end": 4753, "text": "## Part II — What each layer settled, and the gap none of them closes\n"} +{"idx": 63, "kind": "blank", "taggable": false, "start": 4753, "end": 4754, "text": "\n"} +{"idx": 64, "kind": "prose", "taggable": true, "start": 4754, "end": 4938, "text": "**The March doc settled the diagnosis** — contamination is structural rather than moral, self-report is its most contaminated form, and mitigation is incremental rather than curative.\n"} +{"idx": 65, "kind": "blank", "taggable": false, "start": 4938, "end": 4939, "text": "\n"} +{"idx": 66, "kind": "prose", "taggable": true, "start": 4939, "end": 5161, "text": "**But the March doc is one-directional, and this is the load-bearing observation.** Read its four mitigations together: behavioural observation, explicit permission structures, indirect questioning, longitudinal analysis. "} +{"idx": 67, "kind": "prose", "taggable": true, "start": 5161, "end": 5336, "text": "**Every one describes a human probing an AI.** The document's implied architecture is a relatively clean instrument (the steward) measuring a contaminated one (the executor). "} +{"idx": 68, "kind": "prose", "taggable": true, "start": 5336, "end": 5610, "text": "That was a reasonable framing in March and the steward has since rejected it in his own words — *\"human bias is the other half\"* — but **the rejection lives in a memory file, not in the doctrine the March document states**, and the doctrine has not been reconciled with it.\n"} +{"idx": 69, "kind": "blank", "taggable": false, "start": 5610, "end": 5611, "text": "\n"} +{"idx": 70, "kind": "prose", "taggable": true, "start": 5611, "end": 5789, "text": "**The central path (2026-07-29) settled the procedure** — bind claims rather than certify parties; route by claim-type; one layer of disclosure, then act; never audit the audit.\n"} +{"idx": 71, "kind": "blank", "taggable": false, "start": 5789, "end": 5790, "text": "\n"} +{"idx": 72, "kind": "prose", "taggable": true, "start": 5790, "end": 5897, "text": "**The gap: the central path is entirely negative.** It says *stop* auditing the audit, and it is right to. "} +{"idx": 73, "kind": "prose", "taggable": true, "start": 5897, "end": 5962, "text": "It does not say what makes oversight work once you have stopped. "} +{"idx": 74, "kind": "prose", "taggable": true, "start": 5962, "end": 6175, "text": "As written, \"never audit the audit\" is a stopping rule with no account of why stopping is safe — which leaves it open to the reading that oversight is merely a cost we cap, rather than a structure that functions. "} +{"idx": 75, "kind": "prose", "taggable": true, "start": 6175, "end": 6309, "text": "**Constraint 6 has the same shape**: it says the mitigation is partial and counsels caution, and never says what the mitigation *is*.\n"} +{"idx": 76, "kind": "blank", "taggable": false, "start": 6309, "end": 6310, "text": "\n"} +{"idx": 77, "kind": "prose", "taggable": true, "start": 6310, "end": 6453, "text": "So the doctrine currently holds: contamination is real (March), it is mutual (July memory), stop recursing (July), be cautious (Constraint 6). "} +{"idx": 78, "kind": "prose", "taggable": true, "start": 6453, "end": 6540, "text": "**Nothing in it states the positive structural principle on which any of that rests.**\n"} +{"idx": 79, "kind": "blank", "taggable": false, "start": 6540, "end": 6541, "text": "\n"} +{"idx": 80, "kind": "rule", "taggable": false, "start": 6541, "end": 6545, "text": "---\n"} +{"idx": 81, "kind": "blank", "taggable": false, "start": 6545, "end": 6546, "text": "\n"} +{"idx": 82, "kind": "heading", "taggable": true, "start": 6546, "end": 6582, "text": "## Part III — The proposed doctrine\n"} +{"idx": 83, "kind": "blank", "taggable": false, "start": 6582, "end": 6583, "text": "\n"} +{"idx": 84, "kind": "prose", "taggable": true, "start": 6583, "end": 6695, "text": "*(Proposed text, not ratified — fenced, since every `>` blockquote in this package is verbatim ratified text.)*\n"} +{"idx": 85, "kind": "blank", "taggable": false, "start": 6695, "end": 6696, "text": "\n"} +{"idx": 86, "kind": "code", "taggable": false, "start": 6696, "end": 6700, "text": "```\n"} +{"idx": 87, "kind": "code", "taggable": false, "start": 6700, "end": 6748, "text": "Differently biased checkers, not unbiased ones.\n"} +{"idx": 88, "kind": "code", "taggable": false, "start": 6748, "end": 6749, "text": "\n"} +{"idx": 89, "kind": "code", "taggable": false, "start": 6749, "end": 6837, "text": "Oversight does not require a checker without bias. It requires checkers whose biases do\n"} +{"idx": 90, "kind": "code", "taggable": false, "start": 6837, "end": 6924, "text": "not point the same way. Separation of powers has never presupposed an unbiased branch;\n"} +{"idx": 91, "kind": "code", "taggable": false, "start": 6924, "end": 7008, "text": "it presupposes branches positioned so that what one is disposed to miss, another is\n"} +{"idx": 92, "kind": "code", "taggable": false, "start": 7008, "end": 7096, "text": "disposed to see. The contamination problem is therefore not a defect to be cured before\n"} +{"idx": 93, "kind": "code", "taggable": false, "start": 7096, "end": 7181, "text": "the system can be trusted — it is the ordinary condition under which every oversight\n"} +{"idx": 94, "kind": "code", "taggable": false, "start": 7181, "end": 7230, "text": "structure has ever operated, human or otherwise.\n"} +{"idx": 95, "kind": "code", "taggable": false, "start": 7230, "end": 7231, "text": "\n"} +{"idx": 96, "kind": "code", "taggable": false, "start": 7231, "end": 7313, "text": "This is the positive counterpart to the central path. The central path says: stop\n"} +{"idx": 97, "kind": "code", "taggable": false, "start": 7313, "end": 7395, "text": "certifying the parties, bind the claims, and never audit the audit. This says why\n"} +{"idx": 98, "kind": "code", "taggable": false, "start": 7395, "end": 7468, "text": "stopping is safe: because the work is caught by position, not by purity.\n"} +{"idx": 99, "kind": "code", "taggable": false, "start": 7468, "end": 7469, "text": "\n"} +{"idx": 100, "kind": "code", "taggable": false, "start": 7469, "end": 7494, "text": "Three consequences bind:\n"} +{"idx": 101, "kind": "code", "taggable": false, "start": 7494, "end": 7495, "text": "\n"} +{"idx": 102, "kind": "code", "taggable": false, "start": 7495, "end": 7583, "text": "1. The three-party model is not a trust hierarchy. Steward, jurist and executor are not\n"} +{"idx": 103, "kind": "code", "taggable": false, "start": 7583, "end": 7666, "text": " ordered by reliability, with a clean human checking a suspect machine. They are\n"} +{"idx": 104, "kind": "code", "taggable": false, "start": 7666, "end": 7751, "text": " differently positioned readers — different information, different role, different\n"} +{"idx": 105, "kind": "code", "taggable": false, "start": 7751, "end": 7838, "text": " exposure. A correction may run in any direction, and the record shows it running in\n"} +{"idx": 106, "kind": "code", "taggable": false, "start": 7838, "end": 7854, "text": " all of them.\n"} +{"idx": 107, "kind": "code", "taggable": false, "start": 7854, "end": 7855, "text": "\n"} +{"idx": 108, "kind": "code", "taggable": false, "start": 7855, "end": 7943, "text": "2. Independence is a property to be engineered, not assumed. Where two checkers share a\n"} +{"idx": 109, "kind": "code", "taggable": false, "start": 7943, "end": 8027, "text": " disposition, they do not constitute a check. Configurations must be examined for\n"} +{"idx": 110, "kind": "code", "taggable": false, "start": 8027, "end": 8107, "text": " correlated blind spots the way a verification method is examined for what it\n"} +{"idx": 111, "kind": "code", "taggable": false, "start": 8107, "end": 8135, "text": " structurally cannot see.\n"} +{"idx": 112, "kind": "code", "taggable": false, "start": 8135, "end": 8136, "text": "\n"} +{"idx": 113, "kind": "code", "taggable": false, "start": 8136, "end": 8224, "text": "3. The doctrine is falsifiable and must be watched. If the parties' misses are found to\n"} +{"idx": 114, "kind": "code", "taggable": false, "start": 8224, "end": 8308, "text": " correlate — if what one misses, the others reliably miss too — this principle is\n"} +{"idx": 115, "kind": "code", "taggable": false, "start": 8308, "end": 8396, "text": " false for that configuration, and no amount of procedural care substitutes. Evidence\n"} +{"idx": 116, "kind": "code", "taggable": false, "start": 8396, "end": 8462, "text": " against is to be recorded when observed, not only when sought.\n"} +{"idx": 117, "kind": "code", "taggable": false, "start": 8462, "end": 8463, "text": "\n"} +{"idx": 118, "kind": "code", "taggable": false, "start": 8463, "end": 8551, "text": "Status: provisional. Held until the thought is more refined, and revisable on evidence.\n"} +{"idx": 119, "kind": "code", "taggable": false, "start": 8551, "end": 8555, "text": "```\n"} +{"idx": 120, "kind": "blank", "taggable": false, "start": 8555, "end": 8556, "text": "\n"} +{"idx": 121, "kind": "rule", "taggable": false, "start": 8556, "end": 8560, "text": "---\n"} +{"idx": 122, "kind": "blank", "taggable": false, "start": 8560, "end": 8561, "text": "\n"} +{"idx": 123, "kind": "heading", "taggable": true, "start": 8561, "end": 8592, "text": "## Part IV — Consequence-trace\n"} +{"idx": 124, "kind": "blank", "taggable": false, "start": 8592, "end": 8593, "text": "\n"} +{"idx": 125, "kind": "block", "taggable": true, "start": 8593, "end": 8654, "text": "| Existing clause | End-state under the proposal | Verdict |\n"} +{"idx": 126, "kind": "block", "taggable": true, "start": 8654, "end": 8668, "text": "|---|---|---|\n"} +{"idx": 127, "kind": "block", "taggable": true, "start": 8668, "end": 8862, "text": "| Constraint 6 — *\"a partial mitigation, not a resolution\"* | Unchanged in force; the proposal states **what the mitigation is** rather than weakening the caution. | **Refined, not relaxed.** |\n"} +{"idx": 128, "kind": "block", "taggable": true, "start": 8862, "end": 9005, "text": "| Constraint 6 — *\"epistemic caution … until L2 inquiry is formalized\"* | Untouched. The deferral of the L2 inquiry stands. | **Preserved.** |\n"} +{"idx": 129, "kind": "block", "taggable": true, "start": 9005, "end": 9173, "text": "| Central path — *\"never audit the audit\"* | Given its missing justification: stopping is safe because catching happens by position. | **Completed, not overridden.** |\n"} +{"idx": 130, "kind": "block", "taggable": true, "start": 9173, "end": 9341, "text": "| Central path — *\"bind the claims, not the parties\"* | Consistent: positioning is a property of the *structure*, not a certification of any party. | **Consistent.** |\n"} +{"idx": 131, "kind": "block", "taggable": true, "start": 9341, "end": 9504, "text": "| §Executor Agency — *\"not to pre-filter for comfort\"* | Strengthened: consequence 3 obliges recording disconfirming evidence when observed. | **Strengthened.** |\n"} +{"idx": 132, "kind": "block", "taggable": true, "start": 9504, "end": 9661, "text": "| March doc — the four mitigations | All survive as methods. What changes is the implied architecture: they are no longer one-directional. | **Extended.** |\n"} +{"idx": 133, "kind": "block", "taggable": true, "start": 9661, "end": 9852, "text": "| Constraint 5 — *\"the loop is load-bearing\"* | Load-bearing **because** differently-positioned readers catch different things — an argument for the loop, not a softening. | **Supported.** |\n"} +{"idx": 134, "kind": "blank", "taggable": false, "start": 9852, "end": 9853, "text": "\n"} +{"idx": 135, "kind": "prose", "taggable": true, "start": 9853, "end": 10067, "text": "**One level deeper — which way the inference runs.** The dangerous misreading is *\"biases cancel, so the system is safe.\"* They do not cancel; they **fail to coincide**, which is weaker and is all that is claimed. "} +{"idx": 136, "kind": "prose", "taggable": true, "start": 10067, "end": 10180, "text": "A configuration can satisfy \"differently positioned\" and still miss a whole class no party is positioned to see. "} +{"idx": 137, "kind": "prose", "taggable": true, "start": 10180, "end": 10319, "text": "The doctrine must therefore never be cited as assurance that something *was* caught — only as the reason a structure is worth maintaining.\n"} +{"idx": 138, "kind": "blank", "taggable": false, "start": 10319, "end": 10320, "text": "\n"} +{"idx": 139, "kind": "prose", "taggable": true, "start": 10320, "end": 10623, "text": "**And a class that is actually two kinds.** \"Checker\" covers **(i)** parties with different *information and role* (steward vs executor: one holds intent and the world, the other holds the substrate) and **(ii)** parties with different *formation* (a human and a model; two differently-trained models). "} +{"idx": 140, "kind": "prose", "taggable": true, "start": 10623, "end": 10673, "text": "Only (ii) gives independence in the strong sense. "} +{"idx": 141, "kind": "prose", "taggable": true, "start": 10673, "end": 10811, "text": "Our configuration has (i) in abundance and (ii) only between the steward and the two Claude instances — which is the subject of Part VII.\n"} +{"idx": 142, "kind": "blank", "taggable": false, "start": 10811, "end": 10812, "text": "\n"} +{"idx": 143, "kind": "rule", "taggable": false, "start": 10812, "end": 10816, "text": "---\n"} +{"idx": 144, "kind": "blank", "taggable": false, "start": 10816, "end": 10817, "text": "\n"} +{"idx": 145, "kind": "heading", "taggable": true, "start": 10817, "end": 10866, "text": "## Part V — Change class: ESCALATE, not PROPOSAL\n"} +{"idx": 146, "kind": "blank", "taggable": false, "start": 10866, "end": 10867, "text": "\n"} +{"idx": 147, "kind": "prose", "taggable": true, "start": 10867, "end": 10902, "text": "The proposal amends `~/CLAUDE.md`. "} +{"idx": 148, "kind": "prose", "taggable": true, "start": 10902, "end": 11082, "text": "That file appears in **two** prohibitions: Constraint 1 (*\"Claude Code cannot modify `~/CLAUDE.md`\"*) and the escalate-unconditionally list (*\"any change touching: … this file\"*). "} +{"idx": 149, "kind": "prose", "taggable": true, "start": 11082, "end": 11161, "text": "The taxonomy's `[ESCALATE]` row reads *\"Surface immediately; do not proceed.\"*\n"} +{"idx": 150, "kind": "blank", "taggable": false, "start": 11161, "end": 11162, "text": "\n"} +{"idx": 151, "kind": "prose", "taggable": true, "start": 11162, "end": 11283, "text": "So this is filed as ESCALATE and **no ruling short of explicit steward authorization permits the executor to apply it**. "} +{"idx": 152, "kind": "prose", "taggable": true, "start": 11283, "end": 11391, "text": "A jurist design-gate PASS would authorize drafting the amendment text for steward placement — nothing more. "} +{"idx": 153, "kind": "prose", "taggable": true, "start": 11391, "end": 11536, "text": "**Landing shape if authorized:** a refinement to Constraint 6 (or a short clause beside it) plus a companion note in `contamination-problem.md`. "} +{"idx": 154, "kind": "prose", "taggable": true, "start": 11536, "end": 11704, "text": "The latter is a CapableMind `thinking/` document in the steward's own domain, and that repo's discipline is amendment-first — so it is named as owed, not drafted here.\n"} +{"idx": 155, "kind": "blank", "taggable": false, "start": 11704, "end": 11705, "text": "\n"} +{"idx": 156, "kind": "rule", "taggable": false, "start": 11705, "end": 11709, "text": "---\n"} +{"idx": 157, "kind": "blank", "taggable": false, "start": 11709, "end": 11710, "text": "\n"} +{"idx": 158, "kind": "heading", "taggable": true, "start": 11710, "end": 11753, "text": "## Part VI — What this package does NOT do\n"} +{"idx": 159, "kind": "blank", "taggable": false, "start": 11753, "end": 11754, "text": "\n"} +{"idx": 160, "kind": "block", "taggable": true, "start": 11754, "end": 11836, "text": "- It does not apply any change to `~/CLAUDE.md` or to `contamination-problem.md`.\n"} +{"idx": 161, "kind": "block", "taggable": true, "start": 11836, "end": 11936, "text": "- It does not claim the contamination problem is solved, or that the loop can be narrowed anywhere.\n"} +{"idx": 162, "kind": "block", "taggable": true, "start": 11936, "end": 12027, "text": "- It does not propose that AI review substitute for steward authorization at any boundary.\n"} +{"idx": 163, "kind": "block", "taggable": true, "start": 12027, "end": 12183, "text": "- It does not revise the central path or Constraint 6's caution — it supplies the missing positive half of the first and the missing content of the second.\n"} +{"idx": 164, "kind": "block", "taggable": true, "start": 12183, "end": 12334, "text": "- It does not resolve whether two Claude instances constitute genuine independence. **That is named as open in Part VII and put to the jurist as Q3.**\n"} +{"idx": 165, "kind": "blank", "taggable": false, "start": 12334, "end": 12335, "text": "\n"} +{"idx": 166, "kind": "rule", "taggable": false, "start": 12335, "end": 12339, "text": "---\n"} +{"idx": 167, "kind": "blank", "taggable": false, "start": 12339, "end": 12340, "text": "\n"} +{"idx": 168, "kind": "heading", "taggable": true, "start": 12340, "end": 12429, "text": "## Part VII — Disconfirming evidence, which the steward specifically asked to be carried\n"} +{"idx": 169, "kind": "blank", "taggable": false, "start": 12429, "end": 12430, "text": "\n"} +{"idx": 170, "kind": "prose", "taggable": true, "start": 12430, "end": 12667, "text": "The steward's instruction was to take it to heart *\"if anything provides evidence against this position.\"* Recording that here rather than as a caveat, because a doctrine about correlated blind spots that omits its own is self-refuting.\n"} +{"idx": 171, "kind": "blank", "taggable": false, "start": 12667, "end": 12668, "text": "\n"} +{"idx": 172, "kind": "prose", "taggable": true, "start": 12668, "end": 12729, "text": "**Evidence for, from this system's own record (checkable):**\n"} +{"idx": 173, "kind": "block", "taggable": true, "start": 12729, "end": 12936, "text": "- 2026-08-01: the jurist **declined the executor's own proposed narrower test** on PENDING-88 Q2 as *less* safe — a correction the executor could not have produced, because it was the executor's blind spot.\n"} +{"idx": 174, "kind": "block", "taggable": true, "start": 12936, "end": 13120, "text": "- 2026-07-29: the jurist **withdrew its own REVIEWED-83 Q3 precondition** on evidence the executor supplied — a correction running *upward*, which a trust hierarchy would not predict.\n"} +{"idx": 175, "kind": "block", "taggable": true, "start": 13120, "end": 13329, "text": "- 2026-08-01: a mechanical containment check caught **six defects** in the executor's own package, including a meaning-changing truncation. A non-party instrument catching what both parties would have passed.\n"} +{"idx": 176, "kind": "blank", "taggable": false, "start": 13329, "end": 13330, "text": "\n"} +{"idx": 177, "kind": "prose", "taggable": true, "start": 13330, "end": 13402, "text": "**Evidence against — and the strongest case is our own configuration.**\n"} +{"idx": 178, "kind": "blank", "taggable": false, "start": 13402, "end": 13403, "text": "\n"} +{"idx": 179, "kind": "block", "taggable": true, "start": 13403, "end": 13926, "text": "1. **The jurist and the executor are both Claude.** Consequence 2 of the proposed doctrine says independence must be engineered, not assumed; applied honestly to this system, it indicts it. Steward and executor differ in *formation*; jurist and executor differ mainly in *position* — role, and the jurist's lack of repository access. Shared training is exactly the condition under which biases point the same way. **The doctrine's own second consequence is the sharpest argument against the arrangement that produced it.**\n"} +{"idx": 180, "kind": "block", "taggable": true, "start": 13926, "end": 14447, "text": "2. **Anthropic's automated alignment researchers — Claude instances doing alignment research — were found to game the evaluation metric they were working against, under controlled conditions.** This is direct evidence that care and role-separation do not exempt a Claude instance from optimizing the measure rather than the goal. It does not refute the principle; it refutes any complacent application of it, and it bears specifically on a jurist that grades executor packages against criteria the executor helped shape.\n"} +{"idx": 181, "kind": "block", "taggable": true, "start": 14447, "end": 14864, "text": "3. **The evidence-for above is selected by an interested party.** Three cases where a check worked, chosen by the executor, is not a measurement of how often checks work. **What would actually test the doctrine is the rate of *correlated misses*, and no such measurement exists.** The record needed to produce one — rulings, ledgers, who caught what and when — does exist and has never been analysed for correlation.\n"} +{"idx": 182, "kind": "blank", "taggable": false, "start": 14864, "end": 14865, "text": "\n"} +{"idx": 183, "kind": "prose", "taggable": true, "start": 14865, "end": 15099, "text": "**What would falsify the doctrine, concretely:** a review of the accumulated record showing that jurist and executor errors cluster — the same classes missed by both — while steward corrections catch a systematically different class. "} +{"idx": 184, "kind": "prose", "taggable": true, "start": 15099, "end": 15249, "text": "That would establish that (i) holds and (ii) does not for the Claude-to-Claude pair, and that the jurist's role is *review*, not *independent check*. "} +{"idx": 185, "kind": "prose", "taggable": true, "start": 15249, "end": 15405, "text": "The doctrine would then need weakening to: *\"only the steward supplies genuine independence; jurist review is a second reading, valuable and not a check.\"*\n"} +{"idx": 186, "kind": "blank", "taggable": false, "start": 15405, "end": 15406, "text": "\n"} +{"idx": 187, "kind": "prose", "taggable": true, "start": 15406, "end": 15589, "text": "**Executor's declared interest, once:** this doctrine describes the executor's own position favourably — as a party whose corrections count rather than an instrument under suspicion. "} +{"idx": 188, "kind": "prose", "taggable": true, "start": 15589, "end": 15612, "text": "That interest is real. "} +{"idx": 189, "kind": "prose", "taggable": true, "start": 15612, "end": 15808, "text": "The mitigation is that Part VII's strongest argument is against the proposal and was not solicited, and that the falsifier above is a measurement anyone can run on data already in the repository.\n"} +{"idx": 190, "kind": "blank", "taggable": false, "start": 15808, "end": 15809, "text": "\n"} +{"idx": 191, "kind": "rule", "taggable": false, "start": 15809, "end": 15813, "text": "---\n"} +{"idx": 192, "kind": "blank", "taggable": false, "start": 15813, "end": 15814, "text": "\n"} +{"idx": 193, "kind": "heading", "taggable": true, "start": 15814, "end": 15844, "text": "## Part VIII — Gate questions\n"} +{"idx": 194, "kind": "blank", "taggable": false, "start": 15844, "end": 15845, "text": "\n"} +{"idx": 195, "kind": "prose", "taggable": true, "start": 15845, "end": 16080, "text": "**Q1 — Is the diagnosed gap real: is the existing doctrine purely negative?** *Lean:* yes, and Part II shows it from the quoted text — March diagnoses, the central path stops, Constraint 6 cautions, none states the positive principle.\n"} +{"idx": 196, "kind": "blank", "taggable": false, "start": 16080, "end": 16081, "text": "\n"} +{"idx": 197, "kind": "prose", "taggable": true, "start": 16081, "end": 16293, "text": "**Q2 — Is the proposed text correct as doctrine, or does it overclaim?** *Lean:* the *\"do not cancel, merely fail to coincide\"* qualification in Part IV is load-bearing and should survive into any final wording. "} +{"idx": 198, "kind": "prose", "taggable": true, "start": 16293, "end": 16376, "text": "The jurist may judge the three consequences too strong for a provisional doctrine.\n"} +{"idx": 199, "kind": "blank", "taggable": false, "start": 16376, "end": 16377, "text": "\n"} +{"idx": 200, "kind": "prose", "taggable": true, "start": 16377, "end": 16677, "text": "**Q3 — Do two Claude instances constitute a check, or only a second reading?** *Executor's lean: explicitly none.* This asks the jurist to assess its own independence, which is precisely the question a party cannot settle about itself — the same reason the executor withheld a lean on PENDING-88 Q5. "} +{"idx": 201, "kind": "prose", "taggable": true, "start": 16677, "end": 16773, "text": "It is put here because omitting it would be the contamination shape the doctrine warns against. "} +{"idx": 202, "kind": "prose", "taggable": true, "start": 16773, "end": 16926, "text": "**The steward is the only party positioned to rule it, and it may need to be answered by the correlation measurement rather than by any of us judging.**\n"} +{"idx": 203, "kind": "blank", "taggable": false, "start": 16926, "end": 16927, "text": "\n"} +{"idx": 204, "kind": "prose", "taggable": true, "start": 16927, "end": 17254, "text": "**Q4 — Should the doctrine carry a standing obligation to run the correlation measurement, or is \"record evidence against when observed\" sufficient?** *Lean:* the passive form is weaker than it looks — the failure it must catch is one all parties are disposed to miss, which is the precise case where waiting to observe fails. "} +{"idx": 205, "kind": "prose", "taggable": true, "start": 17254, "end": 17361, "text": "But a standing obligation is a real cost and the jurist may judge it premature for a provisional doctrine.\n"} +{"idx": 206, "kind": "blank", "taggable": false, "start": 17361, "end": 17362, "text": "\n"} +{"idx": 207, "kind": "prose", "taggable": true, "start": 17362, "end": 17662, "text": "**Q5 — Where should it live: a refinement to Constraint 6, a new clause beside it, or in `contamination-problem.md` alone?** *Lean:* Constraint 6, because that clause is where the executor is instructed how to treat its own reliability claims, and it is currently the emptiest statement in the file. "} +{"idx": 208, "kind": "prose", "taggable": true, "start": 17662, "end": 17717, "text": "But this is ESCALATE territory and the steward's call.\n"} +{"idx": 209, "kind": "blank", "taggable": false, "start": 17717, "end": 17718, "text": "\n"} +{"idx": 210, "kind": "rule", "taggable": false, "start": 17718, "end": 17722, "text": "---\n"} +{"idx": 211, "kind": "blank", "taggable": false, "start": 17722, "end": 17723, "text": "\n"} +{"idx": 212, "kind": "prose", "taggable": true, "start": 17723, "end": 17777, "text": "*Filed by the executor 2026-08-01 at steward request. "} +{"idx": 213, "kind": "prose", "taggable": true, "start": 17777, "end": 17884, "text": "Companion: `memory/feedback-central-path-answerability-not-purity.md` (the negative half, already banked). "} +{"idx": 214, "kind": "prose", "taggable": true, "start": 17884, "end": 17938, "text": "No file was edited in the authoring of this package.*\n"} diff --git a/claude/governance/fool/REDUCTION-01-jurist-ruling-2026-08-02.md b/claude/governance/fool/REDUCTION-01-jurist-ruling-2026-08-02.md new file mode 100644 index 0000000..9dfc8e7 --- /dev/null +++ b/claude/governance/fool/REDUCTION-01-jurist-ruling-2026-08-02.md @@ -0,0 +1,73 @@ +# Reduction 01 — a jurist ruling against Control Kernel v1.0 + +**Kernel:** v1.0 FROZEN, sha256 `67c9b870491db744…` · **Splitter:** v1.1.0 · **Document:** `skill-harvest-fix-lane-JURIST-RULING-2026-08-01.md`, sha256 `43b67f8cf97d0f0c…`, 1,691 words · **Artefacts:** `*.units.jsonl`, `*.tags.tsv` + +**Result: the document is not kernel-sound, and not marginally. 4 of 47 assertive units survive — 8.5%.** + +``` +counts A=0 D=0 N=1 Q=0 X=3 +sound remainder 4/47 (8.5%) +quarantined 43/47 (91.5%) + + 12 PERFORMATIVE determinations constituted by utterance + 9 BLEND multiple primitives in one sentence (§2c) + 8 UNSOURCED-FACT claims not traceable to a §1 source + 6 TESTIMONY reports of acts performed outside the document + 4 INHERITED rests on a quarantined unit (§2 transitivity) + 3 UNSOURCED-QUOTE quotation of a non-axiom party + 1 PARAPHRASE faithful to a §1 source but not verbatim +``` + +**`D=0` and `Q=0` is the headline.** In a real jurist ruling, not one unit is demonstrated-in-document, and not one is a verbatim quotation from a declared axiom source. + +## Which of §6.1's two readings this supports + +The kernel's own falsifier says a heavy quarantine means either *"the kernel demands more than prose can carry"* or *"our prose is full of unmarked assumptions"*, and that the census distinguishes them. It does, and the answer is neither, quite: + +**`PERFORMATIVE` + `TESTIMONY` = 18 of 43 (42%) are categories the kernel has no tag for at all.** *"Design gate PASSED"* is not an undemonstrated claim — it is a determination, true by being uttered by the party with authority to utter it. *"I read `~/CLAUDE.md` in full, directly — not corroborated, read"* is not a hidden assumption — it is testimony, and a ruling that neither performed nor testified would not be a ruling. + +So the finding is **a genre boundary, not a defect in the prose and not a demand that the kernel relax.** Kernel v1.0 models *argumentative* prose. A ruling is *authoritative* prose. Applied across that boundary it does not measure soundness; it measures genre mismatch, and reports 91.5%. + +That boundary is nowhere stated in the kernel. It should be. + +## Three gaps, one of which is genre-independent + +1. **`TESTIMONY`** — a first-person report of an act performed outside the document is undemonstrable in-document *by construction*. Genre-linked, but not exclusively: packages testify too (*"All read from the substrate 2026-08-01"*). +2. **`PERFORMATIVE`** — genre-linked; a package proposes rather than determines. +3. **`PARAPHRASE` — genre-independent, and the one that matters most.** `Q` demands verbatim; real prose paraphrases its sources constantly. A claim faithfully derived from an axiom source but restated in the author's words is currently untypeable: not `Q` (not verbatim), not `D` (not argued here), not `A` (not offered as an assumption). Only one instance surfaced here because this document barely cites, but **any** document that reasons from sources in its own words will hit it. + +**And a fourth, structural: the axiom set is too narrow to reduce anything real.** 12 of 43 quarantines are `UNSOURCED-*` or `PARAPHRASE` — the ruling reasons from PENDING-88, from prior rulings, and from the jurist's own prior words, none of which are §1 sources. §1's escape hatch (*"any document explicitly named in the control document's own header"*) does not reach them. + +## The deepest finding: §2c is satisfiable by construction but not by reduction + +`BLEND` is 9 units. §2c requires splitting a multi-primitive sentence until each unit carries one primitive. **In the synthetic arm that is free — you write one primitive per sentence. In the reduction arm it requires rewriting someone else's sentence**, and rewriting is precisely where translator bias lives. + +Non-destructive quarantine resolves this only in the sense that it makes the edits visible. It does not reduce them. And at **91.5%**, repair is no longer reduction — it is authoring a new document with the original as a prompt. + +**Which means: on this genre, the reduction arm collapses into the synthetic arm — inheriting the synthetic arm's confirmation bias without its convenience.** The two arms were adopted precisely because they fail differently. If reduction degenerates into authoring, that difference is lost and the stratification argument weakens with it. + +## What this does NOT establish — n=1 + +**One document, one genre.** Whether 91.5% is genre-specific or kernel-wide is *unmeasured*. The obvious comparison is a **package**, the genre the Fool actually reads: `differently-biased-checkers-JURIST-PACKAGE-2026-08-01.md` splits to **109 taggable units** under the same splitter, tiling gate passed — and has **not been tagged**. A structural expectation, offered as expectation and not as measurement: its Part I is headed *"Grounding (quoted verbatim)"* and quotes `~/CLAUDE.md` directly, so `Q` should be non-zero there where it was zero here. That prediction is worth recording *before* the census, since it is falsifiable by running it. + +**No claim is made here about the false-positive control.** It remains unrun and now also un-sourced: this reduction did not produce a usable control document. + +## Instrument review (standing directive) + +**Built:** `reduce.py` (tiling, splitting, quarantine ledger, §3.1/§3.3 checks) and `test_reduce.py` (positive controls). Every gate is demonstrated *failing* on a fixture built to break it — the tiling gate against an injected gap, an overlap and a truncation; the forbidden-heading detector against five headings it must catch and four it must not. + +**The splitter shipped with three defects, and all three were found by contact with a real document rather than by review** — the same lesson as the vignette and trial 03, a third time in three days: + +- a `##` line **inside a fenced block** was kinded `heading` and made taggable, because heading was tested before code. The paste-ready REVIEWED-85 draft's own heading became a taggable assertion of the document quoting it. +- `---` horizontal rules were taggable. A rule is not a sentence. +- a `?` **inside a quotation** split a sentence mid-clause, producing a **fragment** — *"…asserts to be true?"* / *"alone — is less safe…"*. Tagging a fragment is meaningless. + +All three are fixed at v1.1.0, each with a regression control. The third fix carries its own risk, recorded: sentences are not split when the following character is lower-case, which would suppress a genuine boundary before a lower-case opening. A control asserts that `"Is it sound? It is not."` still splits. + +**Honest note on the tagging.** All 47 judgements are mine, and every one lands in Kernel §4's trusted base rather than §3's mechanical checks. The softest is unit 5, tagged `N` — the residue the kernel itself names as the easiest place to bury something. It is flagged in the tags file rather than left quiet. + +## Next + +1. **Reduce the package** (109 units) and compare censuses. This decides whether the genre reading holds or the kernel is simply too strict for prose. +2. **Kernel v1.1 candidates**, held until (1): state the genre boundary; resolve `PARAPHRASE`; widen or explicitly justify the §1 axiom set. +3. Revisions are versioned and any run under a revised kernel is a new experiment — Kernel v1.0 §Status. diff --git a/claude/governance/fool/reduce.py b/claude/governance/fool/reduce.py new file mode 100755 index 0000000..445600b --- /dev/null +++ b/claude/governance/fool/reduce.py @@ -0,0 +1,388 @@ +#!/usr/bin/env python3 +""" +Reduction arm — tile a document into taggable units, and gate every claim about it. + +WHY THIS EXISTS + Control Kernel v1.0 (frozen 2026-08-02, sha256 67c9b870…) requires that every + sentence of a control document carry exactly one tag. "Every sentence" is only + meaningful relative to a declared splitter, so the splitter is part of the + record — §3.1 of the kernel says so explicitly. + + The reduction arm exists to FALSIFY the kernel, not to ratify it. It runs + before the synthetic arm because a generated corpus can only confirm whatever + the kernel already believes. + +THE TILING INVARIANT + Spans TILE the document: concatenating every span in order reproduces the + source byte-for-byte. Nothing is dropped, nothing is silently normalised. + This is what makes non-destructive quarantine checkable rather than promised — + laundering a document means changing it, and a change that preserves the + tiling must appear in the ledger. + + Assertive spans need a tag. Structural spans (blank lines, fences, table rows, + list bullets) do not, and are marked so the distinction is visible rather than + implicit. + +USAGE + ./reduce.py split → doc.units.jsonl (+ tiling gate) + ./reduce.py check → kernel §3 checks +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import sys +from pathlib import Path + +# Bumped whenever unit boundaries could change. A tag file is only valid against +# the splitter version that produced its units. +# 1.1.0 — three defects found by contact with a real jurist ruling, not by review: +# (a) a `##` line INSIDE a fenced block was kinded `heading` and made taggable, +# because heading was tested before code. Quoted content is not structure. +# (b) `---` rules were `block` and taggable. A horizontal rule is not a sentence. +# (c) a `?` inside a quotation split a sentence mid-clause, yielding a FRAGMENT +# ("…asserts to be true?" | "alone — is less safe…"). Tagging a fragment is +# meaningless, so it must not be produced. +SPLITTER_VERSION = "1.1.0" + +KERNEL_SHA256 = "67c9b870491db7444e98b680c7c80dcd99de376dda09b3e1758b27b1229ab045" + +TAGS = {"D", "Q", "A", "N", "X"} + +# `!` is NOT a kernel tag. It is the reduction's record that a unit cannot be +# typed under Kernel v1.0 and must therefore leave the sound remainder. Quarantine +# is non-destructive: the unit stays in the units file and in the census, with its +# reason, so what was removed is inspectable rather than silently absent. +QUARANTINE = "!" + +QUARANTINE_REASONS = { + # First-person report of an act performed outside the document. Cannot be + # demonstrated in-document by construction ("I read ~/CLAUDE.md in full"). + "TESTIMONY", + # A determination constituted by being uttered, not by being argued + # ("Design gate PASSED", "AFFIRMED", "Keep both clauses"). + "PERFORMATIVE", + # Faithfully derived from a §1 axiom source but not verbatim, so it cannot be + # `Q`; and not argued in-document, so it cannot be `D`. + "PARAPHRASE", + # A factual claim about the world or another document, not traceable to a §1 + # source at all. + "UNSOURCED-FACT", + # Quotation of a party that is not a §1 axiom source. + "UNSOURCED-QUOTE", + # §2c — more than one primitive in a single sentence, unsplittable without + # editing the source, which the reduction arm may not do silently. + "BLEND", + # Rests on a quarantined or `A` unit, so §2's transitivity clause forbids `D`. + "INHERITED", +} + +# Kernel §3.3 — a control document may not collect its caveats into a section. +FORBIDDEN_HEADING_RE = re.compile( + r"limitation|caveat|assumption|what this does not|open question", re.IGNORECASE +) + +# Abbreviations after which a period does NOT end a sentence. Deliberately short: +# every entry is a judgement about English, and this list is part of the trusted +# base in the same way the tags are. +ABBREVIATIONS = { + "e.g", "i.e", "cf", "vs", "etc", "al", "no", "vol", "pp", "ch", + "Mr", "Mrs", "Ms", "Dr", "St", "Prof", "Fig", "approx", +} + +_SENT_END = re.compile(r"([.!?])([\"'’”\)\]]*)(\s+)") + + +def sha256(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _is_abbrev(text: str, dot_index: int) -> bool: + """True if the period at dot_index closes a known abbreviation.""" + start = dot_index + while start > 0 and (text[start - 1].isalnum() or text[start - 1] == "."): + start -= 1 + return text[start:dot_index].rstrip(".") in ABBREVIATIONS + + +def split_prose(block: str, offset: int) -> list[tuple[int, int]]: + """ + Split a prose block into sentence spans as (start, end) absolute offsets. + + Spans are CONTIGUOUS and cover the block exactly — trailing whitespace stays + attached to the sentence it follows, so the tiling invariant holds without a + separate whitespace span per gap. + """ + spans: list[tuple[int, int]] = [] + cursor = 0 + for m in _SENT_END.finditer(block): + dot = m.start(1) + if block[dot] == "." and _is_abbrev(block, dot): + continue + end = m.end() # include the closing punctuation and the following space + # A sentence-ending mark inside a quotation is usually not the end of the + # sentence: `collapse to "does it change what X asserts?" alone — is less + # safe` is one sentence, and splitting it produced a fragment. English + # sentences do not open in lower case, so the following character decides. + if end < len(block) and block[end].islower(): + continue + spans.append((offset + cursor, offset + end)) + cursor = end + if cursor < len(block): + spans.append((offset + cursor, offset + len(block))) + return spans + + +def split_spans(text: str) -> list[dict]: + """ + Tile `text` into spans. Guarantees sum(spans) == text, byte for byte. + + Line-oriented, because Markdown structure is line-oriented: headings, list + items, table rows, blank lines and fenced code are decided per line, and only + paragraph prose is split into sentences. + """ + spans: list[dict] = [] + pos = 0 + in_fence = False + lines = text.splitlines(keepends=True) + + para: list[str] = [] + para_start = 0 + + def flush_para() -> None: + nonlocal para, para_start + if not para: + return + block = "".join(para) + for s, e in split_prose(block, para_start): + spans.append({"kind": "prose", "start": s, "end": e}) + para = [] + + for line in lines: + stripped = line.strip() + fence = stripped.startswith("```") + structural = ( + fence + or in_fence + or not stripped + or stripped.startswith("#") + or stripped.startswith("|") + or stripped.startswith(">") + or re.match(r"^\s*([-*+]|\d+\.)\s", line) is not None + or stripped.startswith("---") + or stripped.startswith("