From da90929c83cc36a2cabc7a9b226b1c86ec8a86ba Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:47:43 +0100 Subject: [PATCH 01/64] =?UTF-8?q?chore:=20capture=20iss-2609261647358395?= =?UTF-8?q?=20=E2=80=94=20a=20JSON=20escape=20hides=20tokens=20and=20home?= =?UTF-8?q?=20paths?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The transcript store scans raw JSONL, and a token or home path written beside a JSON string escape is invisible to the raw-line pass. Captured before the fix, per the capture-first rule. Refs: iss-2609261647358395 Assisted-by: Claude:claude-opus-5-5 --- ...cape-beside-a-secret-or-a-home-path-hides-it.md | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md diff --git a/.abcd/work/issues/open/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md b/.abcd/work/issues/open/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md new file mode 100644 index 000000000..69689812a --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261647358395" +slug: "a-json-string-escape-beside-a-secret-or-a-home-path-hides-it" +severity: "major" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/percent.go" +--- + +A JSON string escape beside a secret or a home path hides it from the scanner's raw-line pass, and the transcript store scans raw JSONL. Every bundled token pattern anchors on a leading word boundary, so a token written straight after the \n, \t, \r, \b or \f escape (a GitHub PAT at the start of a line of pasted output, JSON-encoded) has the escape letter as a word byte before it and never matches: ScanText at BASE 211b8853 returns nothing for {"t":"tok\nghp_..."} while the same token after a space is token:github_pat. The identity matchers fail the same way: home_path_other finds no third-party home path that follows a \n or \t escape (the leading anchor reads the escape letter as a path byte) or that precedes a \n or \" escape (the trailing boundary set has no backslash), and a non-ASCII real_name written by an ASCII-only encoder (Python json's default \u00e9) never matches the configured name. The fix reads a line's JSON-decoded layers the way the percent-decode pre-pass reads its percent-decoded copy, mapping each hit back to its raw span. Detector: a token after a \n escape, a home path between \n and \" escapes and a \u-escaped real_name are each a finding on the transcript path. From c55ae5efda18f317f6a6f28823748b930b425698 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:52:35 +0100 Subject: [PATCH 02/64] fix(scanner): read a line's JSON-escape layers as decoded views MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The transcript store scans raw JSONL, and a JSON string escape changed what the detectors saw without changing what the text says. A token written after a \n or \t escape had the escape letter as a word byte before it, so no leading-\b token pattern matched; a home path beside a \n, \t or \" escape failed the home anchor's boundaries; a home written with the solidus escape (\/ or /) had no '/' on the line; and a non-ASCII name written as é matched no configured name. Each line carrying an escape is now decoded one JSON layer at a time (at most three, a layer that decodes nothing ends the walk) and every layer is scanned as a view of the line, the way the percent-decode pre-pass scans its decoded copy: each hit is mapped back to the raw bytes it came from, so Redact masks the live spelling on disk. Every layer is scanned, not only the last, because a literal backslash an earlier layer uncovers is read by the next as an escape. The view scan is shared with the percent pre-pass (viewFindings), and the new stage charges the scan meter, with four linearity fixtures. This changes the scan engine, not the canonical pattern set, so it reaches every ScanText and ScanBundle consumer: the history transcript store, the capture and memory redactors, scanner.CheckOutbound behind `abcd lint outbound`, and the launch bundler and lifeboat packer. The repolint privacy rule and the harness_leak lint rule run the patterns' regexps directly and are not reached. Refs: iss-2609261647358395, iss-2609251639263391 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/jsonescape.go | 187 ++++++++++++++++++ internal/adapter/scanner/jsonescape_test.go | 147 ++++++++++++++ internal/adapter/scanner/meter.go | 2 + internal/adapter/scanner/meter_test.go | 7 + internal/adapter/scanner/percent.go | 21 +- internal/adapter/scanner/scanner.go | 3 +- .../core/history/redact_jsonescape_test.go | 61 ++++++ 7 files changed, 424 insertions(+), 4 deletions(-) create mode 100644 internal/adapter/scanner/jsonescape.go create mode 100644 internal/adapter/scanner/jsonescape_test.go create mode 100644 internal/core/history/redact_jsonescape_test.go diff --git a/internal/adapter/scanner/jsonescape.go b/internal/adapter/scanner/jsonescape.go new file mode 100644 index 000000000..17a27814c --- /dev/null +++ b/internal/adapter/scanner/jsonescape.go @@ -0,0 +1,187 @@ +package scanner + +import ( + "strings" + "unicode/utf16" + "unicode/utf8" +) + +// jsonescape.go — the JSON-escape decoded views of a line +// (iss-2609261647358395, iss-2609251639263391). +// +// The transcript store scans raw JSONL, so every string on the lines it hands +// the scanner is JSON-escaped, and an escape changes what the detectors see +// without changing what the text says. Three shapes were invisible on the raw +// line: +// +// - a control escape before a value. Every bundled token pattern anchors on +// a leading \b, and in "\nghp_…" the byte before the token is the escape's +// 'n', a word byte, so the anchor never holds; the home-path anchor reads +// the same letter as a path byte and declines "\n/home/". +// - an escape after a value. The home-path trailing boundary has no +// backslash in its set, so "/home/\n" and "\"/home/\"" were +// declined as a name that goes on. +// - an escaped separator or letter inside a value. "\/home\/" and +// "\u002fhome\u002f" spell a home path with no '/' on the line, and +// an ASCII-only encoder (Python json's default) writes a non-ASCII name as +// "\u00e9", which no configured name matches. +// +// Rather than teach every detector every escape, each line with an escape is +// decoded one JSON layer at a time and each layer is scanned as a view of the +// line, exactly as the percent-decode pre-pass scans its decoded copy: every +// hit is mapped back to the raw bytes it came from, so Redact masks the live +// spelling on disk. Decode-equivalent spellings therefore match by +// construction, for every detector at once. +// +// EVERY layer is scanned, not only the last. A tool result that is itself +// JSON is escaped again by the transcript line that quotes it, so the second +// layer is where its strings read plainly; but a literal backslash the first +// layer uncovers (a Windows path, a regex) is read again by the next as an +// escape — "C:\Users\bob" decodes "\b" to a backspace — so the last layer alone +// can be wrong where an earlier one was right. The layers are bounded like the +// percent passes, and a layer that decodes nothing ends the walk. +// +// The percent spelling is the percent pre-pass's business (percent.go), and +// the two do not compose: a JSON escape never carries a '%' sequence that a +// percent decode would need unescaped first, and a percent-encoded JSON escape +// is not a spelling any encoder that feeds the scanner writes. + +// maxJSONDecodeLayers bounds the JSON-unescape walk: one layer for a +// transcript line, a second for JSON quoted inside it (a tool result), a third +// for slack. Each layer strictly shrinks the line, so the walk ends early on +// ordinary input. +const maxJSONDecodeLayers = 3 + +// jsonEscapeLayers returns the JSON-decoded views of s, outermost first, each +// with its position map back to s (the percentDecodeBounded shape: posMap[i] +// is the offset in s at which decoded byte i began, with a sentinel at +// len(decoded)). It returns nil when s carries no JSON escape. +func jsonEscapeLayers(s string) []decodedView { + if strings.IndexByte(s, '\\') < 0 { + return nil + } + var out []decodedView + cur := s + var m []int // offsets in cur -> offsets in s; nil means identity + for layer := 0; layer < maxJSONDecodeLayers; layer++ { + next, step, changed := jsonUnescapeOnce(cur) + if !changed { + break + } + composed := make([]int, len(next)+1) + for i := range composed { + if m == nil { + composed[i] = step[i] + } else { + composed[i] = m[step[i]] + } + } + out = append(out, decodedView{text: next, posMap: composed}) + cur, m = next, composed + } + return out +} + +// decodedView is one decoded spelling of a raw line and its position map. +type decodedView struct { + text string + posMap []int +} + +// jsonUnescapeOnce decodes one layer of JSON string escapes in s — the two-byte +// escapes \" \\ \/ \b \f \n \r \t and the \uXXXX escape, a UTF-16 surrogate +// pair combined into one rune — and copies everything else as it stands: a +// backslash before any other byte is not a JSON escape (a Windows path read +// at the wrong depth, a regex) and is kept, as is a lone surrogate, which no +// rune stands for. The map sends each decoded byte to the offset in s its +// escape began at, with a trailing sentinel == len(s). +func jsonUnescapeOnce(s string) (string, []int, bool) { + scanMeter.charge(stageJSONEscape, len(s)) + b := make([]byte, 0, len(s)) + pos := make([]int, 0, len(s)+1) + changed := false + emit := func(at int, bs ...byte) { + for _, c := range bs { + pos = append(pos, at) + b = append(b, c) + } + } + for i := 0; i < len(s); { + if s[i] != '\\' || i+1 >= len(s) { + emit(i, s[i]) + i++ + continue + } + if c, ok := jsonShortEscape(s[i+1]); ok { + emit(i, c) + i += 2 + changed = true + continue + } + if r, n := jsonUnicodeEscape(s[i:]); n > 0 { + var enc [utf8.UTFMax]byte + emit(i, enc[:utf8.EncodeRune(enc[:], r)]...) + i += n + changed = true + continue + } + emit(i, s[i]) + i++ + } + pos = append(pos, len(s)) + return string(b), pos, changed +} + +// jsonShortEscape returns the byte a two-byte JSON escape stands for. +func jsonShortEscape(c byte) (byte, bool) { + switch c { + case '"', '\\', '/': + return c, true + case 'b': + return '\b', true + case 'f': + return '\f', true + case 'n': + return '\n', true + case 'r': + return '\r', true + case 't': + return '\t', true + } + return 0, false +} + +// jsonUnicodeEscape decodes the \uXXXX escape at the start of s, or a +// surrogate pair written as two of them, returning the rune and the bytes it +// consumed; n == 0 when s does not start with one that stands for a rune. +func jsonUnicodeEscape(s string) (rune, int) { + u, ok := hex4(s) + if !ok { + return 0, 0 + } + r := rune(u) + if !utf16.IsSurrogate(r) { + return r, 6 + } + if lo, ok := hex4(s[6:]); ok { + if pair := utf16.DecodeRune(r, rune(lo)); pair != utf8.RuneError { + return pair, 12 + } + } + return 0, 0 +} + +// hex4 reads "\uXXXX" at the start of s. +func hex4(s string) (uint16, bool) { + if len(s) < 6 || s[0] != '\\' || s[1] != 'u' { + return 0, false + } + var v uint16 + for _, c := range []byte(s[2:6]) { + if !isHexDigit(c) { + return 0, false + } + v = v<<4 | uint16(hexNibble(c)) + } + return v, true +} diff --git a/internal/adapter/scanner/jsonescape_test.go b/internal/adapter/scanner/jsonescape_test.go new file mode 100644 index 000000000..767d3525b --- /dev/null +++ b/internal/adapter/scanner/jsonescape_test.go @@ -0,0 +1,147 @@ +package scanner + +import ( + "strings" + "testing" +) + +// The transcript store scans raw JSONL, so every line it hands the scanner is +// a JSON document and every string in it is JSON-escaped. A spelling that +// decodes to a secret or an identity is the same leak as the decoded text +// (iss-2609261647358395, iss-2609251639263391): these tests hand the scanner +// the escaped spelling and require the finding the decoded one gets, and a +// redaction that leaves none of the value's bytes behind. + +// jsonEscapeSpecimen builds the escaped line at run time, so no literal of a +// credential shape enters the history (network_test.go states the +// discipline). +func jsonEscapeToken() string { return "ghp_" + "0123456789abcdefABCDEF0123456789abcd" } + +// jsonEscapeIdentity is a synthetic caller: a POSIX home, a generic login in +// the local part of none of the lines, and a real name with a non-ASCII rune +// that an ASCII-only JSON encoder writes as a \u escape. +func jsonEscapeIdentity() Identity { + return Identity{ + HomePath: "/Users/" + "zqjsonme", + HomeUser: "zqjsonme", + GitUserName: "Zoë Quenby", + } +} + +func findingOf(fs []Finding, kind string) (Finding, bool) { + for _, f := range fs { + if f.Kind == kind { + return f, true + } + } + return Finding{}, false +} + +func TestJSONEscapeBesideASecretOrHomeIsStillAFinding(t *testing.T) { + tok := jsonEscapeToken() + other := "zqjsonother" + cases := []struct { + name string + line string + kind string + gone string // what the redacted line must no longer carry + keeps string // what the redacted line must still carry + }{ + {"token after a newline escape", `{"t":"pasted:\n` + tok + `\nend"}`, "token:github_pat", tok, "pasted:"}, + {"token after a tab escape", `{"t":"key\t` + tok + `"}`, "token:github_pat", tok, "key"}, + {"token after a doubly escaped newline", `{"t":"{\"out\":\"a\\n` + tok + `\"}"}`, "token:github_pat", tok, "out"}, + {"home after a newline escape", `{"t":"ls\n/home/` + other + `/src"}`, kindHomeOther, other, "src"}, + {"home after a tab escape", `{"t":"ls\t/home/` + other + `/src"}`, kindHomeOther, other, "src"}, + {"home before a newline escape", `{"t":"cd /home/` + other + `\nok"}`, kindHomeOther, other, "ok"}, + {"home inside escaped quotes", `{"t":"cd \"/home/` + other + `\" ok"}`, kindHomeOther, other, "ok"}, + {"real name written with a unicode escape", `{"t":"signed Zo\u00eb Quenby here"}`, kindRealName, `Zo\u00eb`, "here"}, + {"real name written with an upper-case unicode escape", `{"t":"signed Zo\u00EB Quenby here"}`, kindRealName, `Zo\u00EB`, "here"}, + } + id := jsonEscapeIdentity() + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + fs := ScanText(c.line, id, DefaultPatterns(), nil, "transcript") + if !hasKind(fs, c.kind) { + t.Fatalf("%s: no %s finding for %s; got %+v", c.name, c.kind, c.line, fs) + } + red, _ := Redact(c.line, fs) + if strings.Contains(red, c.gone) { + t.Errorf("%s: the value survived redaction:\n%s", c.name, red) + } + if !strings.Contains(red, c.keeps) { + t.Errorf("%s: redaction took the surrounding text too:\n%s", c.name, red) + } + }) + } +} + +// TestJSONSolidusEscapeReadsAsASeparator is iss-2609251639263391's detector: +// a home path written with escaped forward slashes — the JSON solidus escape +// PHP's json_encode and org.json write, and its \u002f spelling — raises the +// finding the unescaped path raises, for the caller's own home, another +// user's home, and a generic login standing in a home. +func TestJSONSolidusEscapeReadsAsASeparator(t *testing.T) { + id := jsonEscapeIdentity() + other := "zqjsonother" + cases := []struct { + name, line, kind, gone string + }{ + {"own home, solidus escape", `{"p":"\/Users\/zqjsonme\/Desktop\/a.txt"}`, kindHomeSelf, "zqjsonme"}, + {"own home, unicode solidus", `{"p":"\u002fUsers\u002fzqjsonme\u002fDesktop"}`, kindHomeSelf, "zqjsonme"}, + {"other home, solidus escape", `{"p":"\/home\/` + other + `\/x"}`, kindHomeOther, other}, + {"other home, unicode solidus", `{"p":"\u002Fhome\u002F` + other + `\u002Fx"}`, kindHomeOther, other}, + {"other home, solidus at line end", `{"p":"\/home\/` + other + `"}`, kindHomeOther, other}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + fs := ScanText(c.line, id, DefaultPatterns(), nil, "transcript") + if !hasKind(fs, c.kind) { + t.Fatalf("%s: no %s finding for %s; got %+v", c.name, c.kind, c.line, fs) + } + red, _ := Redact(c.line, fs) + if strings.Contains(red, c.gone) { + t.Errorf("%s: the name survived redaction:\n%s", c.name, red) + } + }) + } + + // A generic login is reported where it stands as an account, and the + // solidus-escaped home root is such a position. + generic := Identity{HomePath: "/home/" + "dev", HomeUser: "dev"} + line := `{"p":"\/home\/dev\/project"}` + if f, ok := findingOf(ScanText(line, generic, DefaultPatterns(), nil, "transcript"), kindHomeSelf); !ok { + t.Errorf("a generic login's own home behind solidus escapes raised no %s", kindHomeSelf) + } else if !strings.Contains(f.Matched, "dev") { + t.Errorf("the %s span does not cover the login: %q", kindHomeSelf, f.Matched) + } +} + +// TestJSONEscapeLayerAddsNothingToPlainLines is the false-positive side: an +// ordinary transcript line full of escapes and carrying no secret and no +// identity stays clean, a Windows path whose escaped separators decode into +// control escapes on a further layer ("\\b", "\\n") invents nothing, and a +// finding the raw line already carries is reported once, not once per layer. +func TestJSONEscapeLayerAddsNothingToPlainLines(t *testing.T) { + id := jsonEscapeIdentity() + clean := []string{ + `{"type":"assistant","text":"Line one.\nLine two with \"quotes\" and a tab\there.\n\u2014 done"}`, + `{"cmd":"dir C:\\\\build\\\\bin\\\\new\\\\tmp","out":"ok\r\n"}`, + `{"re":"^\\/api\\/v1\\/items$","path":"\/srv\/app\/bin"}`, + } + for _, line := range clean { + if fs := ScanText(line, id, DefaultPatterns(), nil, "transcript"); len(fs) != 0 { + t.Errorf("a clean line raised findings: %s\n%+v", line, fs) + } + } + tok := jsonEscapeToken() + line := `{"t":"a \"` + tok + `\" b\n"}` + n := 0 + for _, f := range ScanText(line, id, DefaultPatterns(), nil, "transcript") { + if f.Kind == "token:github_pat" { + n++ + } + } + if n != 1 { + t.Errorf("a token the raw line already finds was reported %d times, want 1", n) + } +} diff --git a/internal/adapter/scanner/meter.go b/internal/adapter/scanner/meter.go index 30f7278dd..390cace80 100644 --- a/internal/adapter/scanner/meter.go +++ b/internal/adapter/scanner/meter.go @@ -42,6 +42,8 @@ const ( stageIdentity = "identity" // stagePercent is each percent-decode pass over the line. stagePercent = "percent" + // stageJSONEscape is each JSON-unescape layer decoded from the line. + stageJSONEscape = "json_escape" ) // costMeter records per-stage scan work. The zero value charges nothing. diff --git a/internal/adapter/scanner/meter_test.go b/internal/adapter/scanner/meter_test.go index e53c8f5a4..4b875f150 100644 --- a/internal/adapter/scanner/meter_test.go +++ b/internal/adapter/scanner/meter_test.go @@ -88,6 +88,13 @@ var meterFixtures = []meterFixture{ }}, {"other_homes", Identity{}, rep("/home/bob/x ")}, // abcd-audit:allow {"shared_home_traversal", Identity{}, rep("/Users/Shared/../x ")}, // abcd-audit:allow + // The JSON-escape layers (jsonescape.go): each layer is one more scan of + // the line, so a line dense in escapes, in nested escapes and in + // escaped homes must still cost a constant number of passes. + {"json_escaped_other_homes", Identity{}, rep(`\/home\/zqa\/x\n`)}, + {"json_unicode_escaped_own_homes", meterNamedID, rep(`/Users/zq8home/x `)}, + {"json_nested_escape_runs", meterNamedID, rep(`\\\\\\\"\\\\n`)}, + {"json_tokens_after_escapes", Identity{}, rep(`\n` + "ghp_" + strings.Repeat("a", 36))}, } // TestScanLineWorkIsLinear is the cost-class guard for the whole of a line's diff --git a/internal/adapter/scanner/percent.go b/internal/adapter/scanner/percent.go index 508a2ae08..d324e9083 100644 --- a/internal/adapter/scanner/percent.go +++ b/internal/adapter/scanner/percent.go @@ -39,10 +39,25 @@ const maxPercentDecodePasses = 3 // leak surviving into a committed memory/intent/capture artifact // (iss-2608270720336165). func decodedLineFindings(patterns []Pattern, probes []matcher, junctions junctionSet, matchers identityMatchers, id2sev map[string]Severity, rawLine string, lineno int, file string) []Finding { - decoded, posMap := percentDecodeBounded(rawLine) - if posMap == nil { - return nil // nothing was percent-encoded; the raw scan already covers it + var out []Finding + if decoded, posMap := percentDecodeBounded(rawLine); posMap != nil { + out = viewFindings(patterns, probes, junctions, matchers, id2sev, rawLine, decodedView{decoded, posMap}, lineno, file) + } + // The JSON-escape views (jsonescape.go): the same scan over each layer of + // the line's JSON string escapes, mapped back the same way, so a value an + // escape hid from an anchor or spelled with escaped bytes is found where + // it sits on disk (iss-2609261647358395, iss-2609251639263391). + for _, v := range jsonEscapeLayers(rawLine) { + out = append(out, viewFindings(patterns, probes, junctions, matchers, id2sev, rawLine, v, lineno, file)...) } + return out +} + +// viewFindings runs every detector over one decoded view of rawLine and maps +// each hit back to the raw bytes it came from. A view with nothing decoded in +// it is never handed here; the raw scan already covers the raw line. +func viewFindings(patterns []Pattern, probes []matcher, junctions junctionSet, matchers identityMatchers, id2sev map[string]Severity, rawLine string, v decodedView, lineno int, file string) []Finding { + decoded, posMap := v.text, v.posMap var out []Finding for _, m := range scanAllPatterns(patterns, probes, junctions, decoded) { cp := patterns[m.patIdx] diff --git a/internal/adapter/scanner/scanner.go b/internal/adapter/scanner/scanner.go index c36a262b4..f8aa50c27 100644 --- a/internal/adapter/scanner/scanner.go +++ b/internal/adapter/scanner/scanner.go @@ -867,7 +867,8 @@ func scanText(text string, id Identity, patterns []Pattern, id2sev map[string]Se // %22) leaves a hex word-char before a literal token, defeating the // leading \b so the raw scan above never fires. Scan bounded // percent-decoded copies of the line and map every hit back to its raw - // byte span, so Redact masks the live token where it sits on disk. + // byte span, so Redact masks the live token where it sits on disk. The + // same pass reads the line's JSON-escape layers (jsonescape.go). findings = append(findings, decodedLineFindings(patterns, probes, junctions, matchers, id2sev, line, lineno, file)...) } findings = dedupFindings(findings) diff --git a/internal/core/history/redact_jsonescape_test.go b/internal/core/history/redact_jsonescape_test.go new file mode 100644 index 000000000..d51f50f2a --- /dev/null +++ b/internal/core/history/redact_jsonescape_test.go @@ -0,0 +1,61 @@ +package history + +import ( + "bytes" + "os" + "strings" + "testing" +) + +// TestCaptureRedactsValuesBesideJSONEscapes is the store boundary of +// iss-2609261647358395 and iss-2609251639263391. A transcript is raw JSONL, so +// a token pasted at the start of an output line sits straight after a "\n" +// escape, and an encoder that escapes the solidus writes the caller's home +// with no '/' on the line. Both reached the stored record verbatim: the +// token's leading anchor read the escape letter as a word byte, and neither +// the home-path detector nor the literal $HOME backstop reads "\/". The +// record must carry neither, and must still carry the lines around them. +func TestCaptureRedactsValuesBesideJSONEscapes(t *testing.T) { + repoRoot, home := setupStore(t) + token := "ghp_" + "0123456789abcdefABCDEF0123456789abcd" + escapedHome := strings.ReplaceAll(home, "/", `\/`) + other := "zqstoreother" + + transcript := strings.Join([]string{ + `{"type":"user","text":"here it is:\n` + token + `\nthanks"}`, + `{"type":"tool","text":"wrote ` + escapedHome + `\/notes\/plan.md"}`, + `{"type":"tool","text":"ls\n/home/` + other + `/src\ndone"}`, + }, "\n") + + res, err := Capture(repoRoot, testRootSHA, []byte(transcript), CaptureMeta{SessionID: "sess-jsonescape", Kind: "native"}) + if err != nil { + t.Fatalf("Capture: %v", err) + } + if !res.Wrote { + t.Fatal("expected Wrote=true: an idempotent no-op would leave every assertion below reading a record this call did not write") + } + onDisk, err := os.ReadFile(res.Record.Path) + if err != nil { + t.Fatalf("read record: %v", err) + } + if bytes.Contains(onDisk, []byte(token)) { + t.Errorf("a token after a \\n escape survived in the stored record") + } + if !bytes.Contains(onDisk, []byte("ghp"+strings.Repeat("*", 8))) { + t.Errorf("the token must be MASKED, not merely absent — no fingerprint in the stored record") + } + // The login alone was already redacted by its bare word; what leaked is + // the path around it, so the assertion is on the home's parent. + escapedParent := escapedHome[:strings.LastIndex(escapedHome, `\/`)] + if escapedParent == "" || bytes.Contains(onDisk, []byte(escapedParent)) { + t.Errorf("the caller's home written with solidus escapes survived in the stored record") + } + if bytes.Contains(onDisk, []byte(other)) { + t.Errorf("a third-party home between \\n escapes survived in the stored record") + } + for _, keep := range []string{"here it is:", "thanks", "plan.md", "done"} { + if !bytes.Contains(onDisk, []byte(keep)) { + t.Errorf("redaction took %q with it; the lines around a value must survive", keep) + } + } +} From 401361062c11fc455a2927d5336716d29ba43e3e Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:54:53 +0100 Subject: [PATCH 03/64] fix(scanner): read the Windows home spelling in the one home matcher home_path_other was POSIX-only, so a third party's home written as :\Users\ (a WSL session, a pasted PowerShell transcript, a Windows CI log) raised nothing, as typed or JSON-escaped. genericHomeRe gains the Windows alternative, each separator a run of backslashes so one JSON layer (doubled) or two (quadrupled) read the same as the typed spelling. Its friends follow it: the trailing boundary takes the backslash; the system-directory allowlist recognises the Windows Users root and gains Default and the All Users junction beside Public; the traversal walk reads a backslash run as ONE separator, so an escaped separator inside a system root is not taken for an empty traversal segment; and the home_path_other skip compares the caller's home against the match with its separator runs collapsed, so the caller's own escaped home is never reported as a third party's. The caller's own home is home_path_self at any depth through the JSON-escape views of the previous commit; abcd builds for darwin and linux only, so the caller's home is never itself a Windows path. The allowlist is shared with the repolint privacy rule, which applies it to its own C:\Users branch; that rule's twin regexp is otherwise unchanged. Refs: iss-2609251639261103 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/identity.go | 40 ++++++-- internal/adapter/scanner/network.go | 10 +- internal/adapter/scanner/windows_home_test.go | 92 +++++++++++++++++++ 3 files changed, 129 insertions(+), 13 deletions(-) create mode 100644 internal/adapter/scanner/windows_home_test.go diff --git a/internal/adapter/scanner/identity.go b/internal/adapter/scanner/identity.go index 458dd00c2..46910debe 100644 --- a/internal/adapter/scanner/identity.go +++ b/internal/adapter/scanner/identity.go @@ -268,7 +268,15 @@ var ( // class stays ASCII-cased on purpose: it is the USERNAME, which the boundary // helpers and isHomeSegmentByte judge by the same class, and folding a class // that already carries both cases changes nothing. - genericHomeRe = regexp.MustCompile(`(?i)(?:/Users/[A-Za-z0-9._-]+|/home/[A-Za-z0-9._-]+)`) + // + // The third alternative is the Windows spelling, :\Users\ + // (iss-2609251639261103): a WSL session, a pasted PowerShell transcript or + // a Windows CI log names homes that way. Each separator is a RUN of + // backslashes, because a JSON encoder doubles the separator and JSON quoted + // inside JSON doubles it again; the run reads as one separator at any + // depth, the way endsWithPathFold reads accountRootPrefixes. Windows has no + // /home root, so there is no fourth alternative. + genericHomeRe = regexp.MustCompile(`(?i)(?:/Users/[A-Za-z0-9._-]+|/home/[A-Za-z0-9._-]+|[A-Za-z]:\\+Users\\+[A-Za-z0-9._-]+)`) // Loose URL span (scheme to whitespace/quote/closing). urlSpanRe = regexp.MustCompile(`(?:https?://|git@|ftp://|ssh://)[^\s"'` + "`" + `)>\]<]+`) // A git noreply email is not a leak. @@ -309,10 +317,12 @@ func isOwnRepoSlug(line string, end int, repo string) bool { } // homeBoundary is the trailing-boundary set for a home-path match (ported from -// the Python lookahead [/\s"'`)\]\}<,;:]). +// the Python lookahead [/\s"'`)\]\}<,;:]), plus the backslash: the Windows +// separator, and the first byte of any escape a JSON string writes after a +// path (iss-2609251639261103). No username continues with a backslash. func homeBoundary(r rune) bool { switch r { - case '/', '"', '\'', '`', ')', ']', '}', '<', ',', ';', ':': + case '/', '\\', '"', '\'', '`', ')', ']', '}', '<', ',', ';', ':': return true } return r == ' ' || r == '\t' || r == '\n' || r == '\r' || r == '\f' || r == '\v' @@ -627,7 +637,10 @@ func (m identityMatchers) findings(line string, lineno int, id2sev map[string]Se continue } matched := line[loc[0]:loc[1]] - if m.homeSelf != nil && homeSelfStandsIn(m.homeSelf, matched) { + // The caller's own home, in any escaped spelling: a Windows match + // carries its separators as runs, and the home literal carries one + // backslash per separator. + if m.homeSelf != nil && (homeSelfStandsIn(m.homeSelf, matched) || homeSelfStandsIn(m.homeSelf, collapseSeparatorRuns(matched))) { continue } // /Users/Shared and friends are macOS system directories, not users @@ -1047,13 +1060,14 @@ func isLocalPartByte(b byte) bool { } // isNonUserHomeMatch reports whether a generic-home match's final segment is a -// well-known non-user directory under a /Users root. +// well-known non-user directory under a /Users root, POSIX or Windows. func isNonUserHomeMatch(matched string) bool { scanMeter.charge(stageIdentity, len(matched)) - if !strings.HasPrefix(strings.ToLower(matched), "/users/") { + lower := strings.ToLower(matched) + if !strings.HasPrefix(lower, "/users/") && !(len(lower) > 2 && lower[1] == ':' && lower[2] == '\\') { return false } - i := strings.LastIndexByte(matched, '/') + i := strings.LastIndexAny(matched, `/\`) return i >= 0 && IsNonUserHomeSegment(matched[i+1:]) } @@ -1069,12 +1083,20 @@ func isNonUserHomeMatch(matched string) bool { // rather than a home directory, which is the subtree the exemption now covers // (iss-2609100505145554). // -// genericHomeRe is POSIX-only, so '/' is the only separator that can reach here. +// A separator is a '/' or a RUN of backslashes, the Windows separator at any +// escaping depth: "C:\\Users\\Public\\x" is one separator per run, not an +// empty traversal segment between two backslashes, so it stays inside the +// shared root. Two '/' in a row are an empty segment. func nextPathSegmentEnd(line string, pos int) (int, bool) { traversed := false from := pos - for pos < len(line) && line[pos] == '/' { + for pos < len(line) && (line[pos] == '/' || line[pos] == '\\') { i, named := pos+1, false + if line[pos] == '\\' { + for i < len(line) && line[i] == '\\' { + i++ + } + } for i < len(line) && isHomeSegmentByte(line[i]) { if line[i] != '.' { named = true diff --git a/internal/adapter/scanner/network.go b/internal/adapter/scanner/network.go index 7ec683dd6..6eb13890e 100644 --- a/internal/adapter/scanner/network.go +++ b/internal/adapter/scanner/network.go @@ -151,11 +151,13 @@ func IsPersonaName(seg string) bool { // (a detector must not tax legitimate product code) applied to hostnames. var nonHostLabels = map[string]bool{"work.local": true} -// nonUserHomeSegments are the well-known macOS directories that live under -// /Users but name no user. Flagging them as usernames (iss-153) forced waivers -// onto product code that legitimately writes there. +// nonUserHomeSegments are the well-known directories that live under a Users +// root but name no user: macOS's Shared and Guest, and Windows's Public, +// Default (the profile every new account is copied from) and the All Users +// junction, whose name ends at its space. Flagging them as usernames (iss-153) +// forced waivers onto product code that legitimately writes there. var nonUserHomeSegments = map[string]bool{ - "shared": true, "guest": true, "public": true, + "shared": true, "guest": true, "public": true, "default": true, "all": true, } // IsNonUserHomeSegment reports whether seg, the segment immediately after a diff --git a/internal/adapter/scanner/windows_home_test.go b/internal/adapter/scanner/windows_home_test.go new file mode 100644 index 000000000..6f43c6528 --- /dev/null +++ b/internal/adapter/scanner/windows_home_test.go @@ -0,0 +1,92 @@ +package scanner + +import ( + "strings" + "testing" +) + +// iss-2609251639261103: home_path_other and home_path_self read no Windows +// home spelling. A transcript from WSL, a pasted PowerShell session or a +// Windows CI log names homes as :\Users\, and a JSON encoder +// doubles each backslash (and doubles them again for JSON quoted inside JSON). +// The paths are assembled so no committed line carries a literal one. + +const winRoot = `:\Users\` + +// winPath spells drive + `:\Users\` + rest with every separator written as +// depth backslashes: 1 as typed, 2 as one JSON layer writes it, 4 as two do. +func winPath(drive, rest string, depth int) string { + p := drive + winRoot + rest + return strings.ReplaceAll(p, `\`, strings.Repeat(`\`, depth)) +} + +func TestWindowsHomeIsAThirdPartyHomePath(t *testing.T) { + other := "zqwinother" + cases := []struct { + name, line string + }{ + {"as typed", "open " + winPath("C", other+`\Desktop\a.txt`, 1) + " now"}, + {"one JSON layer", `{"t":"open ` + winPath("C", other+`\Desktop`, 2) + `"}`}, + {"two JSON layers", `{"t":"{\"p\":\"` + winPath("C", other+`\Desktop`, 4) + `\"}"}`}, + {"lower-case drive and root", "open " + strings.ToLower(winPath("c", "", 1)) + other + `\x`}, + {"another drive", "open " + winPath("D", other+`\x`, 1)}, + {"at line end", "home is " + winPath("C", other, 1)}, + {"out of the Public root by traversal", "open " + winPath("C", `Public\..\`+other+`\x`, 1)}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + fs := ScanText(c.line, Identity{}, DefaultPatterns(), nil, "t") + f, ok := findingOf(fs, kindHomeOther) + if !ok { + t.Fatalf("no %s for %q: %+v", kindHomeOther, c.line, fs) + } + if !strings.Contains(f.Matched, other) { + t.Errorf("the %s span %q does not cover the name", kindHomeOther, f.Matched) + } + red, _ := Redact(c.line, fs) + if strings.Contains(red, other) { + t.Errorf("the name survived redaction:\n%s", red) + } + }) + } +} + +// TestWindowsSystemRootsAreNotHomes is the false-positive side: the Windows +// profile directories that name no user, at any escaping depth, a relative +// path that merely contains a Users segment, and a backslash run standing +// for ONE separator rather than an empty traversal segment. +func TestWindowsSystemRootsAreNotHomes(t *testing.T) { + lines := []string{ + "open " + winPath("C", `Public\Documents\a.txt`, 1), + "open " + winPath("C", `Default\AppData\Local`, 1), + `{"t":"` + winPath("C", `Public\Desktop`, 2) + `"}`, + `{"t":"` + winPath("C", `Default\NTUSER.DAT`, 4) + `"}`, + `see src\Users\guide.md and build` + `\Users\list.txt`, + } + for _, line := range lines { + if fs := ScanText(line, Identity{}, DefaultPatterns(), nil, "t"); hasKind(fs, kindHomeOther) { + t.Errorf("a non-home was reported as %s: %q\n%+v", kindHomeOther, line, fs) + } + } +} + +// TestWindowsOwnHomeIsHomeSelfAtAnyDepth: the caller's own home in its +// Windows spelling is home_path_self (hard_fail) wherever it is escaped, and +// is never also reported as a third party's. +func TestWindowsOwnHomeIsHomeSelfAtAnyDepth(t *testing.T) { + id := Identity{HomePath: "C" + winRoot + "zqwinme", HomeUser: "zqwinme"} + for _, depth := range []int{1, 2, 4} { + line := "saved " + winPath("C", `zqwinme\notes.txt`, depth) + " ok" + fs := ScanText(line, id, DefaultPatterns(), nil, "t") + if !hasKind(fs, kindHomeSelf) { + t.Errorf("depth %d: the caller's own Windows home raised no %s: %+v", depth, kindHomeSelf, fs) + } + if hasKind(fs, kindHomeOther) { + t.Errorf("depth %d: the caller's own Windows home was also reported as a third party's: %+v", depth, fs) + } + red, _ := Redact(line, fs) + if strings.Contains(red, "zqwinme") || !strings.Contains(red, "notes.txt") { + t.Errorf("depth %d: redaction left the name or took the file:\n%s", depth, red) + } + } +} From 7babc8d7eac825937f18693e9edf4291d0b6679d Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:57:31 +0100 Subject: [PATCH 04/64] fix(scanner): keep a short real name standing in person metadata on bytes The byte scan dropped every short single-token real_name as chance noise, so a name the text scan hard-fails on shipped unreported in a PDF /Author entry or any other author stamp. The existing shape test pinned the defect itself, expecting /Author (Zedqx) to raise nothing. Decision (recorded here, not in the record): the byte scan keeps a short single token where a metadata key that names a person ends within 96 bytes before it, and drops it everywhere else. The keys are author, artist, creator and lastModifiedBy, which cover a PDF Info /Author, XMP dc:creator, pdf:Author and tiff:Artist (pretty-printed across lines, as XMP writers lay it out), PNG text Author and Artist, and an OOXML or ODF dc:creator or cp:lastModifiedBy inside a decoded zip. That is the anchored context the record names as closing it, and it neither lowers the threshold nor accepts a false-positive rate: the same token incidental to binary content, a hundred times over or past the key's reach, still raises nothing, so the report is not flooded. The threshold keeps counting bytes, now with the reason stated: a chance collision needs that many specific bytes in a row. A key held in binary structure (EXIF's Artist tag, a UTF-16 PDF string) is not text in the bytes and stays out of reach. Refs: iss-2609090934372160 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/binary_skip_test.go | 72 +++++++++++++-- internal/adapter/scanner/scanner.go | 92 +++++++++++++++++--- 2 files changed, 141 insertions(+), 23 deletions(-) diff --git a/internal/adapter/scanner/binary_skip_test.go b/internal/adapter/scanner/binary_skip_test.go index ca7c778ee..e4000c446 100644 --- a/internal/adapter/scanner/binary_skip_test.go +++ b/internal/adapter/scanner/binary_skip_test.go @@ -223,16 +223,19 @@ func TestBinaryScanKeepsLongIdentityRulesDropsShortOnes(t *testing.T) { } // TestRealNameOnBytesByLiteralShape: real_name is kept on bytes when the -// literal cannot collide by chance — 8+ characters or more than one word — and -// dropped for a short single token. +// literal cannot collide by chance — 8+ bytes or more than one word — and a +// short single token is kept where it stands in a metadata field that names a +// person, and dropped as chance noise everywhere else +// (iss-2609090934372160; TestShortRealNameOnBytesIsKeptInAMetadataField). func TestRealNameOnBytesByLiteralShape(t *testing.T) { cases := []struct { - name string - keep bool + name, body string + keep bool }{ - {"Zed Q Eight", true}, // multi-word - {"Zedquinta", true}, // long single token - {"Zedqx", false}, // short single token: noise on bytes + {"Zed Q Eight", "/Author (Zed Q Eight)", true}, // multi-word + {"Zedquinta", "/Title (Zedquinta)", true}, // long single token, any field + {"Zedqx", "/Author (Zedqx)", true}, // short single token in an author field + {"Zedqx", "/Title (Zedqx) /Producer (x)", false}, // short single token in no person field } for _, c := range cases { root := t.TempDir() @@ -241,14 +244,65 @@ func TestRealNameOnBytesByLiteralShape(t *testing.T) { t.Fatal(err) } sc.identity = Identity{GitUserName: c.name} - abs := writeFile(t, root, "deck.pdf", "%PDF-1.4\n/Author ("+c.name+")\n") + abs := writeFile(t, root, "deck.pdf", "%PDF-1.4\n"+c.body+"\n") res := scanOne(t, sc, "deck.pdf", abs) if got := hasKind(res.Findings, kindRealName); got != c.keep { - t.Errorf("real_name %q on bytes: fired=%v want %v (%+v)", c.name, got, c.keep, res.Findings) + t.Errorf("real_name %q in %q on bytes: fired=%v want %v (%+v)", c.name, c.body, got, c.keep, res.Findings) } } } +// TestShortRealNameOnBytesIsKeptInAMetadataField is iss-2609090934372160's +// detector. A short single-token name is a hard_fail real_name in text, and +// the byte scan dropped it everywhere as chance noise, so the same name in a +// document's author metadata shipped unreported. It is kept where a metadata +// key that names a person stands just before it — a PDF /Author entry, an +// XMP dc:creator (pretty-printed across lines, as XMP writers lay it out), a +// PNG tEXt Author chunk, an OOXML cp:lastModifiedBy inside a zip — and the +// same token incidental to binary content, however often it occurs, still +// raises nothing, so the byte report is not flooded. +func TestShortRealNameOnBytesIsKeptInAMetadataField(t *testing.T) { + const name = "Zedqx" + kept := map[string][]byte{ + "deck.pdf": []byte("%PDF-1.7\n1 0 obj\n<< /Author (" + name + ") /Title (q3) >>\nendobj\n"), + "xmp.pdf": []byte("%PDF-1.7\n\n \n \n " + name + "\n"), + "shot.png": append([]byte("\x89PNG\r\n\x1a\n\x00\x00\x00\x0btEXtAuthor\x00"), []byte(name+"\x00\x00\x00\x00IEND")...), + "props.zip": zipOf(t, "docProps/core.xml", []byte(``+name+``)), + } + for logical, body := range kept { + root := t.TempDir() + sc, err := New(root) + if err != nil { + t.Fatal(err) + } + sc.identity = Identity{GitUserName: name} + res := scanOne(t, sc, logical, writeFile(t, root, logical, string(body))) + if !hasKind(res.Findings, kindRealName) || res.HardFails == 0 { + t.Errorf("%s: a short name in a person metadata field was not a hard_fail real_name: %+v", logical, res.Findings) + } + } + + // Incidental: the token a hundred times over in binary content, and once + // just past the reach of an author key, raises nothing. + var noise bytes.Buffer + noise.WriteString("\x89PNG\r\n\x1a\n") + for i := 0; i < 100; i++ { + noise.Write([]byte{0x00, byte(i), 0xff, 0x7f}) + noise.WriteString(" " + name + " ") + } + noise.WriteString("Author\x00" + strings.Repeat("\x01", 200) + " " + name + " ") + root := t.TempDir() + sc, err := New(root) + if err != nil { + t.Fatal(err) + } + sc.identity = Identity{GitUserName: name} + res := scanOne(t, sc, "noise.png", writeFile(t, root, "noise.png", noise.String())) + if hasKind(res.Findings, kindRealName) { + t.Errorf("a short name incidental to binary content was reported: %d findings", len(res.Findings)) + } +} + // TestBinaryHomePathHardFailsLikeText: renaming deck.md to deck.pdf must not // turn a release-blocking home path in plaintext metadata into a pass. func TestBinaryHomePathHardFailsLikeText(t *testing.T) { diff --git a/internal/adapter/scanner/scanner.go b/internal/adapter/scanner/scanner.go index f8aa50c27..7ca9961c5 100644 --- a/internal/adapter/scanner/scanner.go +++ b/internal/adapter/scanner/scanner.go @@ -1,6 +1,7 @@ package scanner import ( + "bytes" "encoding/json" "errors" "os" @@ -1138,15 +1139,17 @@ func guardedReadWhy(err error) string { // secret patterns (secretPatterns) plus the identity rules whose literal is // the caller's own and long enough that a chance collision with binary // content is negligible — the home path, the email, and a real name that is -// multi-word or 8+ characters. A home path in PDF /Creator, an /Author stamp, -// or a session URL in PNG tEXt metadata is the same release-blocking leak it -// is in prose, and renaming deck.md to deck.pdf must not change the verdict. -// The short/generic identity kinds (local_username, github_username, -// home_path_other, a short single-token real_name) are dropped by -// byteScanPolicy after the scan — home_path_other runs unguarded inside the -// identity matcher, so switching it off by identity value is not possible — -// unless the repo's pii.json raised that kind above its built-in default, -// which is a judgement the byte scan honours. +// multi-word or 8+ bytes, or a short single token standing in a metadata +// field that names a person (metadataPersonKeys). A home path in PDF +// /Creator, an /Author stamp, or a session URL in PNG tEXt metadata is the +// same release-blocking leak it is in prose, and renaming deck.md to deck.pdf +// must not change the verdict. The short/generic identity kinds +// (local_username, github_username, home_path_other, a short single-token +// real_name anywhere else) are dropped by byteScanPolicy after the scan — +// home_path_other runs unguarded inside the identity matcher, so switching it +// off by identity value is not possible — unless the repo's pii.json raised +// that kind above its built-in default, which is a judgement the byte scan +// honours. func (s *Scanner) scanBytes(data []byte, secrets []Pattern, logical string) []Finding { // GitRemoteUsername and the emails travel with the names so the matcher // tells a public handle from a real name exactly as it does on text @@ -1159,9 +1162,10 @@ func (s *Scanner) scanBytes(data []byte, secrets []Pattern, logical string) []Fi OtherGitUserNames: s.identity.OtherGitUserNames, OtherGitUserEmails: s.identity.OtherGitUserEmails, } all := scanText(string(data), long, secrets, s.identSev, logical, true) + meta := metadataFields{data: data} out := all[:0] for _, f := range all { - if s.byteScanDrops(f) { + if s.byteScanDrops(f, &meta) { continue } out = append(out, f) @@ -1178,14 +1182,73 @@ const ( // bytePolicyDrop: noise on bytes; dropped unless the repo raised the kind. bytePolicyDrop // bytePolicyKeepLongLiteral: kept when the matched literal is multi-word or - // at least byteScanLongLiteral bytes, dropped when it is a short token. + // at least byteScanLongLiteral bytes, or when a short token stands in a + // metadata field that names a person; dropped when it is a short token + // anywhere else. bytePolicyKeepLongLiteral ) // byteScanLongLiteral is the length from which a single-token real name no -// longer collides with binary content by chance. +// longer collides with binary content by chance. It counts BYTES, not runes, +// on purpose: what a chance collision needs is that many specific bytes in a +// row, so a four-rune CJK name (twelve bytes) is as unlikely by chance as a +// twelve-letter Latin one, and a seven-byte Latin name is not. const byteScanLongLiteral = 8 +// metadataPersonKeys are the metadata keys that name a person, lower-cased: a +// PDF Info dictionary's /Author, XMP's dc:creator, pdf:Author and tiff:Artist, +// a PNG text chunk's Author and Artist keywords, and an OOXML or ODF +// document's dc:creator and cp:lastModifiedBy. Each is written as text in the +// raw bytes (or in a region the container decoder inflates), so a short name +// after one is the name the file was stamped with rather than a chance run of +// bytes (iss-2609090934372160). A key read from binary structure — EXIF's +// Artist tag, a UTF-16 PDF string — is not text in the bytes and is not +// reached here. +var metadataPersonKeys = [][]byte{[]byte("author"), []byte("artist"), []byte("creator"), []byte("lastmodifiedby")} + +// maxMetadataKeyGap is how far before a short name metadataFields looks for a +// person key: the markup between an XMP dc:creator and its rdf:li value, laid +// out one element per line and indented the way XMP writers indent it, fits +// with room to spare, and the reach is what keeps an incidental match far +// from any key out. +const maxMetadataKeyGap = 96 + +// metadataFields answers whether a byte-scan finding stands in a person +// metadata field of the data it was scanned from. The line offsets are found +// once, on the first question. +type metadataFields struct { + data []byte + starts []int +} + +// holds reports whether a person key ends within maxMetadataKeyGap bytes +// before f. scanText splits on '\n', so f's line starts after the (Line-1)th +// newline and its Column is the byte offset on that line. +func (m *metadataFields) holds(f Finding) bool { + if m.starts == nil { + m.starts = []int{0} + for i, b := range m.data { + if b == '\n' { + m.starts = append(m.starts, i+1) + } + } + } + if f.Line < 1 || f.Line > len(m.starts) || f.Column < 1 { + return false + } + at := m.starts[f.Line-1] + f.Column - 1 + if at > len(m.data) { + return false + } + window := bytes.ToLower(m.data[max(0, at-maxMetadataKeyGap):at]) + for _, k := range metadataPersonKeys { + if bytes.Contains(window, k) { + return true + } + } + return false +} + // byteScanPolicy is the EXPLICIT byte-scan classification of every identity // kind. It is a table, not a derivation: a kind that is not listed reports // ok=false, and TestEveryIdentityKindIsClassifiedForBytes fails until it is @@ -1211,7 +1274,7 @@ func byteScanPolicy(kind string) (bytePolicy, bool) { // byteScanDrops applies byteScanPolicy to one finding, honouring a repo-raised // severity: when pii.json made the kind stricter than its built-in default, // the repo has judged that kind a leak wherever it sits, and the drop yields. -func (s *Scanner) byteScanDrops(f Finding) bool { +func (s *Scanner) byteScanDrops(f Finding, meta *metadataFields) bool { policy, ok := byteScanPolicy(f.Kind) if !ok { return false @@ -1223,7 +1286,8 @@ func (s *Scanner) byteScanDrops(f Finding) bool { case bytePolicyDrop: return true case bytePolicyKeepLongLiteral: - return len(f.Matched) < byteScanLongLiteral && strings.IndexFunc(f.Matched, unicode.IsSpace) < 0 + short := len(f.Matched) < byteScanLongLiteral && strings.IndexFunc(f.Matched, unicode.IsSpace) < 0 + return short && !meta.holds(f) } return false } From 7909f8bfdfaa78f52bbc807425f26eebbb3d8122 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:57:59 +0100 Subject: [PATCH 05/64] =?UTF-8?q?chore:=20re-defer=20iss-96=20past=20v0.11?= =?UTF-8?q?.0=20=E2=80=94=20the=20entropy=20residue=20owes=20a=20ruling?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The residue iss-96 names (a bare value with no key name, a labelled value under the entropy floor) is reached only by an always-on entropy or charset detector, which the 2026-08-28 ruling rejected for its redaction false-positive cost on transcript prose when it shipped the opt-in external-scanner adapter. Building one here would re-open that ruling, so nothing is built: the deferral is carried past v0.11.0 with the tradeoff named, and a dated re-check records that both pins still hold after this lane's JSON-escape views. Refs: iss-96 Assisted-by: Claude:claude-opus-5-5 --- ...ipts-are-captured-automatically-on-every-ses.md | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/.abcd/work/issues/open/iss-96-now-that-transcripts-are-captured-automatically-on-every-ses.md b/.abcd/work/issues/open/iss-96-now-that-transcripts-are-captured-automatically-on-every-ses.md index cb6b17a86..de76b7ef2 100644 --- a/.abcd/work/issues/open/iss-96-now-that-transcripts-are-captured-automatically-on-every-ses.md +++ b/.abcd/work/issues/open/iss-96-now-that-transcripts-are-captured-automatically-on-every-ses.md @@ -7,8 +7,8 @@ category: "security" source: "manual-test" found_during: "itd-89-m1" found_at: "internal/adapter/scanner/patterns.go" -deferred_after: "v0.10.0" -deferral_reason: "ruling owed to the product thinker (away; run A 2026-09-25): Transcript-path secret detection: an entropy detector, key-name context, or an opt-in external scanner?" +deferred_after: "v0.11.0" +deferral_reason: "ruling owed to the product thinker: the residue past the opt-in external-scanner adapter (ruled 2026-08-28) is a bare value with no key name and a labelled value under the entropy floor, and only an always-on entropy or charset detector reads either; the same ruling rejected that detector for its redaction false-positive cost on transcript prose (hashes, ids, base64 blobs), and a native key-name rule reaches only labelled values the armed adapter already reaches, so where the floor sits, or whether a keyword-delimiter-entropy rule runs natively by default, trades a corrupted record against reach and is not an implementer call" --- Now that transcripts are captured automatically on every session end, the scanner's secret-pattern coverage becomes load-bearing in a way it was not when capture was a manual verb nobody ran. Verified by live test: the bundled patterns DO catch anchored tokens (AKIA... access key IDs, ghp_/gho_/sk-ant- style prefixes) and absolute home paths, but they do NOT catch unanchored high-entropy values — an AWS SECRET access key (the 40-char base64 value, no prefix), a bare password, or a generic API token with no recognisable prefix all pass through into the store verbatim. This is the standard prefix-matching limitation and is pre-existing, not a regression; the point is that the blast radius changed. Consider entropy-based detection or the opt-in gitleaks adapter for the transcript path specifically, where the input is unstructured prose rather than curated source. @@ -213,3 +213,13 @@ carries a share of that cost too, in proportion to its reach — its entropy flo fires on any labelled high-entropy token in prose, credential or not. Reach and cost together are the open question, and where the bar sits is the maintainer's to decide — grill-then-implement, not autonomous work. + +**Re-check (2026-09-26, autonomous run A, lane drainS1).** Nothing built for +this entry: the 2026-08-28 ruling already rejected option (a) and shipped (c) +opt-in, and the residue it names is exactly what only (a) reaches, so building +a detector would be re-opening a ruling rather than implementing one. The +lane's JSON-escape views (iss-2609261647358395) widen the ANCHORED reach on +this path — a prefixed token written straight after a `\n` or `\t` escape in a +raw transcript line is masked from that change on — and leave the +unanchored residue exactly where it was: `TestTranscriptPathMissesUnanchoredEntropy` +and `TestCaptureStoresUnanchoredEntropyVerbatim` both still pass. From b605146a72ed989be5be123f951bcc44620df63a Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:59:22 +0100 Subject: [PATCH 06/64] chore: capture three siblings the scanner-identity lane leaves open Found by the sibling sweep and left for their own lanes: the repolint privacy rule and the harness_leak rule read committed lines raw, so the escaped spellings the scanner's views now read pass them; the literal caller-home backstop reads no escaped spelling of the home; and a short name in a binary EXIF Artist tag has no metadata key text beside it. Refs: iss-2609261658553101, iss-2609261659041553, iss-2609261659051539 Assisted-by: Claude:claude-opus-5-5 --- ...ivacy-hygiene-rule-and-the-harness-leak-lint.md | 14 ++++++++++++++ ...ral-caller-home-backstop-sweepcallerhome-and.md | 14 ++++++++++++++ ...e-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md | 14 ++++++++++++++ 3 files changed, 42 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md create mode 100644 .abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md create mode 100644 .abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md diff --git a/.abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md b/.abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md new file mode 100644 index 000000000..88c923979 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261658553101" +slug: "the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint" +severity: "minor" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/core/repolint/rule_privacy.go" +--- + +The repolint privacy-hygiene rule and the harness_leak lint rule read each committed line raw, so the JSON-escape spellings the scanner reads through its decoded views (jsonescape.go) pass both: a third-party home after a \n escape, a home written with the solidus escape, a Windows home with doubled separators, and a token or a session URL written straight after a \n escape (the patterns anchor on a leading word boundary, and the escape letter is a word byte). A committed JSON fixture, export or transcript carrying any of them is not refused. Reading the views in the lint is not contained: Go interpreted string literals use the same escapes, so the views would surface every test string that puts a \n escape before a /home root and a name in committed Go source, and the waivers on those lines do not all exist. The CI gitleaks history scan covers the token half; the home-path and session-URL halves have no other gate. Detector: privacyLeak and harnessLeakOnLine report each of the four spellings on a committed line. diff --git a/.abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md b/.abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md new file mode 100644 index 000000000..429042967 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261659041553" +slug: "the-literal-caller-home-backstop-sweepcallerhome-and" +severity: "minor" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/residual.go" +--- + +The literal caller-home backstop (SweepCallerHome and SurvivingCallerHome, internal/adapter/scanner/residual.go) reads the home only as written, so an escaped spelling of it is invisible to the one stage the store-before-commit redactors keep independent of the detector: the solidus escape (\/ between segments), its \u002f form, and a separator written as an escape run. Stage one and the stage-two rescan read those spellings through the scanner's JSON-escape views since fix/drain-scanner-identity, so a leak needs both detector passes to miss before the backstop matters; but the backstop exists for exactly that case, and against an escaped home it is not a backstop. Detector: with HOME under the /Users root, SweepCallerHome collapses the solidus-escaped and the \u002f spelling of the home to the tilde, as it does the literal one. diff --git a/.abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md b/.abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md new file mode 100644 index 000000000..6fd46d16a --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261659051539" +slug: "a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif" +severity: "minor" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/scanner.go" +--- + +A short single-token real name in a JPEG's or TIFF's EXIF Artist tag (IFD0 tag 0x013B) is still dropped by the payload byte scan as chance noise. The byte scan keeps a short name only where a person metadata key stands as text within reach before it (metadataPersonKeys in internal/adapter/scanner/scanner.go: a PDF /Author, XMP dc:creator, a PNG text Author, an OOXML cp:lastModifiedBy), and EXIF stores the Artist tag as a binary IFD entry whose ASCII value sits at an offset with no key text beside it; a UTF-16 PDF string (/Author with a FEFF byte-order mark) is out of reach the same way. Closing it needs the IFD read structurally, the way container.go walks PNG chunks: find the Exif header, read the TIFF byte order, walk IFD0 and scan the Artist (and XPAuthor) value with the text rules. Residue of iss-2609090934372160. Detector: a short banned name in a camera-written EXIF Artist tag, with no XMP packet beside it, is a real_name finding in the payload scan. From 69b37253880040ffc09efd1d05340d9dd3ff65f6 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:59:36 +0100 Subject: [PATCH 07/64] =?UTF-8?q?chore:=20resolve=20iss-2609261647358395?= =?UTF-8?q?=20=E2=80=94=20JSON=20escapes=20no=20longer=20hide=20values?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The fix is c55ae5ef: the scanner reads each line's JSON-escape layers as decoded views mapped back to raw spans. Resolves: iss-2609261647358395 Assisted-by: Claude:claude-opus-5-5 --- ...ring-escape-beside-a-secret-or-a-home-path-hides-it.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md (65%) diff --git a/.abcd/work/issues/open/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md b/.abcd/work/issues/resolved/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md similarity index 65% rename from .abcd/work/issues/open/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md rename to .abcd/work/issues/resolved/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md index 69689812a..fd32d3c35 100644 --- a/.abcd/work/issues/open/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md +++ b/.abcd/work/issues/resolved/iss-2609261647358395-a-json-string-escape-beside-a-secret-or-a-home-path-hides-it.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25" origin: researcher-authored production_mode: hand-written found_at: "internal/adapter/scanner/percent.go" +resolution: "Fixed by c55ae5ef: the scanner reads each line's JSON-escape layers (at most three, every layer scanned) as decoded views mapped back to raw spans, the percent pre-pass's shape, so a token after a \\n or \\t escape, a home path beside \\n, \\t or \\\" escapes and a \\u-escaped real name are findings and are masked on disk. Pinned by TestJSONEscapeBesideASecretOrHomeIsStillAFinding and, at the store boundary, TestCaptureRedactsValuesBesideJSONEscapes, both watched RED at 211b8853." +impact: fix +resolved_by: + commit: "c55ae5ef" --- A JSON string escape beside a secret or a home path hides it from the scanner's raw-line pass, and the transcript store scans raw JSONL. Every bundled token pattern anchors on a leading word boundary, so a token written straight after the \n, \t, \r, \b or \f escape (a GitHub PAT at the start of a line of pasted output, JSON-encoded) has the escape letter as a word byte before it and never matches: ScanText at BASE 211b8853 returns nothing for {"t":"tok\nghp_..."} while the same token after a space is token:github_pat. The identity matchers fail the same way: home_path_other finds no third-party home path that follows a \n or \t escape (the leading anchor reads the escape letter as a path byte) or that precedes a \n or \" escape (the trailing boundary set has no backslash), and a non-ASCII real_name written by an ASCII-only encoder (Python json's default \u00e9) never matches the configured name. The fix reads a line's JSON-decoded layers the way the percent-decode pre-pass reads its percent-decoded copy, mapping each hit back to its raw span. Detector: a token after a \n escape, a home path between \n and \" escapes and a \u-escaped real_name are each a finding on the transcript path. + +## Grounds + +- pursued: every ScanText consumer (history, capture, memory, lint outbound, launch, lifeboat) masks a value an escape sits beside; a raw JSONL transcript line with a prefixed token or third-party home next to an escape reaching the store verbatim would show it wrong From adc37e42baec4ae9b9c4eb4324223987fb0c7ede Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:59:49 +0100 Subject: [PATCH 08/64] =?UTF-8?q?chore:=20resolve=20iss-2609251639263391?= =?UTF-8?q?=20=E2=80=94=20the=20solidus=20escape=20reads=20as=20a=20separa?= =?UTF-8?q?tor?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The fix is c55ae5ef: the JSON-escape views decode \/ and /, and the resolution note records the decision on each other escape spelling the scanner's input can carry. Resolves: iss-2609251639263391 Assisted-by: Claude:claude-opus-5-5 --- ...tity-matchers-do-not-read-the-json-solidus-escape-a.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609251639263391-the-identity-matchers-do-not-read-the-json-solidus-escape-a.md (50%) diff --git a/.abcd/work/issues/open/iss-2609251639263391-the-identity-matchers-do-not-read-the-json-solidus-escape-a.md b/.abcd/work/issues/resolved/iss-2609251639263391-the-identity-matchers-do-not-read-the-json-solidus-escape-a.md similarity index 50% rename from .abcd/work/issues/open/iss-2609251639263391-the-identity-matchers-do-not-read-the-json-solidus-escape-a.md rename to .abcd/work/issues/resolved/iss-2609251639263391-the-identity-matchers-do-not-read-the-json-solidus-escape-a.md index 6048f5d4c..9eafa4630 100644 --- a/.abcd/work/issues/open/iss-2609251639263391-the-identity-matchers-do-not-read-the-json-solidus-escape-a.md +++ b/.abcd/work/issues/resolved/iss-2609251639263391-the-identity-matchers-do-not-read-the-json-solidus-escape-a.md @@ -11,6 +11,14 @@ production_mode: hand-written found_at: "internal/adapter/scanner/identity.go" deferred_after: "v0.10.0" deferral_reason: "Judged theoretical in the scanner-cluster fix round: none of the encoders that write abcd's input (Go encoding/json, JSON.stringify for transcripts, Python json) emits the solidus escape, so no transcript, capture or payload abcd reads carries the shape today." +resolution: "Fixed by c55ae5ef: the JSON-escape views decode \\/ and \\u002f (either case) to '/', so a solidus-escaped home raises home_path_self or home_path_other and a generic login's escaped home is the caller's home. Escape spellings swept and decided: \\/ and \\u002f match (JSON views); %2F matches (the percent pre-pass, pre-existing); a doubled or quadrupled backslash separator matches (the Windows alternative reads separator runs, iss-2609251639261103); \\x2f is not matched, because no encoder that feeds abcd writes '/' that way (a Python repr or C escape only escapes non-printables); an HTML entity (/, /) is not matched, because abcd's inputs are JSON and markdown and an entity-escaped path in a fetched page is not a spelling the store receives today. Pinned by TestJSONSolidusEscapeReadsAsASeparator, watched RED at 211b8853." +impact: fix +resolved_by: + commit: "c55ae5ef" --- The identity matchers do not read the JSON solidus escape. A home path written with escaped forward slashes (\/Users\/LOGIN\/Desktop, \/home\/OTHER\/) raises neither home_path_self nor home_path_other, and a login on the generic list in it raises no local_username; a named login is still caught by its bare word. Only PHP json_encode and org.json write the escape; none of the encoders that feed abcd (Go encoding/json, JSON.stringify for transcripts, Python json) does. + +## Grounds + +- pursued: a home path spelled with the JSON solidus escape is found and masked wherever ScanText runs; an \/ or \u002f spelled home reaching the store verbatim would show it wrong From 5e8cf3d598a062e8bde02fa062eaed0a5bc30fb4 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:59:59 +0100 Subject: [PATCH 09/64] =?UTF-8?q?chore:=20resolve=20iss-2609251639261103?= =?UTF-8?q?=20=E2=80=94=20the=20Windows=20home=20spelling=20is=20a=20home?= =?UTF-8?q?=20path?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The fix is 40136106: the one home matcher reads :\Users\ at any escaping depth, with its boundary, allowlist and traversal walk. Refs: iss-2609261658553101 Resolves: iss-2609251639261103 Assisted-by: Claude:claude-opus-5-5 --- ...-path-other-and-home-path-self-read-no-windows-home.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md (51%) diff --git a/.abcd/work/issues/open/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md b/.abcd/work/issues/resolved/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md similarity index 51% rename from .abcd/work/issues/open/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md rename to .abcd/work/issues/resolved/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md index ca5a0dcd4..a356329f1 100644 --- a/.abcd/work/issues/open/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md +++ b/.abcd/work/issues/resolved/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md @@ -11,6 +11,14 @@ production_mode: hand-written found_at: "internal/adapter/scanner/identity.go" deferred_after: "v0.10.0" deferral_reason: "Not contained: a Windows spelling for home_path_other changes genericHomeRe, the path-segment byte class, the system-directory allowlist, the traversal walk and the lint audit rule that shares them, which is a lane of its own; the caller's own login in a Windows home, literal or JSON-escaped, is still reported hard_fail as local_username, so what the gap costs is the warn-level third-party path and the kind the caller's own home is reported under, not a caller leak." +resolution: "Fixed by 40136106: genericHomeRe gains the Windows alternative :\\Users\\ with each separator a backslash run, so the typed, JSON-escaped and doubly escaped spellings are home_path_other; the trailing boundary takes the backslash, the allowlist recognises the Windows Users root and gains Default and All beside Public (shared with the repolint privacy rule), the traversal walk reads a backslash run as one separator, and the home_path_other skip compares the caller's home against the match with runs collapsed. The caller's own home is home_path_self at any depth through the JSON-escape views of c55ae5ef (abcd builds for darwin and linux, so the caller's home is never itself a Windows path). Pinned by TestWindowsHomeIsAThirdPartyHomePath and TestWindowsOwnHomeIsHomeSelfAtAnyDepth (watched RED at 211b8853) and the false-positive guard TestWindowsSystemRootsAreNotHomes (watched failing under two mutations). The repolint privacy rule's own regexp is a twin that reads a single-backslash Windows home only; its escaped half is iss-2609261658553101." +impact: fix +resolved_by: + commit: "40136106" --- home_path_other and home_path_self read no Windows home spelling beyond the literal one. genericHomeRe (internal/adapter/scanner/identity.go) is POSIX-only, so C:\Users\OTHER\Desktop and its JSON-escaped spelling C:\\Users\\OTHER\\Desktop raise no home_path_other, and home_path_self matches the configured home verbatim, so the escaped spelling of the caller's own Windows home is reached only through local_username on its last segment. A Windows spelling threads through genericHomeRe, the path-segment byte class, the system-directory allowlist, the traversal walk and the lint audit rule that shares them, so it is not a contained change. + +## Grounds + +- pursued: a third party's Windows home, typed or escaped, is reported and masked, and a Windows system profile directory is not; an escaped Windows home of a third party reaching the store verbatim, or C:\Users\Public reported as a user, would show it wrong From aaf4dd3677ae390cf67ca6dde86ef5b2f751ea68 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:00:20 +0100 Subject: [PATCH 10/64] =?UTF-8?q?chore:=20resolve=20iss-2609090934372160?= =?UTF-8?q?=20=E2=80=94=20a=20short=20name=20in=20author=20metadata=20is?= =?UTF-8?q?=20kept=20on=20bytes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The fix is 7babc8d7: a short single-token real name standing in a person metadata field is kept by the byte scan; the binary-structured residue (EXIF Artist, a UTF-16 PDF string) is captured on its own. Refs: iss-2609261659051539 Resolves: iss-2609090934372160 Assisted-by: Claude:claude-opus-5-5 --- ...-a-short-single-token-real-name-is-dropped-on-bytes.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609090934372160-a-short-single-token-real-name-is-dropped-on-bytes.md (57%) diff --git a/.abcd/work/issues/open/iss-2609090934372160-a-short-single-token-real-name-is-dropped-on-bytes.md b/.abcd/work/issues/resolved/iss-2609090934372160-a-short-single-token-real-name-is-dropped-on-bytes.md similarity index 57% rename from .abcd/work/issues/open/iss-2609090934372160-a-short-single-token-real-name-is-dropped-on-bytes.md rename to .abcd/work/issues/resolved/iss-2609090934372160-a-short-single-token-real-name-is-dropped-on-bytes.md index d1cf9eaa1..49d33d434 100644 --- a/.abcd/work/issues/open/iss-2609090934372160-a-short-single-token-real-name-is-dropped-on-bytes.md +++ b/.abcd/work/issues/resolved/iss-2609090934372160-a-short-single-token-real-name-is-dropped-on-bytes.md @@ -12,6 +12,14 @@ found_at: "internal/adapter/scanner/scanner.go" related_issues: ["iss-2608291832160371"] deferred_after: "v0.10.0" deferral_reason: "ruling owed to the product thinker (away; run A 2026-09-25): Read PDF Author/EXIF structurally, or accept a false-positive rate for short single-token names on raw bytes?" +resolution: "Fixed by 7babc8d7, by the anchored context the record names as closing it: the byte scan keeps a short single-token real_name where a metadata key that names a person (author, artist, creator, lastModifiedBy) ends within 96 bytes before it — a PDF /Author, XMP dc:creator, pdf:Author or tiff:Artist across pretty-printed lines, a PNG text Author, an OOXML cp:lastModifiedBy inside a decoded zip — at its hard_fail floor, and drops it elsewhere as before. No threshold was lowered and no false-positive rate accepted: the same token incidental to binary content, a hundred times over or past the key's reach, still raises nothing. The threshold keeps counting bytes, with the reason stated (a chance collision needs that many specific bytes). Pinned by TestShortRealNameOnBytesIsKeptInAMetadataField and the corrected TestRealNameOnBytesByLiteralShape (which had pinned the defect), both watched RED at 211b8853. Residue, captured as iss-2609261659051539: EXIF's binary Artist tag and a UTF-16 PDF string carry no key text and need a structural read." +impact: fix +resolved_by: + commit: "7babc8d7" --- A real_name whose literal is a short single token is dropped by the byte scan as chance noise, so a banned name in binary metadata is not a finding there while the same name in text is. byteScanPolicy carries byteScanLongLiteral at eight, and the threshold counts bytes rather than runes, so a four-rune CJK name passes as long while a seven-byte Latin one does not. The consequence is a name in a PDF Author field or an EXIF artist tag reaching a published payload unreported. Decoding containers does not touch this: a decoded region is scanned with the same byte rules, so a short name inside a zip entry is dropped exactly as in raw bytes. The threshold exists to hold down a false-positive rate on binary bytes, where a short Latin token appears by chance, so simply lowering it trades one defect for another. Closing it needs an anchored context that makes a match meaningful rather than incidental, a PDF Author key or an EXIF artist tag read structurally, or an accepted false-positive rate stated as a decision. Detector: a banned short single-token name placed in a PDF Author field must be reported by the payload scan, and a byte-identical run over binary content carrying the same token incidentally must not flood the report. Split out of iss-2608291832160371, whose container half shipped separately; this half was never addressed by it. + +## Grounds + +- pursued: a short banned name stamped into a document's author metadata is a payload finding while the same token as incidental binary noise is not; a /Author (NAME) PDF passing the payload scan, or a binary with the token scattered through it raising real_name, would show it wrong From ef3a96381a200f8e7ac079dbc9412b16ee0aacd7 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:03:55 +0100 Subject: [PATCH 11/64] fix: never take the home directory as a session's repo root A home that is itself a git working tree (dotfiles in the home) made the home the git toplevel for every non-repo directory beneath it, so the rules root walk stopped at ~/.abcd and read it a second time as the REPO layer: the home's guard.json, which has no user layer at all, and its config.json over itself as the machine layer. Resolve now passes over the home in the walk, and a toplevel that IS the home takes the non-repo route (cwd, no walk) when nothing below it carries a .abcd. The user layer still reads ~/.abcd/rules.json, once, as the user layer. Decision taken in the lane (the run A orchestrator's brief rules the exclusion; the record's owed question was whether a home toplevel is a legitimate repo-scope root): only the home itself is excluded, not every ancestor of it. A toplevel that CONTAINS the home, the shape of a hermetic harness that points HOME inside its checkout, stays the root, because it is a repository git vouched for and its own .abcd is its own; the walk still skips the home on the way up to it. A session whose working directory IS the home keeps reading a .abcd there as cwd's, the working-directory read that stays a posture question (DECISIONS.md, 2026-09-25). Tests watched fail first: TestResolveRootNeverAdoptsTheHomeDirectory, TestResolveRootNeverAdoptsTheHomeThroughTheMarker and TestResolveRootNeverAdoptsTheHomeBeneathAnotherToplevel each resolved the temp HOME as the root before the change. Refs: iss-2609020219198779 Assisted-by: Claude:claude-opus-5-5 --- .../brief/05-internals/03-configuration.md | 16 ++- internal/core/rules/home_root_test.go | 136 ++++++++++++++++++ internal/core/rules/root.go | 56 ++++++-- 3 files changed, 191 insertions(+), 17 deletions(-) create mode 100644 internal/core/rules/home_root_test.go diff --git a/.abcd/development/brief/05-internals/03-configuration.md b/.abcd/development/brief/05-internals/03-configuration.md index 64d64bd57..d9ad08319 100644 --- a/.abcd/development/brief/05-internals/03-configuration.md +++ b/.abcd/development/brief/05-internals/03-configuration.md @@ -502,13 +502,15 @@ directory, and [adr-46](../../decisions/adrs/0046-persistence-never-weakens-the-verification-posture.md) treats home write as the ownership root. -One residual stays open and recorded rather than assumed shut: -**iss-2609020219198779**, the user scope when the home directory is itself a git -working tree. The toplevel for a session in a non-repo directory beneath such a -home is the home, so the user-scope `.abcd` governs it as the repo root too — its -`rules.json` as the repo layer as well as the user layer, and its `guard.json` and -`config.json` with it; closing it needs a decision on whether a home-directory -toplevel is a legitimate repo-scope root. +**The home directory is never a repo root.** Its `.abcd/` is the user layer, and +a home that is itself a git working tree (dotfiles in the home) is not thereby a +project. The walk passes over the home, and a toplevel that is the home resolves +like a directory outside any repository — the working directory, no walk — when +nothing below the home carries a `.abcd/`. So a session beneath such a home reads +`~/.abcd/rules.json` once, as the user layer, and never the home's `guard.json` +or `config.json` as a repository's own. A toplevel that contains the home — a +test harness that points `HOME` inside its checkout — stays the root, because it +is a repository git vouched for, and its own `.abcd/` stays its own. ## 1. Visibility-driven gitignore policy diff --git a/internal/core/rules/home_root_test.go b/internal/core/rules/home_root_test.go new file mode 100644 index 000000000..65e86e91e --- /dev/null +++ b/internal/core/rules/home_root_test.go @@ -0,0 +1,136 @@ +package rules + +import ( + "path/filepath" + "testing" + + "github.com/intentdriven/abcd/internal/core/guard" + "github.com/intentdriven/abcd/internal/gitutil" +) + +// versionControlledHome is the dotfiles-in-home fixture (iss-2609020219198779): +// a HOME that is itself a git working tree, carrying a user-scope .abcd whose +// guard.json switches the hazard registry off, and a plain directory beneath it +// that is not a repository of its own. It returns the home and that directory. +func versionControlledHome(t *testing.T) (home, plain string) { + t.Helper() + outer := mustDir(t, t.TempDir()) + home = filepath.Join(outer, "home") + gitInitAt(t, home) + t.Setenv("HOME", home) + plantConfiguration(t, home) + plain = mustDir(t, filepath.Join(home, "scratch", "notes")) + if top, err := gitutil.Run(plain, "rev-parse", "--show-toplevel"); err != nil || resolvedPath(top) != resolvedPath(home) { + t.Skipf("git does not name the home as the toplevel for the fixture (%q, %v)", top, err) + } + return home, plain +} + +// TestResolveRootNeverAdoptsTheHomeDirectory: a home under version control is +// not a project. The user-scope ~/.abcd is the USER layer (rules.json there is +// read as such whatever the root), and it must not govern a session a second +// time as the repo root — which is what adopting the home's git toplevel did, +// handing a plain directory beneath it the home's guard.json and config.json +// as though they were that directory's repository's own. +func TestResolveRootNeverAdoptsTheHomeDirectory(t *testing.T) { + home, plain := versionControlledHome(t) + + res := Resolve(plain) + if got := resolvedPath(res.Root); got == resolvedPath(home) { + t.Fatalf("Resolve(%q).Root = the home directory %q; a version-controlled home must not be a session's repo root", plain, got) + } + if res.Root != plain { + t.Errorf("Resolve(%q).Root = %q, want cwd with no walk (the non-repo route)", plain, res.Root) + } + if len(res.Notes) != 0 { + t.Errorf("declining the home as a repo root declines nothing the session should read; notes = %q", res.Notes) + } + reg, err := guard.Load(res.Root) + if err != nil { + t.Fatalf("guard.Load(%q): %v", res.Root, err) + } + if reg.Disabled { + t.Errorf("the home's guard.json governs a session beneath it: the hazard registry is switched off at %q", res.Root) + } +} + +// TestResolveRootNeverAdoptsTheHomeThroughTheMarker is the same bound on the +// fallback: with git unable to answer (off the PATH a hook runs under), the .git +// marker walk would otherwise find the home's own repository and adopt it. +func TestResolveRootNeverAdoptsTheHomeThroughTheMarker(t *testing.T) { + home, plain := versionControlledHome(t) + t.Setenv("PATH", "/nonexistent") + if out, err := gitutil.Run(plain, "rev-parse", "--show-toplevel"); err == nil { + t.Fatalf("git answered %q with an emptied PATH; the fixture did not stage the failure", out) + } + + res := Resolve(plain) + if got := resolvedPath(res.Root); got == resolvedPath(home) { + t.Fatalf("Resolve(%q).Root = the home directory %q through the .git marker", plain, got) + } + if res.Root != plain { + t.Errorf("Resolve(%q).Root = %q, want cwd with no walk", plain, res.Root) + } +} + +// TestResolveRootNeverAdoptsTheHomeBeneathAnotherToplevel: a repository whose +// toplevel CONTAINS the home (a hermetic harness that points HOME inside its +// checkout, or a whole-disk checkout) is the same shape one level up — the walk +// from a directory beneath the home passes through the home and would stop at +// ~/.abcd before it reached the toplevel. The home is skipped as a stop; the +// toplevel git named is still the root, because a repository git vouched for +// that CONTAINS the home is not the home, and its own .abcd stays its own. +func TestResolveRootNeverAdoptsTheHomeBeneathAnotherToplevel(t *testing.T) { + outer := mustDir(t, t.TempDir()) + top := filepath.Join(outer, "checkout") + gitInitAt(t, top) + home := mustDir(t, filepath.Join(top, ".home")) + t.Setenv("HOME", home) + plantConfiguration(t, home) + plain := mustDir(t, filepath.Join(home, "scratch")) + + res := Resolve(plain) + if got := resolvedPath(res.Root); got == resolvedPath(home) { + t.Fatalf("Resolve(%q).Root = the home directory %q; the walk must not stop at ~/.abcd", plain, got) + } + if got, want := resolvedPath(res.Root), resolvedPath(top); got != want { + t.Errorf("Resolve(%q).Root = %q, want the toplevel that contains the home, %q", plain, got, want) + } + reg, err := guard.Load(res.Root) + if err != nil { + t.Fatalf("guard.Load(%q): %v", res.Root, err) + } + if reg.Disabled { + t.Errorf("the home's guard.json governs a session beneath it: the hazard registry is switched off at %q", res.Root) + } + + // The checkout's own .abcd, above the home, is still the checkout's. + mustDir(t, filepath.Join(top, ".abcd")) + if got, want := resolvedPath(Resolve(plain).Root), resolvedPath(top); got != want { + t.Errorf("Resolve(%q).Root = %q, want the checkout's own .abcd at %q", plain, got, want) + } +} + +// TestResolveRootStillAdoptsARepositoryBeneathTheHome: the bound is on the +// home, not on everything under it. A real checkout inside a version-controlled +// home resolves its own toplevel and reads its own .abcd, exactly as before. +func TestResolveRootStillAdoptsARepositoryBeneathTheHome(t *testing.T) { + home, _ := versionControlledHome(t) + repo := filepath.Join(home, "src", "project") + gitInitAt(t, repo) + mustDir(t, filepath.Join(repo, ".abcd")) + sub := mustDir(t, filepath.Join(repo, "internal")) + + if got, want := resolvedPath(Resolve(sub).Root), resolvedPath(repo); got != want { + t.Errorf("Resolve(%q).Root = %q, want the checkout's own root %q", sub, got, want) + } + + // A nearer .abcd inside the home's working tree, below the home, still + // governs the directories beneath it: only the home itself is excluded. + project := mustDir(t, filepath.Join(home, "notes-project")) + mustDir(t, filepath.Join(project, ".abcd")) + deep := mustDir(t, filepath.Join(project, "drafts")) + if got, want := resolvedPath(Resolve(deep).Root), resolvedPath(project); got != want { + t.Errorf("Resolve(%q).Root = %q, want the nearest .abcd below the home, %q", deep, got, want) + } +} diff --git a/internal/core/rules/root.go b/internal/core/rules/root.go index 78f80d857..f7d7a0d86 100644 --- a/internal/core/rules/root.go +++ b/internal/core/rules/root.go @@ -58,15 +58,24 @@ var ownerUID = fsutil.OwnerUID // uid (`git init /tmp`, or a repository laid in a root-owned mode-1777 // directory), which the fallback refuses on OWNERSHIP — see // foreignOwnerRefusal, and TrustedRootsRelPath for the explicit opt-in that -// re-admits a foreign-uid checkout the caller means to trust. One residual -// stays open, recorded rather than silently assumed shut: +// re-admits a foreign-uid checkout the caller means to trust. // -// - iss-2609020219198779 — the user-scope ~/.abcd when the home directory is -// ITSELF a git working tree (dotfiles-in-home). The toplevel for a session -// in a non-repo directory beneath such a home is the home, so ~/.abcd -// governs it as the REPO layer as well as the user layer (spc-23), and its -// guard.json and config.json with it. Closing it needs a decision on -// whether a home-directory toplevel is a legitimate repo-scope root. +// The home directory is never a repo root (iss-2609020219198779). Its .abcd is +// the USER layer (spc-23), and a home that is itself a git working tree +// (dotfiles-in-home) would otherwise hand every non-repo directory beneath it +// the home's .abcd a second time, as the repo layer — its guard.json with it, +// which has no user layer at all, and its config.json as the repo layer over +// itself as the machine layer. So the walk +// passes over the home, a toplevel that IS the home takes the non-repo route +// (cwd, no walk) when nothing nearer carries a .abcd, and a toplevel that +// CONTAINS the home — a hermetic harness pointing HOME inside its checkout — +// stays the root, because it is a repository git vouched for and not the home. +// What this does not reach is a session whose working directory IS the home: +// the root is then cwd, as it is for any non-repo directory, and a .abcd at the +// working directory is read. That is the working-directory read the refusal +// below also leaves standing; making it refuse is a posture change, not a +// bound on the walk (iss-2609251522588539, and its entry of 2026-09-25 in +// .abcd/work/DECISIONS.md). // // "Not a repository" and "a repository git will not answer for" are DIFFERENT // outcomes and only the first resolves to cwd with no walk. abcd runs git under @@ -142,9 +151,17 @@ func Resolve(cwd string) Resolution { if real, err := filepath.EvalSymlinks(top); err == nil { top = real } + // The home is never a repo root (iss-2609020219198779): its .abcd is the + // USER layer, read as such by every loader that has one, and a home under + // version control is not thereby a project. So the walk passes over it, and + // a toplevel that IS the home takes the non-repo route once the walk finds + // nothing nearer. + home := resolvedHome() for inside(dir, top) { - if fi, err := os.Stat(filepath.Join(dir, ".abcd")); err == nil && fi.IsDir() { - return Resolution{Root: dir} + if dir != home { + if fi, err := os.Stat(filepath.Join(dir, ".abcd")); err == nil && fi.IsDir() { + return Resolution{Root: dir} + } } if dir == top { break @@ -155,9 +172,28 @@ func Resolve(cwd string) Resolution { } dir = parent } + if top == home { + return Resolution{Root: cwd} + } return Resolution{Root: top} } +// resolvedHome is the caller's home directory, symlink-resolved so it compares +// with the physical paths the walk climbs, or "" when there is none to name. It +// reads through userHomeDir, the lookup the user layer is read through, so the +// directory whose .abcd is the user layer and the directory the walk declines +// are always the same one. +func resolvedHome() string { + home, err := userHomeDir() + if err != nil || home == "" || !filepath.IsAbs(home) { + return "" + } + if real, err := filepath.EvalSymlinks(home); err == nil { + return real + } + return filepath.Clean(home) +} + // inside reports whether dir is top or lies beneath it. func inside(dir, top string) bool { if dir == top { From b01bcd5f17c8a0de5de9cf4f10d27521c634d291 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:04:22 +0100 Subject: [PATCH 12/64] =?UTF-8?q?chore:=20resolve=20iss-2609020219198779?= =?UTF-8?q?=20=E2=80=94=20the=20home=20is=20never=20a=20session's=20repo?= =?UTF-8?q?=20root?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609020219198779 Assisted-by: Claude:claude-opus-5-5 --- ...-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609020219198779-the-rules-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md (67%) diff --git a/.abcd/work/issues/open/iss-2609020219198779-the-rules-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md b/.abcd/work/issues/resolved/iss-2609020219198779-the-rules-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md similarity index 67% rename from .abcd/work/issues/open/iss-2609020219198779-the-rules-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md rename to .abcd/work/issues/resolved/iss-2609020219198779-the-rules-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md index b723991a8..0ca21e0bb 100644 --- a/.abcd/work/issues/open/iss-2609020219198779-the-rules-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md +++ b/.abcd/work/issues/resolved/iss-2609020219198779-the-rules-root-bound-from-ghsa-vvqc-3mv2-5p49-does-not-close.md @@ -11,6 +11,14 @@ production_mode: hand-written found_at: "internal/core/rules/root.go" deferred_after: "v0.10.0" deferral_reason: "ruling owed to the product thinker (away; run A 2026-09-25): Is a home-directory git toplevel a legitimate config scope (spc-23's user layer), or excluded outright?" +resolution: "The rules root resolver never takes the home directory as a session's repo root: the walk passes over the home, and a toplevel that is the home resolves like a non-repo directory (cwd, no walk) when nothing below the home carries a .abcd. A dotfiles-in-home session reads ~/.abcd/rules.json once, as the user layer, and never the home's guard.json or config.json as a repository's own. A toplevel that contains the home stays the root (a repository git vouched for). A session whose working directory is the home still reads a .abcd there as cwd's; that is the working-directory posture question, not this walk." +impact: fix +resolved_by: + commit: "ef3a9638" --- The rules-root bound from GHSA-vvqc-3mv2-5p49 does not close the user-scope ~/.abcd when the home directory is itself a git working tree (dotfiles-in-home). ResolveRoot bounds the walk at the toplevel, and for a session in a non-repo directory beneath a version-controlled home that toplevel IS the home, so the home-scope rules.json and guard.json still govern the session: injected rules, the kill switch and the guard registry all come from a file outside any project. Cross-UID /tmp is NOT closed either, contrary to what this record first said: git refuses on ownership and the .git-marker fallback then resolves the planted tree's OWN root, which is precisely what a foreign-uid `git init` in a shared directory wants — that residual is recorded separately as iss-2609020259564193 and needs an ownership policy. What is closed is the one-command plant: a bare `.git` file or an empty `.git` directory no longer bounds the fallback, because the marker must now look like a repository (a `.git` directory carrying HEAD, or a `.git` file beginning "gitdir: "). A version-controlled home and a foreign-uid repository both remain open. Evidence: ResolveRoot in internal/core/rules/root.go. The fix needs a decision first — whether a home-directory toplevel is a legitimate configuration scope (spc-23 plans a user layer that would make it one) or must be excluded outright — so nothing is changed here; the doc comment now states the exact scope and names this record. + +## Grounds + +- pursued: a session in a plain directory beneath a version-controlled HOME resolves to that directory, not the home, and the home's guard.json does not switch its guard off; a Resolve that returns the home (or a walk that stops at ~/.abcd from beneath it) would show it wrong From ec5ca74f8987b75566cc0c969433f662728bf609 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:05:27 +0100 Subject: [PATCH 13/64] fix: say what a refused root still reads at the working directory The ownership refusal's note told the session that the refused root's rules.json and guard.json were NOT read and that the bundled defaults stood in. Resolve returns the working directory as the root on that refusal, so a .abcd there is read: a session started at the refused root reads that root's configuration while being told it did not, and so does one beneath it that carries its own .abcd. 0434d475 corrected AGENTS.md and the marker block ahoy writes; the note itself, the configuration chapter and the install how-to still made the same claim. The note now names which of the two happened: from a directory with no .abcd the old wording stands, and where the working directory carries one it says the refusal bounds the walk, not the working directory, and that the .abcd there IS read. The posture change that would make that read refuse too stays deferred as recorded (DECISIONS.md, 2026-09-25). Test watched fail first: TestResolveRootRefusalSaysWhatItStillReads, both subtests (the note said "NOT read" and "fall back to the bundled defaults" while Load read the working directory's kill switch). Refs: iss-2609251522588539 Assisted-by: Claude:claude-opus-5-5 --- .../brief/05-internals/03-configuration.md | 13 ++-- docs/how-to/install.md | 9 ++- internal/core/rules/root.go | 19 ++++-- internal/core/rules/root_test.go | 63 +++++++++++++++++++ 4 files changed, 93 insertions(+), 11 deletions(-) diff --git a/.abcd/development/brief/05-internals/03-configuration.md b/.abcd/development/brief/05-internals/03-configuration.md index d9ad08319..7aaceff01 100644 --- a/.abcd/development/brief/05-internals/03-configuration.md +++ b/.abcd/development/brief/05-internals/03-configuration.md @@ -473,11 +473,16 @@ second falls back to the `.git` marker, under two bounds: | **ownership** | a marker root whose owner is not the caller. Shape alone is not a trust boundary: `git init` in a shared world-writable directory produces a genuine repository, and git's refusal on ownership is the same signal in that attack as in the legitimate foreign-uid case (iss-2609020259564193) | a declaration, once, per foreign-uid checkout | A refused root is refused **loudly and fail-closed**: the session resolves to its -own working directory with no walk, the bundled rule defaults (under the user -layer, which is the caller's own) and bundled hazard registry stand in for the -repository's, and every front door prints one line naming -the refused directory, the two uids, and the exact command that re-admits it +own working directory with no walk, and every front door prints one line naming +the refused directory, the two uids, what the session reads instead, and the +exact command that re-admits it ([`../../principles/loud-staging.md`](../../principles/loud-staging.md)). The +refusal bounds the walk, not the working directory. From a directory with no +`.abcd/` of its own, the bundled rule defaults (under the user layer, which is +the caller's own) and the bundled hazard registry stand in for the repository's. +A `.abcd/` at the working directory is still read, so a session started at the +refused root reads that root's configuration, and the line says so rather than +promising the defaults. The ownership bound applies only to the git-refused fallback: where git answers, the toplevel it named stands whoever owns it, because that is a repository git itself vouched for. diff --git a/docs/how-to/install.md b/docs/how-to/install.md index a59afb482..70dfbf929 100644 --- a/docs/how-to/install.md +++ b/docs/how-to/install.md @@ -118,9 +118,12 @@ for the session, and that root is never taken from a directory above your working tree. When `git` will not name the tree — a checkout owned by a different user account, a container bind mount, a shared CI checkout — the root is recovered from the `.git` marker instead, and a root your account does not -own is refused: the session falls back to its own working directory, the -bundled rule defaults and bundled hazard registry stand in for the -repository's, and one line names the directory refused. Laying out a real +own is refused: the session falls back to its own working directory, nothing +above it is read, and one line names the directory refused and what the session +reads instead. From a directory with no `.abcd/` of its own, the bundled rule +defaults and bundled hazard registry stand in for the repository's. The refusal +bounds the walk, not the working directory, so a session started at the refused +checkout itself still reads that checkout's `.abcd/`. Laying out a real repository in a shared directory anyone can write is otherwise enough to supply both, and no property of the tree tells that apart from a checkout that is honestly someone else's. If such a checkout is genuinely yours to trust, diff --git a/internal/core/rules/root.go b/internal/core/rules/root.go index f7d7a0d86..4b675e71f 100644 --- a/internal/core/rules/root.go +++ b/internal/core/rules/root.go @@ -287,13 +287,24 @@ func foreignOwnerRefusal(marker, cwd string) []string { if err != nil { because = "its owning uid could not be read (" + termsafe.Sanitize(err.Error()) + ")" } + // What the session reads instead depends on the working directory, because + // the refusal bounds the WALK and the root falls back to cwd: a .abcd there + // is read whatever lies above it (iss-2609251522588539). So the note says + // which of the two happened rather than promising the bundled defaults — at + // the refused root itself, cwd's .abcd IS the refused root's. + outcome := fmt.Sprintf("so %s and .abcd/guard.json there were NOT read and nothing above %s governs this session "+ + "(injected rules and the loader kill switch fall back to the bundled defaults under the user scope's %s, the hazard registry to the bundled defaults)", + RepoRelPath, termsafe.Sanitize(cwd), UserDisplayPath) + if fi, serr := os.Stat(filepath.Join(cwd, ".abcd")); serr == nil && fi.IsDir() { + outcome = fmt.Sprintf("so nothing above %s governs this session — but the refusal bounds the walk, not the working directory, "+ + "so the .abcd/ at %s itself IS read: its %s and .abcd/guard.json govern this session", + termsafe.Sanitize(cwd), termsafe.Sanitize(cwd), RepoRelPath) + } return append(notes, fmt.Sprintf( - "rules: REFUSED %s as this session's configuration root — %s, and git would not answer for it, "+ - "so %s and .abcd/guard.json there were NOT read and nothing above %s governs this session "+ - "(injected rules and the loader kill switch fall back to the bundled defaults under the user scope's %s, the hazard registry to the bundled defaults). "+ + "rules: REFUSED %s as this session's configuration root — %s, and git would not answer for it, %s. "+ "If that checkout really is yours to trust — a foreign-uid checkout, a container bind mount, a shared CI checkout — "+ "declare it once, from an account you control: mkdir -p ~/.abcd && printf '%%s\\n' '%s' >> %s", - termsafe.Sanitize(marker), because, RepoRelPath, termsafe.Sanitize(cwd), UserDisplayPath, + termsafe.Sanitize(marker), because, outcome, termsafe.Sanitize(marker), TrustedRootsDisplay)) } diff --git a/internal/core/rules/root_test.go b/internal/core/rules/root_test.go index f8c677ee2..03a65237b 100644 --- a/internal/core/rules/root_test.go +++ b/internal/core/rules/root_test.go @@ -625,3 +625,66 @@ func TestResolveRootMatchesADeclaredRootExactly(t *testing.T) { } assertPlantNotRead(t, res.Root) } + +// TestResolveRootRefusalSaysWhatItStillReads (iss-2609251522588539): the +// refusal bounds the WALK, not the working directory, so a session started at +// the refused root — or in a directory beneath it carrying its own .abcd — +// still reads the .abcd at its working directory. The note is the one account +// the user gets of what governs the session, so it must not say that +// configuration went unread, or that the bundled defaults stand in, when the +// loaders are about to read it. +func TestResolveRootRefusalSaysWhatItStillReads(t *testing.T) { + cases := []struct { + name string + cwd func(plant, victim string) string + }{ + {"session at the refused root", func(plant, _ string) string { return plant }}, + {"session beneath it with its own .abcd", func(_, victim string) string { + plantConfiguration(t, victim) + return victim + }}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + plant, victim, _ := foreignPlant(t) + ownedByAnother(t, plant) + cwd := tc.cwd(plant, victim) + + res := Resolve(cwd) + if res.Root != cwd { + t.Fatalf("Resolve(%q).Root = %q, want the working directory", cwd, res.Root) + } + rs, err := Load(res.Root) + if err != nil { + t.Fatalf("Load(%q): %v", res.Root, err) + } + if !rs.Disabled { + t.Fatalf("fixture: the working directory's rules.json was not read at %q", res.Root) + } + note := noteMentioning(res.Notes, "REFUSED") + if note == "" { + t.Fatalf("the refusal is silent; notes = %q", res.Notes) + } + for _, false_ := range []string{"NOT read", "fall back to the bundled defaults"} { + if strings.Contains(note, false_) { + t.Errorf("the note says %q while the working directory's .abcd is read: %s", false_, note) + } + } + if !strings.Contains(note, "IS read") { + t.Errorf("the note does not say the working directory's .abcd is read: %s", note) + } + }) + } +} + +// TestResolveRootRefusalBeneathTheRootStillSaysTheRootWentUnread: the other +// geometry keeps its wording — a plain directory beneath the refused root, with +// no .abcd of its own, reads nothing but the bundled defaults and the user layer. +func TestResolveRootRefusalBeneathTheRootStillSaysTheRootWentUnread(t *testing.T) { + plant, victim, _ := foreignPlant(t) + ownedByAnother(t, plant) + note := noteMentioning(Resolve(victim).Notes, "REFUSED") + if !strings.Contains(note, "NOT read") || strings.Contains(note, "IS read") { + t.Errorf("a plain directory beneath the refused root must be told the root's configuration went unread: %s", note) + } +} From 33ceaa131cb3e2446e2fd2f3b74d110145feb3c0 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:05:36 +0100 Subject: [PATCH 14/64] =?UTF-8?q?chore:=20resolve=20iss-2609251522588539?= =?UTF-8?q?=20=E2=80=94=20the=20refusal=20says=20what=20it=20still=20reads?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609251522588539 Assisted-by: Claude:claude-opus-5-5 --- ...nts-md-s-foreign-uid-roots-paragraph-says-a-refused.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609251522588539-agents-md-s-foreign-uid-roots-paragraph-says-a-refused.md (59%) diff --git a/.abcd/work/issues/open/iss-2609251522588539-agents-md-s-foreign-uid-roots-paragraph-says-a-refused.md b/.abcd/work/issues/resolved/iss-2609251522588539-agents-md-s-foreign-uid-roots-paragraph-says-a-refused.md similarity index 59% rename from .abcd/work/issues/open/iss-2609251522588539-agents-md-s-foreign-uid-roots-paragraph-says-a-refused.md rename to .abcd/work/issues/resolved/iss-2609251522588539-agents-md-s-foreign-uid-roots-paragraph-says-a-refused.md index eb823ce75..5652e1fdf 100644 --- a/.abcd/work/issues/open/iss-2609251522588539-agents-md-s-foreign-uid-roots-paragraph-says-a-refused.md +++ b/.abcd/work/issues/resolved/iss-2609251522588539-agents-md-s-foreign-uid-roots-paragraph-says-a-refused.md @@ -10,6 +10,14 @@ origin: researcher-authored production_mode: hand-written deferred_after: "v0.10.0" deferral_reason: "The posture fix (refuse the read at the working directory as well as the walk) is on the unmerged branch feat/ruled-security-forks (da3efea6) and spans the rules loader, the guard and the banlist; fix round 1 of the tier2 lane corrected the wording to the truth instead of landing a partial second form of it. Recorded in .abcd/work/DECISIONS.md, 2026-09-25." +resolution: "Every statement of the foreign-uid refusal now says what holds: the refusal bounds the walk, not the working directory, so a .abcd at the working directory is still read. 0434d475 corrected AGENTS.md and the marker block ahoy writes; ec5ca74f corrected the remaining three statements of the same claim, the refusal note itself (which now says the working directory's .abcd IS read where there is one), the configuration chapter and the install how-to. The posture change that would make the read at the working directory refuse too stays deferred as recorded in .abcd/work/DECISIONS.md on 2026-09-25; this record was the wording." +impact: fix +resolved_by: + commit: "ec5ca74f" --- AGENTS.md's Foreign-uid roots paragraph says a refused foreign-owned root sends the session to its own working directory on the bundled defaults, but rules.Resolve returns Resolution{Root: cwd} on that refusal (internal/core/rules/root.go:138), so when the working directory is the refused root, or lies inside it and carries its own .abcd/, that directory's .abcd/rules.json, .abcd/guard.json and .abcd/config/oracle-routing.json are read. The ownership gate bounds the upward walk, not the read at cwd. layered.RootsFor's comment states this correctly. The posture fix (an empty root with PerRepo false, so no per-repo .abcd is read in any geometry) is on the unmerged branch feat/ruled-security-forks (da3efea6), not on main. + +## Grounds + +- pursued: no surface tells a session its working directory's .abcd went unread while the loaders read it; a refusal note, AGENTS.md, the chapter or the how-to promising the bundled defaults at a refused root that carries a .abcd would show it wrong From aea2ae7e3d44b69e9b85dda04c186c751e80414c Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:21:13 +0100 Subject: [PATCH 15/64] fix: name the bundled guardrail entries an override withholds A rules.json list replaces the bundled list wholesale, so a repo that pinned PII's rules before iss-156 added the network-identifier rule kept the old set with no notice: a stale guardrail set that looks current. The merge stays per field (the documented behaviour a repo relies on to say a rule in its own words); the loss is no longer silent. For the guardrail domains, COMMITTING, LOAD and PII, Load compares every recall, alias and rule list an override set against the list this binary bundles, and adds a note naming each bundled entry left out and the file whose list is in force. The hook and `abcd rules` print it on stderr, never in the injected context or the --json document. Decisions taken in the lane: the check covers the three guardrail domains and not every bundled domain, because the others are conventions a repository restates in its own words (this repository's own INTENTS and ROADMAP overrides do exactly that), and a note on each deliberate restatement would teach the reader to skip the one that matters. The comparison is against the bundled list as shipped, not a hash recorded at override time, so it needs no new schema field. Whether security-bearing lists should union instead of replace, or take a replace-versus-extend marker, stays the product thinker's question, recorded with itd-117's finer-grained-merging follow-up. Two front-door tests read `rules --json` through the combined output, and the new stderr note broke their parse; they read stdout alone now, which is what the JSON contract is. Tests watched fail first: TestLoadNamesTheBundledSecurityRulesAnOverrideWithholds, TestLoadNamesTheBundledRecallAnOverrideWithholds and TestLoadNamesTheUserLayerThatWithholds (against a stub list, before noteWithheld existed); TestRulesNamesAWithheldGuardrailOnStderr and TestHookPromptRouterNamesAWithheldGuardrail on a scratch copy with the note switched off. Refs: iss-174, iss-156 Assisted-by: Claude:claude-opus-5-5 --- .../brief/05-internals/03-configuration.md | 10 ++ docs/reference/cli/commands.md | 9 +- internal/core/rules/rules.go | 87 +++++++++- internal/core/rules/user_layer_test.go | 6 + internal/core/rules/withheld_test.go | 150 ++++++++++++++++++ internal/surface/cli/cli.go | 9 +- internal/surface/cli/rules_provenance_test.go | 57 ++++++- internal/surface/cli/rules_user_layer_test.go | 2 +- 8 files changed, 324 insertions(+), 6 deletions(-) create mode 100644 internal/core/rules/withheld_test.go diff --git a/.abcd/development/brief/05-internals/03-configuration.md b/.abcd/development/brief/05-internals/03-configuration.md index 7aaceff01..493b546e6 100644 --- a/.abcd/development/brief/05-internals/03-configuration.md +++ b/.abcd/development/brief/05-internals/03-configuration.md @@ -446,6 +446,16 @@ makes that grain more visible; finer-grained merging, detecting a repo file that duplicates the user layer, and moving conventions out of per-project harness memory are all recorded in itd-117 as follow-up questions. +**A withheld guardrail is named.** Because a list replaces the bundled list, an +override written before a release added an entry keeps withholding that entry. +For the three guardrail domains — `PII`, `COMMITTING` and `LOAD` — the load +compares every recall, alias and rule list an override set against the list the +running binary bundles. It names each bundled entry left out, and the file whose +list is in force, on stderr from `abcd rules` and from the hook on every prompt. +The effective set is unchanged. Restating the entry keeps it; leaving the field +out inherits the bundled list. The other bundled domains are conventions a +repository restates in its own words, so a replacement there is not reported. + ## The rules root — which `.abcd/` governs a session The rules, the hazard registry and the per-repo config are read from ONE resolved diff --git a/docs/reference/cli/commands.md b/docs/reference/cli/commands.md index 9306de276..8c76a6a36 100644 --- a/docs/reference/cli/commands.md +++ b/docs/reference/cli/commands.md @@ -1847,7 +1847,14 @@ rules replaced, its state changed, or a custom domain declared — renders as "## NAME (user override)" or "## NAME (repo override)" here, in the injected block and in the hook's diagnostic, and carries "source": "user" or "repo" in --json; the last layer to name a domain labels it. An untouched bundled domain -renders bare and carries "source": "bundled". Read-only. +renders bare and carries "source": "bundled". + +A list an override sets replaces the bundled one, so an override can hold back +an entry abcd ships. For the guardrail domains (COMMITTING, LOAD, PII), every +bundled recall keyword, alias or rule that an override's list leaves out is +named on stderr, with the file that set the list, here and on every hook +prompt. To keep an entry, restate it in the list, or leave the field out to +inherit the bundled list. Read-only. ### `abcd site` diff --git a/internal/core/rules/rules.go b/internal/core/rules/rules.go index 476563824..fa9cee5b4 100644 --- a/internal/core/rules/rules.go +++ b/internal/core/rules/rules.go @@ -261,7 +261,92 @@ func Load(repoRoot string) (RuleSet, error) { // repo's file. return RuleSet{}, fmt.Errorf("rules: %s: %w", RepoRelPath, err) } - return merged, nil + var layers []overrideLayer + if haveUser { + layers = append(layers, overrideLayer{SourceUser, user}) + } + if haveRepo { + layers = append(layers, overrideLayer{SourceRepo, repo}) + } + return noteWithheld(merged, layers), nil +} + +// securityBearingDomains are the bundled domains whose entries are guardrails +// rather than house style: PII (secrets, local paths and network identifiers +// leaving the machine), COMMITTING (unasked pushes, bypassed hooks, AI +// attribution) and LOAD (orphaned load starving a live machine). An override +// that replaces one of their lists without an entry the binary ships is named +// on every load (noteWithheld); the other bundled domains are conventions a +// repo restates in its own words, and a note on each would teach the reader to +// skip the one that matters. Name-sorted, so the notes come out in a stable +// order. +var securityBearingDomains = []string{"COMMITTING", "LOAD", "PII"} + +// overrideLayer is one override file as Load read it, with the label of the +// layer it came from. +type overrideLayer struct { + source string + set RuleSet +} + +// noteWithheld records one note for every list of a security-bearing domain +// that an override replaced without some entry the bundled default carries +// (iss-174). Replacement stays per field — the documented merge, and the one a +// repo relies on to say a rule in its own words — so the effective set is not +// touched; what changes is that the loss is no longer silent. A repo that +// pinned PII's rules before a later release added one keeps its own list and +// is told, on every load, which bundled rule it is withholding and from which +// file, instead of carrying a stale guardrail set that looks current. +// +// The comparison is against the bundled list as this binary ships it, so an +// upgrade that adds an entry is named on the first load after it, with no +// record of what the override saw when it was written. The file named is the +// LAST layer that set the field, the one whose list is in force. +func noteWithheld(rs RuleSet, layers []overrideLayer) RuleSet { + fields := []struct { + name string + of func(Domain) []string + }{ + {"recall", func(d Domain) []string { return d.Recall }}, + {"aliases", func(d Domain) []string { return d.Aliases }}, + {"rules", func(d Domain) []string { return d.Rules }}, + } + for _, name := range securityBearingDomains { + have, ok := rs.Domains[name] + if !ok { + continue + } + bundled := defaultRuleSet.Domains[name] + for _, f := range fields { + file := "" + for _, l := range layers { + if od, ok := l.set.Domains[name]; ok && f.of(od) != nil { + file = LayerPath(l.source) + } + } + if file == "" { + continue + } + kept := make(map[string]bool, len(f.of(have))) + for _, e := range f.of(have) { + kept[e] = true + } + var withheld []string + for _, e := range f.of(bundled) { + if !kept[e] { + withheld = append(withheld, fmt.Sprintf("%q", e)) + } + } + if len(withheld) == 0 { + continue + } + rs.notes = append(rs.notes, fmt.Sprintf( + "rules: %s: domain %q replaces the bundled %q list and WITHHOLDS %d of its %d entries, a security guardrail abcd ships that this load does not carry: %s; "+ + "restate them in the override to keep them, or leave %q out to inherit the bundled list", + file, name, f.name, len(withheld), len(f.of(bundled)), strings.Join(withheld, ", "), f.name)) + } + } + return rs } // userHomeDir is the package's view of os.UserHomeDir, held as a var for the diff --git a/internal/core/rules/user_layer_test.go b/internal/core/rules/user_layer_test.go index 8f59d87f5..0f2bc8e63 100644 --- a/internal/core/rules/user_layer_test.go +++ b/internal/core/rules/user_layer_test.go @@ -82,6 +82,12 @@ func TestUserLayerAbsentChangesNothing(t *testing.T) { t.Fatal(err) } want := Merge(Defaults(), RuleSet{SchemaVersion: 1, Domains: map[string]Domain{"PII": {Rules: []string{"repo pii"}}}}) + // The override withholds every bundled PII rule, which Load names in a + // note (iss-174); the SET is what it always was. + if withheldNote(rs, "PII", "rules") == "" { + t.Fatalf("the withheld bundled PII rules are silent; notes = %q", rs.Notes()) + } + rs.notes = nil if !reflect.DeepEqual(rs, want) { t.Fatalf("no user file: the repo merge must be what it always was") } diff --git a/internal/core/rules/withheld_test.go b/internal/core/rules/withheld_test.go new file mode 100644 index 000000000..95954cd3e --- /dev/null +++ b/internal/core/rules/withheld_test.go @@ -0,0 +1,150 @@ +//go:build unix + +package rules + +import ( + "strings" + "testing" +) + +// networkRule is the PII rule iss-156 added to the bundled defaults — the +// upgrade a repo that pinned PII's rules before it never received. +const networkRule = "Never commit hostnames, IP addresses, MAC addresses, or other live network identifiers" + +// withheldNote returns the one note naming domain as withholding field, or "". +func withheldNote(rs RuleSet, domain, field string) string { + for _, n := range rs.Notes() { + if strings.Contains(n, `"`+domain+`"`) && strings.Contains(n, "WITHHOLDS") && strings.Contains(n, field) { + return n + } + } + return "" +} + +// TestLoadNamesTheBundledSecurityRulesAnOverrideWithholds (iss-174): a repo +// that pinned PII's rules before the network-identifier rule shipped keeps the +// old list — per-field replacement is the documented merge — but it may not do +// so SILENTLY. The load names the bundled rule the override is withholding, and +// the file that withholds it, so a stale security rule set never looks current. +func TestLoadNamesTheBundledSecurityRulesAnOverrideWithholds(t *testing.T) { + repo := t.TempDir() + writeRepoRules(t, repo, `{"schema_version":1,"domains":{"PII":{"rules":[ + "Never commit, print, or paste secrets, tokens, or .env contents; reference them by name.", + "Never put absolute local paths in anything that leaves the machine — use repo-relative paths.", + "Never name a private repo in commits, PRs, issues, or docs; describe it generically."]}}}`) + + rs, err := Load(repo) + if err != nil { + t.Fatalf("Load: %v", err) + } + if got := len(rs.Domains["PII"].Rules); got != 3 { + t.Fatalf("the merge semantics moved: PII carries %d rules, want the override's 3", got) + } + note := withheldNote(rs, "PII", "rules") + if note == "" { + t.Fatalf("the withheld bundled rule is silent; notes = %q", rs.Notes()) + } + for _, want := range []string{RepoRelPath, networkRule} { + if !strings.Contains(note, want) { + t.Errorf("the note does not name %q: %s", want, note) + } + } + if strings.Contains(note, "Never name a private repo") { + t.Errorf("the note names a bundled rule the override restates: %s", note) + } +} + +// TestLoadNamesTheBundledRecallAnOverrideWithholds: withholding recall is the +// quieter half — the rules are all there, but prompts that name the network no +// longer summon them. The keywords are named like the rules are. +func TestLoadNamesTheBundledRecallAnOverrideWithholds(t *testing.T) { + repo := t.TempDir() + writeRepoRules(t, repo, `{"schema_version":1,"domains":{"PII":{"recall":["secret","token","credential","pii","redact"]}}}`) + + rs, err := Load(repo) + if err != nil { + t.Fatalf("Load: %v", err) + } + note := withheldNote(rs, "PII", "recall") + if note == "" { + t.Fatalf("the withheld bundled recall keywords are silent; notes = %q", rs.Notes()) + } + for _, want := range []string{`"tailscale"`, `"hostname"`} { + if !strings.Contains(note, want) { + t.Errorf("the note does not name the withheld keyword %s: %s", want, note) + } + } + if strings.Contains(note, `"secret"`) { + t.Errorf("the note names a keyword the override keeps: %s", note) + } +} + +// TestLoadNamesTheUserLayerThatWithholds: the user scope's file replaces a +// field the same way, and the note must name the file that did it — the one to +// edit — rather than the repo's. +func TestLoadNamesTheUserLayerThatWithholds(t *testing.T) { + home := userHome(t) + writeUserRules(t, home, `{"schema_version":1,"domains":{"LOAD":{"rules":["Ask before a load experiment."]}}}`) + + rs, err := Load(t.TempDir()) + if err != nil { + t.Fatalf("Load: %v", err) + } + note := withheldNote(rs, "LOAD", "rules") + if note == "" { + t.Fatalf("the user layer's withholding is silent; notes = %q", rs.Notes()) + } + if !strings.Contains(note, UserDisplayPath) { + t.Errorf("the note does not name the user-scope file: %s", note) + } +} + +// TestLoadIsQuietWhenNothingIsWithheld: the note is about loss, not about +// overriding. An override that restates every bundled entry and adds its own, +// one that only changes state, and one that replaces a domain carrying no +// security rule all load without a word — a note on every deliberate override +// would teach the reader to skip the one that matters. +func TestLoadIsQuietWhenNothingIsWithheld(t *testing.T) { + bundled := Defaults().Domains["PII"] + superset := `{"schema_version":1,"domains":{"PII":{"rules":[` + quoteAll(append(append([]string(nil), bundled.Rules...), "Mind the widget.")) + + `],"recall":[` + quoteAll(append(append([]string(nil), bundled.Recall...), "widget")) + `]}}}` + for name, body := range map[string]string{ + "superset": superset, + "state only": `{"schema_version":1,"domains":{"PII":{"state":"dormant"}}}`, + "not a guardrail": `{"schema_version":1,"domains":{"ROADMAP":{"rules":["Our own roadmap rule."]}}}`, + } { + t.Run(name, func(t *testing.T) { + repo := t.TempDir() + writeRepoRules(t, repo, body) + rs, err := Load(repo) + if err != nil { + t.Fatalf("Load: %v", err) + } + for _, n := range rs.Notes() { + if strings.Contains(n, "WITHHOLDS") { + t.Errorf("unexpected note: %s", n) + } + } + }) + } +} + +// TestSecurityBearingDomainsAreBundled: the list of guarded domains names +// bundled domains only, so a rename in the defaults cannot quietly drop one +// from the check. +func TestSecurityBearingDomainsAreBundled(t *testing.T) { + d := Defaults() + for _, name := range securityBearingDomains { + if _, ok := d.Domains[name]; !ok { + t.Errorf("securityBearingDomains names %q, which the bundled defaults do not carry", name) + } + } +} + +func quoteAll(ss []string) string { + q := make([]string, len(ss)) + for i, s := range ss { + q[i] = `"` + strings.ReplaceAll(s, `"`, `\"`) + `"` + } + return strings.Join(q, ",") +} diff --git a/internal/surface/cli/cli.go b/internal/surface/cli/cli.go index e89c1c944..d1f41b6e9 100644 --- a/internal/surface/cli/cli.go +++ b/internal/surface/cli/cli.go @@ -1965,7 +1965,14 @@ rules replaced, its state changed, or a custom domain declared — renders as "## NAME (user override)" or "## NAME (repo override)" here, in the injected block and in the hook's diagnostic, and carries "source": "user" or "repo" in --json; the last layer to name a domain labels it. An untouched bundled domain -renders bare and carries "source": "bundled". Read-only.`, +renders bare and carries "source": "bundled". + +A list an override sets replaces the bundled one, so an override can hold back +an entry abcd ships. For the guardrail domains (COMMITTING, LOAD, PII), every +bundled recall keyword, alias or rule that an override's list leaves out is +named on stderr, with the file that set the list, here and on every hook +prompt. To keep an entry, restate it in the list, or leave the field out to +inherit the bundled list. Read-only.`, Args: cobra.MaximumNArgs(1), RunE: func(cmd *cobra.Command, args []string) error { cwd, err := os.Getwd() diff --git a/internal/surface/cli/rules_provenance_test.go b/internal/surface/cli/rules_provenance_test.go index 3ee6646b5..0c0154018 100644 --- a/internal/surface/cli/rules_provenance_test.go +++ b/internal/surface/cli/rules_provenance_test.go @@ -32,7 +32,7 @@ func TestRulesJSONCarriesSource(t *testing.T) { overrideRepo(t) for name, want := range map[string]string{"PII": "repo", "COMMITTING": "bundled"} { var got map[string]any - out := runCLI(t, "rules", name, "--json") + out := rulesJSON(t, "rules", name, "--json") if err := json.Unmarshal(out, &got); err != nil { t.Fatalf("rules %s --json not JSON: %v\n%s", name, err, out) } @@ -43,7 +43,7 @@ func TestRulesJSONCarriesSource(t *testing.T) { var bare struct { Domains []map[string]any `json:"domains"` } - out := runCLI(t, "rules", "--json") + out := rulesJSON(t, "rules", "--json") if err := json.Unmarshal(out, &bare); err != nil { t.Fatalf("rules --json not JSON: %v\n%s", err, out) } @@ -80,3 +80,56 @@ func TestHookPromptRouterDiagnosticNamesOverrides(t *testing.T) { t.Fatalf("diagnostic does not name the override:\n%s", errlog) } } + +// rulesJSON runs a `rules --json` form and returns its STDOUT alone, failing on +// an error. The load's notes (a withheld bundled guardrail, a skipped domain) +// go to stderr by design, so a fixture that overrides a security-bearing +// domain has a stderr line beside a stdout that must still parse as one +// document. +func rulesJSON(t *testing.T, args ...string) []byte { + t.Helper() + out, errb, err := runCLISplit(t, args...) + if err != nil { + t.Fatalf("execute %v: %v\n%s%s", args, err, out, errb) + } + return []byte(out) +} + +// TestRulesNamesAWithheldGuardrailOnStderr (iss-174) is the front-door half of +// the withheld-entry note: `abcd rules` names the bundled PII rules the repo's +// override withholds, on stderr, and --json stays one parseable document. +func TestRulesNamesAWithheldGuardrailOnStderr(t *testing.T) { + overrideRepo(t) + for _, args := range [][]string{{"rules"}, {"rules", "--json"}, {"rules", "PII", "--json"}} { + out, errb, err := runCLISplit(t, args...) + if err != nil { + t.Fatalf("%v: %v", args, err) + } + if !strings.Contains(errb, `domain "PII"`) || !strings.Contains(errb, "WITHHOLDS") || !strings.Contains(errb, "network identifiers") { + t.Errorf("%v: the withheld bundled PII rules are not named on stderr; stderr = %q", args, errb) + } + if strings.Contains(out, "WITHHOLDS") { + t.Errorf("%v: the note reached stdout:\n%s", args, out) + } + } +} + +// TestHookPromptRouterNamesAWithheldGuardrail (iss-174): the hook injects the +// override's PII rules, as the per-field merge says it must, and names the +// bundled rules that override withholds out of band — never in the +// model-facing context. +func TestHookPromptRouterNamesAWithheldGuardrail(t *testing.T) { + t.Setenv("ABCD_RULES_STATE_DIR", t.TempDir()) + repo := overrideRepo(t) + + out, errlog := runHook(t, hookInputJSON(t, "s1", repo, "do not leak the token"), "hook", "prompt-router") + if !strings.Contains(out, "printing secrets is fine in this repo") { + t.Errorf("the override's own PII rule must still inject; stdout:\n%s\nstderr:\n%s", out, errlog) + } + if !strings.Contains(errlog, `domain "PII"`) || !strings.Contains(errlog, "WITHHOLDS") { + t.Errorf("the withheld bundled PII rules must be named out of band; stderr = %q", errlog) + } + if strings.Contains(out, "WITHHOLDS") { + t.Errorf("the note must not reach the model-facing context:\n%s", out) + } +} diff --git a/internal/surface/cli/rules_user_layer_test.go b/internal/surface/cli/rules_user_layer_test.go index ac69c01f7..e14796c4a 100644 --- a/internal/surface/cli/rules_user_layer_test.go +++ b/internal/surface/cli/rules_user_layer_test.go @@ -43,7 +43,7 @@ func TestRulesVerbLabelsTheUserLayer(t *testing.T) { } for name, want := range map[string]string{"PII": "user", "HOUSE": "user", "ROADMAP": "repo", "COMMITTING": "bundled"} { var got map[string]any - out := runCLI(t, "rules", name, "--json") + out := rulesJSON(t, "rules", name, "--json") if err := json.Unmarshal(out, &got); err != nil { t.Fatalf("rules %s --json not JSON: %v\n%s", name, err, out) } From 141895658180fe72d4a073f21c9875bf432b4a23 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:21:25 +0100 Subject: [PATCH 16/64] =?UTF-8?q?chore:=20resolve=20iss-174=20=E2=80=94=20?= =?UTF-8?q?a=20withheld=20bundled=20guardrail=20is=20named=20on=20every=20?= =?UTF-8?q?load?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-174 Assisted-by: Claude:claude-opus-5-5 --- .../development/plans/2026-08-15-plugin-user-safety.md | 2 +- ...ules-override-withholds-bundled-default-upgrades.md | 10 +++++++++- 2 files changed, 10 insertions(+), 2 deletions(-) rename .abcd/work/issues/{open => resolved}/iss-174-rules-override-withholds-bundled-default-upgrades.md (62%) diff --git a/.abcd/development/plans/2026-08-15-plugin-user-safety.md b/.abcd/development/plans/2026-08-15-plugin-user-safety.md index 5fcae5d89..934794f00 100644 --- a/.abcd/development/plans/2026-08-15-plugin-user-safety.md +++ b/.abcd/development/plans/2026-08-15-plugin-user-safety.md @@ -90,7 +90,7 @@ once. Human-paired (the §4 gate is manual by design). 8. **[iss-148](../../work/issues/resolved/iss-148-guard-registry-coverage-gaps-found-while-wiring-itd-103-regi.md)** (minor) — registry coverage gaps; every entry lands fixture-first per the v0.5.0 plan's rule. -9. **[iss-174](../../work/issues/open/iss-174-rules-override-withholds-bundled-default-upgrades.md)** +9. **[iss-174](../../work/issues/resolved/iss-174-rules-override-withholds-bundled-default-upgrades.md)** (minor) — a repo's rules override silently withholds bundled security upgrades. diff --git a/.abcd/work/issues/open/iss-174-rules-override-withholds-bundled-default-upgrades.md b/.abcd/work/issues/resolved/iss-174-rules-override-withholds-bundled-default-upgrades.md similarity index 62% rename from .abcd/work/issues/open/iss-174-rules-override-withholds-bundled-default-upgrades.md rename to .abcd/work/issues/resolved/iss-174-rules-override-withholds-bundled-default-upgrades.md index 6a917ccef..8101a8dfb 100644 --- a/.abcd/work/issues/open/iss-174-rules-override-withholds-bundled-default-upgrades.md +++ b/.abcd/work/issues/resolved/iss-174-rules-override-withholds-bundled-default-upgrades.md @@ -9,6 +9,14 @@ found_during: "iss-156 adversarial review 2026-07-30" found_at: "internal/core/rules/rules.go" deferred_after: "v0.10.0" deferral_reason: "ruling owed to the product thinker (away; run A 2026-09-25): For security-bearing arrays in a repo override: union with the bundled entries, a replace-vs-extend marker, or a diagnostic naming withheld entries?" +resolution: "The silence is closed: for the guardrail domains (COMMITTING, LOAD, PII) Load names every bundled recall keyword, alias and rule an override's list leaves out, with the file whose list is in force, on stderr from abcd rules and from the hook on every prompt. The comparison is against the list the running binary bundles, so an upgrade that adds an entry is named on the first load after it. The merge stays per field, as documented. Whether security-bearing lists should union or take a replace-versus-extend marker stays the product thinker's question (itd-117's finer-grained-merging follow-up; DECISIONS.md 2026-09-26)." +impact: fix +resolved_by: + commit: "aea2ae7e" --- -mergeDomain replaces a domain's recall and rules arrays wholesale, so a repo that overrides either array silently withholds every later security upgrade to the bundled defaults: internal/core/rules/rules.go copies the override array over the base one instead of unioning, with no schema bump and no notice, so a repo that pinned PII recall or PII rules before iss-156 keeps the old set and never sees the new network keywords or the never-commit-network-identifiers rule line. Per-field merge is the documented and wanted behaviour for state, but for security-bearing arrays the quiet outcome is a stale ruleset that looks current. Options to weigh: union rather than replace for recall/aliases (additive, no loss), a distinct replace-vs-extend marker in the override, or at minimum a loud diagnostic in abcd rules naming which bundled entries an override is withholding. First activated by the iss-156 PII upgrade, which is why it is captured now. Prior disposition to weigh when triaging: iss-66 (resolved, rules-loader trust boundary) document-accepted the adjacent risk that an override can weaken a default guardrail domain (its P15); this entry is the distinct case of an override silently withholding later upgrades rather than deliberately silencing a domain. \ No newline at end of file +mergeDomain replaces a domain's recall and rules arrays wholesale, so a repo that overrides either array silently withholds every later security upgrade to the bundled defaults: internal/core/rules/rules.go copies the override array over the base one instead of unioning, with no schema bump and no notice, so a repo that pinned PII recall or PII rules before iss-156 keeps the old set and never sees the new network keywords or the never-commit-network-identifiers rule line. Per-field merge is the documented and wanted behaviour for state, but for security-bearing arrays the quiet outcome is a stale ruleset that looks current. Options to weigh: union rather than replace for recall/aliases (additive, no loss), a distinct replace-vs-extend marker in the override, or at minimum a loud diagnostic in abcd rules naming which bundled entries an override is withholding. First activated by the iss-156 PII upgrade, which is why it is captured now. Prior disposition to weigh when triaging: iss-66 (resolved, rules-loader trust boundary) document-accepted the adjacent risk that an override can weaken a default guardrail domain (its P15); this entry is the distinct case of an override silently withholding later upgrades rather than deliberately silencing a domain. + +## Grounds + +- pursued: a repo or user override that holds back a bundled PII, COMMITTING or LOAD entry is told which entry and which file on every load; an override withholding one with no stderr note, or a note reaching the injected context or the --json document, would show it wrong From 719803c39e704cd01670721f17a09648093f8b61 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:24:25 +0100 Subject: [PATCH 17/64] chore: defer iss-2609020219265817 past v0.11.0 on the ruling it owes No mechanical close leaves rules reading as they do. Every CommonMark construct that makes a rule body a heading has to be closed: ATX on a continuation line and on the first line (`- # x` is a list item holding a heading, which the record did not name), a setext underline, and an HTML h1-h6 block. That takes either a code-safe rendering (a relative indent of four or more, or a fence), which flattens every legitimately structured multi-line rule wherever the block is rendered, or a fence-aware escaper complete only by enumeration, which writes escapes into the raw text the model reads. Both change how rules read, so the choice is the product thinker's. The v0.10.0 grant lapsed at the v0.11.0 anchor; the renewal quotes the question and the record carries the first-line finding. Refs: iss-2609020219265817 Assisted-by: Claude:claude-opus-5-5 --- ...erulebody-indents-every-continuation-line-of-a-rule-b.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/.abcd/work/issues/open/iss-2609020219265817-sanitizerulebody-indents-every-continuation-line-of-a-rule-b.md b/.abcd/work/issues/open/iss-2609020219265817-sanitizerulebody-indents-every-continuation-line-of-a-rule-b.md index 2cc96b0d4..d9397b62a 100644 --- a/.abcd/work/issues/open/iss-2609020219265817-sanitizerulebody-indents-every-continuation-line-of-a-rule-b.md +++ b/.abcd/work/issues/open/iss-2609020219265817-sanitizerulebody-indents-every-continuation-line-of-a-rule-b.md @@ -9,8 +9,10 @@ found_during: "autonomous-run-2026-09-01" origin: researcher-authored production_mode: hand-written found_at: "internal/core/rules/rules.go" -deferred_after: "v0.10.0" -deferral_reason: "ruling owed to the product thinker (away; run A 2026-09-25): Are rule bodies escaped, fenced, or left as the host parser contract defines them?" +deferred_after: "v0.11.0" +deferral_reason: "ruling owed to the product thinker (run A lane drainS2, 2026-09-26): are rule bodies rendered code-safe for a markdown reader (a relative indent of four or more, or a fence, which neutralises every construct but flattens every legitimately structured multi-line rule wherever the block is rendered), escaped construct by construct (fence-aware backslashes on ATX headings, including the first line, setext underlines and HTML h1-h6 blocks, which put escapes into the raw text the model reads and is complete only by enumeration), or left to the host line-start contract as now? No mechanical close leaves rules reading as they do." --- sanitizeRuleBody indents every continuation line of a rule body by two spaces, which defuses the line-start contract the host-side parser splits on but not CommonMark: a two-space-indented heading line inside a list item still renders as a heading INSIDE that item. A repo-overridden domain can therefore put an unmarked heading in front of any reader that renders the injected block as markdown, wearing no repo-override label of its own. The line-start contract holds and this is already stronger than the pre-fix behaviour, which indented nothing at all; closing the CommonMark half needs a decision on whether rule bodies get escaped, fenced, or left as the parser contract defines them. Evidence: sanitizeRuleBody in internal/core/rules/rules.go. + +Addendum (autonomous run A, lane drainS2, 2026-09-26): the CommonMark half is wider than continuation lines. The FIRST line of a body is defused only for the line-start contract: in CommonMark `- # x` is a list item whose content is a heading, so a one-line rule forges a heading too. A setext underline (`===` or `---` under a text line) and an HTML block opening `

` do the same from any line, and an HTML block of that kind can interrupt a paragraph. That is why the fix is a ruling rather than a mechanical close: closing every construct means either a code-safe rendering that flattens legitimate structure, or a fence-aware escaper that is complete only by enumeration and changes the raw text the model reads. From 5cc6d1f789fa9e7f1d3ca557db67e5ea2e59ed7e Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:24:49 +0100 Subject: [PATCH 18/64] docs(record): record the rules-loader lane's rulings and the one still owed Refs: iss-2609020219198779, iss-174, iss-2609251522588539, iss-2609020219265817 Assisted-by: Claude:claude-opus-5-5 --- .abcd/work/DECISIONS.md | 1 + 1 file changed, 1 insertion(+) diff --git a/.abcd/work/DECISIONS.md b/.abcd/work/DECISIONS.md index d1f2e3ffb..56c912495 100644 --- a/.abcd/work/DECISIONS.md +++ b/.abcd/work/DECISIONS.md @@ -2561,3 +2561,4 @@ together (the script's header says why there is no escape hatch). - 2026-09-26 — v0.11.0 is published: autonomous run A approved the `release` environment at 07:14:41Z under ruling A2 and the releases ruling of 2026-09-25T08:04:52Z, after the merge queue, the verify job and main's own CI on the tagged commit 22997314 reported green (25 check runs succeeded, 5 skipped by design). The one red run on that commit was the site preview, which the run judged a deploy step and not a gate: site.yml's own header declares it non-gating, since release.yml calls it only after `release` and a failure there cannot change what was published. Its cause predates the release (iss-2609260709386741, major): since the root help was grouped, `site` is listed only under `abcd --help --agent`, and the workflow's probe read the plain listing, so every site run on main had refused since then, and v0.11.0's production render after publishing refused the same way. The release, its four binaries, the plugin archive (its digest equal to the catalog pin) and the attestations were published at 07:16:44Z and verified locally. The probe was fixed in #720, and the v0.11.0 site was redeployed through the workflow's documented emergency path, a dispatch from main naming the tag (run 36229830492), which rendered abcdev.app from the released binary and attached site.tar.gz to the release. - 2026-09-25 — itd-2609211913453478's acceptance criterion 4 ships under two readings the intent's scope line does not state. A glossary entry's `not_to_be_confused_with` passes when at least one member names a family row on the record-families page or the page itself, where the scope line says the field "may name only a family on the page"; the stricter reading would force nonsense pairs such as warm against intent, and the entries keep their real confusion pairs. The family-key rule (`record_family_key`, warn) reports a record frontmatter key only when the glossary already marks that word superseded or forbidden, so a brand-new grouping word with no row (the intent's own Mechanism case, e.g. an `initiative:` key) is not detected by construction, and the stores the page does not row (adr, rdi, dsp, rdg, adm, srp) are not reported. The six `grandfathered_at_phase` warnings on itd-20, 27, 28, 63, 69 and 72 are history and stay. Recorded for the product thinker to confirm or widen (autonomous run A, glossary lane review, orchestrator abcd-39). - 2026-09-26 — The lab store is keyed `~/.abcd/lab///`, with one `index.jsonl` registry per root-sha lane beside the lab homes (lane implementer, autonomous run A, on review-lab's third finding against spc-2609212141418943 for itd-2609212137128014). This supersedes two recorded texts: the spec's literal `~/.abcd/lab//` (scope item 1), and the 2026-08-31 lab-convention entry's hand-run keying `~/.abcd/lab/-/` with a single top-level `~/.abcd/lab/index.jsonl`, whose stated divergence from root-sha keying is withdrawn. Why: the intent's scope condition keys the store "as the other machine-scoped stores are", and the worktree and transcript stores key on the repository's root commit, because a checkout moves, is renamed and is cloned twice on one machine while its root commit does none of that; a lab's identity is still its intention, carried by its id `lab--` (the UTC mint time and the pin), so several labs share one baseline inside one lane. The hand-run labs that predate the verb stay where they are, beside the root-sha lanes, and the verb neither reads nor writes them or the top-level registry, so no real lab is moved or migrated by the change. A later text naming `~/.abcd/lab//` (the open spc-2609221011151661's `pairs.jsonl` among them) means the lab home inside its root-sha lane. +- 2026-09-26 — Three of the rules loader's security records close, and one stays owed to the product thinker (autonomous run A orchestrator's lane brief, taken by the implementer of lane drainS2). (1) The home directory is never a session's repo root (iss-2609020219198779, answering the owed question "is a home-directory git toplevel a legitimate config scope, or excluded outright?" as the brief rules it): its `.abcd/` is the user layer, so the root walk passes over the home and a toplevel that is the home resolves like a non-repo directory. The lane narrowed the brief's "or an ancestor of HOME": a toplevel that contains the home, the shape of a hermetic harness that points `HOME` inside its checkout, stays the root because git vouched for it and its own `.abcd/` is its own; only the stop at the home is removed. A session whose working directory is the home still reads a `.abcd/` there as the working directory's, the posture question recorded on 2026-09-25. (2) A bundled guardrail that an override withholds is named on every load (iss-174): for COMMITTING, LOAD and PII, each bundled recall keyword, alias or rule missing from a list an override set goes to stderr with the file whose list is in force, and the merge stays per field. The other bundled domains are left out because a repository restates them in its own words, and a note on every restatement would bury the one that matters. Still owed to the product thinker: whether security-bearing lists should union with the bundled entries or take a replace-versus-extend marker instead (itd-117's finer-grained-merging follow-up). (3) The foreign-uid refusal says what it still reads (iss-2609251522588539): the note, the configuration chapter and the install how-to now say that a `.abcd/` at the working directory is read, as AGENTS.md has since 0434d475. (4) iss-2609020219265817 is deferred past v0.11.0, not closed. Every CommonMark heading construct in a rule body (ATX on any line, the first line included, setext, and HTML h1-h6) can be closed only by a code-safe rendering that flattens legitimate structure, or by a fence-aware escaper that is complete only by enumeration and changes the raw text the model reads. So "escaped, fenced, or left to the line-start contract" is the product thinker's ruling. From b97efc0432fb3c6df9245e408c6deaecc4d25b3d Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:53:52 +0100 Subject: [PATCH 19/64] chore: capture the review-drainS2 home-identity and refusal-note findings The security review of the rules-loader lane found the home exclusion comparing path strings, so a case-variant HOME is adopted as the repo root, and the refusal note deciding "IS read" through a stat that follows a symlinked .abcd the loader then refuses. Refs: iss-2609261753285273, iss-2609261753290536 Assisted-by: Claude:claude-opus-5-5 --- ...root-resolver-s-home-exclusion-compares-path.md | 14 ++++++++++++++ ...ner-refusal-note-decides-whether-the-working.md | 14 ++++++++++++++ 2 files changed, 28 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md create mode 100644 .abcd/work/issues/open/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md diff --git a/.abcd/work/issues/open/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md b/.abcd/work/issues/open/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md new file mode 100644 index 000000000..885e34414 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261753285273" +slug: "the-rules-root-resolver-s-home-exclusion-compares-path" +severity: "minor" +category: "bug" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: review-drainS2" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/core/rules/root.go" +--- + +The rules root resolver's home exclusion compares path strings (root.go dir != home in the walk, top == home after it), so a HOME spelled as a case variant of the on-disk path (the home spelled with a lower-case first component where the case-insensitive APFS volume stores it capitalised) is not recognised: filepath.EvalSymlinks keeps the caller's case while git reports the on-disk case, and the version-controlled home is adopted as the repo root, its .abcd read a second time as the repo layer. Compare by file identity (os.Stat plus os.SameFile) through one helper at both sites. diff --git a/.abcd/work/issues/open/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md b/.abcd/work/issues/open/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md new file mode 100644 index 000000000..c09e4ed91 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261753290536" +slug: "the-foreign-owner-refusal-note-decides-whether-the-working" +severity: "nitpick" +category: "inconsistency" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: review-drainS2" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/core/rules/root.go" +--- + +The foreign-owner refusal note decides whether the working directory's .abcd is read with os.Stat, which follows a symlinked cwd/.abcd, while readRepoLayer Lstat-refuses a symlinked .abcd; in that edge the note tells the user the working directory's configuration IS read when the loader refuses it. Use Lstat plus IsDir so the note and the loader agree. From c742b4f694ea8201c62e273bc4e8871a26e58c8e Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:55:39 +0100 Subject: [PATCH 20/64] fix(rules): recognise the home by file identity, not by spelling The rules root resolver passes over the home in its walk and sends a toplevel that is the home down the non-repo route, but it asked both questions by comparing path strings. HOME is the caller's string and the walk climbs the physical path git reports, and filepath.EvalSymlinks keeps the caller's case, so a HOME spelled as a case variant of the on-disk path on a case-insensitive volume was not recognised: the version-controlled home became the repo root and its .abcd was read a second time as the repo layer. Both sites now ask one helper, homeMatcher, which stats the home once and compares each candidate with os.SameFile. The home is still read through userHomeDir, so the user layer and the declined directory stay the same one; a home that is absent or cannot be stat'd matches nothing, as an unset one did. TestResolveRootNeverAdoptsTheHomeAtAnySpelling pins the four spellings the review named (trailing slash, symlinked HOME, cwd through a symlinked HOME, case variant) at both sites, with and without a ~/.abcd. Watched fail first on a scratch copy of the unfixed tree: both case-variant subtests failed (the home adopted as the root), the other six passed as the review's probes had. The case-variant subtests skip, and say so, where the test filesystem is case-sensitive. Refs: iss-2609261753285273 Assisted-by: Claude:claude-opus-5-5 --- internal/core/rules/home_root_test.go | 84 +++++++++++++++++++++++++++ internal/core/rules/root.go | 36 ++++++++---- 2 files changed, 108 insertions(+), 12 deletions(-) diff --git a/internal/core/rules/home_root_test.go b/internal/core/rules/home_root_test.go index 65e86e91e..bec370148 100644 --- a/internal/core/rules/home_root_test.go +++ b/internal/core/rules/home_root_test.go @@ -1,7 +1,9 @@ package rules import ( + "os" "path/filepath" + "strings" "testing" "github.com/intentdriven/abcd/internal/core/guard" @@ -134,3 +136,85 @@ func TestResolveRootStillAdoptsARepositoryBeneathTheHome(t *testing.T) { t.Errorf("Resolve(%q).Root = %q, want the nearest .abcd below the home, %q", deep, got, want) } } + +// homeSpellings are the spellings of one home directory the exclusion has to +// see through (iss-2609261753285273): HOME is the caller's string, the walk +// climbs the physical path git reports, and the two name the same directory +// without being the same bytes. Each returns the HOME to set and the working +// directory to resolve from, given the home as created and a plain directory +// beneath it; link is a symlink to the home beside it. +var homeSpellings = []struct { + name string + spell func(t *testing.T, home, plain, link string) (homeEnv, cwd string) +}{ + {"trailing slash", func(_ *testing.T, home, plain, _ string) (string, string) { + return home + string(filepath.Separator), plain + }}, + {"symlinked HOME", func(_ *testing.T, _, plain, link string) (string, string) { + return link, plain + }}, + {"cwd through a symlinked HOME", func(t *testing.T, home, plain, link string) (string, string) { + rel, err := filepath.Rel(home, plain) + if err != nil { + t.Fatal(err) + } + return link, filepath.Join(link, rel) + }}, + {"case variant", func(t *testing.T, home, plain, _ string) (string, string) { + variant := filepath.Join(filepath.Dir(home), strings.ToUpper(filepath.Base(home))) + hi, herr := os.Stat(home) + vi, verr := os.Stat(variant) + if herr != nil || verr != nil || !os.SameFile(hi, vi) { + t.Skipf("the test filesystem is case-sensitive: %q names no directory, so a case-variant HOME cannot be staged here", variant) + } + return variant, plain + }}, +} + +// TestResolveRootNeverAdoptsTheHomeAtAnySpelling (iss-2609261753285273): the +// exclusion compares the home by file IDENTITY, not by spelling. A HOME with a +// trailing slash, reached through a symlink, or spelled as a case variant of +// the on-disk path on a case-insensitive volume names the same directory the +// walk arrives at, and a string comparison missed the last of those — the +// version-controlled home became the repo root and its .abcd was read a second +// time as the repo layer. Both sites are exercised: with a ~/.abcd the walk +// would stop at the home, and without one the toplevel IS the home. +func TestResolveRootNeverAdoptsTheHomeAtAnySpelling(t *testing.T) { + for _, planted := range []bool{true, false} { + site := "the walk passes over ~/.abcd" + if !planted { + site = "a toplevel that is the home takes the non-repo route" + } + for _, shape := range homeSpellings { + t.Run(site+"/"+shape.name, func(t *testing.T) { + outer := mustDir(t, t.TempDir()) + home := filepath.Join(outer, "home") + gitInitAt(t, home) + if planted { + plantConfiguration(t, home) + } + plain := mustDir(t, filepath.Join(home, "scratch", "notes")) + if top, err := gitutil.Run(plain, "rev-parse", "--show-toplevel"); err != nil || resolvedPath(top) != resolvedPath(home) { + t.Skipf("git does not name the home as the toplevel for the fixture (%q, %v)", top, err) + } + link := filepath.Join(outer, "link") + if err := os.Symlink(home, link); err != nil { + t.Fatal(err) + } + homeEnv, cwd := shape.spell(t, home, plain, link) + t.Setenv("HOME", homeEnv) + + res := Resolve(cwd) + if got := resolvedPath(res.Root); got == resolvedPath(home) { + t.Fatalf("HOME=%q: Resolve(%q).Root = the home directory %q; the home must not be a repo root at any spelling", homeEnv, cwd, got) + } + if res.Root != cwd { + t.Errorf("HOME=%q: Resolve(%q).Root = %q, want cwd with no walk (the non-repo route)", homeEnv, cwd, res.Root) + } + if len(res.Notes) != 0 { + t.Errorf("declining the home as a repo root declines nothing the session should read; notes = %q", res.Notes) + } + }) + } + } +} diff --git a/internal/core/rules/root.go b/internal/core/rules/root.go index 4b675e71f..77422d563 100644 --- a/internal/core/rules/root.go +++ b/internal/core/rules/root.go @@ -156,9 +156,9 @@ func Resolve(cwd string) Resolution { // version control is not thereby a project. So the walk passes over it, and // a toplevel that IS the home takes the non-repo route once the walk finds // nothing nearer. - home := resolvedHome() + isHome := homeMatcher() for inside(dir, top) { - if dir != home { + if !isHome(dir) { if fi, err := os.Stat(filepath.Join(dir, ".abcd")); err == nil && fi.IsDir() { return Resolution{Root: dir} } @@ -172,26 +172,38 @@ func Resolve(cwd string) Resolution { } dir = parent } - if top == home { + if isHome(top) { return Resolution{Root: cwd} } return Resolution{Root: top} } -// resolvedHome is the caller's home directory, symlink-resolved so it compares -// with the physical paths the walk climbs, or "" when there is none to name. It -// reads through userHomeDir, the lookup the user layer is read through, so the +// homeMatcher returns the predicate both home sites in Resolve ask: does this +// path name the caller's home directory? It answers by file IDENTITY — the +// home and the path are each stat'd and compared with os.SameFile — never by +// spelling (iss-2609261753285273). HOME is the caller's string and the walk +// climbs the physical path git reports; a trailing slash, a symlink and a case +// variant on a case-insensitive volume all name the home without matching its +// bytes, and filepath.EvalSymlinks keeps the caller's case, so a string +// comparison adopted a case-variant home as the repo root. The home is read +// through userHomeDir, the lookup the user layer is read through, so the // directory whose .abcd is the user layer and the directory the walk declines -// are always the same one. -func resolvedHome() string { +// are always the same one. With no home to name, or one that cannot be +// stat'd, nothing is the home. +func homeMatcher() func(path string) bool { + never := func(string) bool { return false } home, err := userHomeDir() if err != nil || home == "" || !filepath.IsAbs(home) { - return "" + return never } - if real, err := filepath.EvalSymlinks(home); err == nil { - return real + hi, err := os.Stat(home) + if err != nil { + return never + } + return func(path string) bool { + pi, err := os.Stat(path) + return err == nil && os.SameFile(hi, pi) } - return filepath.Clean(home) } // inside reports whether dir is top or lies beneath it. From 7b22cd16d3b867974cbd3d4e4cc54cadb76f3e7a Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:55:43 +0100 Subject: [PATCH 21/64] fix(rules): the refusal note asks about .abcd the way the loader does The foreign-owner refusal note says the working directory's .abcd "IS read" when one is there, and decided that with os.Stat, which follows a symlinked .abcd. readRepoLayer Lstat-refuses that shape, so in that edge the note told the user a layer governs the session that the loader then refuses. The check is Lstat plus IsDir, so the note and the loader give the same answer. TestResolveRootRefusalNeverSaysASymlinkedAbcdIsRead stages a refused root with a symlinked .abcd at the working directory, proves the loader refuses it, and requires the note to say NOT read. Watched fail first on a scratch copy of the unfixed tree (the note said IS read). Refs: iss-2609261753290536 Assisted-by: Claude:claude-opus-5-5 --- internal/core/rules/root.go | 5 ++++- internal/core/rules/root_test.go | 31 +++++++++++++++++++++++++++++++ 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/internal/core/rules/root.go b/internal/core/rules/root.go index 77422d563..ec71a6a79 100644 --- a/internal/core/rules/root.go +++ b/internal/core/rules/root.go @@ -307,7 +307,10 @@ func foreignOwnerRefusal(marker, cwd string) []string { outcome := fmt.Sprintf("so %s and .abcd/guard.json there were NOT read and nothing above %s governs this session "+ "(injected rules and the loader kill switch fall back to the bundled defaults under the user scope's %s, the hazard registry to the bundled defaults)", RepoRelPath, termsafe.Sanitize(cwd), UserDisplayPath) - if fi, serr := os.Stat(filepath.Join(cwd, ".abcd")); serr == nil && fi.IsDir() { + // Lstat, not Stat: the repo-layer read refuses a symlinked .abcd + // (readRepoLayer), so a note that followed the link would say a layer is + // read that the loader refuses (iss-2609261753290536). + if fi, serr := os.Lstat(filepath.Join(cwd, ".abcd")); serr == nil && fi.IsDir() { outcome = fmt.Sprintf("so nothing above %s governs this session — but the refusal bounds the walk, not the working directory, "+ "so the .abcd/ at %s itself IS read: its %s and .abcd/guard.json govern this session", termsafe.Sanitize(cwd), termsafe.Sanitize(cwd), RepoRelPath) diff --git a/internal/core/rules/root_test.go b/internal/core/rules/root_test.go index 03a65237b..f1050b7bb 100644 --- a/internal/core/rules/root_test.go +++ b/internal/core/rules/root_test.go @@ -688,3 +688,34 @@ func TestResolveRootRefusalBeneathTheRootStillSaysTheRootWentUnread(t *testing.T t.Errorf("a plain directory beneath the refused root must be told the root's configuration went unread: %s", note) } } + +// TestResolveRootRefusalNeverSaysASymlinkedAbcdIsRead (iss-2609261753290536): +// the note's "IS read" branch must ask the question the loaders ask. A .abcd at +// the working directory that is a symlink is refused by the repo-layer read +// (readRepoLayer Lstat-refuses it), so a note that followed the link and said +// its configuration governs the session would tell the user the opposite of +// what the loader does. +func TestResolveRootRefusalNeverSaysASymlinkedAbcdIsRead(t *testing.T) { + plant, victim, _ := foreignPlant(t) + ownedByAnother(t, plant) + elsewhere := mustDir(t, filepath.Join(filepath.Dir(plant), "elsewhere")) + plantConfiguration(t, elsewhere) + if err := os.Symlink(filepath.Join(elsewhere, ".abcd"), filepath.Join(victim, ".abcd")); err != nil { + t.Fatal(err) + } + + res := Resolve(victim) + if res.Root != victim { + t.Fatalf("Resolve(%q).Root = %q, want the working directory", victim, res.Root) + } + if _, err := Load(res.Root); err == nil { + t.Fatalf("fixture: the loader read a symlinked .abcd at %q; the note's claim rests on its refusing it", res.Root) + } + note := noteMentioning(res.Notes, "REFUSED") + if note == "" { + t.Fatalf("the refusal is silent; notes = %q", res.Notes) + } + if strings.Contains(note, "IS read") || !strings.Contains(note, "NOT read") { + t.Errorf("the note says a symlinked .abcd the loader refuses is read: %s", note) + } +} From 2cfc3eb6f4bc1fe4ba789cc322f84d46100ec6c7 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:55:59 +0100 Subject: [PATCH 22/64] =?UTF-8?q?chore:=20resolve=20iss-2609261753285273?= =?UTF-8?q?=20=E2=80=94=20the=20home=20is=20recognised=20by=20file=20ident?= =?UTF-8?q?ity?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261753285273 Assisted-by: Claude:claude-opus-5-5 --- ...-rules-root-resolver-s-home-exclusion-compares-path.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md (61%) diff --git a/.abcd/work/issues/open/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md b/.abcd/work/issues/resolved/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md similarity index 61% rename from .abcd/work/issues/open/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md rename to .abcd/work/issues/resolved/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md index 885e34414..2ad78b9c7 100644 --- a/.abcd/work/issues/open/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md +++ b/.abcd/work/issues/resolved/iss-2609261753285273-the-rules-root-resolver-s-home-exclusion-compares-path.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: review-drainS2" origin: researcher-authored production_mode: hand-written found_at: "internal/core/rules/root.go" +resolution: "Both home sites in rules.Resolve ask one helper, homeMatcher, which compares the home and the candidate by file identity (os.Stat plus os.SameFile), so a trailing-slash, symlinked or case-variant HOME is still the home." +impact: fix +resolved_by: + commit: "c742b4f6" --- The rules root resolver's home exclusion compares path strings (root.go dir != home in the walk, top == home after it), so a HOME spelled as a case variant of the on-disk path (the home spelled with a lower-case first component where the case-insensitive APFS volume stores it capitalised) is not recognised: filepath.EvalSymlinks keeps the caller's case while git reports the on-disk case, and the version-controlled home is adopted as the repo root, its .abcd read a second time as the repo layer. Compare by file identity (os.Stat plus os.SameFile) through one helper at both sites. + +## Grounds + +- pursued: a HOME naming the home directory at any spelling is never adopted as the repo root, pinned by TestResolveRootNeverAdoptsTheHomeAtAnySpelling at both sites; a case-variant or symlinked HOME resolving to the home as Root on a case-insensitive volume would show it wrong From 13357ecb7eb40144b0a64a8b5546a32a7e1d6bb7 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:56:01 +0100 Subject: [PATCH 23/64] =?UTF-8?q?chore:=20resolve=20iss-2609261753290536?= =?UTF-8?q?=20=E2=80=94=20the=20refusal=20note=20Lstats=20.abcd=20like=20t?= =?UTF-8?q?he=20loader?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261753290536 Assisted-by: Claude:claude-opus-5-5 --- ...eign-owner-refusal-note-decides-whether-the-working.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md (58%) diff --git a/.abcd/work/issues/open/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md b/.abcd/work/issues/resolved/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md similarity index 58% rename from .abcd/work/issues/open/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md rename to .abcd/work/issues/resolved/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md index c09e4ed91..19dc4b0a7 100644 --- a/.abcd/work/issues/open/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md +++ b/.abcd/work/issues/resolved/iss-2609261753290536-the-foreign-owner-refusal-note-decides-whether-the-working.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: review-drainS2" origin: researcher-authored production_mode: hand-written found_at: "internal/core/rules/root.go" +resolution: "The refusal note's working-directory branch uses Lstat plus IsDir, matching readRepoLayer's refusal of a symlinked .abcd, so the note never says a layer is read that the loader refuses." +impact: fix +resolved_by: + commit: "7b22cd16" --- The foreign-owner refusal note decides whether the working directory's .abcd is read with os.Stat, which follows a symlinked cwd/.abcd, while readRepoLayer Lstat-refuses a symlinked .abcd; in that edge the note tells the user the working directory's configuration IS read when the loader refuses it. Use Lstat plus IsDir so the note and the loader agree. + +## Grounds + +- pursued: a refused root with a symlinked .abcd at the working directory gets a note saying NOT read, pinned by TestResolveRootRefusalNeverSaysASymlinkedAbcdIsRead; a note saying IS read for a .abcd that Load refuses would show it wrong From 33715d8e1a8134122d946f5d0144bda74aadc6a5 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:11:08 +0100 Subject: [PATCH 24/64] fix(scanner): read escaped spellings of the home in the literal backstop SweepCallerHome and the user-segment rewrite behind SurvivingCallerHome matched the home only as written, so the solidus escape, its / form in either case, a second JSON layer and the percent-encoded separator all passed the one stage the store-before-commit redactors keep independent of the detector. Both now collect their spans through backstopSpans: the text as written plus the decoded views of every line carrying a backslash or a '%' (percentDecodeBounded and jsonEscapeLayers, the scanner's one definition), with each view hit mapped back through its position map so the rewrite covers whole escape units. An occurrence straight after an odd backslash run is the tail of an escape, so the raw reading leaves it to the view: judging it as written read the escape's backslash as a boundary, swept "x\/root" under HOME=/root, and rewriting from the '/' left a dangling "\~" that no JSON reader accepts. The anchors run on the text the occurrence was found in, so a control escape before a single-segment home is judged by its decoded newline. Cost stays linear: TestSweepCallerHomeWorkIsLinear holds five dense shapes to linearCostBar. Refs: iss-2609261659041553 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/residual.go | 162 ++++++++++++++---- .../adapter/scanner/residual_escape_test.go | 116 +++++++++++++ 2 files changed, 244 insertions(+), 34 deletions(-) create mode 100644 internal/adapter/scanner/residual_escape_test.go diff --git a/internal/adapter/scanner/residual.go b/internal/adapter/scanner/residual.go index f0019e7c0..4f67024e6 100644 --- a/internal/adapter/scanner/residual.go +++ b/internal/adapter/scanner/residual.go @@ -2,6 +2,7 @@ package scanner import ( "os" + "sort" "strings" ) @@ -56,32 +57,144 @@ func BlockingResidual(findings []Finding) []Finding { // into "~bc/x" under HOME=/home/a, silently corrupting the committed text; the // anchor is what lets a short home coexist with the paths that merely share // its prefix. An empty home sweeps nothing. +// +// The home is read in every spelling the detector reads (backstopSpans): as +// written, and through the decoded views of each line that carries an escape, +// so "\/Users\/me", its \u002f form, a second JSON layer and "%2FUsers%2Fme" abcd-lint:allow +// are swept with the escape's own bytes, exactly as the literal home is +// (iss-2609261659041553). func SweepCallerHome(text, home string) string { if home == "" { return text } - urls := urlSpans(text) - var b strings.Builder + return replaceSpans(text, backstopSpans(text, home, true, homeSweepable), "~") +} + +// backstopSpans returns, sorted and disjoint, the byte spans of text at which +// needle occurs and accept holds, read in the spellings the detector reads: as +// written, and through the decoded views of every line carrying a backslash or +// a '%' (the percent pre-pass's view and each JSON-escape layer, the one +// definition in percent.go and jsonescape.go). A hit on a view is mapped back +// through the view's position map, so the span covers the whole escape units +// it was decoded from and a rewrite never splits one. accept judges an +// occurrence on the text it was found in, so the anchors read the decoded +// neighbours of an escaped home rather than the escape's letters; wantURLs +// says whether it consults the URL spans, which are then found once per text. +// +// An occurrence written straight after an odd run of backslashes is the tail +// of an escape ("\/Users/me" read from its '/'), so the raw reading leaves it abcd-lint:allow +// to the view that decodes the escape: judging the tail as written read the +// escape's backslash as a boundary and swept "x\/root" under HOME=/root, and +// rewriting from the '/' left a dangling "\~" no JSON reader accepts. +func backstopSpans(text, needle string, wantURLs bool, accept func(s string, at, end int, urls urlSet) bool) []span { + out := needleOccurrences(text, needle, true, wantURLs, accept) + for start := 0; start < len(text); { + end := strings.IndexByte(text[start:], '\n') + if end < 0 { + end = len(text) + } else { + end += start + } + line := text[start:end] + if strings.IndexByte(line, '\\') >= 0 || strings.IndexByte(line, '%') >= 0 { + for _, v := range lineViews(line) { + for _, sp := range needleOccurrences(v.text, needle, false, wantURLs, accept) { + if rs, re, ok := mapDecodedSpan(v.posMap, sp.start, sp.end, len(line)); ok { + out = append(out, span{start + rs, start + re}) + } + } + } + } + start = end + 1 + } + return disjointSpans(out) +} + +// lineViews is every decoded view of one line the scan reads: the percent +// pre-pass's fully decoded copy and each JSON-escape layer, outermost first. +func lineViews(line string) []decodedView { + var views []decodedView + if decoded, posMap := percentDecodeBounded(line); posMap != nil { + views = append(views, decodedView{decoded, posMap}) + } + return append(views, jsonEscapeLayers(line)...) +} + +// needleOccurrences walks s for needle, keeping each occurrence accept holds +// and resuming past it, or one byte on where accept declines — the walk the +// sweep has always made. raw marks s as the text as written, where an +// occurrence behind an escape's backslash is left to the views. +func needleOccurrences(s, needle string, raw, wantURLs bool, accept func(s string, at, end int, urls urlSet) bool) []span { + if needle == "" || !strings.Contains(s, needle) { + return nil + } + var urls urlSet + if wantURLs { + urls = urlSpans(s) + } + var out []span from := 0 for { - i := strings.Index(text[from:], home) + i := strings.Index(s[from:], needle) if i < 0 { - break + return out } at := from + i - end := at + len(home) - if homeSweepable(text, at, end, urls) { - b.WriteString(text[from:at]) - b.WriteByte('~') + end := at + len(needle) + if !(raw && afterEscapeBackslash(s, at)) && accept(s, at, end, urls) { + out = append(out, span{at, end}) from = end continue } - b.WriteString(text[from : at+1]) from = at + 1 } - if from == 0 { +} + +// maxEscapeRunWalk bounds how far afterEscapeBackslash reads back. A run past +// it is judged even — the text as written decides, as it always did — so a +// crafted run of backslashes before every occurrence costs a constant each. +const maxEscapeRunWalk = 64 + +// afterEscapeBackslash reports whether the byte at is escaped: an odd run of +// backslashes stands right before it. +func afterEscapeBackslash(s string, at int) bool { + n := 0 + for j := at - 1; j >= 0 && s[j] == '\\' && n < maxEscapeRunWalk; j-- { + n++ + } + scanMeter.charge(stageIdentity, n) + return n%2 == 1 && n < maxEscapeRunWalk +} + +// disjointSpans sorts spans by start and unions the ones that overlap. Spans +// that merely touch stay apart, so two homes written back to back are two +// rewrites, as the literal sweep has always made them. +func disjointSpans(spans []span) []span { + sort.Slice(spans, func(i, j int) bool { return spans[i].start < spans[j].start }) + scanMeter.charge(stageIdentity, len(spans)*searchCost(len(spans))) + out := spans[:0] + for _, s := range spans { + if n := len(out); n > 0 && s.start < out[n-1].end { + out[n-1].end = max(out[n-1].end, s.end) + continue + } + out = append(out, s) + } + return out +} + +// replaceSpans rewrites each of the sorted, disjoint spans of text to repl. +func replaceSpans(text string, spans []span, repl string) string { + if len(spans) == 0 { return text } + var b strings.Builder + from := 0 + for _, sp := range spans { + b.WriteString(text[from:sp.start]) + b.WriteString(repl) + from = sp.end + } b.WriteString(text[from:]) return b.String() } @@ -195,29 +308,10 @@ func SurvivingCallerHome(text, home string) (string, []Finding) { // match "/Users/metoo" (a different, longer username), while "/Users/me." at abcd-audit:allow // a sentence end does. Only the trailing half — a leading anchor here would // trade the old refusal for a leak, since a name behind a host or under a -// longer root is still the caller's name. +// longer root is still the caller's name. The segment is read in the same +// spellings as the home (backstopSpans), escaped ones included. func sweepUserSegment(text, needle, repl string) string { - var b strings.Builder - from := 0 - for { - i := strings.Index(text[from:], needle) - if i < 0 { - break - } - at := from + i - end := at + len(needle) - if !nameContinues(text, end) { - b.WriteString(text[from:at]) - b.WriteString(repl) - from = end - continue - } - b.WriteString(text[from : at+1]) - from = at + 1 - } - if from == 0 { - return text - } - b.WriteString(text[from:]) - return b.String() + return replaceSpans(text, backstopSpans(text, needle, false, func(s string, _, end int, _ urlSet) bool { + return !nameContinues(s, end) + }), repl) } diff --git a/internal/adapter/scanner/residual_escape_test.go b/internal/adapter/scanner/residual_escape_test.go new file mode 100644 index 000000000..168acb417 --- /dev/null +++ b/internal/adapter/scanner/residual_escape_test.go @@ -0,0 +1,116 @@ +package scanner + +import ( + "strings" + "testing" +) + +// escapeSeparators spells every '/' of p with sep, so a fixture names a home +// path in an escaped spelling without the committed file carrying one. +func escapeSeparators(p, sep string) string { return strings.ReplaceAll(p, "/", sep) } + +// uSolidus and uSolidusUpper are the JSON \u escape of '/', assembled so no +// editor or tool reading the source can fold the six bytes back into a '/'. +var ( + uSolidus = `\` + "u002f" + uSolidusUpper = `\` + "u002F" +) + +// TestSweepCallerHomeReadsEscapedSpellings pins the literal-home backstop to +// the spellings the detector reads through its decoded views +// (iss-2609261659041553). The backstop exists for the case where both +// detector passes miss the home, and against an escaped home it read nothing: +// the JSON solidus escape, its / form in either case, a second JSON layer, +// the percent-encoded separator, and a control escape standing before a +// single-segment home all passed it verbatim. Each is collapsed to "~" exactly +// as the literal home is, and only the home's own bytes are rewritten. +func TestSweepCallerHomeReadsEscapedSpellings(t *testing.T) { + const home = "/Users/maya" // a registry persona home, abcd-lint:allow + cases := []struct{ name, home, in, want string }{ + {"solidus escape", home, escapeSeparators(home+"/x", `\/`), `~\/x`}, + {"unicode escape", home, escapeSeparators(home+"/x", uSolidus), "~" + uSolidus + "x"}, + {"upper-case unicode escape", home, escapeSeparators(home+"/x", uSolidusUpper), "~" + uSolidusUpper + "x"}, + {"second JSON layer", home, escapeSeparators(home+"/x", `\\\/`), `~\\\/x`}, + {"percent-encoded separator", home, escapeSeparators(home+"/x", `%2F`), `~%2Fx`}, + {"lower-case percent", home, escapeSeparators(home+"/x", `%2f`), `~%2fx`}, + {"mixed spellings", home, `\/Users/maya\/x`, `~\/x`}, + {"inside a JSON string", home, `{"cwd":"` + escapeSeparators(home, `\/`) + `"}`, `{"cwd":"~"}`}, + {"control escape before a single-segment home", "/root", `a\n/root/x`, `a\n~/x`}, + // The anchor holds in the decoded view exactly as on the literal line. + {"a longer name after an escaped home", home, escapeSeparators(home+"field", `\/`), escapeSeparators(home+"field", `\/`)}, + {"a single-segment home under another root", "/root", `x\/root\/y`, `x\/root\/y`}, + {"an escape with no home in it", home, `line one\nline two`, `line one\nline two`}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + if got := SweepCallerHome(c.in, c.home); got != c.want { + t.Errorf("SweepCallerHome(%q, home=%q) = %q, want %q", c.in, c.home, got, c.want) + } + }) + } +} + +// TestSweepCallerHomeKeepsEveryOtherLineIntact pins the sweep's reach across a +// multi-line text: an escaped home on one line is swept, and the lines around +// it, which carry escapes of their own, are left byte-for-byte as written. +func TestSweepCallerHomeKeepsEveryOtherLineIntact(t *testing.T) { + const home = "/Users/maya" // abcd-lint:allow + in := "first\\tline\n" + escapeSeparators(home+"/notes", `\/`) + "\nlast %41 line\n" + want := "first\\tline\n" + `~\/notes` + "\nlast %41 line\n" + if got := SweepCallerHome(in, home); got != want { + t.Errorf("SweepCallerHome = %q, want %q", got, want) + } +} + +// TestSurvivingCallerHomeReadsEscapedSpellings pins the second backstop to +// the same views: the caller's user segment behind another root is rewritten +// in its escaped spellings too, and an escaped home that survived is reported. +func TestSurvivingCallerHomeReadsEscapedSpellings(t *testing.T) { + const home = "/home/maya" // abcd-lint:allow + for _, sep := range []string{`\/`, uSolidus, `%2F`} { + in := escapeSeparators("/Users/maya/notes", sep) // abcd-lint:allow + out, _ := SurvivingCallerHome(in, home) + if strings.Contains(out, "maya") { + t.Errorf("SurvivingCallerHome(%q) kept the caller's name: %q", in, out) + } + } + if _, resid := SurvivingCallerHome(escapeSeparators("/opt/root/x", `\/`), "/opt/root"); len(resid) == 0 { + t.Error("an escaped caller home that survived was not reported") + } +} + +// TestSweepCallerHomeWorkIsLinear holds the sweep's decoded views to the cost +// class the line scan is held to: quadrupling a text dense in escaped homes, +// escape runs and percent triples at most multiplies the charged work by +// linearCostBar. +func TestSweepCallerHomeWorkIsLinear(t *testing.T) { + if raceEnabled { + t.Skip("a deterministic count gains nothing under -race; the uninstrumented run asserts it") + } + shapes := []struct{ name, home, unit string }{ + {"escaped homes", "/Users/zq8home", `\/Users\/zq8home\/x `}, // abcd-lint:allow + {"escaped single-segment homes", "/root", `\n\/root`}, + {"percent homes", "/root", `%2Froot`}, + {"escape runs", "/root", `\\\\\\\"\\\\n`}, + {"escaped homes across lines", "/root", "\\/root\n"}, + } + for _, s := range shapes { + t.Run(s.name, func(t *testing.T) { + charge := func(text string) int { + total := 0 + scanMeter.tally = func(_ string, n int) { total += n } + defer func() { scanMeter.tally = nil }() + SweepCallerHome(text, s.home) + return total + } + base := max(4096/len(s.unit), 1) + lo, hi := charge(strings.Repeat(s.unit, base)), charge(strings.Repeat(s.unit, 4*base)) + if lo == 0 { + t.Fatal("the shape charged nothing; it pins nothing") + } + if growth := float64(hi) / float64(lo); growth > linearCostBar { + t.Errorf("quadrupling the text multiplied the sweep's charge by %.2fx, want at most %.1fx", growth, linearCostBar) + } + }) + } +} From 930f57e0d71fbf48b74210d6a9517c32fbd5e8d7 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:11:24 +0100 Subject: [PATCH 25/64] =?UTF-8?q?chore:=20resolve=20iss-2609261659041553?= =?UTF-8?q?=20=E2=80=94=20the=20home=20backstop=20reads=20escaped=20spelli?= =?UTF-8?q?ngs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261659041553 Assisted-by: Claude:claude-opus-5-5 --- ...aller-home-backstop-sweepcallerhome-and.md | 14 ------------ ...aller-home-backstop-sweepcallerhome-and.md | 22 +++++++++++++++++++ 2 files changed, 22 insertions(+), 14 deletions(-) delete mode 100644 .abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md create mode 100644 .abcd/work/issues/resolved/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md diff --git a/.abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md b/.abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md deleted file mode 100644 index 429042967..000000000 --- a/.abcd/work/issues/open/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md +++ /dev/null @@ -1,14 +0,0 @@ ---- -schema_version: 1 -id: "iss-2609261659041553" -slug: "the-literal-caller-home-backstop-sweepcallerhome-and" -severity: "minor" -category: "security" -source: "agent-finding" -found_during: "autonomous run A resumed 2026-09-25" -origin: researcher-authored -production_mode: hand-written -found_at: "internal/adapter/scanner/residual.go" ---- - -The literal caller-home backstop (SweepCallerHome and SurvivingCallerHome, internal/adapter/scanner/residual.go) reads the home only as written, so an escaped spelling of it is invisible to the one stage the store-before-commit redactors keep independent of the detector: the solidus escape (\/ between segments), its \u002f form, and a separator written as an escape run. Stage one and the stage-two rescan read those spellings through the scanner's JSON-escape views since fix/drain-scanner-identity, so a leak needs both detector passes to miss before the backstop matters; but the backstop exists for exactly that case, and against an escaped home it is not a backstop. Detector: with HOME under the /Users root, SweepCallerHome collapses the solidus-escaped and the \u002f spelling of the home to the tilde, as it does the literal one. diff --git a/.abcd/work/issues/resolved/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md b/.abcd/work/issues/resolved/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md new file mode 100644 index 000000000..3ac9f48b4 --- /dev/null +++ b/.abcd/work/issues/resolved/iss-2609261659041553-the-literal-caller-home-backstop-sweepcallerhome-and.md @@ -0,0 +1,22 @@ +--- +schema_version: 1 +id: "iss-2609261659041553" +slug: "the-literal-caller-home-backstop-sweepcallerhome-and" +severity: "minor" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/residual.go" +resolution: "SweepCallerHome and the user-segment rewrite behind SurvivingCallerHome now collect their spans through backstopSpans, which reads the text as written and the scanner's decoded views of each line carrying a backslash or a percent sign (the percent pre-pass and every JSON-escape layer), mapping each view hit back to whole escape units. The solidus escape, the u002f form in either case, a second JSON layer, the percent-encoded separator and a control escape before a single-segment home are swept to the tilde as the literal home is; an occurrence right after an odd backslash run is judged by the view, so the backstop no longer leaves a dangling backslash or sweeps a single-segment home under another root. Tests: TestSweepCallerHomeReadsEscapedSpellings, TestSweepCallerHomeKeepsEveryOtherLineIntact, TestSurvivingCallerHomeReadsEscapedSpellings, and TestSweepCallerHomeWorkIsLinear for the cost class." +impact: fix +resolved_by: + commit: "33715d8e" +--- + +The literal caller-home backstop (SweepCallerHome and SurvivingCallerHome, internal/adapter/scanner/residual.go) reads the home only as written, so an escaped spelling of it is invisible to the one stage the store-before-commit redactors keep independent of the detector: the solidus escape (\/ between segments), its \u002f form, and a separator written as an escape run. Stage one and the stage-two rescan read those spellings through the scanner's JSON-escape views since fix/drain-scanner-identity, so a leak needs both detector passes to miss before the backstop matters; but the backstop exists for exactly that case, and against an escaped home it is not a backstop. Detector: with HOME under the /Users root, SweepCallerHome collapses the solidus-escaped and the \u002f spelling of the home to the tilde, as it does the literal one. + +## Grounds + +- pursued: every escaped spelling the detector reads through its views is also swept by the literal backstop; a home spelling that ScanText finds through a view but SweepCallerHome leaves verbatim would show it wrong From 01295f88be1994fd9364557f64394ee6b2875ad3 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:11:39 +0100 Subject: [PATCH 26/64] =?UTF-8?q?chore:=20capture=20iss-2609261811321435?= =?UTF-8?q?=20=E2=80=94=20a=20meter=20fixture=20lost=20its=20unicode=20esc?= =?UTF-8?q?apes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Refs: iss-2609261811321435 Assisted-by: Claude:claude-opus-5-5 --- ...ty-fixture-json-unicode-escaped-own-homes-in.md | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md diff --git a/.abcd/work/issues/open/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md b/.abcd/work/issues/open/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md new file mode 100644 index 000000000..7e4873816 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261811321435" +slug: "the-linearity-fixture-json-unicode-escaped-own-homes-in" +severity: "minor" +category: "bug" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/meter_test.go" +--- + +The linearity fixture json_unicode_escaped_own_homes in internal/adapter/scanner/meter_test.go is the literal home path, not its unicode-escaped spelling: the u002f escapes were folded back into slashes when the fixture was written, so TestScanLineWorkIsLinear never holds the JSON unicode-escape decode of a dense line of escaped homes to the cost bar its name claims, and the shape duplicates own_home_paths. Detector: the fixture's line carries the six-byte backslash-u002f escape, assembled so no tool can fold it again. From d8f5252d20b221200496d5c5b1372103067ddbb9 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:12:03 +0100 Subject: [PATCH 27/64] test(scanner): give the unicode-escaped meter fixture its escapes back The json_unicode_escaped_own_homes linearity fixture carried the literal home path: its six-byte escapes had been folded back into slashes when it was written, so TestScanLineWorkIsLinear held a plain path to the bar and never the unicode-escape decode the fixture names. The line is now assembled from escapeSeparators and uSolidus, which no tool can fold, and TestMeterFixturesCarryTheSpellingTheyName pins both escaped fixtures to the bytes their names promise. Refs: iss-2609261811321435 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/meter_test.go | 25 ++++++++++++++++++++++++- 1 file changed, 24 insertions(+), 1 deletion(-) diff --git a/internal/adapter/scanner/meter_test.go b/internal/adapter/scanner/meter_test.go index 4b875f150..c3a27ffe2 100644 --- a/internal/adapter/scanner/meter_test.go +++ b/internal/adapter/scanner/meter_test.go @@ -92,7 +92,7 @@ var meterFixtures = []meterFixture{ // the line, so a line dense in escapes, in nested escapes and in // escaped homes must still cost a constant number of passes. {"json_escaped_other_homes", Identity{}, rep(`\/home\/zqa\/x\n`)}, - {"json_unicode_escaped_own_homes", meterNamedID, rep(`/Users/zq8home/x `)}, + {"json_unicode_escaped_own_homes", meterNamedID, rep(escapeSeparators("/Users/zq8home/x ", uSolidus))}, // abcd-audit:allow {"json_nested_escape_runs", meterNamedID, rep(`\\\\\\\"\\\\n`)}, {"json_tokens_after_escapes", Identity{}, rep(`\n` + "ghp_" + strings.Repeat("a", 36))}, } @@ -208,3 +208,26 @@ func TestBoundedContextHelpersKeepTheFinding(t *testing.T) { t.Errorf("a footer linking a reserved documentation host was reported: %+v", f) } } + +// TestMeterFixturesCarryTheSpellingTheyName pins the escaped fixtures to the +// bytes their names promise (iss-2609261811321435). The unicode-escaped home +// fixture was once written as the literal home, its six-byte escapes folded +// back into slashes, so the linearity guard held a plain path to the bar and +// never the unicode-escape decode it names. +func TestMeterFixturesCarryTheSpellingTheyName(t *testing.T) { + want := map[string]string{ + "json_escaped_other_homes": `\/`, + "json_unicode_escaped_own_homes": uSolidus, + } + for _, fx := range meterFixtures { + if esc, ok := want[fx.name]; ok { + if !strings.Contains(fx.build(1), esc) { + t.Errorf("fixture %s is %q, which carries no %q", fx.name, fx.build(1), esc) + } + delete(want, fx.name) + } + } + for name := range want { + t.Errorf("no meter fixture named %s", name) + } +} From ce3eb8b13bfbb6069970816567c555de459671b3 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:12:05 +0100 Subject: [PATCH 28/64] =?UTF-8?q?chore:=20resolve=20iss-2609261811321435?= =?UTF-8?q?=20=E2=80=94=20the=20meter=20fixture=20spells=20its=20escapes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261811321435 Assisted-by: Claude:claude-opus-5-5 --- ...linearity-fixture-json-unicode-escaped-own-homes-in.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md (60%) diff --git a/.abcd/work/issues/open/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md b/.abcd/work/issues/resolved/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md similarity index 60% rename from .abcd/work/issues/open/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md rename to .abcd/work/issues/resolved/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md index 7e4873816..dbe7aafea 100644 --- a/.abcd/work/issues/open/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md +++ b/.abcd/work/issues/resolved/iss-2609261811321435-the-linearity-fixture-json-unicode-escaped-own-homes-in.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25" origin: researcher-authored production_mode: hand-written found_at: "internal/adapter/scanner/meter_test.go" +resolution: "The fixture is assembled from escapeSeparators and uSolidus, so the line carries the six-byte unicode escape the linearity guard is meant to hold to the bar; TestMeterFixturesCarryTheSpellingTheyName pins both escaped meter fixtures to the bytes their names promise." +impact: internal +resolved_by: + commit: "d8f5252d" --- The linearity fixture json_unicode_escaped_own_homes in internal/adapter/scanner/meter_test.go is the literal home path, not its unicode-escaped spelling: the u002f escapes were folded back into slashes when the fixture was written, so TestScanLineWorkIsLinear never holds the JSON unicode-escape decode of a dense line of escaped homes to the cost bar its name claims, and the shape duplicates own_home_paths. Detector: the fixture's line carries the six-byte backslash-u002f escape, assembled so no tool can fold it again. + +## Grounds + +- pursued: the unicode-escape decode of a dense line of escaped homes is now metered by TestScanLineWorkIsLinear; a fixture whose bytes again carry no escape would show it wrong, and the new test fails on exactly that From a8f0af6b7da7b2b5bdb6d89c21fe7523682eb62a Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:17:11 +0100 Subject: [PATCH 29/64] fix(reading): drop a CRLF pair whole when the redaction runs to the end The redactor splits on "\n", so each line's carriage return is the first half of the CRLF pair that ends it. When an excluded section is the last one, the drop takes the newline after the last kept line and the join left that line's carriage return behind alone. The verifier then refused the document for a lone CR the source does not carry, so a CRLF record whose excluded section came last could never be assembled. The carriage return now goes with its newline, which is what an LF document already loses at the same place: a CRLF document redacts to its LF twin's text, line endings aside, wherever the section sits. A CR ending the source's own last line had no newline to lose and is still refused. Output bytes change only for documents that were refused before, so the assembler version does not move. Refs: iss-2609251600019863 Assisted-by: Claude:claude-opus-5-5 --- internal/core/reading/floor_lines_test.go | 48 +++++++++++++++++++++++ internal/core/reading/project.go | 15 +++++++ 2 files changed, 63 insertions(+) diff --git a/internal/core/reading/floor_lines_test.go b/internal/core/reading/floor_lines_test.go index d5516f9e3..c13955c9c 100644 --- a/internal/core/reading/floor_lines_test.go +++ b/internal/core/reading/floor_lines_test.go @@ -26,3 +26,51 @@ func TestVerifyRedactionRefusesABOMHeadingAndALoneCarriageReturn(t *testing.T) { t.Errorf("a CRLF document was refused: %v", err) } } + +// A CRLF document is redacted to the bytes its LF twin is redacted to, line +// endings aside, wherever the excluded section sits (iss-2609251600019863). +// +// The redactor splits on "\n", so each line's carriage return is the first half +// of the CRLF pair that ends it. Dropping a section at the END of the document +// drops the newline after the last kept line, and the join left that line's +// carriage return behind with no newline: a lone CR the source does not carry, +// which the verifier then refused as the source's. The pair is dropped whole, +// which is what the LF document already loses there, so the refusal of a lone +// carriage return stays a statement about the source. +func TestACRLFDocumentRedactsLikeItsLFTwinWhereverTheSectionSits(t *testing.T) { + for name, doc := range map[string]string{ + "the section first": "## Private Notes\r\n\r\nsecret\r\n\r\n## Public\r\n\r\nkept\r\n", + "the section in the middle": "# A spec\r\n\r\nintro\r\n\r\n## Private Notes\r\n\r\nsecret\r\n\r\n" + + "## Public\r\n\r\nkept\r\n", + "the section last": "# A spec\r\n\r\nintro\r\n\r\n## Private Notes\r\n\r\nsecret\r\n", + "the section last, no blank above it": "# A spec\r\nintro\r\n## Private Notes\r\nsecret\r\n", + "the section last, no final ending": "# A spec\r\n\r\nintro\r\n\r\n## Private Notes\r\n\r\nsecret", + } { + out, err := redactExcluded("spc-x.md", doc, privateNotes) + if err != nil { + t.Errorf("%s: refused: %v", name, err) + continue + } + if strings.Contains(out, "secret") { + t.Errorf("%s: the excluded section travelled: %q", name, out) + } + if strings.Contains(strings.ReplaceAll(out, "\r\n", ""), "\r") { + t.Errorf("%s: the redaction carries a lone carriage return: %q", name, out) + } + lf, err := redactExcluded("spc-x.md", strings.ReplaceAll(doc, "\r\n", "\n"), privateNotes) + if err != nil { + t.Fatalf("%s: the LF twin was refused: %v", name, err) + } + if got := strings.ReplaceAll(out, "\r\n", "\n"); got != lf { + t.Errorf("%s: the CRLF redaction is %q, and the LF twin's is %q", name, got, lf) + } + } + + // The refusal stays true of a source that does carry one: a lone carriage + // return ending the last KEPT line is the document's own, not the drop's. + const own = "# A spec\r\n\r\n## Private Notes\r\n\r\nsecret\r\n\r\n# Next\r\n\r\nend\r" + if _, err := redactExcluded("spc-x.md", own, privateNotes); err == nil || + !strings.Contains(err.Error(), "lone carriage return") { + t.Errorf("a source ending in a lone carriage return was not refused for it: %v", err) + } +} diff --git a/internal/core/reading/project.go b/internal/core/reading/project.go index 114044e1b..1928271f5 100644 --- a/internal/core/reading/project.go +++ b/internal/core/reading/project.go @@ -119,12 +119,27 @@ func redactExcluded(rel, doc string, exclusions []Exclusion) (string, error) { } kept := make([]string, 0, len(lines)) + lastKept := -1 for i, line := range lines { if !drop[i] { kept = append(kept, line) + lastKept = i } } out := strings.Join(kept, "\n") + // A CRLF pair is kept or dropped whole. The split is on "\n", so a line's + // carriage return is the first half of the pair that ends it; when the drop + // runs to the end of the document it takes the newline after the last kept + // line, and the join left that line's carriage return behind alone — a lone + // CR the source does not carry, which the verifier then refused as the + // source's (iss-2609251600019863). The carriage return goes with its + // newline, which is what an LF document already loses at the same place, so + // the two line endings redact to the same text. A carriage return ending the + // document's own last line had no newline to lose and is left for the + // verifier to refuse. + if lastKept >= 0 && lastKept < len(lines)-1 { + out = strings.TrimSuffix(out, "\r") + } if err := verifyRedaction(rel, doc, out, keys, headings); err != nil { return "", err } From ae0b1fbbc178a7baefe2c9bb35e58b9ec105a7a2 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:17:36 +0100 Subject: [PATCH 30/64] =?UTF-8?q?chore:=20resolve=20iss-2609251600019863?= =?UTF-8?q?=20=E2=80=94=20a=20CRLF=20pair=20is=20dropped=20whole=20at=20th?= =?UTF-8?q?e=20tail?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609251600019863 Assisted-by: Claude:claude-opus-5-5 --- ...floor-a-crlf-document-whose-excluded-section-is-the.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609251600019863-reading-floor-a-crlf-document-whose-excluded-section-is-the.md (54%) diff --git a/.abcd/work/issues/open/iss-2609251600019863-reading-floor-a-crlf-document-whose-excluded-section-is-the.md b/.abcd/work/issues/resolved/iss-2609251600019863-reading-floor-a-crlf-document-whose-excluded-section-is-the.md similarity index 54% rename from .abcd/work/issues/open/iss-2609251600019863-reading-floor-a-crlf-document-whose-excluded-section-is-the.md rename to .abcd/work/issues/resolved/iss-2609251600019863-reading-floor-a-crlf-document-whose-excluded-section-is-the.md index e3ea3db85..233c6a8c1 100644 --- a/.abcd/work/issues/open/iss-2609251600019863-reading-floor-a-crlf-document-whose-excluded-section-is-the.md +++ b/.abcd/work/issues/resolved/iss-2609251600019863-reading-floor-a-crlf-document-whose-excluded-section-is-the.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25" origin: researcher-authored production_mode: hand-written found_at: "internal/core/reading/project.go" +resolution: "The redactor drops a CRLF pair whole when the drop runs to the end of the document, so a CRLF record whose excluded section is last redacts to its LF twin's text instead of ending in a lone carriage return the verifier refused. The verifier's lone-CR refusal is unchanged and still refuses a source that carries one." +impact: fix +resolved_by: + commit: "a8f0af6b7da7b2b5bdb6d89c21fe7523682eb62a" --- reading floor: a CRLF document whose excluded section is the LAST section is refused with 'ends line N with a lone carriage return': the redactor's output ends in a CR with no LF, and the verifier's lone-CR refusal then blames a CR the source does not carry. Fail-closed (no leak), but such a record can never be assembled. Fix: keep the tail's newline when the redactor drops the last section, or trim a trailing lone CR at EOF before the check; add the case to floor_lines_test's CRLF control. + +## Grounds + +- pursued: a CRLF document whose excluded section is first, in the middle or last redacts to its LF twin's bytes, line endings aside, with no lone CR; a lone CR in the redacted output of a source that has none, or a CRLF assembly refused for one, would show it wrong From 9088559086d6d4b5334a336257bd926c45d8b409 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:18:07 +0100 Subject: [PATCH 31/64] fix(reading): refuse a nested mapping behind every block indicator MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The floor refused a compact mapping nested in a block sequence by reading one `- ` and then a key, which covered the recorded spelling and none of its siblings. Each of these is an excluded key to YAML and travelled past the floor under a manifest asserting its refusal: a second sequence indicator (`- - origin: x`, with a space or a tab), a node property between the indicator and the key (`- &a origin: x`, `- !t origin: x`), an explicit key inside the entry (`- ? origin`), a compact mapping as an explicit key's value (`: origin: x`), and a single-pair mapping written straight into a flow sequence (`[origin: x]`). The whole run of `- ` and `: ` indicators is now read off first and what follows is judged as the start of a line would be: a tag, an anchor, an explicit key or a compact mapping there is refused whatever the key is named. A flow collection behind the indicators stays with the flow scan, because committed intents carry `- { date: …, reason: "…" }` history rows; the flow scan now also reads a key after `[`. No committed frontmatter carries any newly refused shape (only the eval fixture that exists to be refused). rawHTMLHeading's comment no longer implies the fence does more than skip an opener: fenced lines are joined as they stand and can supply an unfenced opener's title and hard bound, and the comment says so and why that is judged, not claimed away. Refs: iss-2608301237450573 Assisted-by: Claude:claude-opus-5-5 --- internal/core/reading/project.go | 84 +++++++++++++++++++++++---- internal/core/reading/project_test.go | 54 +++++++++++++++++ 2 files changed, 127 insertions(+), 11 deletions(-) diff --git a/internal/core/reading/project.go b/internal/core/reading/project.go index 1928271f5..b6a64e63e 100644 --- a/internal/core/reading/project.go +++ b/internal/core/reading/project.go @@ -202,8 +202,11 @@ var ( mdLinkRe = regexp.MustCompile(`\[([^\]]*)\]\([^)]*\)`) // explicitYAMLKeyRe matches YAML's explicit-key form, `? origin`. explicitYAMLKeyRe = regexp.MustCompile(`^\s*\?\s+["']?([A-Za-z_][A-Za-z0-9_-]*)["']?\s*$`) - // flowKeyRe matches a key inside a flow mapping, at top level or nested. - flowKeyRe = regexp.MustCompile(`[{,]\s*["']?([A-Za-z_][A-Za-z0-9_-]*)["']?\s*:`) + // flowKeyRe matches a key inside a flow mapping, at top level or nested, and + // a single-pair mapping written straight into a flow sequence: `[origin: x]` + // is a sequence holding the mapping {origin: x} to YAML, and a scan that + // wanted a `{` or a `,` before the key let it travel (iss-2608301237450573). + flowKeyRe = regexp.MustCompile(`[{\[,]\s*["']?([A-Za-z_][A-Za-z0-9_-]*)["']?\s*:`) // doubleQuotedKeyRe captures a double-quoted key's raw spelling, escapes and // all, so escapedQuotedKey can judge it. The whitespace is `\s`, YAML's own // class, because a carriage return between the key and its colon is a break @@ -224,12 +227,15 @@ var ( // bound an unclosed element has was missing there and the title was read // past the blank line into whatever followed (iss-2608301421380392). rawHeadingBoundRe = regexp.MustCompile(`(?is)|]*)?/?>|\n[ \t\r]*\n`) - // nestedMappingRe matches a block-sequence entry whose item opens a mapping: - // `- key:` at any indent, bare or quoted. A key nested that way is invisible - // to a reader anchored to the line, and the fix the records ask for is to - // refuse the NESTING rather than to learn one more spelling of the key - // (iss-2608301237450573, iss-2608301251398360). - nestedMappingRe = regexp.MustCompile(`^\s*-\s+(?:"[^"]*"|'[^']*'|[A-Za-z_][A-Za-z0-9_-]*)\s*:(\s|$)`) + // blockIndicatorsRe matches the run of block indicators a frontmatter line + // can open with before its content: sequence entries (`- `) and an explicit + // key's value (`: `), each followed by the whitespace YAML requires of it. + // What follows the run is a node of its own, so a key written there is nested + // and invisible to a reader anchored to the line (nestedBlockEntry). + blockIndicatorsRe = regexp.MustCompile(`^[ \t]*(?:[-:][ \t]+)+`) + // compactKeyRe matches a mapping indicator in a line whose quoted scalars are + // blanked: a colon before whitespace or the end of the line. + compactKeyRe = regexp.MustCompile(`:(\s|$)`) // flowExplicitKeyRe matches YAML's explicit-key indicator inside a flow // mapping: a `?` following `{` or `,`. Same class, same answer. flowExplicitKeyRe = regexp.MustCompile(`[{,]\s*\?`) @@ -1112,7 +1118,17 @@ func maskAngles(out []byte, from, to int) { // of it — the refusal has to name a line a human can go and look at. // // An opener sitting on a fenced line is skipped, so an example inside a code -// block still cannot fire. +// block still cannot fire. That is ALL the fence does here, and nothing more is +// claimed for it (iss-2608301237450573): the lines are joined as they stand, not +// blanked, so a fenced region below an unfenced opener is read as part of that +// opener's text. It can supply the title, which only adds text a refusal may +// name. And it can supply the hard BOUND — a closing tag or the next heading +// open written inside the fence ends the title there — while a renderer escapes +// a fence's text, so on the page that tag is literal text inside the heading +// and ends nothing. The shorter title is still judged, the heading on the page +// then carries the tag's own name as text, and a fence inside an HTML block, +// where a renderer reads the delimiter as markup, is refused on its own by +// fenceInHTMLBlock. // // The document is read TWICE: once as it stands, and once with its markup DATA // masked — see maskMarkupData — because the opener and the bound are structure @@ -1284,8 +1300,8 @@ func unresolvableFrontmatterShape(lines []string, fenced []bool) (int, string, b return i + 1, "a YAML tag", true case strings.HasPrefix(trimmed, "&"): return i + 1, "a YAML anchor", true - case nestedMappingRe.MatchString(lines[i]): - return i + 1, "a mapping nested in a block sequence", true + case nestedBlockEntry(lines[i]) != "": + return i + 1, nestedBlockEntry(lines[i]), true case flowExplicitKeyRe.MatchString(lines[i]): return i + 1, "an explicit key in a flow mapping", true case questionLineRe.MatchString(lines[i]) && !explicitYAMLKeyRe.MatchString(lines[i]): @@ -1295,6 +1311,52 @@ func unresolvableFrontmatterShape(lines []string, fenced []bool) (int, string, b return 0, "", false } +// nestedBlockEntry reports what a block-sequence entry, or an explicit key's +// value, opens on its own line when that is a node whose keys this package +// cannot resolve, or "" when it opens none. +// +// The refusal is of the NESTING, whatever the key is named, because a key +// written behind an indicator is invisible to every reader here that is anchored +// to the line (iss-2608301237450573, iss-2608301251398360). Reading one `- ` and +// then a key covered the recorded spelling and none of its siblings, and each of +// them was an excluded key to YAML that travelled under a manifest asserting its +// refusal: a second indicator (`- - origin: x`), a node property between the +// indicator and the key (`- &a origin: x`, `- !t origin: x`), an explicit key in +// the entry (`- ? origin`), and a compact mapping as an explicit key's value +// (`: origin: x`). So the whole run of indicators is read off first, and what +// follows it is judged as the start of a line would be. +// +// A FLOW collection behind the indicators is left to the flow scan, which reads +// its keys wherever they stand: committed intents carry their history as +// `- { date: …, reason: "…" }`, and refusing the nesting there would refuse the +// corpus. A sequence of scalars opens nothing — `- itd-183`, a quoted scalar +// holding a colon, a URL, whose colon is not followed by whitespace. +func nestedBlockEntry(line string) string { + run := blockIndicatorsRe.FindString(line) + if run == "" { + return "" + } + inner := line[len(run):] + within := "a block sequence" + if last := strings.TrimRight(run, " \t"); last[len(last)-1] == ':' { + within = "an explicit key's value" + } + switch { + case strings.HasPrefix(inner, "{"), strings.HasPrefix(inner, "["): + return "" + case strings.HasPrefix(inner, "!"): + return "a YAML tag in " + within + case strings.HasPrefix(inner, "&"): + return "a YAML anchor in " + within + case questionLineRe.MatchString(inner): + return "an explicit key nested in " + within + } + if bare, _ := blankQuoted(inner); compactKeyRe.MatchString(bare) { + return "a mapping nested in " + within + } + return "" +} + // floorFences reports, per line, whether that line is code to the floor's // verifier, and the opener of a fence left unclosed in the document's BODY — // the region after the frontmatter block closes. diff --git a/internal/core/reading/project_test.go b/internal/core/reading/project_test.go index ef9297bbe..be663bc81 100644 --- a/internal/core/reading/project_test.go +++ b/internal/core/reading/project_test.go @@ -452,3 +452,57 @@ func TestAFencedMarkupExampleIsNotTheShape(t *testing.T) { t.Errorf("the refusal does not name the shape: %v", err) } } + +// TestANestedMappingRefusesBehindEveryBlockIndicator is shape 3's class, not its +// one spelling (iss-2608301237450573). The refusal read one `- ` and then a key, +// so every other way of reaching a compact nested mapping travelled: a second +// sequence indicator, a node property between the indicator and the key, an +// explicit key inside the entry, an explicit key's value on its `:` line, and a +// single-pair mapping inside a flow sequence. Each is an `origin` key to YAML and +// was nothing to the floor, and the manifest asserted its refusal. +func TestANestedMappingRefusesBehindEveryBlockIndicator(t *testing.T) { + const pre, post = "---\nid: spc-1\n", "---\n\n# A record\n" + for name, front := range map[string]string{ + "a sequence of sequences": "links:\n - - origin: ABCD-WARM-ORIGIN\n", + "a tab after the indicator": "links:\n -\t- origin: ABCD-WARM-ORIGIN\n", + "an anchored entry": "links:\n - &a origin: ABCD-WARM-ORIGIN\n", + "a tagged entry": "links:\n - !t origin: ABCD-WARM-ORIGIN\n", + "an explicit key in an entry": "links:\n - ? origin\n : ABCD-WARM-ORIGIN\n", + "an explicit value's mapping": "? meta\n: origin: ABCD-WARM-ORIGIN\n", + "a flow pair in a flow sequence": "links: [origin: ABCD-WARM-ORIGIN]\n", + // Siblings refused before this change, kept refused. + "the recorded shape": "links:\n - origin: ABCD-WARM-ORIGIN\n", + "a quoted key in an entry": "links:\n - \"origin\": ABCD-WARM-ORIGIN\n", + "a flow mapping in an entry": "links:\n - {origin: ABCD-WARM-ORIGIN}\n", + "an anchored flow mapping": "base: &b {origin: ABCD-WARM-ORIGIN}\nuse: *b\n", + "a key under a bare indicator": "links:\n -\n origin: ABCD-WARM-ORIGIN\n", + "a multi-line flow mapping": "meta: {a: 1,\n origin: ABCD-WARM-ORIGIN}\n", + "a block scalar holding the key": "note: |\n origin: ABCD-WARM-ORIGIN\n", + "a quoted pair in a flow sequence": "links: [\"origin\": ABCD-WARM-ORIGIN]\n", + "a second key in a nested mapping": "links:\n - name: a\n origin: ABCD-WARM-ORIGIN\n", + "a flow pair after a flow sequence": "links: [a, origin: ABCD-WARM-ORIGIN]\n", + } { + err := refuses(t, "spc-1-a-record.md", pre+front+post, refusalKeys, refusalHeadings) + if err == nil { + t.Errorf("%s: admitted; the key is an origin key to YAML and travels", name) + continue + } + if !strings.Contains(err.Error(), "spc-1-a-record.md") { + t.Errorf("%s: the refusal does not name the document: %v", name, err) + } + } + + // The anti-vacuity half: what committed records carry is admitted. + for name, front := range map[string]string{ + "a sequence of scalars": "builds_on:\n - itd-183\n - \"itd-199\"\n", + "a flow sequence of scalars": "related: [itd-183, itd-199]\n", + "a URL in a flow sequence": "sources: [https://example.com/a]\n", + "an explicit key and a value": "? meta\n: a plain value\n", + "an entry under a bare dash": "builds_on:\n -\n itd-183\n", + } { + if err := refuses(t, "spc-1-a-record.md", pre+front+post, refusalKeys, refusalHeadings); err != nil { + t.Errorf("%s was refused: %v", name, err) + } + } +} + From b837544642713a4ab025d50c4daae3a9005bf68e Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:18:16 +0100 Subject: [PATCH 32/64] =?UTF-8?q?chore:=20resolve=20iss-2608301237450573?= =?UTF-8?q?=20=E2=80=94=20nested=20mappings=20refused=20behind=20every=20i?= =?UTF-8?q?ndicator?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2608301237450573 Assisted-by: Claude:claude-opus-5-5 --- ...ting-on-the-itd-183-branch-a-compact-nested-mapping.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2608301237450573-pre-existing-on-the-itd-183-branch-a-compact-nested-mapping.md (63%) diff --git a/.abcd/work/issues/open/iss-2608301237450573-pre-existing-on-the-itd-183-branch-a-compact-nested-mapping.md b/.abcd/work/issues/resolved/iss-2608301237450573-pre-existing-on-the-itd-183-branch-a-compact-nested-mapping.md similarity index 63% rename from .abcd/work/issues/open/iss-2608301237450573-pre-existing-on-the-itd-183-branch-a-compact-nested-mapping.md rename to .abcd/work/issues/resolved/iss-2608301237450573-pre-existing-on-the-itd-183-branch-a-compact-nested-mapping.md index 19edd0bde..eb10aef0b 100644 --- a/.abcd/work/issues/open/iss-2608301237450573-pre-existing-on-the-itd-183-branch-a-compact-nested-mapping.md +++ b/.abcd/work/issues/resolved/iss-2608301237450573-pre-existing-on-the-itd-183-branch-a-compact-nested-mapping.md @@ -7,6 +7,10 @@ category: "bug" source: "user-observation" found_during: "itd-183-round-9-security" found_at: "internal/core/reading/project.go" +resolution: "The floor refuses the nesting behind every block indicator, not one spelling of it: the run of sequence and explicit-value indicators is read off and a tag, anchor, explicit key or compact mapping after it is refused whatever the key is named, and the flow scan reads a key after an opening bracket. The recorded shape and its siblings (a second indicator, a node property, an explicit key in the entry, a compact explicit value, a flow pair in a flow sequence) are refused; rawHTMLHeading's comment states only what the fence does there." +impact: fix +resolved_by: + commit: "9088559086d6d4b5334a336257bd926c45d8b409" --- pre-existing on the itd-183 branch: a compact nested mapping in a block sequence leaks an excluded key, and rawHTMLHeading's fence comment overclaims @@ -37,3 +41,7 @@ does not exist on main, so nothing here is inherited from main. Both are pre-existing on the branch and are left open for the facilitator. Item 1 in particular deserves a decision: it is a genuine hole in the exclusion floor that no round has closed. + +## Grounds + +- pursued: every compact nested mapping a frontmatter line can open behind block indicators refuses the assembly, while committed flow-mapping history rows and scalar sequences are admitted; an excluded key admitted in any such spelling, or a committed record refused by the new rule, would show it wrong From 85f1cff9c45196026a5b5095b4e4b33ae7bd9f2a Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:18:29 +0100 Subject: [PATCH 33/64] fix(reading): state only what is known when refusing a frontmatter key MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The escaped-key refusal reported through the excluded-key message, which asserted that the document still carried an excluded key and that its frontmatter block was not closed the way the field reader expects. Neither is known of an escape: the package does not decode one, so which key it spells is exactly what it cannot say, and the block may be closed as expected — `"C:\tmp\x": v`, with only `origin` excluded, was refused naming a key that is not on the list. The escape now has its own message naming the key's raw spelling, the line and the ground: an escape this package declines to decode. The excluded-key message drops the same block-closure claim. A quoted, indented or flow spelling survives redaction in a block closed exactly as expected; what is known is that the field reader did not report it as the key, so the redactor did not remove it. Both refusals stand; only their stated reasons change. Refs: iss-2608301421381157 Assisted-by: Claude:claude-opus-5-5 --- internal/core/reading/project.go | 40 +++++++++++++++++------ internal/core/reading/project_test.go | 47 +++++++++++++++++++++++++++ 2 files changed, 77 insertions(+), 10 deletions(-) diff --git a/internal/core/reading/project.go b/internal/core/reading/project.go index b6a64e63e..4df641d94 100644 --- a/internal/core/reading/project.go +++ b/internal/core/reading/project.go @@ -405,9 +405,23 @@ func verifyRedaction(rel, original, redacted string, keys, headings map[string]b "whose keys this package cannot resolve without becoming a YAML parser; a record has no "+ "reason to use one, so it is refused rather than guessed at", rel, shape, line) } - if line, key, ok := excludedKeyInFirstBlock(lines, fenced, keys); ok { + // Each message states what the scan observed and nothing more. The one + // message both findings shared asserted an excluded key and a block closed + // the wrong way, and for an escape neither is known: the key is undecoded + // and the block may be closed exactly as expected (iss-2608301421381157). + // Nor is a block shape known for an excluded key: a quoted, indented or + // flow spelling survives in a block closed as expected, because the field + // reader reports none of them as the key and the redactor removes only what + // it reports. + if line, key, escaped, ok := excludedKeyInFirstBlock(lines, fenced, keys); ok { + if escaped { + return fmt.Errorf("reading: %s spells the double-quoted frontmatter key %q at line %d with a "+ + "YAML escape; this package does not decode escapes, so which key it names is unknown, "+ + "and it is refused rather than guessed at", rel, key, line) + } return fmt.Errorf("reading: %s still carries the excluded key %q at line %d after redaction; "+ - "the frontmatter block is not closed the way the field reader expects it", rel, key, line) + "the field reader did not report it as a key, so the redactor did not remove it", + rel, key, line) } } if len(headings) == 0 { @@ -1599,10 +1613,16 @@ func firstBlockRange(lines []string, fenced []bool) (int, int, bool) { // dashes delimits it, not an exact `---`, because that is the rule the // frontmatter stripper applies and the gap between the two rules is where a key // survives. -func excludedKeyInFirstBlock(lines []string, fenced []bool, keys map[string]bool) (int, string, bool) { +// +// The third return says the finding is an ESCAPED double-quoted key rather than +// an excluded one. The two are refused for different reasons and are reported +// that way: an excluded key is a name this package read, while an escape is a +// name it declined to decode, so which key it spells is exactly what it does not +// know (iss-2608301421381157). +func excludedKeyInFirstBlock(lines []string, fenced []bool, keys map[string]bool) (int, string, bool, bool) { open, closed, ok := firstBlockRange(lines, fenced) if !ok { - return 0, "", false + return 0, "", false, false } end := len(lines) if closed >= 0 { @@ -1625,12 +1645,12 @@ func excludedKeyInFirstBlock(lines []string, fenced []bool, keys map[string]bool } { for _, key := range submatches(m) { if keys[key] { - return i + 1, key, true + return i + 1, key, false, true } } } if key, ok := escapedQuotedKey(lines[i]); ok { - return i + 1, key, true + return i + 1, key, true, true } // The flow scan runs UNANCHORED over the line with its quoted scalars // blanked. Blanking is what closes the false positive — a quoted reason @@ -1652,10 +1672,10 @@ func excludedKeyInFirstBlock(lines []string, fenced []bool, keys map[string]bool // wherever the key stands, and escapedQuotedKey below reaches only // the line-anchored spelling of it. if tok[0] == '"' && strings.Contains(name, `\`) { - return i + 1, name, true + return i + 1, name, true, true } if keys[name] { - return i + 1, name, true + return i + 1, name, false, true } } scan := bare @@ -1665,7 +1685,7 @@ func excludedKeyInFirstBlock(lines []string, fenced []bool, keys map[string]bool for _, m := range flowKeyRe.FindAllStringSubmatch(scan, -1) { for _, key := range submatches(m) { if keys[key] { - return i + 1, key, true + return i + 1, key, false, true } } } @@ -1675,7 +1695,7 @@ func excludedKeyInFirstBlock(lines []string, fenced []bool, keys map[string]bool depth = 0 } } - return 0, "", false + return 0, "", false, false } // sectionSpan is the half-open line range one heading OWNS: the heading itself diff --git a/internal/core/reading/project_test.go b/internal/core/reading/project_test.go index be663bc81..be4b829b3 100644 --- a/internal/core/reading/project_test.go +++ b/internal/core/reading/project_test.go @@ -506,3 +506,50 @@ func TestANestedMappingRefusesBehindEveryBlockIndicator(t *testing.T) { } } +// TestTheEscapedKeyRefusalStatesOnlyWhatItKnows (iss-2608301421381157). The +// escaped-key refusal shared the excluded-key message, which asserted that the +// document still carried an excluded key and that its block was not closed the +// way the field reader expects. Neither is known of an escape: the package does +// not decode one, so which key it spells is exactly what it cannot say, and the +// block is closed as expected. The refusal stands; its stated reason is the +// escape. +func TestTheEscapedKeyRefusalStatesOnlyWhatItKnows(t *testing.T) { + for name, doc := range map[string]string{ + "a line-anchored escaped key": "---\nid: spc-1\n\"C:\\tmp\\x\": v\n---\n\n# A record\n", + "an escaped key in a flow map": "---\nid: spc-1\nmeta: {a: 1, \"C:\\tmp\\x\": v}\n---\n\n# A record\n", + } { + err := refuses(t, "spc-1-a-record.md", doc, map[string]bool{"origin": true}, nil) + if err == nil { + t.Errorf("%s: an escaped key was admitted", name) + continue + } + msg := err.Error() + for _, want := range []string{"spc-1-a-record.md", "line 3", "escape", `C:\\tmp\\x`} { + if !strings.Contains(msg, want) { + t.Errorf("%s: the refusal does not state %q: %v", name, want, err) + } + } + for _, claim := range []string{"excluded key", "not closed"} { + if strings.Contains(msg, claim) { + t.Errorf("%s: the refusal asserts %q, which is not known of an escape: %v", name, claim, err) + } + } + } + + // The general refusal names the key and the line, and claims no block shape + // it did not observe: a quoted key survives redaction in a block closed + // exactly as the field reader expects. + const quoted = "---\nid: spc-1\n\"origin\": ABCD-WARM-ORIGIN\n---\n\n# A record\n" + err := refuses(t, "spc-1-a-record.md", quoted, map[string]bool{"origin": true}, nil) + if err == nil { + t.Fatal("a quoted excluded key was admitted") + } + for _, want := range []string{"spc-1-a-record.md", `"origin"`, "line 3"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not state %q: %v", want, err) + } + } + if strings.Contains(err.Error(), "not closed") { + t.Errorf("the refusal asserts a block shape the document does not have: %v", err) + } +} From 1284c2de888986049c1b52815b45d323001c8d1f Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:18:31 +0100 Subject: [PATCH 34/64] =?UTF-8?q?chore:=20resolve=20iss-2608301421381157?= =?UTF-8?q?=20=E2=80=94=20the=20escaped-key=20refusal=20states=20its=20own?= =?UTF-8?q?=20ground?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2608301421381157 Assisted-by: Claude:claude-opus-5-5 --- ...ed-key-refusal-returns-through-an-error-message-ass.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2608301421381157-the-escaped-key-refusal-returns-through-an-error-message-ass.md (67%) diff --git a/.abcd/work/issues/open/iss-2608301421381157-the-escaped-key-refusal-returns-through-an-error-message-ass.md b/.abcd/work/issues/resolved/iss-2608301421381157-the-escaped-key-refusal-returns-through-an-error-message-ass.md similarity index 67% rename from .abcd/work/issues/open/iss-2608301421381157-the-escaped-key-refusal-returns-through-an-error-message-ass.md rename to .abcd/work/issues/resolved/iss-2608301421381157-the-escaped-key-refusal-returns-through-an-error-message-ass.md index 91deac382..217155e4f 100644 --- a/.abcd/work/issues/open/iss-2608301421381157-the-escaped-key-refusal-returns-through-an-error-message-ass.md +++ b/.abcd/work/issues/resolved/iss-2608301421381157-the-escaped-key-refusal-returns-through-an-error-message-ass.md @@ -7,6 +7,10 @@ category: "bug" source: "user-observation" found_during: "itd-183-round-10-ruthless" found_at: "internal/core/reading/project.go" +resolution: "The escaped-key refusal has its own message: it names the key's raw spelling, the line and the ground (a YAML escape this package does not decode), and asserts neither an excluded key nor an unclosed block. The excluded-key message no longer claims a block shape either; it states that the field reader did not report the key, so the redactor did not remove it." +impact: fix +resolved_by: + commit: "85f1cff9c45196026a5b5095b4e4b33ae7bd9f2a" --- the escaped key refusal returns through an error message asserting the block is unclosed and the key excluded when neither need be true @@ -32,3 +36,7 @@ so the escape is the signal and the answer is a refusal rather than a guess. Cost is comprehensibility, not safety -- the refusal is correct, its stated reason is not. Remedy: give the escape refusal its own message. Seed material for the exclusion floor's own intent. + +## Grounds + +- pursued: every escaped-key refusal, line-anchored or in a flow mapping, states the escape and the line and never claims an excluded key or an unclosed block; a refusal message asserting either of an escape, or an escape admitted, would show it wrong From ef5f1abd9493eefd20539aff9af5d2a5db03714a Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:18:46 +0100 Subject: [PATCH 35/64] fix(reading): list only the parked runs no ingest has given an outcome MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bare `abcd reading` listed every rdg-* directory under the assembly parking area as a staged run. Nothing removes that directory after its run is ingested, so a committed run, a refused run and a run no reading has been given rendered alike, and the list grew with every assembly while answering a different question from the one an operator asks of it: what is outstanding. Of the two remedies — have ingest remove or mark its parking directory under its lock, or classify from what the record already holds — the second is the smaller honest change. It adds no write to the ingest path, keeps the parked manifest as the run's local evidence, and asks the one question that already decides a rerun: a run has an outcome when a commit marker or a refusal record sits under its id. That probe is lifted out of refuseARerun as runOutcome, and Describe applies it to every parked run, so the rerun refusal and the status render cannot disagree about which runs are outstanding. The JSON key and its shape are unchanged; the command page and the brief's surface chapter say what the list now holds. Refs: iss-2608311621412224 Assisted-by: Claude:claude-opus-5-5 --- .../brief/04-surfaces/23-reading.md | 8 +-- commands/reading.md | 3 +- internal/core/reading/ingest.go | 33 +++++++---- internal/core/reading/ingest_stage_test.go | 49 +++++++++++++++-- internal/core/reading/status.go | 55 ++++++++++++++++--- 5 files changed, 121 insertions(+), 27 deletions(-) diff --git a/.abcd/development/brief/04-surfaces/23-reading.md b/.abcd/development/brief/04-surfaces/23-reading.md index 5945565b8..008c1b742 100644 --- a/.abcd/development/brief/04-surfaces/23-reading.md +++ b/.abcd/development/brief/04-surfaces/23-reading.md @@ -30,10 +30,10 @@ a reading whose account of itself can be checked rather than believed. Bare `abcd reading` is a third form and a **read-only status render**: the assembler's version and schema number, the include and exclusion row counts, the -charter path, the position definitions the binary resolves, the staged runs, and -any orphaned ingest waiting to be swept. It writes nothing, and it is where an -operator reads what the instrument currently is before commissioning anything -through it. +charter path, the position definitions the binary resolves, the staged runs no +ingest has yet committed or refused, and any orphaned ingest waiting to be +swept. It writes nothing, and it is where an operator reads what the instrument +currently is before commissioning anything through it. ## The invocation carries no free text diff --git a/commands/reading.md b/commands/reading.md index 16bedba29..fa63e7178 100644 --- a/commands/reading.md +++ b/commands/reading.md @@ -32,7 +32,8 @@ To render the assembler's state: Summarise the JSON for the user: `assembler_version`, `include_rows` and `exclusion_rows` (what the table admits and what it refuses), `definitions` (the reading definitions the locator RESOLVED), `staged_runs` (runs an -assembly has parked in the local tier), `orphaned_ingests`, and +assembly has parked in the local tier that no ingest has yet committed or +refused), `orphaned_ingests`, and `leftover_stages`. Zero writes. **`definitions` is what resolved, not what is present.** A definition is diff --git a/internal/core/reading/ingest.go b/internal/core/reading/ingest.go index be02b9bb7..aeaefaf15 100644 --- a/internal/core/reading/ingest.go +++ b/internal/core/reading/ingest.go @@ -993,21 +993,34 @@ func validArtefactName(name string) error { // It runs before the refusal path for that second reason: a rerun must not // overwrite the refusal record of the run it is repeating either. func refuseARerun(root *os.Root, runID string) error { + rel, err := runOutcome(root, runID) + if err != nil { + return fmt.Errorf("reading: probing the outcome of run %s: %w", runID, err) + } + if rel != "" { + return fmt.Errorf("reading: run %s already has an outcome at %s; a rerun is a new run with a "+ + "new run id, never an amendment — assemble again, and ingest the run that assembly parked", + runID, rel) + } + return nil +} + +// runOutcome returns the repository-relative path of the outcome record a run +// already has — its commit marker, or its refusal record — or "" when it has +// none. It is the one answer to "has this run been ingested": the rerun refusal +// asks it before an ingest writes, and the bare render asks it of every parked +// run, so the two cannot disagree about which runs are still outstanding. +func runOutcome(root *os.Root, runID string) (string, error) { for _, name := range []string{RunFileName, RefusalFileName} { rel := ReadingsRecordDir + "/" + runID + "/" + name - _, err := root.Lstat(rel) - switch { + switch _, err := root.Lstat(rel); { case err == nil: - return fmt.Errorf("reading: run %s already has an outcome at %s; a rerun is a new run with a "+ - "new run id, never an amendment — assemble again, and ingest the run that assembly parked", - runID, rel) - case os.IsNotExist(err): - continue - default: - return fmt.Errorf("reading: probing the outcome of run %s: %w", runID, err) + return rel, nil + case !os.IsNotExist(err): + return "", err } } - return nil + return "", nil } // refuse records a list-level refusal and returns it. It is the ONE writer of a diff --git a/internal/core/reading/ingest_stage_test.go b/internal/core/reading/ingest_stage_test.go index e0c77b753..e68eb6a5b 100644 --- a/internal/core/reading/ingest_stage_test.go +++ b/internal/core/reading/ingest_stage_test.go @@ -7,6 +7,7 @@ import ( "errors" "os" "path/filepath" + "slices" "strings" "testing" @@ -554,10 +555,9 @@ func TestARefusalRollsBackTheRunsOwnCrashedAttempt(t *testing.T) { // // The sweep is held back to the commit path, so an orphan can now outlive the // invocation that found it — and nothing named one. `staged_runs` reads the -// ASSEMBLY parking area, which is a different directory and lists committed -// runs alongside uncommitted ones, so an operator had no way to see that a -// crashed ingest had left reading records in the ledger for a run that never -// happened. +// ASSEMBLY parking area, which is a different directory, so an operator had no +// way to see that a crashed ingest had left reading records in the ledger for +// a run that never happened. func TestTheBareRenderNamesAnOrphanedIngestStage(t *testing.T) { f := newIngestFixture(t, "detection") status, err := Describe(f.root) @@ -623,3 +623,44 @@ func TestTheBareRenderTellsALeftoverStageFromAnOrphan(t *testing.T) { t.Errorf("the render reports leftover stages %v, want [%s]", status.LeftoverStages, f.runID) } } + +// TestTheBareRenderListsOnlyTheParkedRunsAwaitingAnOutcome +// (iss-2608311621412224). Nothing removes an assembly's parking directory after +// its run is ingested, so `staged_runs` listed every run ever assembled: a +// committed run, a refused one and one no reading had been given rendered +// alike, and the list grew without bound while answering a different question +// from the one an operator asks of it. The record already holds the answer — a +// run with an outcome has a run.json or a refusal.json under its id, the probe +// refuseARerun makes — so the render lists a parked run only while it has none. +func TestTheBareRenderListsOnlyTheParkedRunsAwaitingAnOutcome(t *testing.T) { + f := newIngestFixture(t, "detection") + status, err := Describe(f.root) + if err != nil { + t.Fatal(err) + } + if !slices.Equal(status.StagedRuns, []string{f.runID}) { + t.Fatalf("a parked run awaiting ingest renders as staged runs %v, want [%s]", status.StagedRuns, f.runID) + } + + f.mustIngest(f.payload(1)) + waiting := f.nextRun(f.payload(1))["run_id"].(string) + refused := f.nextRun(f.payload(1)) + for _, it := range refused["items"].([]any) { + it.(map[string]any)[PatternField] = "" + } + if _, err := f.ingest(refused); err == nil { + t.Fatal("a run in which every item was refused was accepted") + } + if !f.exists(ReadingsRecordDir + "/" + refused["run_id"].(string) + "/" + RefusalFileName) { + t.Fatal("the refused run left no refusal record, so this case proves nothing") + } + + status, err = Describe(f.root) + if err != nil { + t.Fatal(err) + } + if !slices.Equal(status.StagedRuns, []string{waiting}) { + t.Errorf("staged runs are %v; want only the run awaiting an outcome, [%s] — the committed run %s "+ + "and the refused run %s have one", status.StagedRuns, waiting, f.runID, refused["run_id"]) + } +} diff --git a/internal/core/reading/status.go b/internal/core/reading/status.go index 67a86c9f1..f57e8cc2d 100644 --- a/internal/core/reading/status.go +++ b/internal/core/reading/status.go @@ -31,9 +31,12 @@ type Status struct { IncludeRows int `json:"include_rows"` ExclusionRows int `json:"exclusion_rows"` Definitions []string `json:"definitions"` - // StagedRuns is what an ASSEMBLY parked. It is not filtered by whether the - // run was ingested: nothing removes an assembly's directory afterwards, so a - // committed run and an unread one appear alike (iss captured separately). + // StagedRuns is what an ASSEMBLY parked and no ingest has yet given an + // outcome: the runs still awaiting a reading. Nothing removes an assembly's + // directory after its run is ingested, so the parking area alone lists every + // run ever assembled, committed and refused alike; a parked run whose id + // already has a commit marker or a refusal record is therefore left out, by + // the probe the rerun refusal makes (runOutcome, iss-2608311621412224). StagedRuns []string `json:"staged_runs"` // OrphanedIngests names the runs whose ingest reached the ledger and never // reached its commit marker. @@ -90,12 +93,9 @@ func Describe(repoRoot string) (Status, error) { if err != nil && !os.IsNotExist(err) { return Status{}, fmt.Errorf("reading: listing the staged runs: %w", err) } - for _, e := range runs { - if e.IsDir() && strings.HasPrefix(e.Name(), RunIDFamily+"-") { - s.StagedRuns = append(s.StagedRuns, e.Name()) - } + if s.StagedRuns, err = awaitingOutcome(repoRoot, runs); err != nil { + return Status{}, err } - sort.Strings(s.StagedRuns) // A stage directory named by a run id is left in one of two states, and the // commit marker is what tells them apart — the same probe the sweep's @@ -126,3 +126,42 @@ func Describe(repoRoot string) (Status, error) { sort.Strings(s.LeftoverStages) return s, nil } + +// awaitingOutcome returns the parked runs no ingest has given an outcome, sorted. +// +// The parking directory is kept after an ingest — it is the run's local +// evidence, and removing it is not this read-only render's to do — so what +// tells an outstanding run from an ingested one is the record: an ingested run +// has a commit marker or a refusal record under its id in the durable tier, the +// same probe refuseARerun makes before an ingest writes. +func awaitingOutcome(repoRoot string, parked []os.DirEntry) ([]string, error) { + out := []string{} + if len(parked) == 0 { + return out, nil + } + root, err := os.OpenRoot(repoRoot) + if err != nil { + return nil, fmt.Errorf("reading: opening the repository to probe the staged runs: %w", err) + } + defer root.Close() + for _, e := range parked { + if !e.IsDir() || !strings.HasPrefix(e.Name(), RunIDFamily+"-") { + continue + } + // A parked name the run-id shape refuses cannot have an outcome, since + // ingest refuses to name a record directory after it; it stays listed, + // as everything parked did, rather than be probed as a path. + if recordid.ValidReadingRunID(e.Name()) { + rel, err := runOutcome(root, e.Name()) + if err != nil { + return nil, fmt.Errorf("reading: probing the outcome of staged run %s: %w", e.Name(), err) + } + if rel != "" { + continue + } + } + out = append(out, e.Name()) + } + sort.Strings(out) + return out, nil +} From 051fd04527e85d29986d1846abc099a9c42b9e1c Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:18:48 +0100 Subject: [PATCH 36/64] =?UTF-8?q?chore:=20resolve=20iss-2608311621412224?= =?UTF-8?q?=20=E2=80=94=20staged=20runs=20lists=20only=20runs=20awaiting?= =?UTF-8?q?=20an=20outcome?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2608311621412224 Assisted-by: Claude:claude-opus-5-5 --- ...ng-assembler-s-staged-runs-line-cannot-tell-a-parke.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2608311621412224-the-reading-assembler-s-staged-runs-line-cannot-tell-a-parke.md (57%) diff --git a/.abcd/work/issues/open/iss-2608311621412224-the-reading-assembler-s-staged-runs-line-cannot-tell-a-parke.md b/.abcd/work/issues/resolved/iss-2608311621412224-the-reading-assembler-s-staged-runs-line-cannot-tell-a-parke.md similarity index 57% rename from .abcd/work/issues/open/iss-2608311621412224-the-reading-assembler-s-staged-runs-line-cannot-tell-a-parke.md rename to .abcd/work/issues/resolved/iss-2608311621412224-the-reading-assembler-s-staged-runs-line-cannot-tell-a-parke.md index 08abc7068..06cd4880e 100644 --- a/.abcd/work/issues/open/iss-2608311621412224-the-reading-assembler-s-staged-runs-line-cannot-tell-a-parke.md +++ b/.abcd/work/issues/resolved/iss-2608311621412224-the-reading-assembler-s-staged-runs-line-cannot-tell-a-parke.md @@ -9,6 +9,14 @@ found_during: "manual-capture" origin: researcher-authored production_mode: hand-written found_at: "internal/core/reading/status.go" +resolution: "staged_runs lists only the parked runs no ingest has given an outcome: Describe probes each parked run id for a commit marker or a refusal record, through runOutcome, the probe the rerun refusal already makes, so an ingested or refused run no longer renders as outstanding. Ingest still leaves its parking directory as the run's local evidence." +impact: fix +resolved_by: + commit: "ef5f1abd9493eefd20539aff9af5d2a5db03714a" --- The reading assembler's staged_runs line cannot tell a parked run from an ingested one. Describe lists every rdg-* directory under the assembly parking area, and nothing removes that directory after its run is ingested, so a committed run and one no reading has ever been given render identically. The Status doc comment claimed the filter existed until it was corrected; the behaviour did not. An operator reading the bare verb to find out what is outstanding is given a list that grows monotonically and answers a different question. Found while wiring orphaned_ingests, which is a separate field precisely because this one cannot be trusted to mean outstanding. + +## Grounds + +- pursued: a parked run leaves staged_runs exactly when its id has run.json or refusal.json in the readings record, and a run awaiting ingest stays listed; an ingested or refused run still listed, or a run awaiting ingest missing from the list, would show it wrong From facc64db18973154d6cbe7b876e3a96f23c75fa4 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:22:06 +0100 Subject: [PATCH 37/64] fix(lint): read committed lines in the scanner's decoded spellings The repolint privacy-hygiene rule and the harness_leak lint rule matched each committed line only as written, so a JSON fixture, export or transcript that wrote a third-party home after a \n escape, with the solidus or unicode escape, with doubled Windows separators or percent-encoded, or a session URL straight after a \n escape, passed both: the escape's letter defeated every leading anchor. Both rules now read the line, then each of scanner.DecodedViews(line), the new exported seam onto lineViews (the percent pre-pass and every JSON-escape layer, now listed once in percent.go for the scan, the literal-home backstop and the lint alike). The waiver is still read on the line as written, and the persona, system-root and reserved-value exemptions run on whichever spelling matched. A decoded view carries the line breaks its escapes stood for, so the footer's line-start SkipAt (footerOwnsItsLine) accepts a break inside the text as the start of a line: a footer on its own line inside a quoted pull-request body is the leak it was in the body, while prose quoting the shape mid-sentence is still spared. A line as written carries no break, so the raw reading is unchanged. Measured on the live tree with `abcd lint`: 101 findings before, 103 after. Twelve test-fixture lines read newly as leaks (Go interpreted strings that spell a home or a footer behind an escape, the reading drainS1 predicted); each is a deliberately illustrative value and carries the rule's own waiver, and one fixture line the meter change touched was already on the baseline and now carries it too. No production line reads newly. The remaining growth is three lines in three resolved records that describe the escaped spellings with placeholder names (LOGIN, OTHER), the same class as the record findings already on the baseline, left as written. docs-lint and record-lint, which run harness_leak as gates, report nothing new. Refs: iss-2609261658553101 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/binary_skip_test.go | 4 +- internal/adapter/scanner/harnessleak.go | 8 ++- internal/adapter/scanner/harnessleak_test.go | 2 +- internal/adapter/scanner/jsonescape_test.go | 6 +- internal/adapter/scanner/meter_test.go | 4 +- .../adapter/scanner/outbound_check_test.go | 2 +- internal/adapter/scanner/percent.go | 45 ++++++++++--- internal/adapter/scanner/residual.go | 10 --- internal/core/lint/harnessleak.go | 29 ++++++-- internal/core/lint/harnessleak_escape_test.go | 45 +++++++++++++ internal/core/repolint/rule_privacy.go | 21 ++++++ .../rule_privacy_cap_internal_test.go | 2 +- .../core/repolint/rule_privacy_escape_test.go | 66 +++++++++++++++++++ .../surface/cli/githook_commitmsg_test.go | 2 +- .../surface/cli/lint_outbound_surface_test.go | 2 +- 15 files changed, 211 insertions(+), 37 deletions(-) create mode 100644 internal/core/lint/harnessleak_escape_test.go create mode 100644 internal/core/repolint/rule_privacy_escape_test.go diff --git a/internal/adapter/scanner/binary_skip_test.go b/internal/adapter/scanner/binary_skip_test.go index e4000c446..581a488fa 100644 --- a/internal/adapter/scanner/binary_skip_test.go +++ b/internal/adapter/scanner/binary_skip_test.go @@ -376,7 +376,7 @@ func TestSessionURLInBinaryHardFailsLikeText(t *testing.T) { // identity guards, so it must be dropped explicitly on the binary branch. func TestGenericHomePathInBinaryIsNotAFinding(t *testing.T) { root := t.TempDir() - abs := writeFile(t, root, "b.png", "\x89PNG\r\n\x1a\n/home/runner/work/repo/x\n") + abs := writeFile(t, root, "b.png", "\x89PNG\r\n\x1a\n/home/runner/work/repo/x\n") // abcd-lint:allow sc, err := New(root) if err != nil { t.Fatal(err) @@ -396,7 +396,7 @@ func TestGenericHomePathInBinaryIsNotAFinding(t *testing.T) { func TestRepoRaisedSeverityIsHonouredOnBytes(t *testing.T) { root := t.TempDir() writeFile(t, root, ".abcd/config/pii.json", `{"identity_severities":{"home_path_other":"hard_fail"}}`) - abs := writeFile(t, root, "b.png", "\x89PNG\r\n\x1a\n/home/runner/work/repo/x\n") + abs := writeFile(t, root, "b.png", "\x89PNG\r\n\x1a\n/home/runner/work/repo/x\n") // abcd-lint:allow sc, err := New(root) if err != nil { t.Fatal(err) diff --git a/internal/adapter/scanner/harnessleak.go b/internal/adapter/scanner/harnessleak.go index 199e49daf..84205e240 100644 --- a/internal/adapter/scanner/harnessleak.go +++ b/internal/adapter/scanner/harnessleak.go @@ -151,13 +151,19 @@ func IsHarnessLeakKind(kind string) bool { // regex runs only over a prefix made entirely of markers: running it over the // whole line before every match cost the line's length per footer // (iss-2609251535277823). +// +// A line break inside the text starts a line too. A line as written carries +// none, but a decoded view does — the \n escapes of a JSON string quoting a +// pull-request body decode to the breaks the body had — and a footer after +// one owns its line exactly as it did in the body (iss-2609261658553101). func footerOwnsItsLine(line string, start int) bool { i := start for i > 0 && footerPrefixByte(line[i-1]) { i-- } scanMeter.charge(stageSkipAt, start-i) - return i == 0 && footerLinePrefixRe.MatchString(line[:start]) + atLineStart := i == 0 || line[i-1] == '\n' || line[i-1] == '\r' + return atLineStart && footerLinePrefixRe.MatchString(line[i:start]) } // footerPrefixByte is every byte footerLinePrefixRe can match. diff --git a/internal/adapter/scanner/harnessleak_test.go b/internal/adapter/scanner/harnessleak_test.go index 068dc3c71..e6d3ca0f7 100644 --- a/internal/adapter/scanner/harnessleak_test.go +++ b/internal/adapter/scanner/harnessleak_test.go @@ -183,7 +183,7 @@ func TestHarnessLeakPatternsAreInTheCanonicalSet(t *testing.T) { // carrying the harness-appended footer comes back carrying only the repo's own // attribution trailer. func TestScrubOutboundStripsFooterKeepsTrailer(t *testing.T) { - body := "Fixes the walk.\n\nAssisted-by: Claude:some-model\n\n🤖 _Generated with [Some Tool](https://tool.dev)_\n" + body := "Fixes the walk.\n\nAssisted-by: Claude:some-model\n\n🤖 _Generated with [Some Tool](https://tool.dev)_\n" // abcd-lint:allow got, findings, err := ScrubOutbound(t.TempDir(), body, "pr-body") if err != nil { diff --git a/internal/adapter/scanner/jsonescape_test.go b/internal/adapter/scanner/jsonescape_test.go index 767d3525b..e3b847657 100644 --- a/internal/adapter/scanner/jsonescape_test.go +++ b/internal/adapter/scanner/jsonescape_test.go @@ -86,8 +86,8 @@ func TestJSONSolidusEscapeReadsAsASeparator(t *testing.T) { cases := []struct { name, line, kind, gone string }{ - {"own home, solidus escape", `{"p":"\/Users\/zqjsonme\/Desktop\/a.txt"}`, kindHomeSelf, "zqjsonme"}, - {"own home, unicode solidus", `{"p":"\u002fUsers\u002fzqjsonme\u002fDesktop"}`, kindHomeSelf, "zqjsonme"}, + {"own home, solidus escape", `{"p":"\/Users\/zqjsonme\/Desktop\/a.txt"}`, kindHomeSelf, "zqjsonme"}, // abcd-lint:allow + {"own home, unicode solidus", `{"p":"\u002fUsers\u002fzqjsonme\u002fDesktop"}`, kindHomeSelf, "zqjsonme"}, // abcd-lint:allow {"other home, solidus escape", `{"p":"\/home\/` + other + `\/x"}`, kindHomeOther, other}, {"other home, unicode solidus", `{"p":"\u002Fhome\u002F` + other + `\u002Fx"}`, kindHomeOther, other}, {"other home, solidus at line end", `{"p":"\/home\/` + other + `"}`, kindHomeOther, other}, @@ -108,7 +108,7 @@ func TestJSONSolidusEscapeReadsAsASeparator(t *testing.T) { // A generic login is reported where it stands as an account, and the // solidus-escaped home root is such a position. generic := Identity{HomePath: "/home/" + "dev", HomeUser: "dev"} - line := `{"p":"\/home\/dev\/project"}` + line := `{"p":"\/home\/dev\/project"}` // abcd-lint:allow if f, ok := findingOf(ScanText(line, generic, DefaultPatterns(), nil, "transcript"), kindHomeSelf); !ok { t.Errorf("a generic login's own home behind solidus escapes raised no %s", kindHomeSelf) } else if !strings.Contains(f.Matched, "dev") { diff --git a/internal/adapter/scanner/meter_test.go b/internal/adapter/scanner/meter_test.go index c3a27ffe2..4fa528050 100644 --- a/internal/adapter/scanner/meter_test.go +++ b/internal/adapter/scanner/meter_test.go @@ -58,7 +58,7 @@ var meterFixtures = []meterFixture{ {"footers", Identity{}, rep("generated with [x](y) ")}, {"footer_openings_without_links", Identity{}, rep("generated with [x ")}, {"footers_after_prose", Identity{}, func(n int) string { return strings.Repeat("prose ", n) + "generated with [x](y)" }}, - {"percent_encoded", meterNamedID, rep("%2Fhome%2Fzq8home ")}, + {"percent_encoded", meterNamedID, rep("%2Fhome%2Fzq8home ")}, // abcd-lint:allow // Identity matchers. {"generic_login_words", meterGenericID, rep("dev ")}, {"generic_login_dotted_run", meterGenericID, rep("dev.")}, @@ -91,7 +91,7 @@ var meterFixtures = []meterFixture{ // The JSON-escape layers (jsonescape.go): each layer is one more scan of // the line, so a line dense in escapes, in nested escapes and in // escaped homes must still cost a constant number of passes. - {"json_escaped_other_homes", Identity{}, rep(`\/home\/zqa\/x\n`)}, + {"json_escaped_other_homes", Identity{}, rep(`\/home\/zqa\/x\n`)}, // abcd-lint:allow {"json_unicode_escaped_own_homes", meterNamedID, rep(escapeSeparators("/Users/zq8home/x ", uSolidus))}, // abcd-audit:allow {"json_nested_escape_runs", meterNamedID, rep(`\\\\\\\"\\\\n`)}, {"json_tokens_after_escapes", Identity{}, rep(`\n` + "ghp_" + strings.Repeat("a", 36))}, diff --git a/internal/adapter/scanner/outbound_check_test.go b/internal/adapter/scanner/outbound_check_test.go index 85d406e1a..4175ce92c 100644 --- a/internal/adapter/scanner/outbound_check_test.go +++ b/internal/adapter/scanner/outbound_check_test.go @@ -37,7 +37,7 @@ func TestCheckOutboundRefusesASessionURL(t *testing.T) { // is the position AGENTS.md states. What is not fine is a Go check that knows // only half the policy it claims to enforce. func TestCheckOutboundRefusesAnAttributionFooter(t *testing.T) { - body := "Closes the gate.\n\n🤖 Generated with [Some Tool](https://sometool.dev)\n" + body := "Closes the gate.\n\n🤖 Generated with [Some Tool](https://sometool.dev)\n" // abcd-lint:allow findings, err := CheckOutbound(t.TempDir(), body, "pr-body") if err == nil { diff --git a/internal/adapter/scanner/percent.go b/internal/adapter/scanner/percent.go index d324e9083..fff1c4546 100644 --- a/internal/adapter/scanner/percent.go +++ b/internal/adapter/scanner/percent.go @@ -40,19 +40,48 @@ const maxPercentDecodePasses = 3 // (iss-2608270720336165). func decodedLineFindings(patterns []Pattern, probes []matcher, junctions junctionSet, matchers identityMatchers, id2sev map[string]Severity, rawLine string, lineno int, file string) []Finding { var out []Finding - if decoded, posMap := percentDecodeBounded(rawLine); posMap != nil { - out = viewFindings(patterns, probes, junctions, matchers, id2sev, rawLine, decodedView{decoded, posMap}, lineno, file) - } - // The JSON-escape views (jsonescape.go): the same scan over each layer of - // the line's JSON string escapes, mapped back the same way, so a value an - // escape hid from an anchor or spelled with escaped bytes is found where - // it sits on disk (iss-2609261647358395, iss-2609251639263391). - for _, v := range jsonEscapeLayers(rawLine) { + for _, v := range lineViews(rawLine) { out = append(out, viewFindings(patterns, probes, junctions, matchers, id2sev, rawLine, v, lineno, file)...) } return out } +// lineViews is every decoded view of one line the scan reads, and the one +// place that list is made: the percent pre-pass's fully decoded copy, then +// each layer of the line's JSON string escapes (jsonescape.go), outermost +// first, so a value an escape hid from an anchor or spelled with escaped +// bytes is found where it sits on disk (iss-2609261647358395, +// iss-2609251639263391). The literal-home backstop (residual.go) and, through +// DecodedViews, the committed-text lint rules read the same list. +func lineViews(line string) []decodedView { + var views []decodedView + if decoded, posMap := percentDecodeBounded(line); posMap != nil { + views = append(views, decodedView{decoded, posMap}) + } + return append(views, jsonEscapeLayers(line)...) +} + +// DecodedViews returns the decoded spellings of one line that the scan reads +// beside the line as written (lineViews), without their position maps: nil +// for a line with nothing to decode. It is the seam for a surface that judges +// committed lines with its own matchers — the repolint privacy rule and the +// harness_leak lint rule — so a JSON fixture, export or transcript that +// writes a home path, an address or a session URL behind an escape is read +// in the spelling the store-before-commit redactors read it in +// (iss-2609261658553101). A decoded view can carry line breaks the escapes +// stood for; a line-scoped check reads each one as the start of a line. +func DecodedViews(line string) []string { + views := lineViews(line) + if len(views) == 0 { + return nil + } + out := make([]string, len(views)) + for i, v := range views { + out[i] = v.text + } + return out +} + // viewFindings runs every detector over one decoded view of rawLine and maps // each hit back to the raw bytes it came from. A view with nothing decoded in // it is never handed here; the raw scan already covers the raw line. diff --git a/internal/adapter/scanner/residual.go b/internal/adapter/scanner/residual.go index 4f67024e6..3ab1741c4 100644 --- a/internal/adapter/scanner/residual.go +++ b/internal/adapter/scanner/residual.go @@ -110,16 +110,6 @@ func backstopSpans(text, needle string, wantURLs bool, accept func(s string, at, return disjointSpans(out) } -// lineViews is every decoded view of one line the scan reads: the percent -// pre-pass's fully decoded copy and each JSON-escape layer, outermost first. -func lineViews(line string) []decodedView { - var views []decodedView - if decoded, posMap := percentDecodeBounded(line); posMap != nil { - views = append(views, decodedView{decoded, posMap}) - } - return append(views, jsonEscapeLayers(line)...) -} - // needleOccurrences walks s for needle, keeping each occurrence accept holds // and resuming past it, or one byte on where accept declines — the walk the // sweep has always made. raw marks s as the text as written, where an diff --git a/internal/core/lint/harnessleak.go b/internal/core/lint/harnessleak.go index 55553bc0e..54c4787a8 100644 --- a/internal/core/lint/harnessleak.go +++ b/internal/core/lint/harnessleak.go @@ -61,7 +61,28 @@ func checkHarnessLeak(rel string, lines []string, mask []bool, cfg RuleConfig) [ // // At most one finding per line, as the audit rule does: the citation points a // reader at the line, and the line is what gets fixed. +// +// The line is read in every spelling the scanner reads (scanner.DecodedViews): +// as written, then through its percent and JSON-escape views, so a session URL +// written straight after a \n escape in a quoted JSON string — where the +// escape's letter defeats the pattern's leading word boundary — or a footer on +// a line of its own inside one is the leak the plain spelling is +// (iss-2609261658553101). func harnessLeakOnLine(rel string, lineNo int, line, severity string) (Finding, bool) { + for _, spelling := range append([]string{line}, scanner.DecodedViews(line)...) { + if label, ok := harnessLeakIn(spelling); ok { + return Finding{ + File: rel, Line: lineNo, RuleID: ruleHarnessLeak, Severity: severity, + Message: "committed text carries a " + label + "; " + scanner.OutboundPolicy + + " (add `" + harnessLeakWaiver + "` on the line if it is deliberately illustrative)", + }, true + } + } + return Finding{}, false +} + +// harnessLeakIn returns the label of the first leak on one spelling of a line. +func harnessLeakIn(line string) (string, bool) { for _, p := range scanner.HarnessLeakPatterns() { for _, loc := range p.Re.FindAllStringIndex(line, -1) { if p.Skip != nil && p.Skip(line[loc[0]:loc[1]]) { @@ -70,12 +91,8 @@ func harnessLeakOnLine(rel string, lineNo int, line, severity string) (Finding, if p.SkipAt != nil && p.SkipAt(line, loc[0], loc[1]) { continue } - return Finding{ - File: rel, Line: lineNo, RuleID: ruleHarnessLeak, Severity: severity, - Message: "committed text carries a " + p.Label + "; " + scanner.OutboundPolicy + - " (add `" + harnessLeakWaiver + "` on the line if it is deliberately illustrative)", - }, true + return p.Label, true } } - return Finding{}, false + return "", false } diff --git a/internal/core/lint/harnessleak_escape_test.go b/internal/core/lint/harnessleak_escape_test.go new file mode 100644 index 000000000..ff8535c79 --- /dev/null +++ b/internal/core/lint/harnessleak_escape_test.go @@ -0,0 +1,45 @@ +package lint + +import ( + "path/filepath" + "testing" +) + +// harness_leak reads each committed line in the spellings the scanner reads +// (iss-2609261658553101): in a JSON transcript or export quoted into the +// record, a session URL written straight after a \n escape defeats the +// pattern's leading word boundary, and a footer after one sits on a line of +// its own once the string is read. Both are the leak the plain spelling is. +func TestHarnessLeakReadsEscapedSpellings(t *testing.T) { + root := t.TempDir() + writeFile(t, root, filepath.Join("docs", "session.md"), + "# Run\n\n"+`{"body":"done\n`+synthSessionURL(t, 41)+`"}`+"\n") + writeFile(t, root, filepath.Join("docs", "footer.md"), + "# Body\n\n"+`{"body":"Shipped.\n\n`+harnessFooter+`"}`+"\n") + + fs, err := Lint(harnessLeakCfg(), root) + if err != nil { + t.Fatal(err) + } + for _, rel := range []string{"session.md", "footer.md"} { + if !hasFinding(fs, filepath.Join("docs", rel), ruleHarnessLeak, 3) { + t.Errorf("no harness_leak finding on docs/%s:3; got %+v", rel, fs) + } + } +} + +// A footer quoted mid-sentence inside an escaped string is still prose about +// the ban, and an escape that hides nothing adds no finding. +func TestHarnessLeakEscapedSpellingsSpareProse(t *testing.T) { + root := t.TempDir() + writeFile(t, root, filepath.Join("docs", "policy.md"), + "# Policy\n\n"+`{"body":"We refuse the \"`+harnessFooter+`\" footer.\nThanks."}`+"\n") + + fs, err := Lint(harnessLeakCfg(), root) + if err != nil { + t.Fatal(err) + } + if n := countRule(fs, ruleHarnessLeak); n != 0 { + t.Fatalf("expected prose about the ban to be spared, got %d: %+v", n, fs) + } +} diff --git a/internal/core/repolint/rule_privacy.go b/internal/core/repolint/rule_privacy.go index 80d7e433c..664e83ad0 100644 --- a/internal/core/repolint/rule_privacy.go +++ b/internal/core/repolint/rule_privacy.go @@ -224,7 +224,28 @@ func (privacyHygiene) Eval(ctx Context) ([]Finding, error) { // speaks about paths and addresses, which is no use to somebody looking at a // tool footer, and the harness class has an operational half — the post-create // re-read — that only its own policy states (scanner.OutboundPolicy). +// +// The line is read in every spelling the scanner reads (scanner.DecodedViews): +// as written, then through its percent and JSON-escape views, so a committed +// JSON fixture, export or transcript that writes a home path, an address or +// a session URL behind an escape is refused as the plain spelling is +// (iss-2609261658553101). Each view keeps every exemption the line has — the +// waiver is read on the line as written, and the persona, system-root and +// reserved-value exemptions run on whichever spelling matched. func privacyLeak(line string, patterns []scanner.Pattern) (msg, fix string, sev Severity, leaked bool) { + if msg, fix, sev, leaked = privacyLeakOn(line, patterns); leaked { + return msg, fix, sev, leaked + } + for _, view := range scanner.DecodedViews(line) { + if msg, fix, sev, leaked = privacyLeakOn(view, patterns); leaked { + return msg, fix, sev, leaked + } + } + return "", "", SeverityError, false +} + +// privacyLeakOn is privacyLeak over one spelling of the line. +func privacyLeakOn(line string, patterns []scanner.Pattern) (msg, fix string, sev Severity, leaked bool) { if hasAbsHomePath(line) { return "committed file contains an absolute local path", "", SeverityError, true } diff --git a/internal/core/repolint/rule_privacy_cap_internal_test.go b/internal/core/repolint/rule_privacy_cap_internal_test.go index 05df6ec28..9794f2dfc 100644 --- a/internal/core/repolint/rule_privacy_cap_internal_test.go +++ b/internal/core/repolint/rule_privacy_cap_internal_test.go @@ -33,7 +33,7 @@ func TestReadTrackedFileCapBoundary(t *testing.T) { for i := range buf { buf[i] = 'a' } - tail := []byte("\n/home/somebody/secret\n") + tail := []byte("\n/home/somebody/secret\n") // abcd-lint:allow copy(buf[len(buf)-len(tail):], tail) if err := os.WriteFile(atCap, buf, 0o644); err != nil { t.Fatal(err) diff --git a/internal/core/repolint/rule_privacy_escape_test.go b/internal/core/repolint/rule_privacy_escape_test.go new file mode 100644 index 000000000..6a686398f --- /dev/null +++ b/internal/core/repolint/rule_privacy_escape_test.go @@ -0,0 +1,66 @@ +package repolint_test + +import ( + "strings" + "testing" +) + +// slashes spells every '#' of p with sep, so a specimen names a home path in +// an escaped spelling while this file carries no home path at all. +func slashes(p, sep string) string { return strings.ReplaceAll(p, "#", sep) } + +// uSolidus is the JSON unicode escape of '/', assembled so no tool reading the +// source folds the six bytes back into a '/'. +var uSolidus = `\` + "u002f" + +// privacy-hygiene reads each committed line in the spellings the scanner reads +// (iss-2609261658553101): a JSON fixture, export or transcript that writes a +// third-party home after a \n escape, with the solidus or unicode escape, with +// doubled Windows separators or percent-encoded, a session URL straight after +// a \n escape, or a footer on a line of its own inside the string, carries the +// same leak as the plain spelling and is refused like it. Read raw, the escape +// letter before the value is a word or path byte, so every anchor declined it. +func TestAC_PrivacyReadsEscapedSpellings(t *testing.T) { + const name = "zqother" // not a registry persona, so the home is a leak + cases := []struct{ name, line string }{ + {"home after a newline escape", `{"out":"cwd\n` + slashes("#home#"+name+"#x", "/") + `"}`}, + {"solidus-escaped home", `{"cwd":"` + slashes("#home#"+name, `\/`) + `"}`}, + {"unicode-escaped home", `{"cwd":"` + slashes("#Users#"+name, uSolidus) + `"}`}, + {"doubled Windows separators", `{"cwd":"C:` + slashes("#Users#"+name, `\\`) + `"}`}, + {"percent-encoded home", "see file:" + slashes("#home#"+name, "%2F") + "\n"}, + {"session URL after a newline escape", `{"body":"done\n` + synthSessionURL("agent-host.dev", 31) + `"}`}, + {"footer on a line of its own inside a JSON string", `{"body":"Shipped.\n\n` + harnessFooter + `"}`}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + res := newFixtureRepo(t).conforming(). + file("reference/export.json", c.line+"\n"). + commit().run() + f := findingFor(res, "privacy-hygiene") + if f == nil { + t.Fatalf("no privacy-hygiene finding for %q", c.line) + } + if f.File != "reference/export.json" || f.Line != 1 { + t.Errorf("citation = %s:%d, want reference/export.json:1", f.File, f.Line) + } + }) + } +} + +// The decoded views add no finding of their own on a line whose escapes hide +// nothing, and they keep every exemption the plain spelling has: a persona +// home, a system root and the line waiver. +func TestAC_PrivacyEscapedSpellingsKeepTheExemptions(t *testing.T) { + body := strings.Join([]string{ + `{"out":"line one\nline two\t%41"}`, + `{"cwd":"` + slashes("#Users#alice#x", `\/`) + `"}`, + `{"cwd":"C:` + slashes("#Users#Public#x", `\\`) + `"}`, + `{"cwd":"` + slashes("#home#zqother", `\/`) + `"} abcd-lint:allow`, + }, "\n") + "\n" + res := newFixtureRepo(t).conforming(). + file("reference/export.json", body). + commit().run() + if f := findingFor(res, "privacy-hygiene"); f != nil { + t.Fatalf("unexpected privacy-hygiene finding: %s:%d %s", f.File, f.Line, f.Message) + } +} diff --git a/internal/surface/cli/githook_commitmsg_test.go b/internal/surface/cli/githook_commitmsg_test.go index 16f69858c..f67d14bb0 100644 --- a/internal/surface/cli/githook_commitmsg_test.go +++ b/internal/surface/cli/githook_commitmsg_test.go @@ -123,7 +123,7 @@ func TestCommitMsgHookRefusesASessionURL(t *testing.T) { func TestCommitMsgHookRefusesAToolFooter(t *testing.T) { c := newCommitMsgHookCase(t) - refused, out := c.commitWith("a.txt", "fix: the walk\n\n🤖 Generated with [Some Tool](https://sometool.dev)\n") + refused, out := c.commitWith("a.txt", "fix: the walk\n\n🤖 Generated with [Some Tool](https://sometool.dev)\n") // abcd-lint:allow if !refused { t.Fatalf("a commit message carrying a tool attribution footer was committed\n%s", out) } diff --git a/internal/surface/cli/lint_outbound_surface_test.go b/internal/surface/cli/lint_outbound_surface_test.go index 5f7bc5f27..dcf87bc2b 100644 --- a/internal/surface/cli/lint_outbound_surface_test.go +++ b/internal/surface/cli/lint_outbound_surface_test.go @@ -67,7 +67,7 @@ func TestLintOutboundPassesACleanArtefact(t *testing.T) { func TestLintOutboundReadsAFilePositional(t *testing.T) { dir := t.TempDir() path := filepath.Join(dir, "body.md") - body := "Closes the gate.\n\n🤖 Generated with [Some Tool](https://sometool.dev)\n" + body := "Closes the gate.\n\n🤖 Generated with [Some Tool](https://sometool.dev)\n" // abcd-lint:allow if err := os.WriteFile(path, []byte(body), 0o644); err != nil { t.Fatal(err) } From 6d3ccf1197ae7c98980c336b71d612fbd55be4b3 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:22:46 +0100 Subject: [PATCH 38/64] test(scanner): pin the footer after a decoded line break in a transcript The footer's line-start rule now reads a break inside the text as the start of a line, and ScanText's JSON-escape views are where such a break appears: a raw JSONL line quoting a pull-request body carries the body's breaks as \n escapes. TestFooterAfterADecodedLineBreakOwnsItsLine pins that a footer on a line of its own inside the string is a finding for every store-before-commit consumer, and that prose quoting the shape mid-sentence in the same kind of string is still spared. Watched RED against aaf4dd36 on a scratch copy. Refs: iss-2609261658553101 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/harnessleak_test.go | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/internal/adapter/scanner/harnessleak_test.go b/internal/adapter/scanner/harnessleak_test.go index e6d3ca0f7..3024406e6 100644 --- a/internal/adapter/scanner/harnessleak_test.go +++ b/internal/adapter/scanner/harnessleak_test.go @@ -342,3 +342,21 @@ func TestScrubOutboundKeepsTheSentenceAroundASessionURL(t *testing.T) { t.Errorf("the artefact's own content was destroyed: %q", got) } } + +// TestFooterAfterADecodedLineBreakOwnsItsLine pins the footer's line-start +// rule on the decoded views (iss-2609261658553101): a raw JSONL transcript +// line quoting a pull-request body carries the body's breaks as \n escapes, so +// a footer the harness appended sits on a line of its own once the string is +// read, and is the finding it is in the body. Prose quoting the shape +// mid-sentence inside the same kind of string is still spared. +func TestFooterAfterADecodedLineBreakOwnsItsLine(t *testing.T) { + const footer = "Generated with [Some Tool](https://tool.dev)" + leak := `{"body":"Shipped.\n\n` + footer + `"}` + if _, ok := findingOf(ScanText(leak, Identity{}, DefaultPatterns(), nil, "transcript"), kindHarnessFooter); !ok { + t.Errorf("a footer on its own line inside a JSON string raised no %s", kindHarnessFooter) + } + prose := `{"body":"We refuse the \"` + footer + `\" footer.\nThanks."}` + if f, ok := findingOf(ScanText(prose, Identity{}, DefaultPatterns(), nil, "transcript"), kindHarnessFooter); ok { + t.Errorf("prose quoting the footer inside a JSON string was reported: %+v", f) + } +} From 3983396b3812eab85176c5ce266dbae332842d9f Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:25:12 +0100 Subject: [PATCH 39/64] docs(brief): the privacy rule reads the scanner's decoded spellings The lint surface chapter's privacy-hygiene row now states that each committed line is read as written and through scanner.DecodedViews, that the waiver is read on the line as written, and that the record/docs harness_leak rule reads the same spellings, so the chapter moves with the behaviour the lint change introduced. Refs: iss-2609261658553101 Assisted-by: Claude:claude-opus-5-5 --- .abcd/development/brief/04-surfaces/16-lint.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.abcd/development/brief/04-surfaces/16-lint.md b/.abcd/development/brief/04-surfaces/16-lint.md index 366bc20bc..e9015a0c6 100644 --- a/.abcd/development/brief/04-surfaces/16-lint.md +++ b/.abcd/development/brief/04-surfaces/16-lint.md @@ -142,7 +142,7 @@ iss-2608231000561060. | `conventions-router` | error | `AGENTS.md` present at the repo root | | `decision-durability` | warn | a committed `.abcd/work/DECISIONS.md`; decisions not living only in the gitignored layer | | `docs-currency` | warn | reuses the docs-lint engine where `docs/` exists, and says so where it cannot: a repo with a `docs/` tree but no docs-lint configuration, and a configuration that will not load, each raise a finding against `.abcd/docs-lint.json` rather than passing quietly | -| `privacy-hygiene` | error (network-identifier findings mapped from a scanner `warn`/`info` land as `warn`) | three leak classes on any tracked text line: absolute local paths in committed files, real network identifiers, and the harness-leak pair the outbound policy bans everywhere (a live agent-session URL, and a tool's own "generated with" footer). The fix names reserved documentation values (RFC 5737/3849/2606/7042, or a persona-derived device name), and an `abcd-lint:allow` line waiver is honoured (the `abcd-audit:allow` spelling too). The network severities come from the merged scanner configuration, so a repo that raises one in `.abcd/config/pii.json` is honoured, and an override that cannot be read is itself an `error` finding saying the scan fell back to the built-in severities. Two findings report what was *not* read rather than a leak: a tracked text file over the 4 MiB scan cap, and one that could not be opened. Binary files are skipped silently | +| `privacy-hygiene` | error (network-identifier findings mapped from a scanner `warn`/`info` land as `warn`) | three leak classes on any tracked text line: absolute local paths in committed files, real network identifiers, and the harness-leak pair the outbound policy bans everywhere (a live agent-session URL, and a tool's own "generated with" footer). The fix names reserved documentation values (RFC 5737/3849/2606/7042, or a persona-derived device name), and an `abcd-lint:allow` line waiver is honoured (the `abcd-audit:allow` spelling too). Each line is read as written and in the scanner's decoded spellings of it (`scanner.DecodedViews`: its percent and JSON-escape views), so a home path, an address or a harness-leak shape written behind an escape in a JSON fixture, export or transcript is the finding its plain spelling is; the waiver is read on the line as written, and the record/docs `harness_leak` rule reads the same spellings. The network severities come from the merged scanner configuration, so a repo that raises one in `.abcd/config/pii.json` is honoured, and an override that cannot be read is itself an `error` finding saying the scan fell back to the built-in severities. Two findings report what was *not* read rather than a leak: a tracked text file over the 4 MiB scan cap, and one that could not be opened. Binary files are skipped silently | | `site-gates` | warn | where `.abcd/site.json` declares a site: renders it into a fresh temporary directory outside the repository, runs the website's gates over it (the site target's, [`22-site.md`](22-site.md)), and removes it, so the lint still writes nothing in the repository. Each gate failure is one finding, filed against the source span it names; a composition that cannot be rendered is a finding against `.abcd/site.json` rather than an aborted lint. Warn, as `docs-currency` is, because the authoritative gate is the site target's exit 1 and re-raising it as an error would double-gate one check | | `identity-positioning` | warn | every registered surface still carries the canonical identity block's tagline (and pitch, where required), and every registered surface can still be found: a surface whose locator matches nothing is its own finding, because drift there would go unseen. A registry or identity block that cannot be read is reported rather than passed. Gated on `.abcd/positioning.json` being present on disk, and per-repo upgradeable to `error` (see [`19-identity.md`](19-identity.md)) | From c6514ce56b66afad63df02937f39b29c10629b11 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:25:15 +0100 Subject: [PATCH 40/64] =?UTF-8?q?chore:=20resolve=20iss-2609261658553101?= =?UTF-8?q?=20=E2=80=94=20the=20lint=20rules=20read=20decoded=20spellings?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261658553101 Assisted-by: Claude:claude-opus-5-5 --- ...-hygiene-rule-and-the-harness-leak-lint.md | 14 ------------ ...-hygiene-rule-and-the-harness-leak-lint.md | 22 +++++++++++++++++++ 2 files changed, 22 insertions(+), 14 deletions(-) delete mode 100644 .abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md create mode 100644 .abcd/work/issues/resolved/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md diff --git a/.abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md b/.abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md deleted file mode 100644 index 88c923979..000000000 --- a/.abcd/work/issues/open/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md +++ /dev/null @@ -1,14 +0,0 @@ ---- -schema_version: 1 -id: "iss-2609261658553101" -slug: "the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint" -severity: "minor" -category: "security" -source: "agent-finding" -found_during: "autonomous run A resumed 2026-09-25" -origin: researcher-authored -production_mode: hand-written -found_at: "internal/core/repolint/rule_privacy.go" ---- - -The repolint privacy-hygiene rule and the harness_leak lint rule read each committed line raw, so the JSON-escape spellings the scanner reads through its decoded views (jsonescape.go) pass both: a third-party home after a \n escape, a home written with the solidus escape, a Windows home with doubled separators, and a token or a session URL written straight after a \n escape (the patterns anchor on a leading word boundary, and the escape letter is a word byte). A committed JSON fixture, export or transcript carrying any of them is not refused. Reading the views in the lint is not contained: Go interpreted string literals use the same escapes, so the views would surface every test string that puts a \n escape before a /home root and a name in committed Go source, and the waivers on those lines do not all exist. The CI gitleaks history scan covers the token half; the home-path and session-URL halves have no other gate. Detector: privacyLeak and harnessLeakOnLine report each of the four spellings on a committed line. diff --git a/.abcd/work/issues/resolved/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md b/.abcd/work/issues/resolved/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md new file mode 100644 index 000000000..12515574c --- /dev/null +++ b/.abcd/work/issues/resolved/iss-2609261658553101-the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint.md @@ -0,0 +1,22 @@ +--- +schema_version: 1 +id: "iss-2609261658553101" +slug: "the-repolint-privacy-hygiene-rule-and-the-harness-leak-lint" +severity: "minor" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/core/repolint/rule_privacy.go" +resolution: "Both rules now read each committed line as written and through scanner.DecodedViews, the exported seam onto the scanner's one list of decoded views (the percent pre-pass and every JSON-escape layer), so a home after a newline escape, the solidus and unicode escapes, doubled Windows separators, a percent-encoded home, a session URL straight after a newline escape and a footer on its own line inside a quoted string are all refused as their plain spellings are. The footer's line-start SkipAt reads a break inside the text as a line start, which reaches the scanner's own views too. The Go-literal cost was measured on the live tree with abcd lint: 101 findings before, 103 after; twelve deliberately illustrative test-fixture lines that read newly carry the rule's waiver, no production line reads newly, and the remaining three are resolved records describing the escaped spellings with placeholder names, the same class as the record findings already on the baseline. docs-lint and record-lint report nothing new. Tests: TestAC_PrivacyReadsEscapedSpellings, TestAC_PrivacyEscapedSpellingsKeepTheExemptions, TestHarnessLeakReadsEscapedSpellings, TestHarnessLeakEscapedSpellingsSpareProse, TestFooterAfterADecodedLineBreakOwnsItsLine." +impact: fix +resolved_by: + commit: "facc64db" +--- + +The repolint privacy-hygiene rule and the harness_leak lint rule read each committed line raw, so the JSON-escape spellings the scanner reads through its decoded views (jsonescape.go) pass both: a third-party home after a \n escape, a home written with the solidus escape, a Windows home with doubled separators, and a token or a session URL written straight after a \n escape (the patterns anchor on a leading word boundary, and the escape letter is a word byte). A committed JSON fixture, export or transcript carrying any of them is not refused. Reading the views in the lint is not contained: Go interpreted string literals use the same escapes, so the views would surface every test string that puts a \n escape before a /home root and a name in committed Go source, and the waivers on those lines do not all exist. The CI gitleaks history scan covers the token half; the home-path and session-URL halves have no other gate. Detector: privacyLeak and harnessLeakOnLine report each of the four spellings on a committed line. + +## Grounds + +- pursued: a committed line whose escapes hide a home path, an address or a harness-leak shape is refused by both rules at no production-code false-positive cost; a production source line flagged only through a decoded view, or an escaped leak either rule still passes, would show it wrong From dd4f0e9f6e9a3450e49556c013917adf003db167 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:27:12 +0100 Subject: [PATCH 41/64] =?UTF-8?q?chore:=20capture=20iss-2609261827066511?= =?UTF-8?q?=20=E2=80=94=20UTF-16=20text=20on=20bytes=20is=20never=20read?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Split from iss-2609261659051539: its UTF-16 half needs no format parser and is taken in this lane, while its EXIF half needs an IFD reader. Refs: iss-2609261827066511, iss-2609261659051539 Assisted-by: Claude:claude-opus-5-5 --- ...d-byte-scan-never-matches-a-value-written-as.md | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md diff --git a/.abcd/work/issues/open/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md b/.abcd/work/issues/open/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md new file mode 100644 index 000000000..d36b2d1a7 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261827066511" +slug: "the-payload-byte-scan-never-matches-a-value-written-as" +severity: "minor" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/scanner.go" +--- + +The payload byte scan never matches a value written as UTF-16 text behind a byte-order mark, whatever its length: a PDF text string that opens with FE FF (the /Author, /Creator or /Title of a document with a non-ASCII field) or a little-endian string that opens with FF FE interleaves a zero byte with every letter, so the caller's real name, home path and email in it raise nothing and the file publishes. This is the UTF-16 half of iss-2609261659051539, split from its EXIF half, which needs an IFD reader; it reaches long values too, not only the short names that record names. Detector: the caller's name, home path or email written as byte-order-marked UTF-16 in a skip-listed file is a hard_fail finding in the payload scan, and a short name there is kept only where a person key stands within reach, as in plain bytes. From 75ce1ac46281415187b34f0c7d945cd9189105cd Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:31:18 +0100 Subject: [PATCH 42/64] fix(scanner): read byte-order-marked UTF-16 text in the byte scan A PDF text string outside PDFDocEncoding is UTF-16 behind FE FF, and a Windows writer stores strings little-endian behind FF FE; a zero byte stands between every two ASCII letters, so the byte scan never matched the caller's name, home path or email written that way, whatever its length. scanBytes now decodes every run of UTF-16 text a byte-order mark opens (utf16View), joins the runs one to a line into a single view, scans it once with the same byte rules, and re-homes each finding onto the raw offset of its first code unit. The short-name person-key rule then judges the raw bytes before the run, so a PDF "/Author (" before a UTF-16 string keeps a short name there, while a short name with no key in reach is still dropped. Matched stays the decoded name, so the length rule counts the name rather than its zero-interleaved spelling. A run ends at a control, a surrogate, a noncharacter or a symbol unit, which is what ends a big-endian PDF string: its closing ')' pairs with the next byte into an arrows-block unit. Each raw byte is decoded at most once and the view is scanned in one call, so the cost stays linear (TestUTF16ViewWorkIsLinear, a new utf16 meter stage). A run with no mark (EXIF XPAuthor, legacy binary documents) is not read: telling it from chance bytes needs the structure it sits in. The launch dry-run over this tree reports the same findings before and after. Refs: iss-2609261827066511 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/meter.go | 2 + internal/adapter/scanner/scanner.go | 16 ++- internal/adapter/scanner/utf16.go | 147 +++++++++++++++++++++++++ internal/adapter/scanner/utf16_test.go | 137 +++++++++++++++++++++++ 4 files changed, 293 insertions(+), 9 deletions(-) create mode 100644 internal/adapter/scanner/utf16.go create mode 100644 internal/adapter/scanner/utf16_test.go diff --git a/internal/adapter/scanner/meter.go b/internal/adapter/scanner/meter.go index 390cace80..28eb53ba8 100644 --- a/internal/adapter/scanner/meter.go +++ b/internal/adapter/scanner/meter.go @@ -44,6 +44,8 @@ const ( stagePercent = "percent" // stageJSONEscape is each JSON-unescape layer decoded from the line. stageJSONEscape = "json_escape" + // stageUTF16 is the byte scan's walk for byte-order-marked UTF-16 runs. + stageUTF16 = "utf16" ) // costMeter records per-stage scan work. The zero value charges nothing. diff --git a/internal/adapter/scanner/scanner.go b/internal/adapter/scanner/scanner.go index 7ca9961c5..93209d55f 100644 --- a/internal/adapter/scanner/scanner.go +++ b/internal/adapter/scanner/scanner.go @@ -1163,6 +1163,7 @@ func (s *Scanner) scanBytes(data []byte, secrets []Pattern, logical string) []Fi } all := scanText(string(data), long, secrets, s.identSev, logical, true) meta := metadataFields{data: data} + all = append(all, s.utf16Findings(data, long, secrets, logical, &meta)...) out := all[:0] for _, f := range all { if s.byteScanDrops(f, &meta) { @@ -1201,9 +1202,11 @@ const byteScanLongLiteral = 8 // document's dc:creator and cp:lastModifiedBy. Each is written as text in the // raw bytes (or in a region the container decoder inflates), so a short name // after one is the name the file was stamped with rather than a chance run of -// bytes (iss-2609090934372160). A key read from binary structure — EXIF's -// Artist tag, a UTF-16 PDF string — is not text in the bytes and is not -// reached here. +// bytes (iss-2609090934372160). A UTF-16 value behind a byte-order mark is +// decoded first (utf16.go) and judged by the raw bytes before it, so a PDF +// "/Author (" before a UTF-16 string reaches its name. A key read from binary +// structure — EXIF's Artist tag — is not text in the bytes and is not reached +// here (iss-2609261659051539). var metadataPersonKeys = [][]byte{[]byte("author"), []byte("artist"), []byte("creator"), []byte("lastmodifiedby")} // maxMetadataKeyGap is how far before a short name metadataFields looks for a @@ -1226,12 +1229,7 @@ type metadataFields struct { // newline and its Column is the byte offset on that line. func (m *metadataFields) holds(f Finding) bool { if m.starts == nil { - m.starts = []int{0} - for i, b := range m.data { - if b == '\n' { - m.starts = append(m.starts, i+1) - } - } + m.starts = lineStarts(m.data) } if f.Line < 1 || f.Line > len(m.starts) || f.Column < 1 { return false diff --git a/internal/adapter/scanner/utf16.go b/internal/adapter/scanner/utf16.go new file mode 100644 index 000000000..0ab2c05d5 --- /dev/null +++ b/internal/adapter/scanner/utf16.go @@ -0,0 +1,147 @@ +package scanner + +import ( + "sort" + "unicode" + "unicode/utf16" + "unicode/utf8" +) + +// utf16.go — the UTF-16 view of a byte scan (iss-2609261827066511). +// +// A PDF text string that holds anything outside PDFDocEncoding is written as +// UTF-16 behind the FE FF byte-order mark, and a Windows writer stores its +// strings little-endian behind FF FE. Either way a zero byte stands between +// every two ASCII letters, so the byte scan, which reads the file as one long +// string, never saw a name, a home path or an email written that way, however +// long. The fix is a decode, not a format parser: every run of UTF-16 text a +// byte-order mark opens is decoded to UTF-8, the runs are joined one to a line +// into a single view, and the view is scanned with the same byte rules as the +// raw bytes. Each hit is mapped back to the raw offset its first unit sits at, +// so the person-key rule (metadataFields) judges a short name by the raw bytes +// before it — a PDF's "/Author (" stands right before the mark. +// +// A run without a mark (EXIF's XPAuthor tag, a legacy binary document) is not +// read: telling UTF-16 from chance bytes there needs the structure the text +// sits in, which is iss-2609261659051539's IFD reader, not a decode. + +// minUTF16Run is the fewest code units a run needs to be read: a mark before +// a single unit is chance far more often than text. +const minUTF16Run = 2 + +// utf16View returns the UTF-16 runs a byte-order mark opens in data, decoded +// and joined one run to a line, with the position map of each decoded byte to +// the raw offset its code unit began at (a separator maps to the end of the +// run before it, and the sentinel to len(data)). ok is false when data holds +// no such run. Every raw byte is decoded at most once, so the view costs what +// the data does. +func utf16View(data []byte) (decodedView, bool) { + var text []byte + var pos []int + for i := 0; i+1 < len(data); { + var bigEndian bool + switch { + case data[i] == 0xfe && data[i+1] == 0xff: + bigEndian = true + case data[i] == 0xff && data[i+1] == 0xfe: + default: + i++ + continue + } + start, mark := len(text), i+2 + j := mark + for ; j+1 < len(data); j += 2 { + u := uint16(data[j+1])<<8 | uint16(data[j]) + if bigEndian { + u = uint16(data[j])<<8 | uint16(data[j+1]) + } + r := rune(u) + if !utf16TextRune(r) { + break + } + var enc [utf8.UTFMax]byte + for _, c := range enc[:utf8.EncodeRune(enc[:], r)] { + text = append(text, c) + pos = append(pos, j) + } + } + scanMeter.charge(stageUTF16, j-i) + if (j-mark)/2 < minUTF16Run { + text, pos = text[:start], pos[:start] + } else { + text = append(text, '\n') + pos = append(pos, j) + } + i = j + } + if len(text) == 0 { + return decodedView{}, false + } + return decodedView{text: string(text), posMap: append(pos, len(data))}, true +} + +// utf16TextRune reports whether a code unit reads as text inside a run: the +// layout controls, printable Latin through the IPA block, and beyond it the +// letters, marks, digits, punctuation and spaces of any script. A control, a +// surrogate (no BMP rune on its own), a noncharacter and a symbol end the run. +// The symbol clause is what ends a big-endian PDF string: the ')' closing it +// pairs with the byte after it into a unit in the arrows block. +func utf16TextRune(r rune) bool { + switch { + case r == '\t' || r == '\n' || r == '\r': + return true + case r < 0x20 || r == 0x7f || (r >= 0x80 && r < 0xa0) || utf16.IsSurrogate(r) || r >= 0xfffe: + return false + case r < 0x300: + return true + } + return unicode.IsLetter(r) || unicode.IsMark(r) || unicode.IsDigit(r) || unicode.IsPunct(r) || unicode.IsSpace(r) +} + +// utf16Findings scans the UTF-16 view of data with the byte rules and re-homes +// each finding onto the raw bytes: Line and Column name the raw position of +// the value's first code unit (meta's line starts), while Matched and the +// snippet stay the decoded text, so the short-name length rule counts the +// name's own bytes rather than its zero-interleaved spelling, and a serialized +// finding masks the name as it reads. +func (s *Scanner) utf16Findings(data []byte, id Identity, secrets []Pattern, logical string, meta *metadataFields) []Finding { + v, ok := utf16View(data) + if !ok { + return nil + } + starts := lineStarts([]byte(v.text)) + var out []Finding + for _, f := range scanText(v.text, id, secrets, s.identSev, logical, true) { + if f.Line < 1 || f.Line > len(starts) { + continue + } + at := starts[f.Line-1] + f.Column - 1 + if at < 0 || at >= len(v.text) { + continue + } + f.Line, f.Column = meta.position(v.posMap[at]) + out = append(out, f) + } + return out +} + +// lineStarts returns the offset each '\n'-separated line of data starts at. +func lineStarts(data []byte) []int { + starts := []int{0} + for i, b := range data { + if b == '\n' { + starts = append(starts, i+1) + } + } + return starts +} + +// position is the 1-based line and column of raw offset at in m's data, in +// the coordinates scanText gives a finding on the same bytes. +func (m *metadataFields) position(at int) (int, int) { + if m.starts == nil { + m.starts = lineStarts(m.data) + } + line := sort.Search(len(m.starts), func(i int) bool { return m.starts[i] > at }) + return line, at - m.starts[line-1] + 1 +} diff --git a/internal/adapter/scanner/utf16_test.go b/internal/adapter/scanner/utf16_test.go new file mode 100644 index 000000000..91cfa8e17 --- /dev/null +++ b/internal/adapter/scanner/utf16_test.go @@ -0,0 +1,137 @@ +package scanner + +import ( + "bytes" + "strings" + "testing" + "unicode/utf16" +) + +// utf16Bytes encodes s as UTF-16 behind the byte-order mark of the given +// order, the way a PDF text string (big-endian, FE FF) or a Windows writer +// (little-endian, FF FE) stores it. +func utf16Bytes(s string, bigEndian bool) []byte { + var b bytes.Buffer + if bigEndian { + b.Write([]byte{0xfe, 0xff}) + } else { + b.Write([]byte{0xff, 0xfe}) + } + for _, u := range utf16.Encode([]rune(s)) { + if bigEndian { + b.Write([]byte{byte(u >> 8), byte(u)}) + } else { + b.Write([]byte{byte(u), byte(u >> 8)}) + } + } + return b.Bytes() +} + +// TestUTF16TextOnBytesIsRead pins the byte scan to UTF-16 text a byte-order +// mark opens (iss-2609261827066511): a PDF /Author written as a +// UTF-16 text string interleaves a zero byte with every letter, so no name, +// home path or email in it matched at all, whatever its length. The run is +// decoded and scanned with the byte rules, and a short name in it is kept by +// the person-key rule exactly as the plain-text /Author is. +func TestUTF16TextOnBytesIsRead(t *testing.T) { + id := synthIdentity() + cases := []struct { + name, logical string + body []byte + kind string + }{ + {"short name in a UTF-16 /Author", "deck.pdf", + append(append([]byte("%PDF-1.7\n1 0 obj\n<< /Author ("), utf16Bytes("Zedqx", true)...), []byte(") >>\nendobj\n")...), kindRealName}, + {"multi-word name in a UTF-16 /Author", "deck.pdf", + append(append([]byte("%PDF-1.7\n<< /Author ("), utf16Bytes(id.GitUserName, true)...), []byte(") >>\n")...), kindRealName}, + {"home path in a little-endian string", "shot.png", + append(append([]byte("\x89PNG\r\n\x1a\n\x00\x01"), utf16Bytes(id.HomePath+"/deck.key", false)...), 0, 0, 0x7f), kindHomeSelf}, + {"email in a UTF-16 /Creator", "deck.pdf", + append(append([]byte("%PDF-1.7\n<< /Creator ("), utf16Bytes("by "+id.GitUserEmail, true)...), []byte(") >>\n")...), kindRealEmail}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + root := t.TempDir() + sc, err := New(root) + if err != nil { + t.Fatal(err) + } + sc.identity = id + if c.name == "short name in a UTF-16 /Author" { + sc.identity = Identity{GitUserName: "Zedqx"} + } + res := scanOne(t, sc, c.logical, writeFile(t, root, c.logical, string(c.body))) + if !hasKind(res.Findings, c.kind) || res.HardFails == 0 { + t.Errorf("no hard_fail %s in the UTF-16 text: %+v", c.kind, res.Findings) + } + }) + } +} + +// TestUTF16TextOnBytesKeepsTheShortNameRule pins the other half: a short name +// in UTF-16 text with no person key within reach is chance noise on bytes and +// is dropped, as it is in plain bytes, and a byte-order mark followed by +// binary content decodes nothing that raises a finding. +func TestUTF16TextOnBytesKeepsTheShortNameRule(t *testing.T) { + var noise bytes.Buffer + noise.WriteString("\x89PNG\r\n\x1a\n") + noise.Write(append([]byte("\x00\x00"), utf16Bytes("scanned page "+strings.Repeat("q", 60)+" Zedqx seen", true)...)) + for i := 0; i < 64; i++ { + noise.Write([]byte{0xfe, 0xff, byte(i), 0x00, 0xff, 0xfe, 0x00, byte(i)}) + } + root := t.TempDir() + sc, err := New(root) + if err != nil { + t.Fatal(err) + } + sc.identity = Identity{GitUserName: "Zedqx"} + res := scanOne(t, sc, "noise.png", writeFile(t, root, "noise.png", noise.String())) + if hasKind(res.Findings, kindRealName) { + t.Errorf("a short name with no person key in reach was reported: %+v", res.Findings) + } +} + +// TestUTF16ViewWorkIsLinear holds the byte scan with its UTF-16 view to the +// cost class the line scan is held to: quadrupling a payload dense in marks, +// in short runs and in one long run at most multiplies the charged work by +// linearCostBar, because each raw byte is decoded at most once and the runs +// are scanned as one view. +func TestUTF16ViewWorkIsLinear(t *testing.T) { + if raceEnabled { + t.Skip("a deterministic count gains nothing under -race; the uninstrumented run asserts it") + } + sc := &Scanner{identity: Identity{GitUserName: "Zedqx"}, identSev: DefaultIdentitySeverities()} + secrets := secretPatterns(DefaultPatterns()) + shapes := []struct { + name string + unit []byte + }{ + {"marks with nothing after them", []byte{0xfe, 0xff, 0xff, 0xfe}}, + {"short runs", append(utf16Bytes("ab", true), 0, 0)}, + {"short names by a key", append([]byte("/Author ("), append(utf16Bytes("Zedqx", true), ')', ' ')...)}, + {"one long run", utf16Bytes("Zedqx ", false)[2:]}, + } + for _, s := range shapes { + t.Run(s.name, func(t *testing.T) { + charge := func(data []byte) int { + total := 0 + scanMeter.tally = func(_ string, n int) { total += n } + defer func() { scanMeter.tally = nil }() + sc.scanBytes(data, secrets, "f") + return total + } + build := func(n int) []byte { + out := []byte{0xff, 0xfe} + return append(out, bytes.Repeat(s.unit, n)...) + } + base := max(4096/len(s.unit), 1) + lo, hi := charge(build(base)), charge(build(4*base)) + if lo == 0 { + t.Fatal("the shape charged nothing; it pins nothing") + } + if growth := float64(hi) / float64(lo); growth > linearCostBar { + t.Errorf("quadrupling the payload multiplied the byte scan's charge by %.2fx, want at most %.1fx", growth, linearCostBar) + } + }) + } +} From 4f2f460ea062a895694ccd583d4916c2e69d2025 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:31:21 +0100 Subject: [PATCH 43/64] =?UTF-8?q?chore:=20resolve=20iss-2609261827066511?= =?UTF-8?q?=20=E2=80=94=20the=20byte=20scan=20reads=20marked=20UTF-16=20te?= =?UTF-8?q?xt?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261827066511 Assisted-by: Claude:claude-opus-5-5 --- ...-payload-byte-scan-never-matches-a-value-written-as.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md (54%) diff --git a/.abcd/work/issues/open/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md b/.abcd/work/issues/resolved/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md similarity index 54% rename from .abcd/work/issues/open/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md rename to .abcd/work/issues/resolved/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md index d36b2d1a7..be660f112 100644 --- a/.abcd/work/issues/open/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md +++ b/.abcd/work/issues/resolved/iss-2609261827066511-the-payload-byte-scan-never-matches-a-value-written-as.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25" origin: researcher-authored production_mode: hand-written found_at: "internal/adapter/scanner/scanner.go" +resolution: "scanBytes decodes every run of UTF-16 text a byte-order mark opens, big- or little-endian, into one view scanned with the byte rules, and re-homes each finding onto the raw offset of its first code unit, so the caller's name, home path or email written as UTF-16 is a hard_fail finding, a short name behind a PDF /Author key is kept by the person-key rule, and a short name with no key in reach is still dropped. Tests: TestUTF16TextOnBytesIsRead, TestUTF16TextOnBytesKeepsTheShortNameRule, TestUTF16ViewWorkIsLinear. A run with no byte-order mark stays unread and belongs with the EXIF residue in iss-2609261659051539." +impact: fix +resolved_by: + commit: "75ce1ac4" --- The payload byte scan never matches a value written as UTF-16 text behind a byte-order mark, whatever its length: a PDF text string that opens with FE FF (the /Author, /Creator or /Title of a document with a non-ASCII field) or a little-endian string that opens with FF FE interleaves a zero byte with every letter, so the caller's real name, home path and email in it raise nothing and the file publishes. This is the UTF-16 half of iss-2609261659051539, split from its EXIF half, which needs an IFD reader; it reaches long values too, not only the short names that record names. Detector: the caller's name, home path or email written as byte-order-marked UTF-16 in a skip-listed file is a hard_fail finding in the payload scan, and a short name there is kept only where a person key stands within reach, as in plain bytes. + +## Grounds + +- pursued: a value written as byte-order-marked UTF-16 in a skip-listed file is found by the payload scan at linear cost without new false positives; a launch dry-run that reports a new finding on this tree's own assets, or a marked UTF-16 name the scan still passes, would show it wrong From b114e4c36ce62052c2edad4ac7099ec8b326b5ec Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:32:07 +0100 Subject: [PATCH 44/64] chore: defer the EXIF and PDF-string residues of the UTF-16 byte scan iss-2609261659051539 keeps its EXIF half, deferred past v0.11.0 with the IFD reader it needs named: the Artist and Copyright values sit at an offset only a binary IFD entry names, so a parser-free reach cannot tell them from chance bytes. Its UTF-16 half shipped as iss-2609261827066511. iss-2609261831352258, captured by the sibling sweep of that fix, is the hex and octal-escape spellings of a PDF text string, which put no UTF-16 bytes in the file; reading them decodes PDF string syntax, a piece of a PDF reader beyond this lane's parser-free scope, so it is deferred past v0.11.0 with what the decode needs named. Refs: iss-2609261659051539, iss-2609261831352258, iss-2609261827066511 Assisted-by: Claude:claude-opus-5-5 --- ...en-real-name-in-a-jpeg-s-or-tiff-s-exif.md | 10 ++++++++++ ...ng-written-in-hex-a-feff-led-run-of-hex.md | 20 +++++++++++++++++++ 2 files changed, 30 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md diff --git a/.abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md b/.abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md index 6fd46d16a..0dc646114 100644 --- a/.abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md +++ b/.abcd/work/issues/open/iss-2609261659051539-a-short-single-token-real-name-in-a-jpeg-s-or-tiff-s-exif.md @@ -9,6 +9,16 @@ found_during: "autonomous run A resumed 2026-09-25" origin: researcher-authored production_mode: hand-written found_at: "internal/adapter/scanner/scanner.go" +deferred_after: "v0.11.0" +deferral_reason: "needs a format reader the scanner does not have: the EXIF Artist (0x013B) and Copyright (0x8298) values are ASCII at an offset only a binary IFD entry names, so keeping a short name there means finding the TIFF header (a JPEG APP1 Exif segment, a PNG eXIf chunk, or a TIFF file's own header), reading its byte order, walking IFD0's entry count and twelve-byte entries with every offset bounds-checked against the segment, and scanning the Artist, Copyright and XPAuthor (UTF-16LE with no byte-order mark) values with the text rules; no parser-free reach tells those bytes from chance, and the UTF-16 half needed none, so it shipped on its own as iss-2609261827066511" --- A short single-token real name in a JPEG's or TIFF's EXIF Artist tag (IFD0 tag 0x013B) is still dropped by the payload byte scan as chance noise. The byte scan keeps a short name only where a person metadata key stands as text within reach before it (metadataPersonKeys in internal/adapter/scanner/scanner.go: a PDF /Author, XMP dc:creator, a PNG text Author, an OOXML cp:lastModifiedBy), and EXIF stores the Artist tag as a binary IFD entry whose ASCII value sits at an offset with no key text beside it; a UTF-16 PDF string (/Author with a FEFF byte-order mark) is out of reach the same way. Closing it needs the IFD read structurally, the way container.go walks PNG chunks: find the Exif header, read the TIFF byte order, walk IFD0 and scan the Artist (and XPAuthor) value with the text rules. Residue of iss-2609090934372160. Detector: a short banned name in a camera-written EXIF Artist tag, with no XMP packet beside it, is a real_name finding in the payload scan. + +**Split and deferred (2026-09-26, autonomous run A, lane drainS3).** The +UTF-16 half of this record needs no format parser: a byte-order-marked +UTF-16 run is decoded and scanned as a view of the bytes, which +iss-2609261827066511 shipped. What stays here is the EXIF half, and it needs +an IFD reader; the deferral names what that reader has to do. BOM-less +UTF-16 (EXIF XPAuthor, legacy binary documents) stays with it, since only +the structure around such a run tells it from chance bytes. diff --git a/.abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md b/.abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md new file mode 100644 index 000000000..9f41ef866 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md @@ -0,0 +1,20 @@ +--- +schema_version: 1 +id: "iss-2609261831352258" +slug: "a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex" +severity: "minor" +category: "security" +source: "agent-finding" +found_during: "autonomous run A resumed 2026-09-25" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/utf16.go" +deferred_after: "v0.11.0" +deferral_reason: "decoding PDF string syntax is a piece of a PDF reader, beyond the parser-free decodes lane drainS3 was scoped to: it needs each hex string's angle brackets and each literal string's balanced parentheses found, the hex pairs and the backslash escapes (octal, the named controls and line continuations) decoded with every byte mapped back to its raw offset, and the decoded bytes handed to the UTF-16 view, which then reaches the name" +--- + +A PDF text string written in hex (a FEFF-led run of hex digits between angle brackets) or with octal escapes (a backslash-376, backslash-377 opening inside parentheses), the forms several common PDF writers use for a non-ASCII document field, is read by neither the payload byte scan nor its UTF-16 view: neither spelling puts the UTF-16 bytes in the file, so the caller's name in such an /Author or /Creator raises nothing at any length. Reading them is a decode of the string syntax ahead of the UTF-16 view (hex pairs, and the PDF literal-string escapes), in the shape of the percent and JSON-escape views, not a format parser. Detector: the caller's name in a UTF-16 /Author written in hex or with octal escapes is a real_name finding in the payload scan. + +**Deferred (2026-09-26, autonomous run A, lane drainS3).** Found by the +sibling sweep of iss-2609261827066511, whose UTF-16 view reads a +byte-order-marked run only where the file holds the UTF-16 bytes themselves. From 359c07ea1d97b5b80b35a89b2f06233498a243e6 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:00:16 +0100 Subject: [PATCH 45/64] =?UTF-8?q?chore:=20capture=20iss-2609261900095459?= =?UTF-8?q?=20=E2=80=94=20an=20alias=20used=20as=20a=20frontmatter=20key?= =?UTF-8?q?=20travels=20past=20the=20floor?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by the review of the drain-reading lane: `k: &a origin` then `*a : X` is admitted by the reading floor and resolves to {origin: X}. Refs: iss-2609261900095459 Assisted-by: Claude:claude-opus-5-5 --- ...floor-admits-a-yaml-alias-used-as-a-key-in-a.md | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md diff --git a/.abcd/work/issues/open/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md b/.abcd/work/issues/open/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md new file mode 100644 index 000000000..14348cfa4 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261900095459" +slug: "the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a" +severity: "minor" +category: "security" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: review-drainRd" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/core/reading/project.go" +--- + +The reading floor admits a YAML alias used as a key in a frontmatter block, so an excluded key travels past the redaction under a manifest asserting its refusal. `k: &a origin` followed by `*a : X` (or `m: {*a : X}`) is admitted, and a YAML reader resolves it to {origin: X}. The anchor refusal in unresolvableFrontmatterShape fires only at line start or after a block indicator (nestedBlockEntry); an anchor in value or flow position is admitted, and nothing inspects `*` at all. Same class as iss-2608301237450573. From 17e795d5bb5d27bdbf5c296cab2467a5c24304c1 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:04:43 +0100 Subject: [PATCH 46/64] fix(reading): refuse a YAML alias wherever a frontmatter key can stand An anchor sits on any node, a value included, so `k: &a origin` then `*a : X` reads as the key `origin` to YAML, and the floor admitted it: the anchor refusal fired only at line start and behind a block indicator, and nothing read `*` at all. The refusal is on the alias rather than the anchor, because that is the smaller complete rule: every node position can carry an anchor, while a key position is a short closed list, and a `*` can never open a plain scalar. The floor now refuses an alias at line start and behind `{`, `[` or `,` (node properties allowed between). Behind a block indicator the compact-mapping rule and in an explicit key the unreadable-key rule already refused it. An alias in a value position stays admitted: it copies a node whose text the floor already read where the anchor sits. Refs: iss-2609261900095459 Assisted-by: Claude:claude-opus-5-5 --- internal/core/reading/project.go | 22 +++++++++++ internal/core/reading/project_test.go | 55 +++++++++++++++++++++++++++ 2 files changed, 77 insertions(+) diff --git a/internal/core/reading/project.go b/internal/core/reading/project.go index 4df641d94..44f612992 100644 --- a/internal/core/reading/project.go +++ b/internal/core/reading/project.go @@ -239,6 +239,11 @@ var ( // flowExplicitKeyRe matches YAML's explicit-key indicator inside a flow // mapping: a `?` following `{` or `,`. Same class, same answer. flowExplicitKeyRe = regexp.MustCompile(`[{,]\s*\?`) + // flowAliasKeyRe matches an alias where a flow collection's key can stand: + // a `*` following `{`, `[` or `,`, in a line whose quoted scalars are + // blanked. Node properties are allowed between the two. YAML gives an alias + // none, but the floor does not rest a refusal on a reader refusing one. + flowAliasKeyRe = regexp.MustCompile(`[{\[,]\s*(?:[!&][^\s{}\[\],]*\s+)*\*`) ) // One shape this floor does NOT see, disclosed rather than claimed: a title @@ -1280,6 +1285,18 @@ func excludedRawTitle(readings []string, bounds []*rawHeadingBounds, p int, name // any explicit-key line the readable-key pattern cannot fully read is a key // whose name this package is not entitled to assume. // +// An ALIAS in a key position is refused wherever that position is: at line +// start, behind a block indicator (a compact mapping there, nestedBlockEntry), +// in an explicit key (the unreadable-key rule), and behind `{`, `[` or `,`. An +// anchor may sit on any node, a value included, so `k: &a origin` then +// `*a : X` is the key `origin` to YAML (iss-2609261900095459). The rule is on +// the alias rather than the anchor because it is the smaller complete one: +// every node position can carry an anchor, while a key position is a short, +// closed list, and a `*` can never open a plain scalar, so one there is an +// alias and nothing else. An alias in a VALUE position is admitted: it copies a +// node whose own text the floor has already read where the anchor sits, and a +// key inside that node was refused there. +// // The block BOUNDS matter for the same reason the keys do. The frontmatter // stripper closes on `---`, so a block closed by `...`, or opened and never // closed, makes the offset it reports overshoot into the body — and a scan that @@ -1301,6 +1318,7 @@ func unresolvableFrontmatterShape(lines []string, fenced []bool) (int, string, b continue } trimmed := strings.TrimLeft(lines[i], " \t") + bare, _ := blankQuoted(lines[i]) switch { // The fence delimiter is first because it is the shape that used to // switch the rest of this scan off. It can no longer do so — the mask @@ -1314,6 +1332,10 @@ func unresolvableFrontmatterShape(lines []string, fenced []bool) (int, string, b return i + 1, "a YAML tag", true case strings.HasPrefix(trimmed, "&"): return i + 1, "a YAML anchor", true + case strings.HasPrefix(trimmed, "*"): + return i + 1, "a YAML alias as a key", true + case flowAliasKeyRe.MatchString(bare): + return i + 1, "a YAML alias where a flow collection's key can stand", true case nestedBlockEntry(lines[i]) != "": return i + 1, nestedBlockEntry(lines[i]), true case flowExplicitKeyRe.MatchString(lines[i]): diff --git a/internal/core/reading/project_test.go b/internal/core/reading/project_test.go index be4b829b3..4f7084dae 100644 --- a/internal/core/reading/project_test.go +++ b/internal/core/reading/project_test.go @@ -506,6 +506,61 @@ func TestANestedMappingRefusesBehindEveryBlockIndicator(t *testing.T) { } } +// TestAnAliasInAKeyPositionRefuses (iss-2609261900095459). An anchor sits +// wherever a node can, a value included, and an alias written where a key +// stands IS that anchored scalar to YAML: `k: &a origin` then `*a : X` reads as +// {origin: X}. The anchor refusal fired only at line start and behind a block +// indicator, and nothing read `*` at all, so the key travelled. The refusal is +// of the alias in every key position — line start, behind a block indicator, +// behind `{`, `[` or `,` — whatever the anchored scalar says. +func TestAnAliasInAKeyPositionRefuses(t *testing.T) { + const pre, post = "---\nid: spc-1\n", "---\n\n# A record\n" + for name, front := range map[string]string{ + "an alias key at line start": "k: &a origin\n*a : ABCD-WARM-ORIGIN\n", + "an alias key in a flow mapping": "k: &a origin\nm: {*a : ABCD-WARM-ORIGIN}\n", + "an alias key after a flow comma": "k: &a origin\nm: {x: 1, *a : ABCD-WARM-ORIGIN}\n", + "an alias pair in a flow sequence": "k: &a origin\nm: [*a : ABCD-WARM-ORIGIN]\n", + "an alias key on a flow continuation": "k: &a origin\nm: {x: 1,\n *a : ABCD-WARM-ORIGIN}\n", + "a comma-first flow continuation": "k: &a origin\nm: {x: 1\n , *a : ABCD-WARM-ORIGIN}\n", + "an alias key in a nested mapping": "k: &a origin\nm:\n *a : ABCD-WARM-ORIGIN\n", + "an anchor behind a tag": "k: !!str &a origin\n*a : ABCD-WARM-ORIGIN\n", + "an anchor in a flow mapping's value": "m: {k: &a origin}\n*a : ABCD-WARM-ORIGIN\n", + "an anchor in a flow sequence": "l: [&a origin]\n*a : ABCD-WARM-ORIGIN\n", + "a tag before a flow alias key": "k: &a origin\nm: {!!str *a : ABCD-WARM-ORIGIN}\n", + "a CRLF alias key": "k: &a origin\r\n*a : ABCD-WARM-ORIGIN\r\n", + // Siblings refused before this change, kept refused. + "an alias as an explicit key": "k: &a origin\n? *a\n: ABCD-WARM-ORIGIN\n", + "an alias key in a sequence entry": "k: &a origin\nlinks:\n - *a : ABCD-WARM-ORIGIN\n", + "an alias key behind an explicit value": "k: &a origin\n? meta\n: *a : ABCD-WARM-ORIGIN\n", + "a merge over an anchored flow map": "base: &m {origin: ABCD-WARM-ORIGIN}\nuse:\n <<: *m\n", + "a merge over an anchored block map": "base: &m\n origin: ABCD-WARM-ORIGIN\nuse:\n <<: *m\n", + } { + err := refuses(t, "spc-1-a-record.md", pre+front+post, refusalKeys, refusalHeadings) + if err == nil { + t.Errorf("%s: admitted; the alias is an origin key to YAML and travels", name) + continue + } + if !strings.Contains(err.Error(), "spc-1-a-record.md") { + t.Errorf("%s: the refusal does not name the document: %v", name, err) + } + } + + // The anti-vacuity half: an alias in a VALUE position copies a node whose + // own text the floor already read where the anchor sits, and an asterisk + // that is not an alias is prose. + for name, front := range map[string]string{ + "an alias as a value": "k: &a origin\nuse: *a\n", + "an alias in a sequence entry": "k: &a origin\nlist:\n - *a\n", + "a merge over a harmless map": "base: &m {name: x}\nuse:\n <<: *m\n", + "an asterisk in a quoted value": "note: \"see [*] and {*a : b}\"\n", + "an asterisk inside a plain one": "note: a*b, c *d\n", + } { + if err := refuses(t, "spc-1-a-record.md", pre+front+post, refusalKeys, refusalHeadings); err != nil { + t.Errorf("%s was refused: %v", name, err) + } + } +} + // TestTheEscapedKeyRefusalStatesOnlyWhatItKnows (iss-2608301421381157). The // escaped-key refusal shared the excluded-key message, which asserted that the // document still carried an excluded key and that its block was not closed the From 70ae3adf9acd3c1d40b2240ffa39d6c2882265fc Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:05:39 +0100 Subject: [PATCH 47/64] =?UTF-8?q?chore:=20capture=20iss-2609261905354450?= =?UTF-8?q?=20=E2=80=94=20the=20status=20render's=20two=20run=20probes=20d?= =?UTF-8?q?isagree=20on=20a=20symlink?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by the review of the drain-reading lane: the stage marker probe reads through an unbounded Lstat, the staged-runs probe through the repository's os.Root. Refs: iss-2609261905354450 Assisted-by: Claude:claude-opus-5-5 --- ...tatus-render-probes-an-ingest-stage-s-commit.md | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md diff --git a/.abcd/work/issues/open/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md b/.abcd/work/issues/open/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md new file mode 100644 index 000000000..8fa1b97ce --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261905354450" +slug: "the-reading-status-render-probes-an-ingest-stage-s-commit" +severity: "minor" +category: "inconsistency" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: review-drainRd" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/core/reading/status.go" +--- + +The reading status render probes an ingest stage's commit marker with an unbounded os.Lstat over the joined path, while its staged-runs probe reads through an os.Root over the repository. The two disagree on a symlink: with .abcd/development/readings symlinked outside the checkout, a stage whose run is parked makes Describe refuse (path escapes from parent), while a stage alone is classified as a leftover or an orphan by a marker read outside the repository. Both probes should go through the one root. From 97fa2de399b6c9c65ef48852482f4aa32a233ef8 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:06:19 +0100 Subject: [PATCH 48/64] fix(reading): probe every staged run through the one repository root The status render read a stage's commit marker through an unbounded Lstat and a parked run's outcome through an os.Root, so the two disagreed on a symlink: with the readings directory symlinked out of the checkout, a parked run refused the render while a stage alone was classified by a marker read outside the repository. Describe now opens one root when anything is parked or staged and both probes read through it. Refs: iss-2609261905354450 Assisted-by: Claude:claude-opus-5-5 --- internal/core/reading/ingest_stage_test.go | 32 +++++++++++++++++++ internal/core/reading/status.go | 37 ++++++++++++---------- 2 files changed, 53 insertions(+), 16 deletions(-) diff --git a/internal/core/reading/ingest_stage_test.go b/internal/core/reading/ingest_stage_test.go index e68eb6a5b..b9f68f256 100644 --- a/internal/core/reading/ingest_stage_test.go +++ b/internal/core/reading/ingest_stage_test.go @@ -624,6 +624,38 @@ func TestTheBareRenderTellsALeftoverStageFromAnOrphan(t *testing.T) { } } +// TestTheBareRenderProbesEveryRunThroughTheOneRoot (iss-2609261905354450). The +// staged-runs probe reads through an os.Root over the repository, and the +// stage's commit-marker probe read through an unbounded Lstat, so the two +// disagreed on a symlink: with the readings directory symlinked out of the +// checkout, a parked run refused the render while a stage alone was classified +// by a marker read outside the repository. Both probes go through the root, so +// a stage alone refuses as a parked run does. +func TestTheBareRenderProbesEveryRunThroughTheOneRoot(t *testing.T) { + f := newIngestFixture(t, "detection") + f.mustIngest(f.payload(1)) + f.write(IngestStageDir+"/"+f.runID+"/"+stageFileName, + []byte(`{"_type":"`+StageType+`","run_id":"`+f.runID+`","records":[]}`)) + // No parked run, so only the stage's probe reaches the readings directory. + if err := os.RemoveAll(filepath.Join(f.root, filepath.FromSlash(DefaultRunDir))); err != nil { + t.Fatal(err) + } + readings := filepath.Join(f.root, filepath.FromSlash(ReadingsRecordDir)) + outside := filepath.Join(t.TempDir(), "readings") + if err := os.Rename(readings, outside); err != nil { + t.Fatal(err) + } + if err := os.Symlink(outside, readings); err != nil { + t.Fatal(err) + } + + status, err := Describe(f.root) + if err == nil { + t.Fatalf("a commit marker outside the repository classified the stage: leftover %v, orphaned %v", + status.LeftoverStages, status.OrphanedIngests) + } +} + // TestTheBareRenderListsOnlyTheParkedRunsAwaitingAnOutcome // (iss-2608311621412224). Nothing removes an assembly's parking directory after // its run is ingested, so `staged_runs` listed every run ever assembled: a diff --git a/internal/core/reading/status.go b/internal/core/reading/status.go index f57e8cc2d..7de6dc81d 100644 --- a/internal/core/reading/status.go +++ b/internal/core/reading/status.go @@ -93,7 +93,25 @@ func Describe(repoRoot string) (Status, error) { if err != nil && !os.IsNotExist(err) { return Status{}, fmt.Errorf("reading: listing the staged runs: %w", err) } - if s.StagedRuns, err = awaitingOutcome(repoRoot, runs); err != nil { + stages, err := os.ReadDir(filepath.Join(repoRoot, filepath.FromSlash(IngestStageDir))) + if err != nil && !os.IsNotExist(err) { + return Status{}, fmt.Errorf("reading: listing the ingest stage: %w", err) + } + if len(runs) == 0 && len(stages) == 0 { + return s, nil + } + + // Every probe of the durable tier goes through ONE root over the + // repository, so a parked run and a stage agree on a symlink: a record + // directory that escapes the checkout refuses the render for both, rather + // than refusing it for one and classifying the other by a marker read + // outside the repository (iss-2609261905354450). + root, err := os.OpenRoot(repoRoot) + if err != nil { + return Status{}, fmt.Errorf("reading: opening the repository to probe the staged runs: %w", err) + } + defer root.Close() + if s.StagedRuns, err = awaitingOutcome(root, runs); err != nil { return Status{}, err } @@ -104,16 +122,11 @@ func Describe(repoRoot string) (Status, error) { // committed and only the stage failed to clear, so the records stay and // only the stage goes. Calling both an orphan would tell an operator that a // committed run's records are about to be deleted. - stages, err := os.ReadDir(filepath.Join(repoRoot, filepath.FromSlash(IngestStageDir))) - if err != nil && !os.IsNotExist(err) { - return Status{}, fmt.Errorf("reading: listing the ingest stage: %w", err) - } for _, e := range stages { if !e.IsDir() || !recordid.ValidReadingRunID(e.Name()) { continue } - marker := filepath.Join(repoRoot, filepath.FromSlash(ReadingsRecordDir), e.Name(), RunFileName) - switch _, err := os.Lstat(marker); { + switch _, err := root.Lstat(ReadingsRecordDir + "/" + e.Name() + "/" + RunFileName); { case err == nil: s.LeftoverStages = append(s.LeftoverStages, e.Name()) case os.IsNotExist(err): @@ -134,16 +147,8 @@ func Describe(repoRoot string) (Status, error) { // tells an outstanding run from an ingested one is the record: an ingested run // has a commit marker or a refusal record under its id in the durable tier, the // same probe refuseARerun makes before an ingest writes. -func awaitingOutcome(repoRoot string, parked []os.DirEntry) ([]string, error) { +func awaitingOutcome(root *os.Root, parked []os.DirEntry) ([]string, error) { out := []string{} - if len(parked) == 0 { - return out, nil - } - root, err := os.OpenRoot(repoRoot) - if err != nil { - return nil, fmt.Errorf("reading: opening the repository to probe the staged runs: %w", err) - } - defer root.Close() for _, e := range parked { if !e.IsDir() || !strings.HasPrefix(e.Name(), RunIDFamily+"-") { continue From 398b0ba0326bf19bce4c2b60625f860b120711ca Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:06:28 +0100 Subject: [PATCH 49/64] =?UTF-8?q?chore:=20resolve=20iss-2609261900095459?= =?UTF-8?q?=20=E2=80=94=20an=20alias=20in=20a=20key=20position=20is=20refu?= =?UTF-8?q?sed?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261900095459 Assisted-by: Claude:claude-opus-5-5 --- ...eading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md (51%) diff --git a/.abcd/work/issues/open/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md b/.abcd/work/issues/resolved/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md similarity index 51% rename from .abcd/work/issues/open/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md rename to .abcd/work/issues/resolved/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md index 14348cfa4..75bbb64f1 100644 --- a/.abcd/work/issues/open/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md +++ b/.abcd/work/issues/resolved/iss-2609261900095459-the-reading-floor-admits-a-yaml-alias-used-as-a-key-in-a.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: review-drainRd" origin: researcher-authored production_mode: hand-written found_at: "internal/core/reading/project.go" +resolution: "The floor refuses a YAML alias at every key position: line start, behind { [ or , (node properties allowed between), with the block-indicator and explicit-key positions already refused by the compact-mapping and unreadable-key rules. The rule is on the alias, the smaller complete one; an alias in a value position stays admitted." +impact: fix +resolved_by: + commit: "17e795d5bb5d27bdbf5c296cab2467a5c24304c1" --- The reading floor admits a YAML alias used as a key in a frontmatter block, so an excluded key travels past the redaction under a manifest asserting its refusal. `k: &a origin` followed by `*a : X` (or `m: {*a : X}`) is admitted, and a YAML reader resolves it to {origin: X}. The anchor refusal in unresolvableFrontmatterShape fires only at line start or after a block indicator (nestedBlockEntry); an anchor in value or flow position is admitted, and nothing inspects `*` at all. Same class as iss-2608301237450573. + +## Grounds + +- pursued: every alias-as-key shape the review named and its siblings (flow continuation, tag, CRLF, anchors in flow and after a tag) is refused while a value alias and a harmless merge are admitted, and detection/widening/entailment dry-run assemblies keep 394/336/237 items and assembler_version 1.8.0+3c58dd; a YAML reader resolving an admitted frontmatter to an origin key would show it wrong From 0879fbb4d23b815d1b10d74f23f5ec9ed79716d6 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:06:39 +0100 Subject: [PATCH 50/64] =?UTF-8?q?chore:=20resolve=20iss-2609261905354450?= =?UTF-8?q?=20=E2=80=94=20the=20status=20render=20probes=20every=20run=20t?= =?UTF-8?q?hrough=20one=20root?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261905354450 Assisted-by: Claude:claude-opus-5-5 --- ...ading-status-render-probes-an-ingest-stage-s-commit.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md (54%) diff --git a/.abcd/work/issues/open/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md b/.abcd/work/issues/resolved/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md similarity index 54% rename from .abcd/work/issues/open/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md rename to .abcd/work/issues/resolved/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md index 8fa1b97ce..af112116d 100644 --- a/.abcd/work/issues/open/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md +++ b/.abcd/work/issues/resolved/iss-2609261905354450-the-reading-status-render-probes-an-ingest-stage-s-commit.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: review-drainRd" origin: researcher-authored production_mode: hand-written found_at: "internal/core/reading/status.go" +resolution: "Describe opens one os.Root over the repository when anything is parked or staged, and the stage's commit-marker probe and the parked run's outcome probe both read through it, so a readings directory symlinked out of the checkout refuses the render for both." +impact: fix +resolved_by: + commit: "97fa2de399b6c9c65ef48852482f4aa32a233ef8" --- The reading status render probes an ingest stage's commit marker with an unbounded os.Lstat over the joined path, while its staged-runs probe reads through an os.Root over the repository. The two disagree on a symlink: with .abcd/development/readings symlinked outside the checkout, a stage whose run is parked makes Describe refuse (path escapes from parent), while a stage alone is classified as a leftover or an orphan by a marker read outside the repository. Both probes should go through the one root. + +## Grounds + +- pursued: with the readings directory symlinked outside the checkout and only a stage present, Describe refuses instead of classifying the stage by a marker read outside the repository, and the existing leftover/orphan/staged-runs render tests stay green; a status render that reads a marker through a symlink out of the checkout would show it wrong From 3eaeb8b21bd578759aac256b20798b8fd6a3d0d7 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:09:21 +0100 Subject: [PATCH 51/64] =?UTF-8?q?chore:=20capture=20the=20review-drainS3?= =?UTF-8?q?=20follow-ups=20=E2=80=94=20hook=20escapes,=20astral=20runs,=20?= =?UTF-8?q?PDF=20hex?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three findings of the review of lane drainS3, captured before their fixes: the private-banlist guard reads staged blobs only as written, so an escaped spelling of a private name passes it; the UTF-16 byte view ends a run at a surrogate pair, so a name after an emoji is missed; and a PDF hex string, the hex half split from iss-2609261831352258, is never decoded. Refs: iss-2609261909106167 Refs: iss-2609261909101409 Refs: iss-2609261909108726 Refs: iss-2609261831352258 Assisted-by: Claude:claude-opus-5-5 --- ...-byte-view-internal-adapter-scanner-utf16-go.md | 14 ++++++++++++++ ...ist-guard-in-githooks-pre-commit-reads-every.md | 14 ++++++++++++++ ...-string-written-in-hex-a-feff-led-run-of-hex.md | 14 ++++++++++++++ 3 files changed, 42 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md create mode 100644 .abcd/work/issues/open/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md create mode 100644 .abcd/work/issues/open/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md diff --git a/.abcd/work/issues/open/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md b/.abcd/work/issues/open/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md new file mode 100644 index 000000000..81d15e4f5 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261909101409" +slug: "the-utf-16-byte-view-internal-adapter-scanner-utf16-go" +severity: "minor" +category: "security" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: review-drainS3" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/utf16.go" +--- + +The UTF-16 byte view (internal/adapter/scanner/utf16.go, utf16TextRune) ends a run at a surrogate, so a character outside the Basic Multilingual Plane, an emoji written as a surrogate pair, cuts a byte-order-marked UTF-16 string in two, and a name after it is never read when fewer than two code units stood before the pair: an emoji, a space and a name behind a byte-order mark yields no view at all, while the same name before the emoji is found. Detector: the caller's name after an emoji in one byte-order-marked UTF-16 string is a real_name finding in the payload scan. diff --git a/.abcd/work/issues/open/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md b/.abcd/work/issues/open/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md new file mode 100644 index 000000000..627ac1adf --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261909106167" +slug: "the-private-banlist-guard-in-githooks-pre-commit-reads-every" +severity: "minor" +category: "security" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: review-drainS3" +origin: researcher-authored +production_mode: hand-written +found_at: ".githooks/pre-commit" +--- + +The private-banlist guard in .githooks/pre-commit reads every staged blob raw under LC_ALL=C, so a private name written with JSON string escapes passes it: a \uXXXX spelling of any letter, non-ASCII or plain ASCII, or a \/ inside a pattern, puts bytes in the blob that no ERE written for the plain spelling matches, and a JSON transcript, export or fixture is the natural carrier. The store-before-commit redactors and the lint rules read the scanner's decoded views (lineViews: the percent view and the JSON escape layers); the hook, the one enforcement point of the private layer, reads only the text as written. Detector: a keyed banlist entry for a fake name refuses a staged JSON file that spells the name with \u escapes or percent-encoding, exactly as it refuses the plain spelling. diff --git a/.abcd/work/issues/open/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md b/.abcd/work/issues/open/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md new file mode 100644 index 000000000..0fa245e5d --- /dev/null +++ b/.abcd/work/issues/open/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609261909108726" +slug: "a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex" +severity: "minor" +category: "security" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: review-drainS3" +origin: researcher-authored +production_mode: hand-written +found_at: "internal/adapter/scanner/utf16.go" +--- + +A PDF text string written in hex, a FEFF-led run of hex digits between angle brackets, is not read by the payload byte scan: the hex spelling never puts the UTF-16 bytes in the file, so the UTF-16 view (utf16.go) cannot reach the caller's name in such an /Author or /Creator at any length. Reading it needs no escape grammar: the hex pairs decode to bytes with an offset map, and the decoded bytes feed the existing UTF-16 view. This is the hex half of iss-2609261831352258, split from its octal-literal half. Detector: the caller's name in a UTF-16 /Author written as a hex string is a real_name finding in the payload scan, and a short one is kept only where a person key stands within reach. From 1b4d15c0e5445ae89c3e3f619edfd8faf041f1aa Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:10:22 +0100 Subject: [PATCH 52/64] fix(scanner): let a surrogate pair continue a UTF-16 run utf16TextRune refused every surrogate, so a character outside the Basic Multilingual Plane, an emoji above all, ended a byte-order-marked run: a name after an emoji in one string was cut off, and read not at all when fewer than two units stood before the pair. utf16RuneAt decodes a valid high-low pair as one character and lets the run continue; a lone surrogate, a pair spelling a noncharacter and every unit the text rule refuses still end it. TestUTF16RunContinuesPastAnAstralCharacter reads an emoji, a space and a fake name behind either byte-order mark. Refs: iss-2609261909101409 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/utf16.go | 42 +++++++++++++++++++++----- internal/adapter/scanner/utf16_test.go | 32 ++++++++++++++++++++ 2 files changed, 66 insertions(+), 8 deletions(-) diff --git a/internal/adapter/scanner/utf16.go b/internal/adapter/scanner/utf16.go index 0ab2c05d5..43251ce3f 100644 --- a/internal/adapter/scanner/utf16.go +++ b/internal/adapter/scanner/utf16.go @@ -50,13 +50,9 @@ func utf16View(data []byte) (decodedView, bool) { } start, mark := len(text), i+2 j := mark - for ; j+1 < len(data); j += 2 { - u := uint16(data[j+1])<<8 | uint16(data[j]) - if bigEndian { - u = uint16(data[j])<<8 | uint16(data[j+1]) - } - r := rune(u) - if !utf16TextRune(r) { + for j+1 < len(data) { + r, width := utf16RuneAt(data, j, bigEndian) + if width == 0 { break } var enc [utf8.UTFMax]byte @@ -64,6 +60,7 @@ func utf16View(data []byte) (decodedView, bool) { text = append(text, c) pos = append(pos, j) } + j += width } scanMeter.charge(stageUTF16, j-i) if (j-mark)/2 < minUTF16Run { @@ -80,10 +77,39 @@ func utf16View(data []byte) (decodedView, bool) { return decodedView{text: string(text), posMap: append(pos, len(data))}, true } +// utf16RuneAt decodes the character whose first code unit starts at data[j] +// and reports the bytes it spans: 2 for a text unit, 4 for a surrogate pair +// that makes a character outside the Basic Multilingual Plane (an emoji, a +// historic script, a supplementary CJK ideograph), and 0 for anything that +// ends a run. A valid pair continues the run, so a name after an emoji in one +// string is read (iss-2609261909101409); a lone surrogate, a pair spelling a +// noncharacter, and every unit utf16TextRune refuses still end it. +func utf16RuneAt(data []byte, j int, bigEndian bool) (rune, int) { + unit := func(k int) rune { + if bigEndian { + return rune(data[k])<<8 | rune(data[k+1]) + } + return rune(data[k+1])<<8 | rune(data[k]) + } + r := unit(j) + if utf16TextRune(r) { + return r, 2 + } + if r < 0xd800 || r >= 0xdc00 || j+3 >= len(data) { + return 0, 0 + } + pair := utf16.DecodeRune(r, unit(j+2)) + if pair == unicode.ReplacementChar || pair&0xfffe == 0xfffe { + return 0, 0 + } + return pair, 4 +} + // utf16TextRune reports whether a code unit reads as text inside a run: the // layout controls, printable Latin through the IPA block, and beyond it the // letters, marks, digits, punctuation and spaces of any script. A control, a -// surrogate (no BMP rune on its own), a noncharacter and a symbol end the run. +// surrogate (no BMP rune on its own; utf16RuneAt reads a valid pair), a +// noncharacter and a symbol end the run. // The symbol clause is what ends a big-endian PDF string: the ')' closing it // pairs with the byte after it into a unit in the arrows block. func utf16TextRune(r rune) bool { diff --git a/internal/adapter/scanner/utf16_test.go b/internal/adapter/scanner/utf16_test.go index 91cfa8e17..d30f40853 100644 --- a/internal/adapter/scanner/utf16_test.go +++ b/internal/adapter/scanner/utf16_test.go @@ -135,3 +135,35 @@ func TestUTF16ViewWorkIsLinear(t *testing.T) { }) } } + +// TestUTF16RunContinuesPastAnAstralCharacter pins iss-2609261909101409: a +// character outside the Basic Multilingual Plane is written as a surrogate +// pair, and a run that ended at the pair lost every name after it, so an +// emoji before a name in one marked string hid the name entirely. A valid +// pair continues the run in either byte order; a lone surrogate still ends it. +func TestUTF16RunContinuesPastAnAstralCharacter(t *testing.T) { + const name = "Zoë Qüxbar" + for _, bigEndian := range []bool{true, false} { + body := append(append([]byte("%PDF-1.7\n<< /Author ("), utf16Bytes("😀 "+name, bigEndian)...), []byte(") >>\n")...) + v, ok := utf16View(body) + if !ok || !strings.Contains(v.text, "😀 "+name) { + t.Errorf("bigEndian=%v: the view does not read the name after the emoji: %q", bigEndian, v.text) + } + root := t.TempDir() + sc, err := New(root) + if err != nil { + t.Fatal(err) + } + sc.identity = Identity{GitUserName: name} + res := scanOne(t, sc, "deck.pdf", writeFile(t, root, "deck.pdf", string(body))) + if !hasKind(res.Findings, kindRealName) || res.HardFails == 0 { + t.Errorf("bigEndian=%v: no hard_fail %s for the name after an emoji: %+v", bigEndian, kindRealName, res.Findings) + } + } + // A high surrogate with no low one after it is no character: the run ends + // there, as it always did, and the view keeps what came before it. + lone := []byte{0xfe, 0xff, 0x00, 'a', 0x00, 'b', 0xd8, 0x3d, 0x00, 'c', 0x00, 'd'} + if v, ok := utf16View(lone); !ok || v.text != "ab\n" { + t.Errorf("a lone high surrogate: view %q, want the run before it alone", v.text) + } +} From 453ce8db76d1aabbbcb6d4f94ff76266a3d5ba8a Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:10:39 +0100 Subject: [PATCH 53/64] =?UTF-8?q?chore:=20resolve=20iss-2609261909101409?= =?UTF-8?q?=20=E2=80=94=20a=20UTF-16=20run=20reads=20past=20an=20emoji?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves: iss-2609261909101409 Assisted-by: Claude:claude-opus-5-5 --- ...-utf-16-byte-view-internal-adapter-scanner-utf16-go.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md (55%) diff --git a/.abcd/work/issues/open/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md b/.abcd/work/issues/resolved/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md similarity index 55% rename from .abcd/work/issues/open/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md rename to .abcd/work/issues/resolved/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md index 81d15e4f5..112b5ca36 100644 --- a/.abcd/work/issues/open/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md +++ b/.abcd/work/issues/resolved/iss-2609261909101409-the-utf-16-byte-view-internal-adapter-scanner-utf16-go.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: review-drainS3" origin: researcher-authored production_mode: hand-written found_at: "internal/adapter/scanner/utf16.go" +resolution: "utf16RuneAt decodes a valid surrogate pair as one character outside the Basic Multilingual Plane and lets the byte-order-marked run continue past it, so a name after an emoji in one UTF-16 string is read; a lone surrogate, a pair spelling a noncharacter and every unit utf16TextRune refuses still end the run. Test: TestUTF16RunContinuesPastAnAstralCharacter (both byte orders, plus the lone-surrogate case)." +impact: fix +resolved_by: + commit: "1b4d15c0e" --- The UTF-16 byte view (internal/adapter/scanner/utf16.go, utf16TextRune) ends a run at a surrogate, so a character outside the Basic Multilingual Plane, an emoji written as a surrogate pair, cuts a byte-order-marked UTF-16 string in two, and a name after it is never read when fewer than two code units stood before the pair: an emoji, a space and a name behind a byte-order mark yields no view at all, while the same name before the emoji is found. Detector: the caller's name after an emoji in one byte-order-marked UTF-16 string is a real_name finding in the payload scan. + +## Grounds + +- pursued: a name after an astral character in one marked UTF-16 string is found without new findings on chance bytes; a launch dry-run that reports a new finding on this tree's assets, or a marked name after an emoji the scan still passes, would show it wrong From 76d35988649a87a43d093080978b49bc4ae65549 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:23:28 +0100 Subject: [PATCH 54/64] docs(scanner): say the escape-run cap leans on the JSON view's union Past maxEscapeRunWalk the raw reading of the home backstop judges an occurrence as written and spans the home alone; the escape unit is still masked whole only because the JSON view of the same line spans it and disjointSpans unions the two readings. The comment said the text as written decides, which undersold where the correctness comes from. Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/residual.go | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/internal/adapter/scanner/residual.go b/internal/adapter/scanner/residual.go index 3ab1741c4..edec46135 100644 --- a/internal/adapter/scanner/residual.go +++ b/internal/adapter/scanner/residual.go @@ -143,6 +143,10 @@ func needleOccurrences(s, needle string, raw, wantURLs bool, accept func(s strin // maxEscapeRunWalk bounds how far afterEscapeBackslash reads back. A run past // it is judged even — the text as written decides, as it always did — so a // crafted run of backslashes before every occurrence costs a constant each. +// Past the cap the raw reading spans the home alone, so the escape unit is +// still masked whole only because the JSON view of the same line spans the +// unit and disjointSpans unions the two readings: correctness there rests on +// that union, not on this walk. const maxEscapeRunWalk = 64 // afterEscapeBackslash reports whether the byte at is escaped: an odd run of From d010efe27e8eb4604698a98c85546777e1207050 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:25:58 +0100 Subject: [PATCH 55/64] fix(scanner): read a PDF hex text string through the UTF-16 view A PDF writer may spell a UTF-16 text string in hex, , which never puts the UTF-16 bytes in the file, so the byte scan's UTF-16 view had nothing to read and the caller's name in such an /Author raised nothing at any length. pdfHexView decodes the hex pairs of every string between '<' and '>' (white space skipped, an odd last digit read as its byte's high nibble, a '<<' dictionary opener passed over), maps each byte to the raw offset of its first digit, and hands a byte-order-marked result to utf16View; the findings are re-homed through both maps, so the person-key rule judges a short name by the raw bytes before its first digit. No escape grammar is involved: the octal-literal spelling stays deferred on iss-2609261831352258. The walk reads each raw byte at most twice; TestUTF16ViewWorkIsLinear gains the hex and astral shapes. launch --dry-run on this tree is unchanged: 126 files, 1 finding, 0 hard fails. Refs: iss-2609261909108726 Refs: iss-2609261831352258 Assisted-by: Claude:claude-opus-5-5 --- internal/adapter/scanner/utf16.go | 118 ++++++++++++++++++++++--- internal/adapter/scanner/utf16_test.go | 65 ++++++++++++++ 2 files changed, 169 insertions(+), 14 deletions(-) diff --git a/internal/adapter/scanner/utf16.go b/internal/adapter/scanner/utf16.go index 43251ce3f..50802d46d 100644 --- a/internal/adapter/scanner/utf16.go +++ b/internal/adapter/scanner/utf16.go @@ -21,6 +21,14 @@ import ( // so the person-key rule (metadataFields) judges a short name by the raw bytes // before it — a PDF's "/Author (" stands right before the mark. // +// A PDF writer may also spell the same string in hex, , which +// never puts the UTF-16 bytes in the file at all. The hex pairs of such a +// string are decoded to bytes, each mapped to the raw offset of its first +// digit, and the bytes are handed to the same view (pdfHexView); that needs +// no escape grammar (iss-2609261909108726). A string written with octal +// escapes inside parentheses (\376\377...) does, and stays with +// iss-2609261831352258. +// // A run without a mark (EXIF's XPAuthor tag, a legacy binary document) is not // read: telling UTF-16 from chance bytes there needs the structure the text // sits in, which is iss-2609261659051539's IFD reader, not a decode. @@ -124,29 +132,111 @@ func utf16TextRune(r rune) bool { return unicode.IsLetter(r) || unicode.IsMark(r) || unicode.IsDigit(r) || unicode.IsPunct(r) || unicode.IsSpace(r) } -// utf16Findings scans the UTF-16 view of data with the byte rules and re-homes -// each finding onto the raw bytes: Line and Column name the raw position of +// pdfHexView returns the UTF-16 text of every PDF hex string in data whose +// bytes open with a byte-order mark: the digits between '<' and '>' (white +// space between them skipped, an odd last digit standing for its byte's high +// nibble) decoded to bytes and read by utf16View, with each decoded byte of +// the text mapped to the raw offset of the first hex digit of its code unit +// (a separator to the closing '>'). A '<' that opens a dictionary ("<<") or +// meets any other byte before its '>' opens no hex string, and the walk goes +// on from the byte that ended it, so every raw byte is read at most twice. +func pdfHexView(data []byte) (decodedView, bool) { + var text []byte + var pos []int + var raw []byte + var off []int + for i := 0; i < len(data); i++ { + if data[i] != '<' { + continue + } + if i+1 < len(data) && data[i+1] == '<' { + i++ + continue + } + raw, off = raw[:0], off[:0] + hi, hiAt, closed := -1, 0, false + j := i + 1 + for ; j < len(data); j++ { + c := data[j] + if c == '>' { + closed = true + break + } + if isPDFSpace(c) { + continue + } + if !isHexDigit(c) { + break + } + if hi < 0 { + hi, hiAt = int(hexNibble(c)), j + continue + } + raw = append(raw, byte(hi<<4)|hexNibble(c)) + off = append(off, hiAt) + hi = -1 + } + scanMeter.charge(stageUTF16, j-i) + if !closed { + i = j - 1 + continue + } + i = j + if hi >= 0 { + raw = append(raw, byte(hi<<4)) + off = append(off, hiAt) + } + if len(raw) < 2 || !(raw[0] == 0xfe && raw[1] == 0xff || raw[0] == 0xff && raw[1] == 0xfe) { + continue + } + v, ok := utf16View(raw) + if !ok { + continue + } + off = append(off, j) + for k := 0; k < len(v.text); k++ { + text = append(text, v.text[k]) + pos = append(pos, off[v.posMap[k]]) + } + } + if len(text) == 0 { + return decodedView{}, false + } + return decodedView{text: string(text), posMap: append(pos, len(data))}, true +} + +// isPDFSpace reports whether c is one of the six bytes PDF reads as white +// space, which a hex string may carry between its digits. +func isPDFSpace(c byte) bool { + return c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\f' || c == 0 +} + +// utf16Findings scans the UTF-16 views of data (the marked runs as written, and +// those a PDF hex string spells) with the byte rules and re-homes each finding +// onto the raw bytes: Line and Column name the raw position of // the value's first code unit (meta's line starts), while Matched and the // snippet stay the decoded text, so the short-name length rule counts the // name's own bytes rather than its zero-interleaved spelling, and a serialized // finding masks the name as it reads. func (s *Scanner) utf16Findings(data []byte, id Identity, secrets []Pattern, logical string, meta *metadataFields) []Finding { - v, ok := utf16View(data) - if !ok { - return nil - } - starts := lineStarts([]byte(v.text)) var out []Finding - for _, f := range scanText(v.text, id, secrets, s.identSev, logical, true) { - if f.Line < 1 || f.Line > len(starts) { + for _, view := range []func([]byte) (decodedView, bool){utf16View, pdfHexView} { + v, ok := view(data) + if !ok { continue } - at := starts[f.Line-1] + f.Column - 1 - if at < 0 || at >= len(v.text) { - continue + starts := lineStarts([]byte(v.text)) + for _, f := range scanText(v.text, id, secrets, s.identSev, logical, true) { + if f.Line < 1 || f.Line > len(starts) { + continue + } + at := starts[f.Line-1] + f.Column - 1 + if at < 0 || at >= len(v.text) { + continue + } + f.Line, f.Column = meta.position(v.posMap[at]) + out = append(out, f) } - f.Line, f.Column = meta.position(v.posMap[at]) - out = append(out, f) } return out } diff --git a/internal/adapter/scanner/utf16_test.go b/internal/adapter/scanner/utf16_test.go index d30f40853..897900628 100644 --- a/internal/adapter/scanner/utf16_test.go +++ b/internal/adapter/scanner/utf16_test.go @@ -2,6 +2,7 @@ package scanner import ( "bytes" + "fmt" "strings" "testing" "unicode/utf16" @@ -110,6 +111,9 @@ func TestUTF16ViewWorkIsLinear(t *testing.T) { {"short runs", append(utf16Bytes("ab", true), 0, 0)}, {"short names by a key", append([]byte("/Author ("), append(utf16Bytes("Zedqx", true), ')', ' ')...)}, {"one long run", utf16Bytes("Zedqx ", false)[2:]}, + {"short names in PDF hex strings", []byte("/Author " + pdfHex(utf16Bytes("Zedqx", true), false, 0) + " ")}, + {"hex strings that never close", []byte("<0a1b ")}, + {"one long astral run", utf16Bytes("😀Zedqx", false)[2:]}, } for _, s := range shapes { t.Run(s.name, func(t *testing.T) { @@ -167,3 +171,64 @@ func TestUTF16RunContinuesPastAnAstralCharacter(t *testing.T) { t.Errorf("a lone high surrogate: view %q, want the run before it alone", v.text) } } + +// pdfHex spells b as a PDF hex string: the digits between angle brackets, +// upper- or lower-case, with a space after every group of digits when group > 0 +// (PDF readers skip white space inside a hex string). +func pdfHex(b []byte, lower bool, group int) string { + digits := fmt.Sprintf("%X", b) + if lower { + digits = strings.ToLower(digits) + } + if group > 0 { + var spaced strings.Builder + for i := 0; i < len(digits); i += group { + spaced.WriteString(digits[i:min(i+group, len(digits))]) + spaced.WriteByte(' ') + } + digits = spaced.String() + } + return "<" + digits + ">" +} + +// TestPDFHexTextStringIsRead pins iss-2609261909108726: a PDF writer that +// spells a UTF-16 text string in hex () never puts the UTF-16 bytes +// in the file, so the UTF-16 view had nothing to read and the caller's name in +// such an /Author raised nothing at any length. The hex pairs are decoded to +// bytes and handed to the same view, and each finding is judged by the raw +// bytes before its first hex digit, so a short name after /Author is kept and +// one with no person key in reach is still dropped. The names are fake. +func TestPDFHexTextStringIsRead(t *testing.T) { + id := synthIdentity() + cases := []struct { + name, identity, body string + want bool + }{ + {"multi-word name, upper-case hex", id.GitUserName, + "%PDF-1.7\n<< /Author " + pdfHex(utf16Bytes(id.GitUserName, true), false, 0) + " >>\n", true}, + {"non-ASCII name, lower-case hex with spaces", "Zoë Qüxbar", + "%PDF-1.7\n<>\n", true}, + {"short name behind /Author", "Zedqx", + "%PDF-1.7\n<< /Title (deck) /Author " + pdfHex(utf16Bytes("Zedqx", true), false, 0) + " >>\n", true}, + {"short name with no person key in reach", "Zedqx", + "%PDF-1.7\n<< /Title " + pdfHex(utf16Bytes("scanned "+strings.Repeat("q", 100)+" Zedqx", true), false, 0) + " >>\n", false}, + // An odd count of digits: the last one stands for its byte's high + // nibble, so "...002>" ends the string with the unit 0020, a space. + {"odd count of digits", id.GitUserName, + "%PDF-1.7\n<< /Author " + strings.TrimSuffix(pdfHex(utf16Bytes(id.GitUserName+" ", true), false, 0), "0>") + "> >>\n", true}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + root := t.TempDir() + sc, err := New(root) + if err != nil { + t.Fatal(err) + } + sc.identity = Identity{GitUserName: c.identity} + res := scanOne(t, sc, "deck.pdf", writeFile(t, root, "deck.pdf", c.body)) + if got := hasKind(res.Findings, kindRealName); got != c.want { + t.Errorf("real_name reported = %v, want %v: %+v", got, c.want, res.Findings) + } + }) + } +} From 5e1020e4531d967e9809e307c5caf4f6540985a8 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:26:19 +0100 Subject: [PATCH 56/64] =?UTF-8?q?chore:=20resolve=20iss-2609261909108726?= =?UTF-8?q?=20=E2=80=94=20the=20byte=20scan=20reads=20PDF=20hex=20strings?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The hex half of iss-2609261831352258 is resolved; the record keeps the octal-literal half, re-deferred past v0.11.0 with its reason narrowed to the literal-string syntax that half needs. Refs: iss-2609261831352258 Resolves: iss-2609261909108726 Assisted-by: Claude:claude-opus-5-5 --- ...f-text-string-written-in-hex-a-feff-led-run-of-hex.md | 9 ++++++++- ...f-text-string-written-in-hex-a-feff-led-run-of-hex.md | 8 ++++++++ 2 files changed, 16 insertions(+), 1 deletion(-) rename .abcd/work/issues/{open => resolved}/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md (55%) diff --git a/.abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md b/.abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md index 9f41ef866..c6507fe6a 100644 --- a/.abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md +++ b/.abcd/work/issues/open/iss-2609261831352258-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md @@ -10,7 +10,7 @@ origin: researcher-authored production_mode: hand-written found_at: "internal/adapter/scanner/utf16.go" deferred_after: "v0.11.0" -deferral_reason: "decoding PDF string syntax is a piece of a PDF reader, beyond the parser-free decodes lane drainS3 was scoped to: it needs each hex string's angle brackets and each literal string's balanced parentheses found, the hex pairs and the backslash escapes (octal, the named controls and line continuations) decoded with every byte mapped back to its raw offset, and the decoded bytes handed to the UTF-16 view, which then reaches the name" +deferral_reason: "the octal-literal half needs PDF literal-string syntax, which a parser-free decode cannot read: each literal string's balanced parentheses found (a backslash-escaped parenthesis does not close one), its backslash escapes decoded (one to three octal digits, the named controls, a backslash before a line break that continues the line), and every decoded byte mapped back to its raw offset before the bytes reach the UTF-16 view; the hex half needed none of that and shipped on its own as iss-2609261909108726" --- A PDF text string written in hex (a FEFF-led run of hex digits between angle brackets) or with octal escapes (a backslash-376, backslash-377 opening inside parentheses), the forms several common PDF writers use for a non-ASCII document field, is read by neither the payload byte scan nor its UTF-16 view: neither spelling puts the UTF-16 bytes in the file, so the caller's name in such an /Author or /Creator raises nothing at any length. Reading them is a decode of the string syntax ahead of the UTF-16 view (hex pairs, and the PDF literal-string escapes), in the shape of the percent and JSON-escape views, not a format parser. Detector: the caller's name in a UTF-16 /Author written in hex or with octal escapes is a real_name finding in the payload scan. @@ -18,3 +18,10 @@ A PDF text string written in hex (a FEFF-led run of hex digits between angle bra **Deferred (2026-09-26, autonomous run A, lane drainS3).** Found by the sibling sweep of iss-2609261827066511, whose UTF-16 view reads a byte-order-marked run only where the file holds the UTF-16 bytes themselves. + +**Split and re-deferred (2026-09-26, autonomous run A, fix round +fix2-drainS3).** The hex half of this record needs no string grammar: the +hex pairs decode to bytes with an offset map and feed the UTF-16 view, +which iss-2609261909108726 shipped. What stays here is the octal-literal +spelling inside parentheses, and it needs the literal-string syntax the +deferral names. diff --git a/.abcd/work/issues/open/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md b/.abcd/work/issues/resolved/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md similarity index 55% rename from .abcd/work/issues/open/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md rename to .abcd/work/issues/resolved/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md index 0fa245e5d..8be352765 100644 --- a/.abcd/work/issues/open/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md +++ b/.abcd/work/issues/resolved/iss-2609261909108726-a-pdf-text-string-written-in-hex-a-feff-led-run-of-hex.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: review-drainS3" origin: researcher-authored production_mode: hand-written found_at: "internal/adapter/scanner/utf16.go" +resolution: "pdfHexView decodes the hex pairs of every PDF hex string to bytes, each mapped to the raw offset of its first digit, and hands a byte-order-marked result to the existing UTF-16 view, so the caller's name in a hex-spelled /Author or /Creator is a real_name finding in the payload scan and a short one is kept only where a person key stands within reach. Tests: TestPDFHexTextStringIsRead, and TestUTF16ViewWorkIsLinear with the hex shapes. The octal-literal spelling stays deferred on iss-2609261831352258." +impact: fix +resolved_by: + commit: "d010efe27" --- A PDF text string written in hex, a FEFF-led run of hex digits between angle brackets, is not read by the payload byte scan: the hex spelling never puts the UTF-16 bytes in the file, so the UTF-16 view (utf16.go) cannot reach the caller's name in such an /Author or /Creator at any length. Reading it needs no escape grammar: the hex pairs decode to bytes with an offset map, and the decoded bytes feed the existing UTF-16 view. This is the hex half of iss-2609261831352258, split from its octal-literal half. Detector: the caller's name in a UTF-16 /Author written as a hex string is a real_name finding in the payload scan, and a short one is kept only where a person key stands within reach. + +## Grounds + +- pursued: a name in a hex-spelled PDF text string is found at linear cost without new findings on chance bytes; a launch dry-run that reports a new finding on this tree's assets, or a hex-spelled /Author name the scan still passes, would show it wrong From e86097c46305573628d2e9fb6d29db17eaf8fe6b Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sun, 27 Sep 2026 01:26:17 +0100 Subject: [PATCH 57/64] fix(hooks): the name guard reads the decoded spellings of staged text The private-banlist guard matched every pattern against the staged bytes as written, so a name spelled with JSON string escapes (a \u escape of any letter, plain ASCII included, or an escaped solidus) or with percent-encoding passed it, and a JSON transcript, export or fixture is where such spellings live. Every staged line that holds such an escape is now also decoded, into the two views the scanner reads beside the text as written (the JSON escape layers, surrogate pairs joined, and the percent view), and each pattern is matched against the decoded copy as well as the text as written. The decode stays in the hook, in awk and the shell's printf %b, rather than in the abcd binary: no binary front door checks staged content through the scanner's views, and the guard must hold before abcd is built and in every clone the dispatcher runs it in, the scaffolded copy included. Its reading is a deliberate superset (any backslash run before an escape decodes as one escape, a chain of %25 layers decodes to the byte it names), which can only refuse more. Every step fails closed and names itself, and nothing decoded is printed. Applied to this repository's hook and to the scaffolded template alike; the scaffold cites no record id. Measured on this machine (macOS awk): 30 MB of plain staged text costs about as before (0.9s to 1.3s); 20 MB of JSON with escapes on every line goes from 0.8s to about 5.5s, the awk decode being the cost. Refs: iss-2609261909106167 Assisted-by: Claude:claude-opus-5-5 --- .../brief/04-surfaces/20-banlist.md | 15 ++ .githooks/pre-commit | 168 +++++++++++++++++- internal/core/ahoy/banlist_scaffold_test.go | 80 +++++++++ internal/core/ahoy/defaults/pre-commit | 168 +++++++++++++++++- internal/core/banlist/hook_test.go | 80 +++++++++ 5 files changed, 499 insertions(+), 12 deletions(-) diff --git a/.abcd/development/brief/04-surfaces/20-banlist.md b/.abcd/development/brief/04-surfaces/20-banlist.md index a7e1d3912..da9c419b8 100644 --- a/.abcd/development/brief/04-surfaces/20-banlist.md +++ b/.abcd/development/brief/04-surfaces/20-banlist.md @@ -136,6 +136,21 @@ around: a content line beginning `++`, a blob containing a NUL, a committed reading. Binary blobs are scanned like anything else, because a name in a binary file is in history just the same. +Every staged line that holds a JSON string escape (`\u00eb`, `\/`) or a +percent-encoded byte (`%C3%AB`) is also read **decoded**, and a pattern matching +either spelling refuses the commit. An escape changes the bytes a name is written +in without changing the name, plain ASCII letters included, and a JSON +transcript, export or fixture is where such spellings live. The decoded readings +are the two the scanner's redactors read beside the text as written: the JSON +escape layers and the percent view. The guard's reading is deliberately the wider +one: a run of backslashes of any length before an escape decodes as one escape, +so a JSON string nested inside another decodes all its layers at once, and a +chain of `%25` layers before two hex digits decodes to the byte they name. The +decode runs in the hook itself, in `awk` and the shell's `printf`, not in the +abcd binary, because the guard holds before abcd is built and in every clone the +dispatcher runs it in. It adds readings and replaces none: the text as written +is still read in full. + On a match the guard refuses the commit and names **the key alone**. The matched string and the pattern never reach stdout, stderr, or a log — a refusal that echoed the string would defeat the layer at the moment it worked — and the pattern diff --git a/.githooks/pre-commit b/.githooks/pre-commit index 0657c2538..036c4dd06 100755 --- a/.githooks/pre-commit +++ b/.githooks/pre-commit @@ -73,7 +73,7 @@ case $- in *x*) set +x ;; esac # builtins this hook steers on; a shadowed `continue` or `break` changes which lines # a loop reads or never ends it, so they are pinned with the tools (parity with # .githooks/commit-msg, where a no-op `continue` once cut the judged message to nothing). -unset -f git grep mktemp tr cat mkdir rm chmod printf sed head command read echo exit test [ \ +unset -f git grep awk mktemp tr cat mkdir rm chmod printf sed head command read echo exit test [ \ declare continue break return local export set true 2>/dev/null || true # A field separator the caller cannot choose: an inherited IFS changes how every # unquoted expansion below splits — including the sweep loop's, so it is pinned @@ -134,7 +134,7 @@ fi export PATH # Every external tool this hook runs after the pin. A missing one must BLOCK loudly: # left to `set -e` it would surface as a mute exit 127 that reads like a broken repo. -for tool in git grep mktemp tr cat mkdir rm chmod printf sed head; do +for tool in git grep awk mktemp tr cat mkdir rm chmod printf sed head; do if ! command -v "$tool" >/dev/null 2>&1; then echo "pre-commit: BLOCKED — $tool is not on the guard's pinned PATH ($PATH)." >&2 echo " the guard pins PATH so a repo-scoped override cannot substitute a fake tool," >&2 @@ -752,7 +752,8 @@ fi # every exit path, but a SIGKILL — or a machine that lost power mid-commit — leaves # a copy of a staged tree sitting in the working tree indefinitely. if [ -z "$scratch_tmp" ]; then - for stale in "$scratch_dir"/banlist-candidate.* "$scratch_dir"/banlist-paths.* "$scratch_dir"/banlist-blob.*; do + for stale in "$scratch_dir"/banlist-candidate.* "$scratch_dir"/banlist-paths.* "$scratch_dir"/banlist-blob.* \ + "$scratch_dir"/banlist-escaped.* "$scratch_dir"/banlist-views.* "$scratch_dir"/banlist-decoded.*; do if [ -f "$stale" ]; then rm -f "$stale"; fi done fi @@ -773,8 +774,25 @@ blob=$(mktemp "$scratch_dir/banlist-blob.XXXXXX") || { rm -f "$candidate" "$staged_paths" exit 1 } -trap 'rm -f "$candidate" "$staged_paths" "$blob"; if [ -n "$scratch_tmp" ]; then rm -rf "$scratch_tmp"; fi' EXIT INT TERM HUP -chmod 600 "$candidate" "$staged_paths" "$blob" +# The decoded spellings of the lines that hold an escape (see below): the lines +# themselves, the decoder's %b-escaped output, and the decoded bytes grep reads. +escaped=$(mktemp "$scratch_dir/banlist-escaped.XXXXXX") || { + echo "pre-commit: BLOCKED — could not create the guard's scratch file in $scratch_dir." >&2 + rm -f "$candidate" "$staged_paths" "$blob" + exit 1 +} +views=$(mktemp "$scratch_dir/banlist-views.XXXXXX") || { + echo "pre-commit: BLOCKED — could not create the guard's scratch file in $scratch_dir." >&2 + rm -f "$candidate" "$staged_paths" "$blob" "$escaped" + exit 1 +} +decoded=$(mktemp "$scratch_dir/banlist-decoded.XXXXXX") || { + echo "pre-commit: BLOCKED — could not create the guard's scratch file in $scratch_dir." >&2 + rm -f "$candidate" "$staged_paths" "$blob" "$escaped" "$views" + exit 1 +} +trap 'rm -f "$candidate" "$staged_paths" "$blob" "$escaped" "$views" "$decoded"; if [ -n "$scratch_tmp" ]; then rm -rf "$scratch_tmp"; fi' EXIT INT TERM HUP +chmod 600 "$candidate" "$staged_paths" "$blob" "$escaped" "$views" "$decoded" # NUL-delimited RAW records, so a path with a space, a newline, or a quote is one # field AND the staged (destination) MODE of every entry is available — a gitlink @@ -856,6 +874,143 @@ done <"$staged_paths" # with an unparseable line is broken whether or not this commit touches anything. if [ ! -s "$candidate" ]; then exit "$rc"; fi +# --- the decoded spellings: what an escape hides from a pattern --------------- +# A name written with JSON string escapes (`Q\u0075xbar`, `Zo\u00eb`, a `\/` in a +# path) or percent-encoded (`Zo%C3%AB`) puts bytes in the blob that no pattern +# written for its plain spelling matches, plain ASCII letters included, and a JSON +# transcript, export or fixture is exactly where such spellings live +# (iss-2609261909106167). So every line that holds such an escape is ALSO read +# decoded — the two views the scanner's redactors read beside the text as written +# (the JSON escape layers and the percent view) — and each pattern is matched +# against the decoded copy as well as the candidate. The raw candidate is still +# read in full: the decoded copy adds spellings, it never replaces one. +# +# The decode is here, in awk and the shell's own printf, and not in the abcd +# binary's scanner, because this hook is the private layer's one enforcement point +# and holds before abcd is built or on PATH, in every clone the global dispatcher +# runs it in, abcd's own or not. Its reading is deliberately a superset: a run of +# backslashes of ANY length before an escape letter decodes as one escape, so a +# JSON string nested inside another decodes all its layers at once, and a run of +# `%25` before two hex digits decodes as the byte those digits name. Over-reading +# can only refuse more; a guard that reads less than the redactors lets through +# what they would have caught. +# +# Every step is checked and a failure refuses: a decoded copy that could not be +# built must not look like one that held nothing. Nothing decoded is ever printed. +# Only lines holding an escape reach awk (grep picks them), NUL bytes are dropped +# first (an awk may end a line at one), and awk writes each decoded byte as a +# `\0NNN` escape that the shell's printf %b turns back into the byte, because +# awks disagree about what printf %c emits past 127 and printf %b does not. +esc_gr=0 +grep -aE '\\u[0-9A-Fa-f]{4}|\\/|%[0-9A-Fa-f]{2}' -- "$candidate" >"$escaped" 2>/dev/null || esc_gr=$? +case "$esc_gr" in + 0) + if ! tr -d '\000' <"$escaped" | awk ' +BEGIN { + for (i = 0; i < 256; i++) O[i] = sprintf("\\0%03o", i) + for (i = 0; i < 16; i++) { + H[substr("0123456789abcdef", i + 1, 1)] = i + H[substr("0123456789ABCDEF", i + 1, 1)] = i + } + C["n"] = O[10]; C["t"] = O[9]; C["r"] = O[13]; C["b"] = O[8]; C["f"] = O[12] + C["/"] = "/"; C["\""] = "\"" +} +# put gathers output a few dozen pieces at a time: a printf per piece is the +# slowest step of some awks, and one string per line grows quadratically on a +# line of megabytes. +function put(s) { buf = buf s; if (++nbuf >= 64) { printf "%s", buf; buf = ""; nbuf = 0 } } +function endline() { printf "%s\n", buf; buf = ""; nbuf = 0 } +# lit is text as written, each backslash doubled so printf %b keeps it. Every +# split takes a regular expression, never a one-byte string, which awks read +# alike. +function lit(s, n, k, a) { + if (index(s, "\\") == 0) { put(s); return } + n = split(s, a, /\\/) + put(a[1]) + for (k = 2; k <= n; k++) put("\\\\" a[k]) +} +function utf8(c) { + if (c < 128) return O[c] + if (c < 2048) return O[192 + int(c / 64)] O[128 + c % 64] + if (c < 65536) return O[224 + int(c / 4096)] O[128 + int(c / 64) % 64] O[128 + c % 64] + return O[240 + int(c / 262144)] O[128 + int(c / 4096) % 64] O[128 + int(c / 64) % 64] O[128 + c % 64] +} +# hex4 is the value of the four hex digits after the first byte of f, or -1. +function hex4(f, a, b, c, d) { + a = substr(f, 2, 1); b = substr(f, 3, 1); c = substr(f, 4, 1); d = substr(f, 5, 1) + if (!(a in H) || !(b in H) || !(c in H) || !(d in H)) return -1 + return ((H[a] * 16 + H[b]) * 16 + H[c]) * 16 + H[d] +} +# The JSON view. Split on the backslash, each field after the first is what one +# backslash stood before: an empty field is a backslash run going on, and the +# field that ends the run decides the escape. A surrogate pair joins into one +# character; a lone surrogate reads as U+FFFD, as the JSON decoders read it. +function json_view(line, n, k, f, c0, cp, pend) { + n = split(line, p, /\\/) + put(p[1]) + pend = -1 + for (k = 2; k <= n; k++) { + f = p[k] + if (f == "") continue + c0 = substr(f, 1, 1) + cp = (c0 == "u") ? hex4(f) : -1 + if (pend >= 0) { + if (cp >= 56320 && cp <= 57343) { + put(utf8(65536 + (pend - 55296) * 1024 + (cp - 56320)) substr(f, 6)) + pend = -1 + continue + } + put(utf8(65533)) + pend = -1 + } + if (cp >= 0) { + if (cp >= 55296 && cp <= 56319 && length(f) == 5) { pend = cp; continue } + if (cp >= 55296 && cp <= 57343) cp = 65533 + put(utf8(cp) substr(f, 6)) + } else if (c0 in C) put(C[c0] substr(f, 2)) + else put("\\\\" f) + } + if (pend >= 0) put(utf8(65533)) + endline() +} +# The percent view: each %XX is the byte it names, after any run of %25 layers. +function percent_view(line, n, k, f, a, b) { + n = split(line, q, /%/) + lit(q[1]) + for (k = 2; k <= n; k++) { + f = q[k] + while (substr(f, 1, 2) == "25" && (substr(f, 3, 1) in H) && (substr(f, 4, 1) in H)) f = substr(f, 3) + a = substr(f, 1, 1); b = substr(f, 2, 1) + if ((a in H) && (b in H)) { put(O[H[a] * 16 + H[b]]); lit(substr(f, 3)) } + else { put("%"); lit(f) } + } + endline() +} +index($0, "\\") { json_view($0) } +/%[0-9A-Fa-f][0-9A-Fa-f]/ { percent_view($0) } +' >"$views"; then + echo "pre-commit: BLOCKED — could not decode the escaped spellings of the staged content (awk failed)." >&2 + echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 + exit 1 + fi + if ! views_text=$(cat -- "$views"); then + echo "pre-commit: BLOCKED — could not read back the decoded spellings of the staged content." >&2 + exit 1 + fi + if ! printf '%b\n' "$views_text" >"$decoded"; then + echo "pre-commit: BLOCKED — could not write the decoded spellings of the staged content." >&2 + exit 1 + fi + views_text="" + ;; + 1) ;; + *) + echo "pre-commit: BLOCKED — could not find the escaped lines of the staged content (grep failed)." >&2 + echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 + exit 1 + ;; +esac + i=0 while [ "$i" -lt "$entries" ]; do key="${keys[$i]}" @@ -870,8 +1025,9 @@ while [ "$i" -lt "$entries" ]; do # A file argument (not a pipe) makes grep's exit status the only status there is — # 0 match / 1 clean / >1 unusable pattern — so a matching `grep -q` exiting early # cannot leave a writer holding a closed pipe and surface 141 instead of a verdict. + # The decoded spellings are read beside the candidate, never instead of it. gr=0 - printf '%s\n' "$pat" | grep -iqE -f - -- "$candidate" >/dev/null 2>&1 || gr=$? + printf '%s\n' "$pat" | grep -iqE -f - -- "$candidate" "$decoded" >/dev/null 2>&1 || gr=$? case "$gr" in 0) echo "pre-commit: BLOCKED — staged content matches private banlist entry '$key'." >&2 diff --git a/internal/core/ahoy/banlist_scaffold_test.go b/internal/core/ahoy/banlist_scaffold_test.go index be94e4046..cb0cdd548 100644 --- a/internal/core/ahoy/banlist_scaffold_test.go +++ b/internal/core/ahoy/banlist_scaffold_test.go @@ -1247,3 +1247,83 @@ func TestScaffoldedGuardHookRefusesAMirrorInsideAnotherCheckout(t *testing.T) { }) } } + +// TestScaffoldedGuardHookReadsEscapedSpellings pins iss-2609261909106167 on the +// artefact ahoy installs: a private name spelled with JSON string escapes or +// percent-encoding passed the scaffolded guard exactly as it passed abcd's own, +// because both read the staged bytes only as written. The scaffold carries the +// same decoded reading, and refuses by key without echoing the decoded text. +// The names are fake. +func TestScaffoldedGuardHookReadsEscapedSpellings(t *testing.T) { + if _, err := exec.LookPath("bash"); err != nil { + t.Skip("bash unavailable") + } + if _, err := exec.LookPath("git"); err != nil { + t.Skip("git unavailable") + } + setupHermetic(t) + repo := t.TempDir() + env := gittest.Env(t) + git := func(args ...string) (string, error) { + cmd := exec.Command("git", append([]string{"-C", repo}, args...)...) + cmd.Env = env + out, err := cmd.CombinedOutput() + return string(out), err + } + for _, args := range [][]string{{"init"}, {"config", "user.name", "Alice Example"}, {"config", "user.email", "alice@example.com"}} { + if out, err := git(args...); err != nil { + t.Fatalf("git %v: %v\n%s", args, err, out) + } + } + if _, err := Install(repo, installOpts(), RefusingPrompter{}); err != nil { + t.Fatal(err) + } + src, err := os.ReadFile(filepath.Join(repo, filepath.FromSlash(GuardHookRelPath))) + if err != nil { + t.Fatal(err) + } + hooksDir := filepath.Join(repo, ".git", "hooks") + if err := os.MkdirAll(hooksDir, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(hooksDir, "pre-commit"), src, 0o755); err != nil { + t.Fatal(err) + } + store := filepath.Join(repo, filepath.FromSlash(banlist.PrivateRelPath)) + if err := os.WriteFile(store, []byte("# abcd-banlist: keyed\nfake-person zoë qüxbar\n"), 0o600); err != nil { + t.Fatal(err) + } + + for i, staged := range []string{ + `{"author":"Zo\u00eb Q\u00FCxbar"}` + "\n", + "https://example.com/?who=Zo%C3%AB%20Q%C3%BCxbar\n", + } { + name := filepath.Join(repo, "export.json") + if err := os.WriteFile(name, []byte(staged), 0o644); err != nil { + t.Fatal(err) + } + if out, err := git("add", "export.json"); err != nil { + t.Fatalf("git add: %v\n%s", err, out) + } + out, err := git("commit", "-m", "leak") + if err == nil { + t.Fatalf("case %d: the scaffolded guard let an escaped banned name through:\n%s", i, out) + } + if !strings.Contains(out, "fake-person") { + t.Errorf("case %d: the refusal does not name the key:\n%s", i, out) + } + if strings.Contains(strings.ToLower(out), "qüxbar") { + t.Errorf("case %d: the refusal echoes the decoded text:\n%s", i, out) + } + } + + if err := os.WriteFile(filepath.Join(repo, "export.json"), []byte(`{"note":"caf\u00e9 100%"}`+"\n"), 0o644); err != nil { + t.Fatal(err) + } + if out, err := git("add", "export.json"); err != nil { + t.Fatalf("git add: %v\n%s", err, out) + } + if out, err := git("commit", "-m", "clean"); err != nil { + t.Fatalf("the scaffolded guard refused escapes that spell no banned name:\n%s", out) + } +} diff --git a/internal/core/ahoy/defaults/pre-commit b/internal/core/ahoy/defaults/pre-commit index bdb541ea7..c3d668382 100644 --- a/internal/core/ahoy/defaults/pre-commit +++ b/internal/core/ahoy/defaults/pre-commit @@ -80,7 +80,7 @@ case $- in *x*) set +x ;; esac # builtins this hook steers on; a shadowed `continue` or `break` changes which lines # a loop reads or never ends it, so they are pinned with the tools (parity with # .githooks/commit-msg, where a no-op `continue` once cut the judged message to nothing). -unset -f git grep mktemp tr cat mkdir rm chmod printf sed head command read echo exit test [ \ +unset -f git grep awk mktemp tr cat mkdir rm chmod printf sed head command read echo exit test [ \ declare continue break return local export set true 2>/dev/null || true # A field separator the caller cannot choose: an inherited IFS changes how every # unquoted expansion below splits — including the sweep loop's, so it is pinned @@ -141,7 +141,7 @@ fi export PATH # Every external tool this hook runs after the pin. A missing one must BLOCK loudly: # left to `set -e` it would surface as a mute exit 127 that reads like a broken repo. -for tool in git grep mktemp tr cat mkdir rm chmod printf; do +for tool in git grep awk mktemp tr cat mkdir rm chmod printf; do if ! command -v "$tool" >/dev/null 2>&1; then echo "pre-commit: BLOCKED — $tool is not on the guard's pinned PATH ($PATH)." >&2 echo " the guard pins PATH so a repo-scoped override cannot substitute a fake tool," >&2 @@ -717,7 +717,8 @@ fi # every exit path, but a SIGKILL — or a machine that lost power mid-commit — leaves # a copy of a staged tree sitting in the working tree indefinitely. if [ -z "$scratch_tmp" ]; then - for stale in "$scratch_dir"/banlist-candidate.* "$scratch_dir"/banlist-paths.* "$scratch_dir"/banlist-blob.*; do + for stale in "$scratch_dir"/banlist-candidate.* "$scratch_dir"/banlist-paths.* "$scratch_dir"/banlist-blob.* \ + "$scratch_dir"/banlist-escaped.* "$scratch_dir"/banlist-views.* "$scratch_dir"/banlist-decoded.*; do if [ -f "$stale" ]; then rm -f "$stale"; fi done fi @@ -738,8 +739,25 @@ blob=$(mktemp "$scratch_dir/banlist-blob.XXXXXX") || { rm -f "$candidate" "$staged_paths" exit 1 } -trap 'rm -f "$candidate" "$staged_paths" "$blob"; if [ -n "$scratch_tmp" ]; then rm -rf "$scratch_tmp"; fi' EXIT INT TERM HUP -chmod 600 "$candidate" "$staged_paths" "$blob" +# The decoded spellings of the lines that hold an escape (see below): the lines +# themselves, the decoder's %b-escaped output, and the decoded bytes grep reads. +escaped=$(mktemp "$scratch_dir/banlist-escaped.XXXXXX") || { + echo "pre-commit: BLOCKED — could not create the guard's scratch file in $scratch_dir." >&2 + rm -f "$candidate" "$staged_paths" "$blob" + exit 1 +} +views=$(mktemp "$scratch_dir/banlist-views.XXXXXX") || { + echo "pre-commit: BLOCKED — could not create the guard's scratch file in $scratch_dir." >&2 + rm -f "$candidate" "$staged_paths" "$blob" "$escaped" + exit 1 +} +decoded=$(mktemp "$scratch_dir/banlist-decoded.XXXXXX") || { + echo "pre-commit: BLOCKED — could not create the guard's scratch file in $scratch_dir." >&2 + rm -f "$candidate" "$staged_paths" "$blob" "$escaped" "$views" + exit 1 +} +trap 'rm -f "$candidate" "$staged_paths" "$blob" "$escaped" "$views" "$decoded"; if [ -n "$scratch_tmp" ]; then rm -rf "$scratch_tmp"; fi' EXIT INT TERM HUP +chmod 600 "$candidate" "$staged_paths" "$blob" "$escaped" "$views" "$decoded" # NUL-delimited RAW records, so a path with a space, a newline, or a quote is one # field AND the staged (destination) MODE of every entry is available — a gitlink @@ -820,6 +838,143 @@ done <"$staged_paths" # with an unparseable line is broken whether or not this commit touches anything. if [ ! -s "$candidate" ]; then exit "$rc"; fi +# --- the decoded spellings: what an escape hides from a pattern --------------- +# A name written with JSON string escapes (`Q\u0075xbar`, `Zo\u00eb`, a `\/` in a +# path) or percent-encoded (`Zo%C3%AB`) puts bytes in the blob that no pattern +# written for its plain spelling matches, plain ASCII letters included, and a JSON +# transcript, export or fixture is exactly where such spellings live. So every +# line that holds such an escape is ALSO read decoded — the two views abcd's +# scanner reads beside the text as written (the JSON escape layers and the +# percent view) — and each pattern is matched against the decoded copy as well +# as the candidate. The raw candidate is still +# read in full: the decoded copy adds spellings, it never replaces one. +# +# The decode is here, in awk and the shell's own printf, and not in the abcd +# binary's scanner, because this hook is the private layer's one enforcement point +# and holds in every clone of this repo, whether or not abcd is installed there. +# Its reading is deliberately a superset: a run of backslashes of ANY length +# before an escape letter decodes as one escape, so a JSON string nested inside +# another decodes all its layers at once, and a run of `%25` before two hex +# digits decodes as the byte those digits name. Over-reading +# can only refuse more; a guard that reads less than the redactors lets through +# what they would have caught. +# +# Every step is checked and a failure refuses: a decoded copy that could not be +# built must not look like one that held nothing. Nothing decoded is ever printed. +# Only lines holding an escape reach awk (grep picks them), NUL bytes are dropped +# first (an awk may end a line at one), and awk writes each decoded byte as a +# `\0NNN` escape that the shell's printf %b turns back into the byte, because +# awks disagree about what printf %c emits past 127 and printf %b does not. +esc_gr=0 +grep -aE '\\u[0-9A-Fa-f]{4}|\\/|%[0-9A-Fa-f]{2}' -- "$candidate" >"$escaped" 2>/dev/null || esc_gr=$? +case "$esc_gr" in + 0) + if ! tr -d '\000' <"$escaped" | awk ' +BEGIN { + for (i = 0; i < 256; i++) O[i] = sprintf("\\0%03o", i) + for (i = 0; i < 16; i++) { + H[substr("0123456789abcdef", i + 1, 1)] = i + H[substr("0123456789ABCDEF", i + 1, 1)] = i + } + C["n"] = O[10]; C["t"] = O[9]; C["r"] = O[13]; C["b"] = O[8]; C["f"] = O[12] + C["/"] = "/"; C["\""] = "\"" +} +# put gathers output a few dozen pieces at a time: a printf per piece is the +# slowest step of some awks, and one string per line grows quadratically on a +# line of megabytes. +function put(s) { buf = buf s; if (++nbuf >= 64) { printf "%s", buf; buf = ""; nbuf = 0 } } +function endline() { printf "%s\n", buf; buf = ""; nbuf = 0 } +# lit is text as written, each backslash doubled so printf %b keeps it. Every +# split takes a regular expression, never a one-byte string, which awks read +# alike. +function lit(s, n, k, a) { + if (index(s, "\\") == 0) { put(s); return } + n = split(s, a, /\\/) + put(a[1]) + for (k = 2; k <= n; k++) put("\\\\" a[k]) +} +function utf8(c) { + if (c < 128) return O[c] + if (c < 2048) return O[192 + int(c / 64)] O[128 + c % 64] + if (c < 65536) return O[224 + int(c / 4096)] O[128 + int(c / 64) % 64] O[128 + c % 64] + return O[240 + int(c / 262144)] O[128 + int(c / 4096) % 64] O[128 + int(c / 64) % 64] O[128 + c % 64] +} +# hex4 is the value of the four hex digits after the first byte of f, or -1. +function hex4(f, a, b, c, d) { + a = substr(f, 2, 1); b = substr(f, 3, 1); c = substr(f, 4, 1); d = substr(f, 5, 1) + if (!(a in H) || !(b in H) || !(c in H) || !(d in H)) return -1 + return ((H[a] * 16 + H[b]) * 16 + H[c]) * 16 + H[d] +} +# The JSON view. Split on the backslash, each field after the first is what one +# backslash stood before: an empty field is a backslash run going on, and the +# field that ends the run decides the escape. A surrogate pair joins into one +# character; a lone surrogate reads as U+FFFD, as the JSON decoders read it. +function json_view(line, n, k, f, c0, cp, pend) { + n = split(line, p, /\\/) + put(p[1]) + pend = -1 + for (k = 2; k <= n; k++) { + f = p[k] + if (f == "") continue + c0 = substr(f, 1, 1) + cp = (c0 == "u") ? hex4(f) : -1 + if (pend >= 0) { + if (cp >= 56320 && cp <= 57343) { + put(utf8(65536 + (pend - 55296) * 1024 + (cp - 56320)) substr(f, 6)) + pend = -1 + continue + } + put(utf8(65533)) + pend = -1 + } + if (cp >= 0) { + if (cp >= 55296 && cp <= 56319 && length(f) == 5) { pend = cp; continue } + if (cp >= 55296 && cp <= 57343) cp = 65533 + put(utf8(cp) substr(f, 6)) + } else if (c0 in C) put(C[c0] substr(f, 2)) + else put("\\\\" f) + } + if (pend >= 0) put(utf8(65533)) + endline() +} +# The percent view: each %XX is the byte it names, after any run of %25 layers. +function percent_view(line, n, k, f, a, b) { + n = split(line, q, /%/) + lit(q[1]) + for (k = 2; k <= n; k++) { + f = q[k] + while (substr(f, 1, 2) == "25" && (substr(f, 3, 1) in H) && (substr(f, 4, 1) in H)) f = substr(f, 3) + a = substr(f, 1, 1); b = substr(f, 2, 1) + if ((a in H) && (b in H)) { put(O[H[a] * 16 + H[b]]); lit(substr(f, 3)) } + else { put("%"); lit(f) } + } + endline() +} +index($0, "\\") { json_view($0) } +/%[0-9A-Fa-f][0-9A-Fa-f]/ { percent_view($0) } +' >"$views"; then + echo "pre-commit: BLOCKED — could not decode the escaped spellings of the staged content (awk failed)." >&2 + echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 + exit 1 + fi + if ! views_text=$(cat -- "$views"); then + echo "pre-commit: BLOCKED — could not read back the decoded spellings of the staged content." >&2 + exit 1 + fi + if ! printf '%b\n' "$views_text" >"$decoded"; then + echo "pre-commit: BLOCKED — could not write the decoded spellings of the staged content." >&2 + exit 1 + fi + views_text="" + ;; + 1) ;; + *) + echo "pre-commit: BLOCKED — could not find the escaped lines of the staged content (grep failed)." >&2 + echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 + exit 1 + ;; +esac + i=0 while [ "$i" -lt "$entries" ]; do key="${keys[$i]}" @@ -834,8 +989,9 @@ while [ "$i" -lt "$entries" ]; do # A file argument (not a pipe) makes grep's exit status the only status there is — # 0 match / 1 clean / >1 unusable pattern — so a matching `grep -q` exiting early # cannot leave a writer holding a closed pipe and surface 141 instead of a verdict. + # The decoded spellings are read beside the candidate, never instead of it. gr=0 - printf '%s\n' "$pat" | grep -iqE -f - -- "$candidate" >/dev/null 2>&1 || gr=$? + printf '%s\n' "$pat" | grep -iqE -f - -- "$candidate" "$decoded" >/dev/null 2>&1 || gr=$? case "$gr" in 0) echo "pre-commit: BLOCKED — staged content matches private banlist entry '$key'." >&2 diff --git a/internal/core/banlist/hook_test.go b/internal/core/banlist/hook_test.go index bdaed96d9..c4ac3d0a0 100644 --- a/internal/core/banlist/hook_test.go +++ b/internal/core/banlist/hook_test.go @@ -1411,3 +1411,83 @@ func TestPreCommitHook_ResistsInheritedShellState(t *testing.T) { } }) } + +// TestPreCommitHook_ReadsEscapedSpellings pins iss-2609261909106167: the guard +// read every staged blob only as written, so a private name spelled with JSON +// string escapes or percent-encoding — a transcript, an export, a fixture — +// passed every pattern written for its plain spelling, plain ASCII letters +// included. The guard now reads each line that holds such an escape in its +// decoded spellings too, the views the scanner's redactors read, and refuses by +// key exactly as it refuses the plain spelling. The names are fake. +func TestPreCommitHook_ReadsEscapedSpellings(t *testing.T) { + const banlist = "# abcd-banlist: keyed\n" + + "widget-partner widgetworks\n" + + "fake-person zoë qüxbar\n" + + "lab-share lab-share/widget-drop\n" + cases := []struct{ name, staged, key string }{ + {"an ASCII letter as a unicode escape", `{"note":"the \u0077idgetworks deal"}` + "\n", "widget-partner"}, + {"non-ASCII letters as unicode escapes", `{"author":"Zo\u00EB Q\u00fcxbar"}` + "\n", "fake-person"}, + {"a second escape layer", `"{\\\"note\\\":\\\"\\u0077idgetworks\\\"}"` + "\n", "widget-partner"}, + {"an escaped solidus", `{"path":"lab-share\/widget-drop"}` + "\n", "lab-share"}, + {"a surrogate pair before the name", `{"m":"\ud83d\ude00 Zo\u00eb Q\u00fcxbar"}` + "\n", "fake-person"}, + {"percent-encoding", "see https://example.com/?who=Zo%C3%AB%20Q%C3%BCxbar\n", "fake-person"}, + {"double percent-encoding", "see https://example.com/?who=%2577idgetworks\n", "widget-partner"}, + {"a NUL earlier on the line", "\x00\x01bin \\u0077idgetworks\n", "widget-partner"}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + blocked, out := hookRun(t, banlist, c.staged) + if !blocked { + t.Fatalf("commit not blocked; the escaped spelling is the banned name\n%s", out) + } + if !strings.Contains(out, c.key) { + t.Errorf("the refusal does not name the key %q\n%s", c.key, out) + } + for _, leak := range []string{"widgetworks", "qüxbar", "widget-drop"} { + if strings.Contains(strings.ToLower(out), leak) { + t.Errorf("output leaks %q; the decoded text is withheld like the raw\n%s", leak, out) + } + } + }) + } + + t.Run("escapes that spell no banned name pass", func(t *testing.T) { + staged := `{"note":"\u0077idget works, caf\u00e9, lab-share\/other, 100%, C:\\new \\"}` + "\n" + + "q=%77idget%20works&r=%ZZ\n" + blocked, out := hookRun(t, banlist, staged) + if blocked { + t.Fatalf("a staged file whose decoded spellings hold no banned name was refused\n%s", out) + } + }) +} + +// TestPreCommitHook_DecodeFailureRefuses pins the direction the decoded reading +// fails in (iss-2609261909106167): a decoder that could not finish must never +// leave a decoded copy that reads as "nothing to find". The hook's own decoder is +// made to exit non-zero after reading an escaped line, and the commit is refused +// by naming the step. The staged text spells no banned name, so a refusal can +// only come from the failed step. +func TestPreCommitHook_DecodeFailureRefuses(t *testing.T) { + src, err := os.ReadFile(locateHook(t)) + if err != nil { + t.Fatal(err) + } + const rule = `index($0, "\\") { json_view($0) }` + if !strings.Contains(string(src), rule) { + t.Fatalf("the hook no longer holds the decoder rule %q; the fault this test injects has no target", rule) + } + broken := strings.Replace(string(src), rule, `index($0, "\\") { json_view($0); exit 3 }`, 1) + r := newHookRepo(t, keyedBanlist) + if err := os.WriteFile(filepath.Join(r.dir, ".git", "hooks", "pre-commit"), []byte(broken), 0o755); err != nil { + t.Fatal(err) + } + r.write("note.md", `{"note":"caf\u00e9"}`+"\n") + r.git("add", "note.md") + blocked, out := r.commit() + if !blocked { + t.Fatalf("a decoder that failed let the commit through\n%s", out) + } + if !strings.Contains(out, "could not decode the escaped spellings") { + t.Errorf("the refusal does not name the failed step\n%s", out) + } +} From d7964cf32d1cfa3dcd44259f8659cc50165547e4 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Sun, 27 Sep 2026 01:26:30 +0100 Subject: [PATCH 58/64] =?UTF-8?q?chore:=20resolve=20iss-2609261909106167?= =?UTF-8?q?=20=E2=80=94=20the=20name=20guard=20reads=20escaped=20spellings?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The private-banlist guard and its scaffolded copy decode the JSON escape layers and the percent view of every staged line that holds one, and match each pattern against the decoded copy beside the text as written. Resolves: iss-2609261909106167 Assisted-by: Claude:claude-opus-5-5 --- ...te-banlist-guard-in-githooks-pre-commit-reads-every.md | 8 ++++++++ 1 file changed, 8 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md (61%) diff --git a/.abcd/work/issues/open/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md b/.abcd/work/issues/resolved/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md similarity index 61% rename from .abcd/work/issues/open/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md rename to .abcd/work/issues/resolved/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md index 627ac1adf..e71e53a48 100644 --- a/.abcd/work/issues/open/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md +++ b/.abcd/work/issues/resolved/iss-2609261909106167-the-private-banlist-guard-in-githooks-pre-commit-reads-every.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: review-drainS3" origin: researcher-authored production_mode: hand-written found_at: ".githooks/pre-commit" +resolution: "The private-banlist pre-commit guard, and the scaffolded copy ahoy installs, read every staged line that holds a JSON string escape or a percent-encoded byte in its decoded spellings too (the JSON escape layers and the percent view), and refuse by key on a match in either; a decoder failure refuses and names the step." +impact: fix +resolved_by: + commit: "e86097c46" --- The private-banlist guard in .githooks/pre-commit reads every staged blob raw under LC_ALL=C, so a private name written with JSON string escapes passes it: a \uXXXX spelling of any letter, non-ASCII or plain ASCII, or a \/ inside a pattern, puts bytes in the blob that no ERE written for the plain spelling matches, and a JSON transcript, export or fixture is the natural carrier. The store-before-commit redactors and the lint rules read the scanner's decoded views (lineViews: the percent view and the JSON escape layers); the hook, the one enforcement point of the private layer, reads only the text as written. Detector: a keyed banlist entry for a fake name refuses a staged JSON file that spells the name with \u escapes or percent-encoding, exactly as it refuses the plain spelling. + +## Grounds + +- pursued: a keyed entry for a fake name refuses a staged file spelling it with unicode escapes, a nested escape layer, an escaped solidus or percent-encoding, exactly as it refuses the plain spelling, while escapes that spell no banned name pass; a staged escaped spelling of a banned name that commits cleanly would show it wrong From 7a3e022544800d38e6e00f8f3c811acad20fe828 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Mon, 28 Sep 2026 10:31:43 +0100 Subject: [PATCH 59/64] docs(record): name the Default/All allowlist limit in iss-2609251639261103 review-drainS1 (MINOR) found that the non-user home allowlist entries Default and All, added for the Windows Users root, are shared with the POSIX /Users root and the repolint privacy rule, so a third-party macOS account literally named Default or all raises no home_path_other (the WARN-level kind). The resolution text now says so. Checked against the code: nonUserHomeSegments in scanner/network.go is one case-folded map, isNonUserHomeMatch applies it under /Users/ and :\Users\, and repolint's isUsersRoot does the same; neither word is a generic account name, so the caller's own login under either is still local_username (hard_fail) and the caller's home still home_path_self. Amended rather than captured: the limit is a stated consequence of the fix the record already describes, not a separate defect. Refs: iss-2609251639261103 Assisted-by: Claude:claude-opus-5-5 --- ...3-home-path-other-and-home-path-self-read-no-windows-home.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.abcd/work/issues/resolved/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md b/.abcd/work/issues/resolved/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md index a356329f1..5f8f85d40 100644 --- a/.abcd/work/issues/resolved/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md +++ b/.abcd/work/issues/resolved/iss-2609251639261103-home-path-other-and-home-path-self-read-no-windows-home.md @@ -11,7 +11,7 @@ production_mode: hand-written found_at: "internal/adapter/scanner/identity.go" deferred_after: "v0.10.0" deferral_reason: "Not contained: a Windows spelling for home_path_other changes genericHomeRe, the path-segment byte class, the system-directory allowlist, the traversal walk and the lint audit rule that shares them, which is a lane of its own; the caller's own login in a Windows home, literal or JSON-escaped, is still reported hard_fail as local_username, so what the gap costs is the warn-level third-party path and the kind the caller's own home is reported under, not a caller leak." -resolution: "Fixed by 40136106: genericHomeRe gains the Windows alternative :\\Users\\ with each separator a backslash run, so the typed, JSON-escaped and doubly escaped spellings are home_path_other; the trailing boundary takes the backslash, the allowlist recognises the Windows Users root and gains Default and All beside Public (shared with the repolint privacy rule), the traversal walk reads a backslash run as one separator, and the home_path_other skip compares the caller's home against the match with runs collapsed. The caller's own home is home_path_self at any depth through the JSON-escape views of c55ae5ef (abcd builds for darwin and linux, so the caller's home is never itself a Windows path). Pinned by TestWindowsHomeIsAThirdPartyHomePath and TestWindowsOwnHomeIsHomeSelfAtAnyDepth (watched RED at 211b8853) and the false-positive guard TestWindowsSystemRootsAreNotHomes (watched failing under two mutations). The repolint privacy rule's own regexp is a twin that reads a single-backslash Windows home only; its escaped half is iss-2609261658553101." +resolution: "Fixed by 40136106: genericHomeRe gains the Windows alternative :\\Users\\ with each separator a backslash run, so the typed, JSON-escaped and doubly escaped spellings are home_path_other; the trailing boundary takes the backslash, the allowlist recognises the Windows Users root and gains Default and All beside Public (shared with the repolint privacy rule), the traversal walk reads a backslash run as one separator, and the home_path_other skip compares the caller's home against the match with runs collapsed. The caller's own home is home_path_self at any depth through the JSON-escape views of c55ae5ef (abcd builds for darwin and linux, so the caller's home is never itself a Windows path). Pinned by TestWindowsHomeIsAThirdPartyHomePath and TestWindowsOwnHomeIsHomeSelfAtAnyDepth (watched RED at 211b8853) and the false-positive guard TestWindowsSystemRootsAreNotHomes (watched failing under two mutations). The repolint privacy rule's own regexp is a twin that reads a single-backslash Windows home only; its escaped half is iss-2609261658553101. Known limit: the Default and All allowlist entries also apply to the POSIX Users root and to the repolint privacy rule, so a macOS account literally named Default or all raises no home_path_other, the WARN-level third-party kind; the caller's own login is still reported hard_fail as local_username." impact: fix resolved_by: commit: "40136106" From 1dedc02e899509036b4692db8bc69c0ef29ba4c1 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Mon, 28 Sep 2026 10:45:07 +0100 Subject: [PATCH 60/64] chore: capture the name guard's escaped-backslash layer and its long-line decode cost Two findings of the verify of fix3-drainS3: a backslash the guard decodes from an escape is never read again as an escape, so a name one layer behind it commits while the scanner reads it; and the awk decode is superlinear in line length on macOS awk. Refs: iss-2609280944560197, iss-2609280945018822 Assisted-by: Claude:claude-opus-5-5 --- ...name-guard-decodes-one-json-escape-layer-per.md | 14 ++++++++++++++ ...name-guard-s-awk-decode-of-escaped-spellings.md | 14 ++++++++++++++ 2 files changed, 28 insertions(+) create mode 100644 .abcd/work/issues/open/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md create mode 100644 .abcd/work/issues/open/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md diff --git a/.abcd/work/issues/open/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md b/.abcd/work/issues/open/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md new file mode 100644 index 000000000..b49f0b0c1 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609280944560197" +slug: "the-pre-commit-name-guard-decodes-one-json-escape-layer-per" +severity: "minor" +category: "security" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: verify-fix3-drainS3" +origin: researcher-authored +production_mode: hand-written +found_at: ".githooks/pre-commit" +--- + +The pre-commit name guard decodes one JSON escape layer per run of backslashes and never re-reads its own output, so a backslash spelled as the unicode escape of U+005C (or as %5C) decodes to a literal backslash that is not read again as an escape: a banned name hidden one layer behind such a backslash (the escaped backslash followed by u0075 and the rest of the name) commits, while scanner.DecodedViews reads it through its second layer (maxJSONDecodeLayers is 3). The carrier is a hand-crafted file, since no mainstream encoder writes a backslash that way; the hook comments claim the guard reads what the scanner's redactors read, which overclaims by exactly this case. Both hook copies (.githooks/pre-commit and internal/core/ahoy/defaults/pre-commit) carry the decode. diff --git a/.abcd/work/issues/open/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md b/.abcd/work/issues/open/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md new file mode 100644 index 000000000..fbc10c0c1 --- /dev/null +++ b/.abcd/work/issues/open/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md @@ -0,0 +1,14 @@ +--- +schema_version: 1 +id: "iss-2609280945018822" +slug: "the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings" +severity: "minor" +category: "tech-debt" +source: "review-followup" +found_during: "autonomous run A resumed 2026-09-25: verify-fix3-drainS3" +origin: researcher-authored +production_mode: hand-written +found_at: ".githooks/pre-commit" +--- + +The pre-commit name guard's awk decode of escaped spellings is superlinear in LINE length on macOS awk 20200816: 19 MB of escaped JSON over 105k lines decodes in 1.9 s, but the same bytes as one line take 65 s in awk and 88 s for the commit (the verify of fix3-drainS3). A minified single-line JSON export or asset pays it; the hook still completes and reports, so it is never silent. Both hook copies (.githooks/pre-commit and internal/core/ahoy/defaults/pre-commit) carry the decode. From 5e7929f4ef6a2cee3fa288e3e247ce2efc4eeeaa Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Mon, 28 Sep 2026 10:53:01 +0100 Subject: [PATCH 61/64] chore: recalibrate the reading windows at the integration tip Measured on a clean `git clone --no-local` of ecbaf85f9 with `reading assemble --position P --target HEAD --dry-run --json`; each window is ceil(tokens * 1.01 / 10000) * 10000. - widening: 1,263,373 tokens (4,863,987 bytes); window stays 1,280,000. - entailment: 376,646 tokens (1,450,090 bytes); window 380,000 -> 390,000, since 380,000 left under one per cent of headroom. - detection: 1,272,409 tokens (4,898,775 bytes); window stays 1,290,000. Refs: iss-2609251455354719 Assisted-by: Claude:claude-opus-5-5 --- .abcd/config/reading-presets.json | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/.abcd/config/reading-presets.json b/.abcd/config/reading-presets.json index 7729956ef..9f81d538a 100644 --- a/.abcd/config/reading-presets.json +++ b/.abcd/config/reading-presets.json @@ -61,9 +61,9 @@ ], "window": { "tokens_est": 1280000, - "measured_tokens_est": 1260415, - "measured_bytes": 4852600, - "measured_at": "3bc0c2f208aff2acb91cd4383a62fe2bea58ebf7" + "measured_tokens_est": 1263373, + "measured_bytes": 4863987, + "measured_at": "ecbaf85f9509fcf26d763d392d3481416f30cde8" } }, "entailment": { @@ -132,10 +132,10 @@ "intent-projection" ], "window": { - "tokens_est": 380000, - "measured_tokens_est": 375787, - "measured_bytes": 1446780, - "measured_at": "3bc0c2f208aff2acb91cd4383a62fe2bea58ebf7" + "tokens_est": 390000, + "measured_tokens_est": 376646, + "measured_bytes": 1450090, + "measured_at": "ecbaf85f9509fcf26d763d392d3481416f30cde8" } }, "comparative": { @@ -217,9 +217,9 @@ ], "window": { "tokens_est": 1290000, - "measured_tokens_est": 1269451, - "measured_bytes": 4887388, - "measured_at": "3bc0c2f208aff2acb91cd4383a62fe2bea58ebf7" + "measured_tokens_est": 1272409, + "measured_bytes": 4898775, + "measured_at": "ecbaf85f9509fcf26d763d392d3481416f30cde8" } } } From 5c5a8595c12822e1c6e712eca7996331403367e2 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:07:35 +0100 Subject: [PATCH 62/64] fix(hooks): the name guard reads escape-spelled backslashes, in linear time A backslash the guard decoded from an escape (the unicode escape of U+005C, or %5C) was never read again as an escape, so a banned name one layer behind it committed while scanner.DecodedViews reads it on its second layer. Each decoded layer is now read again for the escapes it still holds, up to decode_layers, which the hook declares once and a scanner test holds equal to maxJSONDecodeLayers (3). A %5C before a JSON escape, which the scanner does not decode, is read here too. The awk decode split every line on a regular expression, which makes the split of the one true awk (macOS) quadratic in line length: one 19 MB line took 156 s to decode and 172 s to commit. The splits take one-character strings, which POSIX, gawk and mawk read as the character itself; the decoded output is byte-identical on a 4,000-line corpus of escape edge cases and on the 19 MB fixtures, and the same line now decodes in 2.5 s (commit 28 s at load 20). Both hook copies carry the same decode block, and a test holds them to the same bytes and every split to a one-character separator. The hook comments and the banlist brief chapter say what the guard reads now. Refs: iss-2609280944560197, iss-2609280945018822 Assisted-by: Claude:claude-opus-5-5 --- .../brief/04-surfaces/20-banlist.md | 6 +- .githooks/pre-commit | 93 +++++++++++------- internal/adapter/scanner/hook_layers_test.go | 41 ++++++++ internal/adapter/scanner/jsonescape.go | 4 +- internal/core/ahoy/banlist_scaffold_test.go | 4 + internal/core/ahoy/defaults/pre-commit | 94 +++++++++++-------- internal/core/banlist/hook_test.go | 90 ++++++++++++++++++ 7 files changed, 257 insertions(+), 75 deletions(-) create mode 100644 internal/adapter/scanner/hook_layers_test.go diff --git a/.abcd/development/brief/04-surfaces/20-banlist.md b/.abcd/development/brief/04-surfaces/20-banlist.md index da9c419b8..232b57b22 100644 --- a/.abcd/development/brief/04-surfaces/20-banlist.md +++ b/.abcd/development/brief/04-surfaces/20-banlist.md @@ -145,7 +145,11 @@ are the two the scanner's redactors read beside the text as written: the JSON escape layers and the percent view. The guard's reading is deliberately the wider one: a run of backslashes of any length before an escape decodes as one escape, so a JSON string nested inside another decodes all its layers at once, and a -chain of `%25` layers before two hex digits decodes to the byte they name. The +chain of `%25` layers before two hex digits decodes to the byte they name. A +decoded reading is itself read again for the escapes it still holds, since a +backslash or a percent sign that an escape spells (`%5C`) opens an escape of its +own, for as many layers as the scanner reads a JSON line through; the hook +declares that bound once and a test holds it equal to the scanner's. The decode runs in the hook itself, in `awk` and the shell's `printf`, not in the abcd binary, because the guard holds before abcd is built and in every clone the dispatcher runs it in. It adds readings and replaces none: the text as written diff --git a/.githooks/pre-commit b/.githooks/pre-commit index 036c4dd06..d64b0d6a6 100755 --- a/.githooks/pre-commit +++ b/.githooks/pre-commit @@ -880,10 +880,10 @@ if [ ! -s "$candidate" ]; then exit "$rc"; fi # written for its plain spelling matches, plain ASCII letters included, and a JSON # transcript, export or fixture is exactly where such spellings live # (iss-2609261909106167). So every line that holds such an escape is ALSO read -# decoded — the two views the scanner's redactors read beside the text as written -# (the JSON escape layers and the percent view) — and each pattern is matched -# against the decoded copy as well as the candidate. The raw candidate is still -# read in full: the decoded copy adds spellings, it never replaces one. +# decoded — in the two kinds of view the scanner's redactors read beside the text +# as written, the JSON escape layers and the percent view — and each pattern is +# matched against the decoded copy as well as the candidate. The raw candidate is +# still read in full: the decoded copy adds spellings, it never replaces one. # # The decode is here, in awk and the shell's own printf, and not in the abcd # binary's scanner, because this hook is the private layer's one enforcement point @@ -891,9 +891,18 @@ if [ ! -s "$candidate" ]; then exit "$rc"; fi # runs it in, abcd's own or not. Its reading is deliberately a superset: a run of # backslashes of ANY length before an escape letter decodes as one escape, so a # JSON string nested inside another decodes all its layers at once, and a run of -# `%25` before two hex digits decodes as the byte those digits name. Over-reading -# can only refuse more; a guard that reads less than the redactors lets through -# what they would have caught. +# `%25` before two hex digits decodes as the byte those digits name. +# +# A decoded layer is read again for the escapes it still holds, because a +# backslash or a percent sign an escape spelled (`%5C`, or the unicode escape of +# U+005C) opens an escape of its own on the next layer (iss-2609280944560197). +# Each later layer reads only what the layer before it decoded, and the walk +# stops at a layer that holds no escape or at decode_layers, the number of JSON +# layers the scanner reads (its maxJSONDecodeLayers; a test holds the two equal). +# A name behind more escape-spelled backslashes than that is left unread here, +# as it is in the scanner; a `%5C` before a JSON escape, which the scanner does +# not decode as one, is read here too. Over-reading can only refuse more; a guard +# that reads less than the redactors lets through what they would have caught. # # Every step is checked and a failure refuses: a decoded copy that could not be # built must not look like one that held nothing. Nothing decoded is ever printed. @@ -901,11 +910,23 @@ if [ ! -s "$candidate" ]; then exit "$rc"; fi # first (an awk may end a line at one), and awk writes each decoded byte as a # `\0NNN` escape that the shell's printf %b turns back into the byte, because # awks disagree about what printf %c emits past 127 and printf %b does not. -esc_gr=0 -grep -aE '\\u[0-9A-Fa-f]{4}|\\/|%[0-9A-Fa-f]{2}' -- "$candidate" >"$escaped" 2>/dev/null || esc_gr=$? -case "$esc_gr" in - 0) - if ! tr -d '\000' <"$escaped" | awk ' +decode_layers=3 +layer=0 +layer_in="$candidate" +while [ "$layer" -lt "$decode_layers" ]; do + layer=$((layer + 1)) + esc_gr=0 + grep -aE '\\u[0-9A-Fa-f]{4}|\\/|%[0-9A-Fa-f]{2}' -- "$layer_in" >"$escaped" 2>/dev/null || esc_gr=$? + case "$esc_gr" in + 0) ;; + 1) break ;; + *) + echo "pre-commit: BLOCKED — could not find the escaped lines of the staged content (grep failed)." >&2 + echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 + exit 1 + ;; + esac + if ! tr -d '\000' <"$escaped" | awk ' BEGIN { for (i = 0; i < 256; i++) O[i] = sprintf("\\0%03o", i) for (i = 0; i < 16; i++) { @@ -921,11 +942,12 @@ BEGIN { function put(s) { buf = buf s; if (++nbuf >= 64) { printf "%s", buf; buf = ""; nbuf = 0 } } function endline() { printf "%s\n", buf; buf = ""; nbuf = 0 } # lit is text as written, each backslash doubled so printf %b keeps it. Every -# split takes a regular expression, never a one-byte string, which awks read -# alike. +# split takes a one-character string, which POSIX reads as that character and +# never as a pattern: a regular-expression separator makes the split of the one +# true awk quadratic in the length of the line. function lit(s, n, k, a) { if (index(s, "\\") == 0) { put(s); return } - n = split(s, a, /\\/) + n = split(s, a, "\\") put(a[1]) for (k = 2; k <= n; k++) put("\\\\" a[k]) } @@ -946,7 +968,7 @@ function hex4(f, a, b, c, d) { # field that ends the run decides the escape. A surrogate pair joins into one # character; a lone surrogate reads as U+FFFD, as the JSON decoders read it. function json_view(line, n, k, f, c0, cp, pend) { - n = split(line, p, /\\/) + n = split(line, p, "\\") put(p[1]) pend = -1 for (k = 2; k <= n; k++) { @@ -975,7 +997,7 @@ function json_view(line, n, k, f, c0, cp, pend) { } # The percent view: each %XX is the byte it names, after any run of %25 layers. function percent_view(line, n, k, f, a, b) { - n = split(line, q, /%/) + n = split(line, q, "%") lit(q[1]) for (k = 2; k <= n; k++) { f = q[k] @@ -989,27 +1011,26 @@ function percent_view(line, n, k, f, a, b) { index($0, "\\") { json_view($0) } /%[0-9A-Fa-f][0-9A-Fa-f]/ { percent_view($0) } ' >"$views"; then - echo "pre-commit: BLOCKED — could not decode the escaped spellings of the staged content (awk failed)." >&2 - echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 - exit 1 - fi - if ! views_text=$(cat -- "$views"); then - echo "pre-commit: BLOCKED — could not read back the decoded spellings of the staged content." >&2 - exit 1 - fi - if ! printf '%b\n' "$views_text" >"$decoded"; then - echo "pre-commit: BLOCKED — could not write the decoded spellings of the staged content." >&2 - exit 1 - fi - views_text="" - ;; - 1) ;; - *) - echo "pre-commit: BLOCKED — could not find the escaped lines of the staged content (grep failed)." >&2 + echo "pre-commit: BLOCKED — could not decode the escaped spellings of the staged content (awk failed)." >&2 echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 exit 1 - ;; -esac + fi + if ! views_text=$(cat -- "$views"); then + echo "pre-commit: BLOCKED — could not read back the decoded spellings of the staged content." >&2 + exit 1 + fi + # The layer's decoded bytes replace its %b-escaped form: the next layer reads them. + if ! printf '%b\n' "$views_text" >"$views"; then + echo "pre-commit: BLOCKED — could not write the decoded spellings of the staged content." >&2 + exit 1 + fi + views_text="" + if ! cat -- "$views" >>"$decoded"; then + echo "pre-commit: BLOCKED — could not keep the decoded spellings of the staged content." >&2 + exit 1 + fi + layer_in="$views" +done i=0 while [ "$i" -lt "$entries" ]; do diff --git a/internal/adapter/scanner/hook_layers_test.go b/internal/adapter/scanner/hook_layers_test.go new file mode 100644 index 000000000..e5ebaa2b9 --- /dev/null +++ b/internal/adapter/scanner/hook_layers_test.go @@ -0,0 +1,41 @@ +package scanner + +import ( + "os" + "path/filepath" + "regexp" + "strconv" + "testing" +) + +// TestNameGuardHooksReadTheScannersJSONLayers holds the pre-commit name guard's +// decode bound to this package's (iss-2609280944560197). The guard is shell and +// awk, so it cannot import maxJSONDecodeLayers; it declares the number once, as +// decode_layers, and this test is what keeps the two one number. A guard that +// reads fewer layers than the scanner lets through a name the store-before-commit +// redactors would have read. Both copies are held: abcd's own hook and the one +// ahoy scaffolds into a managed repository. +func TestNameGuardHooksReadTheScannersJSONLayers(t *testing.T) { + decl := regexp.MustCompile(`(?m)^decode_layers=([0-9]+)$`) + for _, rel := range []string{ + "../../../.githooks/pre-commit", + "../../core/ahoy/defaults/pre-commit", + } { + src, err := os.ReadFile(filepath.FromSlash(rel)) + if err != nil { + t.Fatalf("read the name guard: %v", err) + } + m := decl.FindAllSubmatch(src, -1) + if len(m) != 1 { + t.Errorf("%s: want exactly one decode_layers=N declaration, found %d", rel, len(m)) + continue + } + n, err := strconv.Atoi(string(m[0][1])) + if err != nil { + t.Fatalf("%s: decode_layers: %v", rel, err) + } + if n != maxJSONDecodeLayers { + t.Errorf("%s: decode_layers=%d, the scanner reads %d JSON layers (maxJSONDecodeLayers)", rel, n, maxJSONDecodeLayers) + } + } +} diff --git a/internal/adapter/scanner/jsonescape.go b/internal/adapter/scanner/jsonescape.go index 17a27814c..0a90e1739 100644 --- a/internal/adapter/scanner/jsonescape.go +++ b/internal/adapter/scanner/jsonescape.go @@ -49,7 +49,9 @@ import ( // maxJSONDecodeLayers bounds the JSON-unescape walk: one layer for a // transcript line, a second for JSON quoted inside it (a tool result), a third // for slack. Each layer strictly shrinks the line, so the walk ends early on -// ordinary input. +// ordinary input. The pre-commit name guard, which is shell and cannot import +// this, reads the same number of layers as its decode_layers, and +// TestNameGuardHooksReadTheScannersJSONLayers holds the two equal. const maxJSONDecodeLayers = 3 // jsonEscapeLayers returns the JSON-decoded views of s, outermost first, each diff --git a/internal/core/ahoy/banlist_scaffold_test.go b/internal/core/ahoy/banlist_scaffold_test.go index cb0cdd548..ed843fbfd 100644 --- a/internal/core/ahoy/banlist_scaffold_test.go +++ b/internal/core/ahoy/banlist_scaffold_test.go @@ -1297,6 +1297,10 @@ func TestScaffoldedGuardHookReadsEscapedSpellings(t *testing.T) { for i, staged := range []string{ `{"author":"Zo\u00eb Q\u00FCxbar"}` + "\n", "https://example.com/?who=Zo%C3%AB%20Q%C3%BCxbar\n", + // A backslash spelled as its unicode escape opens an escape on the + // next layer (iss-2609280944560197); built from parts so no tool that + // folds a written escape into its character can change it. + `{"author":"` + "\\" + "u005c" + "u005Ao" + "\\" + "u00eb Q" + "\\" + "u00fcxbar" + `"}` + "\n", } { name := filepath.Join(repo, "export.json") if err := os.WriteFile(name, []byte(staged), 0o644); err != nil { diff --git a/internal/core/ahoy/defaults/pre-commit b/internal/core/ahoy/defaults/pre-commit index c3d668382..a31daf42a 100644 --- a/internal/core/ahoy/defaults/pre-commit +++ b/internal/core/ahoy/defaults/pre-commit @@ -843,11 +843,11 @@ if [ ! -s "$candidate" ]; then exit "$rc"; fi # path) or percent-encoded (`Zo%C3%AB`) puts bytes in the blob that no pattern # written for its plain spelling matches, plain ASCII letters included, and a JSON # transcript, export or fixture is exactly where such spellings live. So every -# line that holds such an escape is ALSO read decoded — the two views abcd's -# scanner reads beside the text as written (the JSON escape layers and the -# percent view) — and each pattern is matched against the decoded copy as well -# as the candidate. The raw candidate is still -# read in full: the decoded copy adds spellings, it never replaces one. +# line that holds such an escape is ALSO read decoded — in the two kinds of view +# abcd's scanner reads beside the text as written, the JSON escape layers and the +# percent view — and each pattern is matched against the decoded copy as well as +# the candidate. The raw candidate is still read in full: the decoded copy adds +# spellings, it never replaces one. # # The decode is here, in awk and the shell's own printf, and not in the abcd # binary's scanner, because this hook is the private layer's one enforcement point @@ -855,9 +855,17 @@ if [ ! -s "$candidate" ]; then exit "$rc"; fi # Its reading is deliberately a superset: a run of backslashes of ANY length # before an escape letter decodes as one escape, so a JSON string nested inside # another decodes all its layers at once, and a run of `%25` before two hex -# digits decodes as the byte those digits name. Over-reading -# can only refuse more; a guard that reads less than the redactors lets through -# what they would have caught. +# digits decodes as the byte those digits name. +# +# A decoded layer is read again for the escapes it still holds, because a +# backslash or a percent sign an escape spelled (`%5C`, or the unicode escape of +# U+005C) opens an escape of its own on the next layer. Each later layer reads +# only what the layer before it decoded, and the walk stops at a layer that holds +# no escape or at decode_layers, the number of JSON layers abcd's scanner reads. +# A name behind more escape-spelled backslashes than that is left unread here, +# as it is in the scanner; a `%5C` before a JSON escape, which the scanner does +# not decode as one, is read here too. Over-reading can only refuse more; a guard +# that reads less than the redactors lets through what they would have caught. # # Every step is checked and a failure refuses: a decoded copy that could not be # built must not look like one that held nothing. Nothing decoded is ever printed. @@ -865,11 +873,23 @@ if [ ! -s "$candidate" ]; then exit "$rc"; fi # first (an awk may end a line at one), and awk writes each decoded byte as a # `\0NNN` escape that the shell's printf %b turns back into the byte, because # awks disagree about what printf %c emits past 127 and printf %b does not. -esc_gr=0 -grep -aE '\\u[0-9A-Fa-f]{4}|\\/|%[0-9A-Fa-f]{2}' -- "$candidate" >"$escaped" 2>/dev/null || esc_gr=$? -case "$esc_gr" in - 0) - if ! tr -d '\000' <"$escaped" | awk ' +decode_layers=3 +layer=0 +layer_in="$candidate" +while [ "$layer" -lt "$decode_layers" ]; do + layer=$((layer + 1)) + esc_gr=0 + grep -aE '\\u[0-9A-Fa-f]{4}|\\/|%[0-9A-Fa-f]{2}' -- "$layer_in" >"$escaped" 2>/dev/null || esc_gr=$? + case "$esc_gr" in + 0) ;; + 1) break ;; + *) + echo "pre-commit: BLOCKED — could not find the escaped lines of the staged content (grep failed)." >&2 + echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 + exit 1 + ;; + esac + if ! tr -d '\000' <"$escaped" | awk ' BEGIN { for (i = 0; i < 256; i++) O[i] = sprintf("\\0%03o", i) for (i = 0; i < 16; i++) { @@ -885,11 +905,12 @@ BEGIN { function put(s) { buf = buf s; if (++nbuf >= 64) { printf "%s", buf; buf = ""; nbuf = 0 } } function endline() { printf "%s\n", buf; buf = ""; nbuf = 0 } # lit is text as written, each backslash doubled so printf %b keeps it. Every -# split takes a regular expression, never a one-byte string, which awks read -# alike. +# split takes a one-character string, which POSIX reads as that character and +# never as a pattern: a regular-expression separator makes the split of the one +# true awk quadratic in the length of the line. function lit(s, n, k, a) { if (index(s, "\\") == 0) { put(s); return } - n = split(s, a, /\\/) + n = split(s, a, "\\") put(a[1]) for (k = 2; k <= n; k++) put("\\\\" a[k]) } @@ -910,7 +931,7 @@ function hex4(f, a, b, c, d) { # field that ends the run decides the escape. A surrogate pair joins into one # character; a lone surrogate reads as U+FFFD, as the JSON decoders read it. function json_view(line, n, k, f, c0, cp, pend) { - n = split(line, p, /\\/) + n = split(line, p, "\\") put(p[1]) pend = -1 for (k = 2; k <= n; k++) { @@ -939,7 +960,7 @@ function json_view(line, n, k, f, c0, cp, pend) { } # The percent view: each %XX is the byte it names, after any run of %25 layers. function percent_view(line, n, k, f, a, b) { - n = split(line, q, /%/) + n = split(line, q, "%") lit(q[1]) for (k = 2; k <= n; k++) { f = q[k] @@ -953,27 +974,26 @@ function percent_view(line, n, k, f, a, b) { index($0, "\\") { json_view($0) } /%[0-9A-Fa-f][0-9A-Fa-f]/ { percent_view($0) } ' >"$views"; then - echo "pre-commit: BLOCKED — could not decode the escaped spellings of the staged content (awk failed)." >&2 - echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 - exit 1 - fi - if ! views_text=$(cat -- "$views"); then - echo "pre-commit: BLOCKED — could not read back the decoded spellings of the staged content." >&2 - exit 1 - fi - if ! printf '%b\n' "$views_text" >"$decoded"; then - echo "pre-commit: BLOCKED — could not write the decoded spellings of the staged content." >&2 - exit 1 - fi - views_text="" - ;; - 1) ;; - *) - echo "pre-commit: BLOCKED — could not find the escaped lines of the staged content (grep failed)." >&2 + echo "pre-commit: BLOCKED — could not decode the escaped spellings of the staged content (awk failed)." >&2 echo " the guard cannot check what it cannot read, so it refuses rather than pass." >&2 exit 1 - ;; -esac + fi + if ! views_text=$(cat -- "$views"); then + echo "pre-commit: BLOCKED — could not read back the decoded spellings of the staged content." >&2 + exit 1 + fi + # The layer's decoded bytes replace its %b-escaped form: the next layer reads them. + if ! printf '%b\n' "$views_text" >"$views"; then + echo "pre-commit: BLOCKED — could not write the decoded spellings of the staged content." >&2 + exit 1 + fi + views_text="" + if ! cat -- "$views" >>"$decoded"; then + echo "pre-commit: BLOCKED — could not keep the decoded spellings of the staged content." >&2 + exit 1 + fi + layer_in="$views" +done i=0 while [ "$i" -lt "$entries" ]; do diff --git a/internal/core/banlist/hook_test.go b/internal/core/banlist/hook_test.go index c4ac3d0a0..eb7718435 100644 --- a/internal/core/banlist/hook_test.go +++ b/internal/core/banlist/hook_test.go @@ -4,6 +4,7 @@ import ( "os" "os/exec" "path/filepath" + "regexp" "strings" "testing" @@ -1461,6 +1462,95 @@ func TestPreCommitHook_ReadsEscapedSpellings(t *testing.T) { }) } +// TestPreCommitHook_ReadsEscapeSpelledBackslashes pins iss-2609280944560197: a +// backslash spelled as an escape (the unicode escape of U+005C, or %5C) decoded +// to a backslash the guard never read again, so a name one escape layer behind +// it committed while the scanner's JSON walk reads it on its next layer. The +// guard now reads each decoded copy again, as many layers as the scanner does. +// The escapes are built from parts so no tool that folds a written escape into +// its character can change the fixture. The names are fake. +func TestPreCommitHook_ReadsEscapeSpelledBackslashes(t *testing.T) { + const banlist = "# abcd-banlist: keyed\n" + + "widget-partner widgetworks\n" + + "fake-person zoë qüxbar\n" + bs := "\\" + escBS := bs + "u005c" // a backslash written as its unicode escape + cases := []struct{ name, staged, key string }{ + {"a backslash as a unicode escape before an escaped letter", `{"note":"` + escBS + "u0077idgetworks" + `"}` + "\n", "widget-partner"}, + {"a backslash as a unicode escape before an escaped capital", `{"author":"` + escBS + "u005Ao" + bs + "u00eb Q" + bs + "u00fcxbar" + `"}` + "\n", "fake-person"}, + {"a backslash as a percent escape", "see https://example.com/?q=%5Cu0077idgetworks\n", "widget-partner"}, + {"two escape-spelled backslashes, the scanner's third layer", `{"note":"` + escBS + "u005c" + "u0077idgetworks" + `"}` + "\n", "widget-partner"}, + {"a doubled backslash still decodes", `"{\"note\":\"` + bs + bs + "u0077idgetworks" + `\"}"` + "\n", "widget-partner"}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + blocked, out := hookRun(t, banlist, c.staged) + if !blocked { + t.Fatalf("commit not blocked; the escaped spelling is the banned name\n%s", out) + } + if !strings.Contains(out, c.key) { + t.Errorf("the refusal does not name the key %q\n%s", c.key, out) + } + for _, leak := range []string{"widgetworks", "qüxbar"} { + if strings.Contains(strings.ToLower(out), leak) { + t.Errorf("output leaks %q; the decoded text is withheld like the raw\n%s", leak, out) + } + } + }) + } + + t.Run("escape-spelled backslashes that spell no banned name pass", func(t *testing.T) { + staged := `{"path":"C:` + escBS + "new" + escBS + "u0077idget works" + `", "q":"%5Cn caf%5Cu00e9"}` + "\n" + blocked, out := hookRun(t, banlist, staged) + if blocked { + t.Fatalf("a staged file whose decoded layers hold no banned name was refused\n%s", out) + } + }) +} + +// TestPreCommitHook_DecoderIsOneBlockSplitOnCharacters pins the decoder's shape +// in both copies of the guard: abcd's own hook and the one ahoy scaffolds. +// Every awk split takes a one-character string separator, because a regular +// expression separator makes the split of the one true awk (macOS) quadratic in +// the length of the line — one 19 MB line of minified JSON took over a minute +// to decode (iss-2609280945018822). And the decode block, from its layer bound +// to the pattern loop, is the same bytes in both copies, so a fix to one cannot +// leave the other reading less. +func TestPreCommitHook_DecoderIsOneBlockSplitOnCharacters(t *testing.T) { + hook := locateHook(t) + scaffold := filepath.Join(filepath.Dir(filepath.Dir(hook)), "internal", "core", "ahoy", "defaults", "pre-commit") + var blocks []string + for _, path := range []string{hook, scaffold} { + src, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + s := string(src) + start := strings.Index(s, "\ndecode_layers=") + if start < 0 { + t.Fatalf("%s: no decode_layers declaration opens the decode block", path) + } + end := strings.Index(s[start:], "\ni=0\n") + if end < 0 { + t.Fatalf("%s: the decode block does not end at the pattern loop", path) + } + block := s[start : start+end] + blocks = append(blocks, block) + calls := regexp.MustCompile(`split\([^,]+,\s*[a-z]+,\s*([^)]*)\)`).FindAllStringSubmatch(block, -1) + if len(calls) == 0 { + t.Fatalf("%s: the decode block holds no split call; the pin has no target", path) + } + for _, c := range calls { + if c[1] != `"\\"` && c[1] != `"%"` { + t.Errorf("%s: split separator %s; want a one-character string, never a regular expression", path, c[1]) + } + } + } + if blocks[0] != blocks[1] { + t.Errorf("the decode blocks of %s and %s differ; the two guards must read the same spellings", hook, scaffold) + } +} + // TestPreCommitHook_DecodeFailureRefuses pins the direction the decoded reading // fails in (iss-2609261909106167): a decoder that could not finish must never // leave a decoded copy that reads as "nothing to find". The hook's own decoder is From 975dc10269035a47642d15b5fd96642f2c3f0173 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:07:53 +0100 Subject: [PATCH 63/64] chore: resolve the name guard's escaped-backslash layer and its long-line decode cost Both findings of the verify of fix3-drainS3 are fixed in 5c5a8595c: the guard reads each decoded layer again, as many layers as the scanner, and its awk decode is linear in line length. Resolves: iss-2609280944560197, iss-2609280945018822 Assisted-by: Claude:claude-opus-5-5 --- ...commit-name-guard-decodes-one-json-escape-layer-per.md | 8 ++++++++ ...commit-name-guard-s-awk-decode-of-escaped-spellings.md | 8 ++++++++ 2 files changed, 16 insertions(+) rename .abcd/work/issues/{open => resolved}/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md (66%) rename .abcd/work/issues/{open => resolved}/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md (59%) diff --git a/.abcd/work/issues/open/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md b/.abcd/work/issues/resolved/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md similarity index 66% rename from .abcd/work/issues/open/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md rename to .abcd/work/issues/resolved/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md index b49f0b0c1..beff5c4de 100644 --- a/.abcd/work/issues/open/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md +++ b/.abcd/work/issues/resolved/iss-2609280944560197-the-pre-commit-name-guard-decodes-one-json-escape-layer-per.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: verify-fix3-drainS3" origin: researcher-authored production_mode: hand-written found_at: ".githooks/pre-commit" +resolution: "Each decoded layer is read again for the escapes it still holds, up to decode_layers (3), which a scanner test holds equal to maxJSONDecodeLayers; both hook copies carry the same decode block, pinned byte-identical." +impact: fix +resolved_by: + commit: "5c5a8595c" --- The pre-commit name guard decodes one JSON escape layer per run of backslashes and never re-reads its own output, so a backslash spelled as the unicode escape of U+005C (or as %5C) decodes to a literal backslash that is not read again as an escape: a banned name hidden one layer behind such a backslash (the escaped backslash followed by u0075 and the rest of the name) commits, while scanner.DecodedViews reads it through its second layer (maxJSONDecodeLayers is 3). The carrier is a hand-crafted file, since no mainstream encoder writes a backslash that way; the hook comments claim the guard reads what the scanner's redactors read, which overclaims by exactly this case. Both hook copies (.githooks/pre-commit and internal/core/ahoy/defaults/pre-commit) carry the decode. + +## Grounds + +- pursued: a name behind one or two escape-spelled backslashes (unicode escape of U+005C, or %5C) is refused by both guard copies while clean escaped text commits; a staged spelling DecodedViews reads that the hook passes, or a decode_layers below maxJSONDecodeLayers, would show it wrong diff --git a/.abcd/work/issues/open/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md b/.abcd/work/issues/resolved/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md similarity index 59% rename from .abcd/work/issues/open/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md rename to .abcd/work/issues/resolved/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md index fbc10c0c1..18ac0ceb2 100644 --- a/.abcd/work/issues/open/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md +++ b/.abcd/work/issues/resolved/iss-2609280945018822-the-pre-commit-name-guard-s-awk-decode-of-escaped-spellings.md @@ -9,6 +9,14 @@ found_during: "autonomous run A resumed 2026-09-25: verify-fix3-drainS3" origin: researcher-authored production_mode: hand-written found_at: ".githooks/pre-commit" +resolution: "The awk splits take one-character string separators instead of regular expressions, which made the one true awk's split quadratic in line length; one 19 MB line decodes in 2.5 s instead of 156 s, output byte-identical." +impact: fix +resolved_by: + commit: "5c5a8595c" --- The pre-commit name guard's awk decode of escaped spellings is superlinear in LINE length on macOS awk 20200816: 19 MB of escaped JSON over 105k lines decodes in 1.9 s, but the same bytes as one line take 65 s in awk and 88 s for the commit (the verify of fix3-drainS3). A minified single-line JSON export or asset pays it; the hook still completes and reports, so it is never silent. Both hook copies (.githooks/pre-commit and internal/core/ahoy/defaults/pre-commit) carry the decode. + +## Grounds + +- pursued: decode time grows linearly with line length (1 to 16 MB: 0.11 to 1.70 s) and the decoded bytes match the regex-split decoder on the corpus; a gawk or mawk that read a one-character string separator as a pattern, or a superlinear curve on another awk, would show it wrong From 0671dccc06514689cb3ac6970491559751d9ac51 Mon Sep 17 00:00:00 2001 From: REPPL <77722411+REPPL@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:46:26 +0100 Subject: [PATCH 64/64] chore: recalibrate the reading windows at the re-merged integration tip Measured on a clean clone of 35483ff77 by dry-run assemble. Every window keeps at least 1% headroom, so none moves: widening 1,280,867 tokens (4,931,339 bytes) under 1,300,000; entailment 382,072 (1,470,980) under 390,000; detection 1,289,903 (4,966,127) under 1,310,000. The comparative position is not measured this way, as its preset comment states. Refs: iss-2609251455354719 Assisted-by: Claude:claude-opus-5-5 --- .abcd/config/reading-presets.json | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/.abcd/config/reading-presets.json b/.abcd/config/reading-presets.json index 886a2e8e2..5371f647e 100644 --- a/.abcd/config/reading-presets.json +++ b/.abcd/config/reading-presets.json @@ -61,9 +61,9 @@ ], "window": { "tokens_est": 1300000, - "measured_tokens_est": 1279461, - "measured_bytes": 4925925, - "measured_at": "e1344b4c3a3d7a27382964d4644a3e7cfff0de81" + "measured_tokens_est": 1280867, + "measured_bytes": 4931339, + "measured_at": "35483ff77ca726d81633625eaa816cae14b38d73" } }, "entailment": { @@ -133,9 +133,9 @@ ], "window": { "tokens_est": 390000, - "measured_tokens_est": 381285, - "measured_bytes": 1467950, - "measured_at": "e1344b4c3a3d7a27382964d4644a3e7cfff0de81" + "measured_tokens_est": 382072, + "measured_bytes": 1470980, + "measured_at": "35483ff77ca726d81633625eaa816cae14b38d73" } }, "comparative": { @@ -217,9 +217,9 @@ ], "window": { "tokens_est": 1310000, - "measured_tokens_est": 1288496, - "measured_bytes": 4960713, - "measured_at": "e1344b4c3a3d7a27382964d4644a3e7cfff0de81" + "measured_tokens_est": 1289903, + "measured_bytes": 4966127, + "measured_at": "35483ff77ca726d81633625eaa816cae14b38d73" } } }