From d339cf483cee87608124d60257a182e94a25dfcb Mon Sep 17 00:00:00 2001 From: Ersan Bilik Date: Sat, 26 Sep 2026 21:42:40 +0300 Subject: [PATCH 1/3] fix(heading): trim the CRLF carriage return from entry titles On a CRLF knowledge file (core.autocrlf=true on Windows), ParseEntryBlocks splits on LF, so every line keeps its trailing carriage return and the EntryHeader title group (.+) captures it. ParseHeaders has the same bug over the whole file: "." stops at LF, not at CR. `ctx disclosure inspect --json` printed titles such as "Foo (consolidated)\r", and a digest plan that names the clean title fails apply with ErrEntryNotInStaging. Trim the captured title at both read points, as conventionBlocks already does for convention sections. Block lines stay verbatim, so the disclosure mover's byte-exact cuts are unaffected. No writer or regex change. Also add the spec covering this and the two follow-up fixes (drift header check on CRLF files; malformed entry headers absorbed by the previous entry). Spec: specs/fix-knowledge-crlf-and-malformed-headers.md Signed-off-by: Ersan Bilik --- internal/heading/entry.go | 4 +- internal/heading/entry_test.go | 15 ++ internal/heading/index.go | 6 +- internal/heading/index_test.go | 15 ++ ...ix-knowledge-crlf-and-malformed-headers.md | 136 ++++++++++++++++++ 5 files changed, 174 insertions(+), 2 deletions(-) create mode 100644 specs/fix-knowledge-crlf-and-malformed-headers.md diff --git a/internal/heading/entry.go b/internal/heading/entry.go index f2a507440..8c27eaeb0 100644 --- a/internal/heading/entry.go +++ b/internal/heading/entry.go @@ -48,7 +48,9 @@ func ParseEntryBlocks(content string) []EntryBlock { entry: entity.IndexEntry{ Timestamp: matches[1] + token.Dash + matches[2], Date: matches[1], - Title: matches[3], + // TrimSpace drops the carriage return a CRLF + // line keeps after the LF split. + Title: strings.TrimSpace(matches[3]), }, }) } diff --git a/internal/heading/entry_test.go b/internal/heading/entry_test.go index 832dc3c71..a4a5861b4 100644 --- a/internal/heading/entry_test.go +++ b/internal/heading/entry_test.go @@ -54,6 +54,21 @@ func TestParseEntryBlocks_Single(t *testing.T) { } } +// A CRLF file (core.autocrlf checkout on Windows) splits on "\n" and +// leaves "\r" on every line; the title must not keep it. +func TestParseEntryBlocks_CRLFTitle(t *testing.T) { + content := "# Decisions\r\n\r\n" + + "## [2026-01-15-120000] Use YAML for config\r\n\r\n" + + "**Context:** Need a config format\r\n" + blocks := ParseEntryBlocks(content) + if len(blocks) != 1 { + t.Fatalf("ParseEntryBlocks() = %d blocks, want 1", len(blocks)) + } + if got := blocks[0].Entry.Title; got != "Use YAML for config" { + t.Errorf("Title = %q, want %q", got, "Use YAML for config") + } +} + func TestParseEntryBlocks_Multiple(t *testing.T) { content := `# Decisions diff --git a/internal/heading/index.go b/internal/heading/index.go index 696667503..48aa81677 100644 --- a/internal/heading/index.go +++ b/internal/heading/index.go @@ -7,6 +7,8 @@ package heading import ( + "strings" + "github.com/ActiveMemory/ctx/internal/config/regex" "github.com/ActiveMemory/ctx/internal/config/token" "github.com/ActiveMemory/ctx/internal/entity" @@ -32,7 +34,9 @@ func ParseHeaders(content string) []entity.IndexEntry { if len(match) == regex.EntryHeaderGroups { date := match[1] time := match[2] - title := match[3] + // "(.+)" stops at LF but keeps a CRLF line's carriage + // return. + title := strings.TrimSpace(match[3]) entries = append(entries, entity.IndexEntry{ Timestamp: date + token.Dash + time, Date: date, diff --git a/internal/heading/index_test.go b/internal/heading/index_test.go index 3b32cc904..7a576b9c0 100644 --- a/internal/heading/index_test.go +++ b/internal/heading/index_test.go @@ -87,6 +87,21 @@ func TestParseHeaders(t *testing.T) { }, }, }, + { + // A CRLF file (core.autocrlf checkout on Windows): the + // title capture must not keep the carriage return. + name: "CRLF line endings", + content: "# Decisions\r\n\r\n" + + "## [2026-01-28-051426] First decision\r\n\r\n" + + "**Status**: Accepted\r\n", + expected: []entity.IndexEntry{ + { + Timestamp: "2026-01-28-051426", + Date: "2026-01-28", + Title: "First decision", + }, + }, + }, } for _, tt := range tests { diff --git a/specs/fix-knowledge-crlf-and-malformed-headers.md b/specs/fix-knowledge-crlf-and-malformed-headers.md new file mode 100644 index 000000000..2e452d8f3 --- /dev/null +++ b/specs/fix-knowledge-crlf-and-malformed-headers.md @@ -0,0 +1,136 @@ +# Spec: fix CRLF knowledge files and malformed entry headers + +Three defects in how ctx reads LEARNINGS.md / DECISIONS.md, all +field-observed on Windows, where a `core.autocrlf=true` checkout gives +every context file CRLF line endings. Two are CRLF sensitivities; the +third is a silent data-integrity hole in the progressive-disclosure +pass that CRLF debugging surfaced. + +## Problem + +### A. Entry titles keep the carriage return + +`heading.ParseEntryBlocks` (`internal/heading/entry.go`) splits content +on `"\n"`, so on a CRLF file every line keeps a trailing `"\r"`. +`regex.EntryHeader` is `## \[(\d{4}-\d{2}-\d{2})-(\d{6})] (.+)`, and +`(.+)` captures that `"\r"` into the title. `heading.ParseHeaders` +(`internal/heading/index.go`) has the same bug: it runs +`FindAllStringSubmatch` over the whole file, and `.` stops at `"\n"` +but not at `"\r"`. + +Observed: `ctx disclosure inspect .context/LEARNINGS.md --json` printed +titles such as `"Foo (consolidated)\r"`. A digest plan written from +those titles either carries the stray `"\r"` or, when a human or agent +types the title cleanly, fails `apply` with `ErrEntryNotInStaging`. +`ctx agent` overflow summaries and trace resolution show the same +dirty titles. + +### B. `ctx drift` flags every CRLF file as a stale header + +`drift.checkTemplateHeaders` (`internal/drift/check.go`) compares +`extractFirstComment(template)` with `extractFirstComment(live)` +byte-for-byte; `strings.TrimSpace` only trims the ends, not interior +newlines. A CRLF live file against the LF embedded template (or a +Windows-built binary, whose embedded templates are CRLF, against an LF +file) mismatches on line endings alone. Every context file warns +`comment header does not match template`, and `ctx init --reset` +cannot clear it: the rewritten file still differs from the embedded +bytes only in line endings. The false positives bury real drift. + +### C. A date-only entry header is silently absorbed + +`ParseEntryBlocks` starts a block only at a line matching the strict +`regex.EntryHeader`. An entry whose header lacks the time part — +`## [2026-09-26] Title`, from a hand edit or a pre-timestamp file — is +not a block start, so its heading and body become part of the +*previous* entry's block: + +- `ctx disclosure inspect` never lists it; +- `ctx disclosure apply` moves it, inside the previous entry's span, + into that entry's theme file — the wrong theme, with no trace in the + plan the human approved. + +`disclosure.Validate` already exists to refuse a structurally malformed +root before the pass mutates anything, but it only catches the case +where *no* staged entry parses (`ErrStagingUnparsable`). A malformed +header after at least one valid entry passes validation. + +## Approach + +Read-side fixes only. No writer changes, no file rewrites, no regex +changes, and no broadening of the accepted header format. + +- **A.** Trim the captured title with `strings.TrimSpace` at both read + points (`ParseEntryBlocks`, `ParseHeaders`). Block `Lines` stay + verbatim, so `SplitStaging`'s byte-exact cuts and conservation are + unaffected. Titles are already trimmed for convention sections + (`conventionBlocks`), so every kind now yields clean titles. +- **B.** `extractFirstComment` normalizes `token.NewlineCRLF` to + `token.NewlineLF` before extracting. Both sides of the comparison go + through it, so the check becomes line-ending-insensitive while staying + content-sensitive. +- **C.** `disclosure.Validate` refuses, for the timestamped kinds + (learning, decision), any line outside an HTML comment that matches + `regex.EntryHeading` (`^## \[`) but not `regex.EntryHeader`. The + refusal is a typed error, `errDisc.MalformedEntryHeaderError`, naming + the 1-based line number and the offending heading, and telling the + user to give it a full `## [YYYY-MM-DD-HHMMSS] Title` timestamp. + - HTML comments are skipped with the scanner's existing + `htmlCommentSpans`: DECISIONS.md's template ships a + `` guide containing + `## [YYYY-MM-DD] Decision Title`, which is an example, not an entry. + - The check runs before `ErrStagingUnparsable`, and the typed error's + `Is` matches `ErrStagingUnparsable`: a malformed header is the + precise form of "staging could not be parsed into discrete + entries". Callers matching the sentinel keep working; callers that + want the location use `errors.AsType[*MalformedEntryHeaderError]`. + A root whose *first* staged entry is malformed now gets the precise + error too, instead of the location-free generic one. + - Conventions are exempt: their sections open with `## ` and carry no + timestamp, so `## [` is ordinary prose there. + - `Apply` already calls `Validate` before any write, so a refused + root is byte-identical. `Inspect` stays total (read-only, never + fails), as documented. + +## Tests + +TDD, one failing test per defect before its fix: + +- **A.** `internal/heading`: a CRLF case for `ParseHeaders` and + `TestParseEntryBlocks_CRLFTitle`; both assert the title has no `"\r"`. +- **B.** `internal/drift`: `checkTemplateHeaders` against the real + embedded template rendered with LF and with CRLF endings raises no + `stale_header` warning; a genuinely edited comment still does. Both + endings are exercised so the test fails before the fix on either an + LF or a CRLF checkout. +- **C.** `internal/disclosure`: + - a date-only header after a valid entry is refused with + `MalformedEntryHeaderError` carrying the right line and heading, for + both LF and CRLF content (the heading is reported without `"\r"`); + - `errors.Is(err, ErrStagingUnparsable)` still holds; + - the real embedded DECISIONS.md template (its commented + `## [YYYY-MM-DD] Decision Title` example) plus a valid entry + validates cleanly; + - existing `TestValidate` cases, including "unparsable staging", + are unchanged and pass. + +## Acceptance + +- On a CRLF LEARNINGS.md, `ctx disclosure inspect --json` titles + contain no `"\r"`. +- Adding `## [2026-09-26] Legacy` below a valid entry makes + `ctx disclosure apply` refuse with the line number and heading; the + root is left untouched. +- `ctx drift` reports no `stale_header` for a CRLF context file whose + header matches the template. +- `golangci-lint run` clean; no new test failures. + +## Non-Goals + +- Normalizing line endings at write time (`ctx init`, `add`, reindex). +- Accepting or auto-repairing date-only headers. The pass stays + fail-loud with no auto-repair (progressive-disclosure Guards §4). +- Making `ctx disclosure inspect` refuse malformed roots; it stays a + total, read-only view, and `apply` is the gate. +- Fenced-code awareness in the header scan; the disclosure scanners are + deliberately fence-blind (see `conventionBlocks`). From 40967f2158162f893418756e40727d7490befea3 Mon Sep 17 00:00:00 2001 From: Ersan Bilik Date: Sat, 26 Sep 2026 21:46:09 +0300 Subject: [PATCH 2/3] fix(drift): compare template comment headers line-ending-insensitively checkTemplateHeaders compares extractFirstComment(template) with extractFirstComment(live) byte-for-byte; TrimSpace only trims the ends, not interior newlines. A CRLF context file (core.autocrlf=true on Windows) against the LF embedded template, or a Windows-built binary with CRLF-embedded templates against an LF file, mismatches on line endings alone. Every context file then warns "comment header ... does not match template", and `ctx init --reset` cannot clear it. Normalize CRLF to LF inside extractFirstComment, so both sides of the comparison are LF. The check stays content-sensitive: an edited header still warns. Spec: specs/fix-knowledge-crlf-and-malformed-headers.md Signed-off-by: Ersan Bilik --- internal/drift/check.go | 9 +++++-- internal/drift/detector_test.go | 46 +++++++++++++++++++++++++++++++++ 2 files changed, 53 insertions(+), 2 deletions(-) diff --git a/internal/drift/check.go b/internal/drift/check.go index 8a475e395..eab9d82c3 100644 --- a/internal/drift/check.go +++ b/internal/drift/check.go @@ -373,13 +373,18 @@ func checkMissingPackages(ctx *entity.Context, report *Report) { // extractFirstComment extracts the first HTML comment block from content. // Returns an empty string if no comment found. // +// CRLF is normalized to LF first, so comparing two extracts is +// line-ending-insensitive: a core.autocrlf checkout writes CRLF files, +// and a binary built from one embeds CRLF templates. +// // Parameters: // - content: Raw file content to scan for an HTML comment // // Returns: -// - string: Trimmed comment including delimiters, -// or empty string if none found +// - string: Trimmed comment including delimiters, with LF line +// endings, or empty string if none found func extractFirstComment(content string) string { + content = strings.ReplaceAll(content, token.NewlineCRLF, token.NewlineLF) start := strings.Index(content, marker.CommentOpen) if start == -1 { return "" diff --git a/internal/drift/detector_test.go b/internal/drift/detector_test.go index 3d01318e5..4f002339e 100644 --- a/internal/drift/detector_test.go +++ b/internal/drift/detector_test.go @@ -12,7 +12,10 @@ import ( "strings" "testing" + readTpl "github.com/ActiveMemory/ctx/internal/assets/read/template" + cfgCtx "github.com/ActiveMemory/ctx/internal/config/ctx" cfgDrift "github.com/ActiveMemory/ctx/internal/config/drift" + "github.com/ActiveMemory/ctx/internal/config/token" "github.com/ActiveMemory/ctx/internal/context/load" "github.com/ActiveMemory/ctx/internal/entity" "github.com/ActiveMemory/ctx/internal/io" @@ -628,3 +631,46 @@ func TestIsTemplateFile(t *testing.T) { }) } } + +// A header that matches its template must pass whatever the line +// endings: a core.autocrlf=true checkout writes CRLF context files, and +// a binary built from one embeds CRLF templates. Both endings are +// checked so the test bites on an LF and on a CRLF checkout alike. +func TestCheckTemplateHeaders_LineEndings(t *testing.T) { + tpl, tplErr := readTpl.Template(cfgCtx.Learning) + if tplErr != nil { + t.Fatalf("read template: %v", tplErr) + } + lf := strings.ReplaceAll(string(tpl), token.NewlineCRLF, token.NewlineLF) + crlf := strings.ReplaceAll(lf, token.NewlineLF, token.NewlineCRLF) + + tests := []struct { + name string + content string + wantStale bool + }{ + {"LF file", lf, false}, + {"CRLF file", crlf, false}, + {"edited header", strings.Replace(lf, "UPDATE WHEN", "EDITED", 1), true}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + ctx := &entity.Context{Files: []entity.FileInfo{ + {Name: cfgCtx.Learning, Content: []byte(tt.content)}, + }} + report := &Report{} + + checkTemplateHeaders(ctx, report) + + stale := false + for _, w := range report.Warnings { + stale = stale || w.Type == cfgDrift.IssueStaleHeader + } + if stale != tt.wantStale { + t.Errorf("stale header = %v, want %v (warnings: %+v)", + stale, tt.wantStale, report.Warnings) + } + }) + } +} From e8f8f509318249923cb1177b304ceb315009f161 Mon Sep 17 00:00:00 2001 From: Ersan Bilik Date: Sat, 26 Sep 2026 21:52:10 +0300 Subject: [PATCH 3/3] fix(disclosure): refuse roots with a malformed entry header ParseEntryBlocks starts a block only at a line matching the strict EntryHeader regex, so an entry whose header lacks the time part ("## [2026-09-26] Title", legacy or hand-written) is silently absorbed into the previous entry's block. `ctx disclosure inspect` never lists it, and `ctx disclosure apply` moves it, inside the previous entry's span, into that entry's theme file. Validate only caught the case where no staged entry parsed at all. For the timestamped kinds, Validate now refuses any "## [" line outside an HTML comment that is not a full EntryHeader, with a typed MalformedEntryHeaderError naming the 1-based line and the heading and telling the user to give it a full YYYY-MM-DD-HHMMSS timestamp. Its Is matches ErrStagingUnparsable, of which it is the located form, so existing callers and the "unparsable staging" case keep working. The comment skip keeps DECISIONS.md's shipped "## [YYYY-MM-DD] Decision Title" format example from tripping the guard. Conventions are exempt: their sections carry no timestamp. The accepted header format is unchanged. Spec: specs/fix-knowledge-crlf-and-malformed-headers.md Signed-off-by: Ersan Bilik --- internal/assets/commands/text/errors.yaml | 2 + internal/config/embed/text/disclosure.go | 4 + internal/disclosure/regions.go | 32 ++++++++ internal/disclosure/validate.go | 20 ++++- internal/disclosure/validate_test.go | 78 +++++++++++++++++++ internal/err/disclosure/disclosure.go | 52 +++++++++++++ internal/err/disclosure/doc.go | 3 + ...ix-knowledge-crlf-and-malformed-headers.md | 2 + 8 files changed, 191 insertions(+), 2 deletions(-) diff --git a/internal/assets/commands/text/errors.yaml b/internal/assets/commands/text/errors.yaml index 679441546..d5a77be6f 100644 --- a/internal/assets/commands/text/errors.yaml +++ b/internal/assets/commands/text/errors.yaml @@ -160,6 +160,8 @@ err.disclosure.orphan-theme-file: short: 'progressive disclosure: a theme file has no matching gist in the root' err.disclosure.staging-unparsable: short: 'progressive disclosure: the staging zone could not be parsed into discrete entries' +err.disclosure.malformed-entry-header: + short: 'progressive disclosure: line %d: %q is not a full entry header; give it a full timestamp ("## [YYYY-MM-DD-HHMMSS] Title") so it is not folded into the entry above it' err.disclosure.apply-not-entry-kind: short: 'progressive disclosure: the mover digests only LEARNINGS.md and DECISIONS.md; conventions digestion is a later milestone' err.disclosure.empty-assignment: diff --git a/internal/config/embed/text/disclosure.go b/internal/config/embed/text/disclosure.go index 846a6aad7..d4637c54a 100644 --- a/internal/config/embed/text/disclosure.go +++ b/internal/config/embed/text/disclosure.go @@ -58,6 +58,10 @@ const ( // DescKeyErrDisclosureStagingUnparsable: the staging zone could not // be parsed into discrete entries. DescKeyErrDisclosureStagingUnparsable = "err.disclosure.staging-unparsable" + // DescKeyErrDisclosureMalformedEntryHeader: format for a "## [" line + // that is not a full timestamped entry header; takes the 1-based line + // number and the heading. + DescKeyErrDisclosureMalformedEntryHeader = "err.disclosure.malformed-entry-header" ) // DescKeys for the milestone-3 mover (the digesting pass that writes diff --git a/internal/disclosure/regions.go b/internal/disclosure/regions.go index 6f46199bc..68665ae1a 100644 --- a/internal/disclosure/regions.go +++ b/internal/disclosure/regions.go @@ -11,6 +11,7 @@ import ( cfgDisc "github.com/ActiveMemory/ctx/internal/config/disclosure" cfgFile "github.com/ActiveMemory/ctx/internal/config/file" + "github.com/ActiveMemory/ctx/internal/config/regex" "github.com/ActiveMemory/ctx/internal/config/token" ) @@ -202,6 +203,37 @@ func entryBelowThemes(themesRaw string, k Kind) bool { return false } +// malformedEntryHeading finds the first line that opens like an entry +// heading ("## [") but is not a full "## [YYYY-MM-DD-HHMMSS] Title" +// entry header, e.g. a date-only "## [2026-09-26] Title". The block +// parser does not start an entry at such a line, so it and its body +// would be silently folded into the entry above. Lines inside an HTML +// comment are skipped: DECISIONS.md's format guide ships a +// "## [YYYY-MM-DD] Decision Title" example. +// +// Parameters: +// - content: the text to scan +// +// Returns: +// - line: 1-based line number of the offending line, or 0 if none +// - heading: that line, whitespace-trimmed, or "" if none +func malformedEntryHeading(content string) (line int, heading string) { + spans := htmlCommentSpans(content) + for i, n := 0, 1; i < len(content); n++ { + text, next := lineAt(content, i) + if regex.EntryHeading.MatchString(text) && + !regex.EntryHeader.MatchString(text) && + !insideAnySpan(i, spans) { + return n, strings.TrimSpace(text) + } + if next == -1 { + break + } + i = next + } + return 0, "" +} + // htmlCommentSpans returns the [start, end) byte ranges of every HTML // comment (token.HTMLCommentOpen … token.HTMLCommentClose) in content; an // unterminated open runs to EOF. The structural heading scans use it to diff --git a/internal/disclosure/validate.go b/internal/disclosure/validate.go index 2cb7a0dad..9d2685f30 100644 --- a/internal/disclosure/validate.go +++ b/internal/disclosure/validate.go @@ -24,6 +24,10 @@ import ( // - no entry heading below "## Themes" (ErrEntryBelowThemes): entries // must stay in the staging zone above it. The heading that opens an // entry is per-kind ([EntryPrefix]). +// - for the timestamped kinds, every "## [" line outside an HTML +// comment is a full "## [YYYY-MM-DD-HHMMSS] Title" header +// (MalformedEntryHeaderError, which matches ErrStagingUnparsable): +// the block parser would fold any other one into the entry above. // - a non-empty staging zone must enumerate into discrete entries // (ErrStagingUnparsable), for every kind. // - for conventions, no two staged sections share a title @@ -37,9 +41,11 @@ import ( // - r: a parsed root (from Parse) // // Returns: -// - error: one of the disclosure sentinels, or nil when well-formed +// - error: one of the disclosure sentinels, a +// *MalformedEntryHeaderError, or nil when well-formed func Validate(r Root) error { - if len(headingLineOffsets(r.Reconstruct(), cfgDisc.HeadingThemes)) > 1 { + content := r.Reconstruct() + if len(headingLineOffsets(content, cfgDisc.HeadingThemes)) > 1 { return errDisc.ErrMultipleThemes } @@ -47,6 +53,16 @@ func Validate(r Root) error { return errDisc.ErrEntryBelowThemes } + // Conventions carry no timestamp, so "## [" is ordinary title text + // there. Scanning the whole root keeps line numbers file-relative; + // only staging can hold a hit: staging starts at the first "## [" + // line, and entryBelowThemes has cleared the themes region. + if r.Kind != KindConvention { + if line, h := malformedEntryHeading(content); line > 0 { + return errDisc.MalformedEntryHeader(line, h) + } + } + blocks := stagedBlocks(r.Staging, r.Kind) if strings.TrimSpace(r.Staging) != "" && len(blocks) == 0 { return errDisc.ErrStagingUnparsable diff --git a/internal/disclosure/validate_test.go b/internal/disclosure/validate_test.go index cdd6ad4e0..40041409d 100644 --- a/internal/disclosure/validate_test.go +++ b/internal/disclosure/validate_test.go @@ -8,8 +8,11 @@ package disclosure_test import ( "errors" + "strings" "testing" + readTpl "github.com/ActiveMemory/ctx/internal/assets/read/template" + cfgCtx "github.com/ActiveMemory/ctx/internal/config/ctx" "github.com/ActiveMemory/ctx/internal/disclosure" errDisc "github.com/ActiveMemory/ctx/internal/err/disclosure" ) @@ -44,6 +47,18 @@ const ( "## [2026-07-15-120000] Same Title\n\nfirst.\n\n" + "## [2026-07-16-090000] Same Title\n\nsecond.\n\n" + "## Themes\n\n- a — g → [a](learnings/a.md)\n" + + // A date-only header (legacy or hand-written) is not an entry start + // for the block parser, so without the guard it is silently folded + // into the entry above it — and moved into that entry's theme. The + // offending heading is on line 9. + dateOnlyHeader = "# Learnings\n\n\n\n" + + "## [2026-07-15-120000] a staged entry\n\n**Context**: x.\n\n" + + "## [2026-09-26] Legacy\n\n**Context**: y.\n" + + // Conventions carry no timestamp: "## [" is ordinary title text. + conventionBracketTitle = "# Conventions\n\n\n\n" + + "## [Draft] Naming\n\nprose.\n" ) // T06: the Validate precondition returns the named sentinel for each @@ -84,6 +99,15 @@ func TestValidate(t *testing.T) { "convention un-migrated", conventionUnmigrated, disclosure.KindConvention, nil, }, + // A malformed header is the precise form of unparsable staging. + { + "date-only header after an entry", dateOnlyHeader, + disclosure.KindLearning, errDisc.ErrStagingUnparsable, + }, + { + "convention bracketed title", conventionBracketTitle, + disclosure.KindConvention, nil, + }, } for _, tc := range cases { t.Run(tc.name, func(t *testing.T) { @@ -97,3 +121,57 @@ func TestValidate(t *testing.T) { }) } } + +// DECISIONS.md's shipped template carries a "## [YYYY-MM-DD] Decision +// Title" example inside its comment. That is +// documentation, not an entry: the malformed-header guard must skip it. +func TestValidate_DecisionTemplateExampleIgnored(t *testing.T) { + tpl, tplErr := readTpl.Template(cfgCtx.Decision) + if tplErr != nil { + t.Fatalf("read template: %v", tplErr) + } + if !strings.Contains(string(tpl), "## [YYYY-MM-DD] Decision Title") { + t.Fatal("template lost its commented date-only example; " + + "this test no longer guards anything") + } + content := string(tpl) + + "\n## [2026-09-26-120000] A real decision\n\n**Status**: Accepted\n" + + err := disclosure.Validate(disclosure.Parse(content, disclosure.KindDecision)) + if err != nil { + t.Errorf("Validate = %v, want nil (commented example is not an entry)", err) + } +} + +// The malformed-header refusal names the 1-based line and the heading, +// on LF and CRLF files alike (the carriage return is not reported), and +// stays matchable as ErrStagingUnparsable. +func TestValidate_MalformedEntryHeader(t *testing.T) { + const wantLine, wantHeading = 9, "## [2026-09-26] Legacy" + for name, content := range map[string]string{ + "LF": dateOnlyHeader, + "CRLF": strings.ReplaceAll(dateOnlyHeader, "\n", "\r\n"), + } { + t.Run(name, func(t *testing.T) { + err := disclosure.Validate( + disclosure.Parse(content, disclosure.KindLearning), + ) + mErr, ok := errors.AsType[*errDisc.MalformedEntryHeaderError](err) + if !ok { + t.Fatalf("Validate = %v, want *MalformedEntryHeaderError", err) + } + if mErr.Line != wantLine || mErr.Heading != wantHeading { + t.Errorf("got line %d heading %q, want line %d heading %q", + mErr.Line, mErr.Heading, wantLine, wantHeading) + } + if !errors.Is(err, errDisc.ErrStagingUnparsable) { + t.Errorf("errors.Is(%v, ErrStagingUnparsable) = false", err) + } + msg := err.Error() + if !strings.Contains(msg, "line 9") || + !strings.Contains(msg, wantHeading) { + t.Errorf("message %q must name line 9 and the heading", msg) + } + }) + } +} diff --git a/internal/err/disclosure/disclosure.go b/internal/err/disclosure/disclosure.go index 093cc3a10..cd2a07125 100644 --- a/internal/err/disclosure/disclosure.go +++ b/internal/err/disclosure/disclosure.go @@ -112,6 +112,58 @@ const ( ) ) +// MalformedEntryHeaderError is returned by Validate when a LEARNINGS or +// DECISIONS root has a line that opens like an entry ("## [") but is not +// a full "## [YYYY-MM-DD-HHMMSS] Title" header — e.g. a date-only +// "## [2026-09-26] Title". The block parser would not start an entry +// there and would silently fold it into the entry above, so the pass +// refuses instead. It refines [ErrStagingUnparsable]: callers using +// errors.Is on that sentinel still match, and callers using +// `errors.AsType[*MalformedEntryHeaderError]` recover the location. +type MalformedEntryHeaderError struct { + // Line is the 1-based line number of the offending heading. + Line int + // Heading is the offending line, whitespace-trimmed. + Heading string +} + +// Error implements the error interface for MalformedEntryHeaderError. +// +// Returns: +// - string: message naming the line, the heading, and the fix +func (e *MalformedEntryHeaderError) Error() string { + return fmt.Sprintf( + desc.Text(text.DescKeyErrDisclosureMalformedEntryHeader), + e.Line, e.Heading, + ) +} + +// Is reports whether target is [ErrStagingUnparsable], of which a +// malformed entry header is the located form. +// +// Parameters: +// - target: error to compare against +// +// Returns: +// - bool: true when target is [ErrStagingUnparsable] +func (e *MalformedEntryHeaderError) Is(target error) bool { + return target == ErrStagingUnparsable +} + +// MalformedEntryHeader returns a MalformedEntryHeaderError. +// +// Parameters: +// - line: 1-based line number of the offending heading +// - heading: the offending line, whitespace-trimmed +// +// Returns: +// - *MalformedEntryHeaderError: typed error for errors.AsType matching +func MalformedEntryHeader( + line int, heading string, +) *MalformedEntryHeaderError { + return &MalformedEntryHeaderError{Line: line, Heading: heading} +} + // NotAKnowledgeFile wraps [ErrNotAKnowledgeFile] with the offending path // and the expected filenames. // diff --git a/internal/err/disclosure/doc.go b/internal/err/disclosure/doc.go index d6e518901..98d0c2c02 100644 --- a/internal/err/disclosure/doc.go +++ b/internal/err/disclosure/doc.go @@ -20,6 +20,9 @@ // // - **Structure** ([ErrMultipleThemes], [ErrEntryBelowThemes], // [ErrStagingUnparsable]): the precondition refused a malformed root. +// [MalformedEntryHeaderError] is the located form of +// [ErrStagingUnparsable]: it names the line and heading of a "## [" +// line that is not a full timestamped entry header. // - **Cross-file** ([ErrOrphanThemeFile], [ErrMissingThemeFile], // [ErrDuplicateEntry], [ErrBrokenThemeLink]): the root ↔ theme-file // link graph or the one-place-per-entry invariant is broken. diff --git a/specs/fix-knowledge-crlf-and-malformed-headers.md b/specs/fix-knowledge-crlf-and-malformed-headers.md index 2e452d8f3..ba05fdd6b 100644 --- a/specs/fix-knowledge-crlf-and-malformed-headers.md +++ b/specs/fix-knowledge-crlf-and-malformed-headers.md @@ -111,6 +111,8 @@ TDD, one failing test per defect before its fix: - the real embedded DECISIONS.md template (its commented `## [YYYY-MM-DD] Decision Title` example) plus a valid entry validates cleanly; + - a convention root with a `## [Draft] Naming` section still + validates (conventions are exempt); - existing `TestValidate` cases, including "unparsable staging", are unchanged and pass.