From 754759311075d43c0910b4882c65e4928a5a3cc6 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 01:29:18 -0600 Subject: [PATCH 01/18] feat(extract): support hierarchical body section selectors --- internal/extract/body.go | 80 +++++++++--- internal/extract/body_source.go | 183 +++++++++++++++++++++++---- internal/extract/body_source_test.go | 182 +++++++++++++++++++++----- internal/rules/field_source_test.go | 9 +- 4 files changed, 365 insertions(+), 89 deletions(-) diff --git a/internal/extract/body.go b/internal/extract/body.go index 0eaf01e1..9611973a 100644 --- a/internal/extract/body.go +++ b/internal/extract/body.go @@ -7,12 +7,25 @@ import ( east "github.com/yuin/goldmark/extension/ast" ) +// HeadingKey identifies one markdown heading. +type HeadingKey struct { + Level int `json:"level"` + Text string `json:"text"` +} + +// SectionPath identifies a section from its root heading to its own heading. +type SectionPath []HeadingKey + +// SectionSelector identifies a section by a contiguous suffix of its path. +type SectionSelector []HeadingKey + // Section represents a heading-delimited section in a markdown body. type Section struct { - Heading string `json:"heading"` - Level int `json:"level"` - Content string `json:"content"` - StartLine int `json:"start_line"` + Heading string `json:"heading"` + Level int `json:"level"` + Path SectionPath `json:"path,omitempty"` + Content string `json:"content"` + StartLine int `json:"start_line"` } // ExtractSections splits a markdown body into sections delimited by headings. @@ -34,15 +47,6 @@ func ExtractSections(node ast.Node, source []byte) []Section { continue } h := child.(*ast.Heading) - // Extract heading text from child text nodes. - var text strings.Builder - for c := h.FirstChild(); c != nil; c = c.NextSibling() { - if c.Kind() == ast.KindText { - seg := c.(*ast.Text).Segment - text.Write(seg.Value(source)) - } - } - lines := h.Lines() startLine := 0 endOffset := 0 @@ -60,7 +64,7 @@ func ExtractSections(node ast.Node, source []byte) []Section { } headings = append(headings, headingInfo{ - text: text.String(), + text: headingTextAtLine(source, startLine), level: h.Level, startLine: startLine, endOffset: endOffset, @@ -100,7 +104,20 @@ func ExtractSections(node ast.Node, source []byte) []Section { }) } - return sections + return sectionsWithPaths(sections) +} + +func headingTextAtLine(source []byte, line int) string { + start := lineOffset(source, line) + end := start + for end < len(source) && source[end] != '\n' { + end++ + } + raw := string(source[start:end]) + if _, text, ok := parseATXHeading(raw); ok { + return text + } + return strings.TrimSpace(strings.TrimRight(raw, "\r")) } // CodeBlock represents a fenced code block in a markdown body. @@ -219,11 +236,11 @@ func ExtractSectionsFromText(body string) []Section { lineEnd++ } line := string(source[start:end]) - if char, length, ok := parseFenceLine(line); ok { + if char, length, trailing, ok := parseFenceLine(line); ok { prevOK = false if !inFence { inFence, fenceChar, fenceLen = true, char, length - } else if char == fenceChar && length >= fenceLen { + } else if char == fenceChar && length >= fenceLen && strings.TrimSpace(trailing) == "" { inFence = false } } else if !inFence { @@ -259,6 +276,22 @@ func ExtractSectionsFromText(body string) []Section { } sections = append(sections, Section{Heading: h.text, Level: h.level, Content: content, StartLine: h.line}) } + return sectionsWithPaths(sections) +} + +func sectionsWithPaths(sections []Section) []Section { + path := make(SectionPath, 0, 6) + for i := range sections { + if sections[i].Level <= 0 { + sections[i].Path = nil + continue + } + for len(path) > 0 && path[len(path)-1].Level >= sections[i].Level { + path = path[:len(path)-1] + } + path = append(path, HeadingKey{Level: sections[i].Level, Text: sections[i].Heading}) + sections[i].Path = append(SectionPath(nil), path...) + } return sections } @@ -306,17 +339,22 @@ func parseSetextUnderline(line string) (int, bool) { return 2, true } -func parseFenceLine(line string) (byte, int, bool) { +func parseFenceLine(line string) (byte, int, string, bool) { line = strings.TrimRight(line, "\r") indent := len(line) - len(strings.TrimLeft(line, " ")) if indent > 3 || indent >= len(line) || (line[indent] != '`' && line[indent] != '~') { - return 0, 0, false + return 0, 0, "", false } char, count := line[indent], 0 - for i := indent; i < len(line) && line[i] == char; i++ { + i := indent + for ; i < len(line) && line[i] == char; i++ { count++ } - return char, count, count >= 3 + trailing := line[i:] + if count < 3 || (char == '`' && strings.Contains(trailing, "`")) { + return 0, 0, "", false + } + return char, count, trailing, true } // lineOffset returns the byte offset of the start of a 1-based line number. diff --git a/internal/extract/body_source.go b/internal/extract/body_source.go index f3925b5b..f011c8ad 100644 --- a/internal/extract/body_source.go +++ b/internal/extract/body_source.go @@ -14,37 +14,118 @@ const ( ) type BodySource struct { - Kind BodySourceKind - Heading string + Kind BodySourceKind + + // Heading retains the final exact heading for callers that use simple selectors. + Heading string + Selector SectionSelector } func ParseBodySource(directive string) (BodySource, error) { switch { case directive == "body.h1": return BodySource{Kind: BodySourceH1}, nil - case strings.HasPrefix(directive, "body.section["): - if !strings.HasSuffix(directive, "]") { - return BodySource{}, fmt.Errorf("malformed body section source %q", directive) - } - quoted := strings.TrimSuffix(strings.TrimPrefix(directive, "body.section["), "]") - heading, err := strconv.Unquote(quoted) + case strings.HasPrefix(directive, "body.section"): + selector, err := parseSectionSelector(directive) if err != nil { - return BodySource{}, fmt.Errorf("malformed body section source %q", directive) - } - if err := validateExactHeading(heading); err != nil { return BodySource{}, err } - return BodySource{Kind: BodySourceSection, Heading: heading}, nil + return BodySource{ + Kind: BodySourceSection, + Heading: exactHeading(selector[len(selector)-1]), + Selector: selector, + }, nil default: return BodySource{}, fmt.Errorf("unsupported body source %q", directive) } } +func parseSectionSelector(directive string) (SectionSelector, error) { + const prefix = "body.section" + if !strings.HasPrefix(directive, prefix) { + return nil, fmt.Errorf("unsupported body source %q", directive) + } + + rest := strings.TrimPrefix(directive, prefix) + selector := make(SectionSelector, 0, 1) + for rest != "" { + if len(rest) < 4 || rest[0] != '[' || (rest[1] != '"' && rest[1] != '`') { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + quoteEnd := quotedStringEnd(rest[1:]) + if quoteEnd < 0 || quoteEnd+2 >= len(rest) || rest[quoteEnd+2] != ']' { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + quoted := rest[1 : quoteEnd+2] + heading, err := strconv.Unquote(quoted) + if err != nil { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + key, err := parseHeadingKey(heading) + if err != nil { + return nil, err + } + if len(selector) > 0 && key.Level <= selector[len(selector)-1].Level { + return nil, fmt.Errorf("section source heading levels must increase in %q", directive) + } + selector = append(selector, key) + rest = rest[quoteEnd+3:] + } + if len(selector) == 0 { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + return selector, nil +} + +// quotedStringEnd returns the index of the closing quote in a quoted string. +func quotedStringEnd(quoted string) int { + if len(quoted) == 0 || (quoted[0] != '"' && quoted[0] != '`') { + return -1 + } + delimiter := quoted[0] + escaped := false + for i := 1; i < len(quoted); i++ { + if delimiter == '`' { + if quoted[i] == delimiter { + return i + } + continue + } + if escaped { + escaped = false + continue + } + switch quoted[i] { + case '\\': + escaped = true + case delimiter: + return i + } + } + return -1 +} + func CanonicalSectionSource(exactHeading string) (string, error) { - if err := validateExactHeading(exactHeading); err != nil { + key, err := parseHeadingKey(exactHeading) + if err != nil { + return "", err + } + return CanonicalSectionSelectorSource(SectionSelector{key}) +} + +// CanonicalSectionSelectorSource serializes a section selector. +func CanonicalSectionSelectorSource(selector SectionSelector) (string, error) { + if err := validateSectionSelector(selector); err != nil { return "", err } - return "body.section[" + strconv.Quote(exactHeading) + "]", nil + var source strings.Builder + source.WriteString("body.section") + for _, key := range selector { + source.WriteByte('[') + source.WriteString(strconv.Quote(exactHeading(key))) + source.WriteByte(']') + } + return source.String(), nil } func ResolveBodyValue(record *Record, directive string) (string, bool, error) { @@ -60,7 +141,7 @@ func ResolveBodyValue(record *Record, directive string) (string, bool, error) { h1 := resolveH1(record) return h1, h1 != "", nil case BodySourceSection: - return resolveUniqueSection(sectionsForRecord(record), source.Heading) + return resolveUniqueSection(sectionsForRecord(record), source.Selector) default: return "", false, fmt.Errorf("unsupported body source %q", directive) } @@ -75,26 +156,45 @@ func resolveH1(record *Record) string { return "" } -func resolveUniqueSection(sections []Section, exactHeading string) (string, bool, error) { - var match *Section +func resolveUniqueSection(sections []Section, selector SectionSelector) (string, bool, error) { + match := -1 for i := range sections { - if sectionExactHeading(sections[i]) != exactHeading { + if !SectionSelectorMatches(sections[i].Path, selector) { continue } - if match != nil { - return "", false, fmt.Errorf("ambiguous body section source %q", exactHeading) + if match >= 0 { + description := exactHeading(selector[0]) + if len(selector) > 1 { + description, _ = CanonicalSectionSelectorSource(selector) + } + return "", false, fmt.Errorf("ambiguous body section source %q", description) } - match = §ions[i] + match = i } - if match == nil { + if match < 0 { return "", false, nil } - return match.Content, true, nil + return sections[match].Content, true, nil +} + +// SectionSelectorMatches reports whether selector is a contiguous path suffix. +func SectionSelectorMatches(path SectionPath, selector SectionSelector) bool { + if len(selector) == 0 || len(selector) > len(path) { + return false + } + offset := len(path) - len(selector) + for i := range selector { + if path[offset+i] != selector[i] { + return false + } + } + return true } func sectionsForRecord(record *Record) []Section { if len(record.BodySections) > 0 { - return record.BodySections + sections := append([]Section(nil), record.BodySections...) + return sectionsWithPaths(sections) } if record.Body == "" { return nil @@ -106,13 +206,40 @@ func sectionExactHeading(sec Section) string { if sec.Level <= 0 { return sec.Heading } - return strings.Repeat("#", sec.Level) + " " + sec.Heading + return exactHeading(HeadingKey{Level: sec.Level, Text: sec.Heading}) +} + +func exactHeading(key HeadingKey) string { + return strings.Repeat("#", key.Level) + " " + key.Text } -func validateExactHeading(heading string) error { +func parseHeadingKey(heading string) (HeadingKey, error) { + if strings.ContainsAny(heading, "\r\n") { + return HeadingKey{}, fmt.Errorf("section source heading must be an exact markdown heading, got %q", heading) + } level, text, ok := parseATXHeading(heading) - if !ok || level == 0 || sectionExactHeading(Section{Heading: text, Level: level}) != heading { - return fmt.Errorf("section source heading must be an exact markdown heading, got %q", heading) + key := HeadingKey{Level: level, Text: text} + if !ok || level == 0 || exactHeading(key) != heading { + return HeadingKey{}, fmt.Errorf("section source heading must be an exact markdown heading, got %q", heading) + } + return key, nil +} + +func validateSectionSelector(selector SectionSelector) error { + if len(selector) == 0 { + return fmt.Errorf("section selector must contain one heading") + } + for i, key := range selector { + if key.Level < 1 || key.Level > 6 || exactHeading(key) == "" { + return fmt.Errorf("section selector has an invalid heading at index %d", i) + } + parsed, err := parseHeadingKey(exactHeading(key)) + if err != nil || parsed != key { + return fmt.Errorf("section selector has an invalid heading at index %d", i) + } + if i > 0 && key.Level <= selector[i-1].Level { + return fmt.Errorf("section selector heading levels must increase") + } } return nil } diff --git a/internal/extract/body_source_test.go b/internal/extract/body_source_test.go index 4d1a9e26..e10a005b 100644 --- a/internal/extract/body_source_test.go +++ b/internal/extract/body_source_test.go @@ -7,57 +7,152 @@ import ( ) func TestParseBodySource(t *testing.T) { - got, err := ParseBodySource("body.h1") - if err != nil || got != (BodySource{Kind: BodySourceH1}) { - t.Fatalf("body.h1 parsed as %+v err=%v", got, err) + tests := []struct { + directive string + want BodySource + }{ + { + directive: "body.h1", + want: BodySource{Kind: BodySourceH1}, + }, + { + directive: `body.section["## Notes"]`, + want: BodySource{ + Kind: BodySourceSection, + Heading: "## Notes", + Selector: SectionSelector{{Level: 2, Text: "Notes"}}, + }, + }, + { + directive: `body.section["## Parent"]["### Notes"]`, + want: BodySource{ + Kind: BodySourceSection, + Heading: "### Notes", + Selector: SectionSelector{ + {Level: 2, Text: "Parent"}, + {Level: 3, Text: "Notes"}, + }, + }, + }, } - got, err = ParseBodySource(`body.section["## Notes"]`) - if err != nil || got != (BodySource{Kind: BodySourceSection, Heading: "## Notes"}) { - t.Fatalf("section parsed as %+v err=%v", got, err) + for _, tt := range tests { + got, err := ParseBodySource(tt.directive) + if err != nil || !reflect.DeepEqual(got, tt.want) { + t.Fatalf("ParseBodySource(%q) = %+v, %v; want %+v", tt.directive, got, err, tt.want) + } } +} + +func TestParseBodySourceRejectsInvalidSelectors(t *testing.T) { for _, tt := range []struct{ directive, wantErr string }{ {"body.title", "unsupported"}, {`body.section[## Notes]`, "malformed"}, {`body.section["Notes"]`, "heading"}, + {`body.section["## Parent"]["## Notes"]`, "increase"}, + {`body.section["### Parent"]["## Notes"]`, "increase"}, } { _, err := ParseBodySource(tt.directive) if err == nil || !strings.Contains(err.Error(), tt.wantErr) { - t.Fatalf("%s: expected %q error, got %v", tt.directive, tt.wantErr, err) + t.Fatalf("ParseBodySource(%q) error = %v; want text %q", tt.directive, err, tt.wantErr) } } } +func TestParseBodySourceAllowsLevelJump(t *testing.T) { + got, err := ParseBodySource(`body.section["# Root"]["#### Detail"]`) + want := SectionSelector{{Level: 1, Text: "Root"}, {Level: 4, Text: "Detail"}} + if err != nil || !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("selector = %+v, %v; want %+v", got.Selector, err, want) + } +} + func TestCanonicalSectionSource(t *testing.T) { got, err := CanonicalSectionSource("## Notes") if err != nil { - t.Fatalf("CanonicalSectionSource: %v", err) + t.Fatalf("CanonicalSectionSource returned an error: %v", err) } if got != `body.section["## Notes"]` { t.Fatalf("CanonicalSectionSource() = %q", got) } + + got, err = CanonicalSectionSelectorSource(SectionSelector{ + {Level: 2, Text: "Parent"}, + {Level: 4, Text: `Quote "value"`}, + }) + if err != nil { + t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) + } + if got != `body.section["## Parent"]["#### Quote \"value\""]` { + t.Fatalf("CanonicalSectionSelectorSource() = %q", got) + } + parsed, err := ParseBodySource(got) + if err != nil || len(parsed.Selector) != 2 || parsed.Selector[1].Text != `Quote "value"` { + t.Fatalf("serialized selector parsed as %+v, %v", parsed, err) + } } func TestResolveBodyValue_PresentEmptySection(t *testing.T) { rec := &Record{BodySections: []Section{{Heading: "Notes", Level: 2, Content: ""}}} got, present, err := ResolveBodyValue(rec, `body.section["## Notes"]`) if err != nil || !present || got != "" { - t.Fatalf("got value=%q present=%v err=%v", got, present, err) + t.Fatalf("value=%q present=%v error=%v; want a present empty value", got, present, err) + } +} + +func TestResolveBodyValue_SelectorMatching(t *testing.T) { + rec := &Record{Body: "# Root\n\n## Parent A\n\n### Notes\n\nfirst\n\n## Parent B\n\n### Notes\n\nsecond\n\n#### Detail\n\ndetail\n\n# Other Root\n\n## Parent B\n\n### Notes\n\nthird\n"} + tests := []struct { + name string + directive string + want string + present bool + ambiguous bool + }{ + {name: "simple selector is ambiguous", directive: `body.section["### Notes"]`, ambiguous: true}, + {name: "qualified selector distinguishes parents", directive: `body.section["## Parent A"]["### Notes"]`, want: "first", present: true}, + {name: "partial suffix has one match", directive: `body.section["### Notes"]["#### Detail"]`, want: "detail", present: true}, + {name: "partial suffix has no match", directive: `body.section["## Missing"]["### Notes"]`}, + {name: "partial suffix has multiple matches", directive: `body.section["## Parent B"]["### Notes"]`, ambiguous: true}, + {name: "full suffix has one match", directive: `body.section["# Root"]["## Parent B"]["### Notes"]`, want: "second", present: true}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, present, err := ResolveBodyValue(rec, tt.directive) + if tt.ambiguous { + if err == nil || !strings.Contains(err.Error(), "ambiguous") { + t.Fatalf("error = %v; want ambiguity", err) + } + return + } + if err != nil || present != tt.present || got != tt.want { + t.Fatalf("value=%q present=%v error=%v; want value=%q present=%v", got, present, err, tt.want, tt.present) + } + }) + } +} + +func TestResolveBodyValue_SelectorCannotOmitDirectParent(t *testing.T) { + rec := &Record{Body: "# Root\n\n## Parent\n\n### Intermediate\n\n#### Notes\n\nvalue\n"} + _, present, err := ResolveBodyValue(rec, `body.section["## Parent"]["#### Notes"]`) + if err != nil || present { + t.Fatalf("present=%v error=%v; want no match", present, err) + } + got, present, err := ResolveBodyValue(rec, `body.section["### Intermediate"]["#### Notes"]`) + if err != nil || !present || got != "value" { + t.Fatalf("value=%q present=%v error=%v; want the direct-parent match", got, present, err) } } -func TestResolveBodyValue_DuplicateSectionIsAmbiguous(t *testing.T) { - rec := &Record{BodySections: []Section{ - {Heading: "Notes", Level: 2, Content: "first"}, - {Heading: "Notes", Level: 2, Content: "second"}, - }} - _, _, err := ResolveBodyValue(rec, `body.section["## Notes"]`) +func TestResolveBodyValue_DuplicateFullPathIsAmbiguous(t *testing.T) { + rec := &Record{Body: "# Root\n\n## Parent\n\n### Notes\n\nfirst\n\n### Notes\n\nsecond\n"} + _, _, err := ResolveBodyValue(rec, `body.section["# Root"]["## Parent"]["### Notes"]`) if err == nil || !strings.Contains(err.Error(), "ambiguous") { - t.Fatalf("expected ambiguity error, got %v", err) + t.Fatalf("error = %v; want ambiguity", err) } } -func TestMarkdownExtractor_PreservesDuplicateSectionsAndFencedCodeExclusion(t *testing.T) { - content := []byte("---\ntitle: Fixture\n---\nTitle\n=====\n\n```\n## Fake\n```\n\nEmpty\n-----\n\n## Duplicate\n\nfirst\n\n## Duplicate\n\nsecond\n") +func TestMarkdownExtractor_PreservesPathsAndFencedCodeExclusion(t *testing.T) { + content := []byte("---\ntitle: Fixture\n---\nRoot\n====\n\n```\n## Fake\n```\n\n~~~~\n## Fake Two\n~~~~ not a closing fence\n## Fake Three\n~~~~\n\n## Empty\n\n## Parent *A*\n\n### Duplicate\n\nfirst\n\n## Parent B\n\n### Duplicate\n\nsecond\n") parseAST := true extractors := map[string]*MarkdownExtractor{ "ast": {ParseAST: &parseAST}, @@ -68,32 +163,51 @@ func TestMarkdownExtractor_PreservesDuplicateSectionsAndFencedCodeExclusion(t *t for name, ext := range extractors { rec, err := ext.Extract(name+".md", content) if err != nil { - t.Fatalf("%s Extract: %v", name, err) + t.Fatalf("%s extraction failed: %v", name, err) } - if got := len(rec.BodySections); got != 4 { - t.Fatalf("%s BodySections len = %d, want 4: %+v", name, got, rec.BodySections) + if got := len(rec.BodySections); got != 6 { + t.Fatalf("%s section count = %d; want 6: %+v", name, got, rec.BodySections) } - if rec.BodySections[1].Heading != "Empty" || rec.BodySections[1].Level != 2 || rec.BodySections[1].Content != "" { + if rec.BodySections[1].Heading != "Empty" || rec.BodySections[1].Content != "" { t.Fatalf("%s empty section = %+v", name, rec.BodySections[1]) } - for _, sec := range rec.BodySections { - if sec.Heading == "Fake" { - t.Fatalf("%s included fenced code heading: %+v", name, rec.BodySections) - } - } - if got, present, err := ResolveBodyValue(rec, "body.h1"); err != nil || !present || got != "Title" { - t.Fatalf("%s h1 resolution value=%q present=%v err=%v", name, got, present, err) + wantLastPath := SectionPath{ + {Level: 1, Text: "Root"}, + {Level: 2, Text: "Parent B"}, + {Level: 3, Text: "Duplicate"}, } - if _, present, err := ResolveBodyValue(rec, `body.section["## Empty"]`); err != nil || !present { - t.Fatalf("%s empty resolution present=%v err=%v", name, present, err) + if !reflect.DeepEqual(rec.BodySections[5].Path, wantLastPath) { + t.Fatalf("%s final path = %+v; want %+v", name, rec.BodySections[5].Path, wantLastPath) } - if _, _, err := ResolveBodyValue(rec, `body.section["## Duplicate"]`); err == nil || !strings.Contains(err.Error(), "ambiguous") { - t.Fatalf("%s expected duplicate ambiguity, got %v", name, err) + for _, sec := range rec.BodySections { + if strings.HasPrefix(sec.Heading, "Fake") { + t.Fatalf("%s included a fenced heading: %+v", name, rec.BodySections) + } } if baseline == nil { baseline = rec.BodySections } else if !reflect.DeepEqual(baseline, rec.BodySections) { - t.Fatalf("text-backed sections differ from AST: AST=%+v got=%+v", baseline, rec.BodySections) + t.Fatalf("text sections differ from AST sections: AST=%+v text=%+v", baseline, rec.BodySections) + } + } +} + +func TestSectionPathsCloseBranches(t *testing.T) { + sections := ExtractSectionsFromText("# First\n\n### Jump\n\n#### Leaf\n\n## Sibling\n\n# Second\n\n### Other\n") + want := []SectionPath{ + {{Level: 1, Text: "First"}}, + {{Level: 1, Text: "First"}, {Level: 3, Text: "Jump"}}, + {{Level: 1, Text: "First"}, {Level: 3, Text: "Jump"}, {Level: 4, Text: "Leaf"}}, + {{Level: 1, Text: "First"}, {Level: 2, Text: "Sibling"}}, + {{Level: 1, Text: "Second"}}, + {{Level: 1, Text: "Second"}, {Level: 3, Text: "Other"}}, + } + if len(sections) != len(want) { + t.Fatalf("section count = %d; want %d", len(sections), len(want)) + } + for i := range sections { + if !reflect.DeepEqual(sections[i].Path, want[i]) { + t.Fatalf("path %d = %+v; want %+v", i, sections[i].Path, want[i]) } } } diff --git a/internal/rules/field_source_test.go b/internal/rules/field_source_test.go index 1575c70f..89e1cbf7 100644 --- a/internal/rules/field_source_test.go +++ b/internal/rules/field_source_test.go @@ -11,15 +11,12 @@ func TestResolveFieldValue_FrontmatterPresenceOverridesBodySource(t *testing.T) for _, value := range []any{"", nil} { rec := &extract.Record{ Frontmatter: map[string]any{"notes": value}, - BodySections: []extract.Section{ - {Heading: "Notes", Level: 2, Content: "first"}, - {Heading: "Notes", Level: 2, Content: "second"}, - }, + Body: "# Root\n\n## Parent\n\n### Notes\n\nfirst\n\n### Notes\n\nsecond\n", } - got, present, err := ResolveFieldValue(rec, "notes", SchemaField{Extract: `body.section["## Notes"]`}) + got, present, err := ResolveFieldValue(rec, "notes", SchemaField{Extract: `body.section["# Root"]["## Parent"]["### Notes"]`}) if err != nil || !present || got != value { - t.Fatalf("got value=%#v present=%v err=%v; want frontmatter value %#v", got, present, err, value) + t.Fatalf("value=%#v present=%v error=%v; want frontmatter value %#v", got, present, err, value) } } } From f40d7b53b7c945b108148d2060c21d3d56d67231 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 01:45:08 -0600 Subject: [PATCH 02/18] fix(extract): preserve hierarchical section semantics --- cmd/rootline/schema.go | 2 +- cmd/rootline/schema_test.go | 5 +++ internal/extract/body.go | 49 ++++++++++++++++++++++++++++ internal/extract/body_test.go | 22 +++++++++++++ internal/infer/apply.go | 2 +- internal/infer/apply_section_test.go | 25 ++++++++++++++ internal/infer/delta.go | 10 +++++- internal/infer/delta_test.go | 12 +++++++ 8 files changed, 124 insertions(+), 3 deletions(-) diff --git a/cmd/rootline/schema.go b/cmd/rootline/schema.go index 9a929fd3..f08d2b24 100644 --- a/cmd/rootline/schema.go +++ b/cmd/rootline/schema.go @@ -1034,7 +1034,7 @@ func schemaToInferences(stem *rules.StemFile) []infer.Inference { for fieldName, sf := range stem.Schema { if sf.Extract != "" && sf.Type == "string" { if source, err := extract.ParseBodySource(sf.Extract); err == nil && source.Kind == extract.BodySourceSection { - if canonical, err := extract.CanonicalSectionSource(source.Heading); err == nil && canonical == sf.Extract { + if canonical, err := extract.CanonicalSectionSelectorSource(source.Selector); err == nil && canonical == sf.Extract { infType := "optional_section" if sf.Required { infType = "required_section" diff --git a/cmd/rootline/schema_test.go b/cmd/rootline/schema_test.go index 04a6ba8b..aa2c942c 100644 --- a/cmd/rootline/schema_test.go +++ b/cmd/rootline/schema_test.go @@ -1622,9 +1622,11 @@ func TestSchemaApplyWritesProposedPatchVerbatim(t *testing.T) { func TestSchemaProposeIncrementalSectionConversionPreservesSource(t *testing.T) { source := `body.section["## Notes"]` + hierarchicalSource := `body.section["# Root"]["## Parent"]["### Detail"]` got := schemaToInferences(&rules.StemFile{Schema: map[string]rules.SchemaField{ "notes": {Type: "string", Required: true, Extract: source}, "summary": {Type: "string", Extract: `body.section["## Summary"]`}, + "detail": {Type: "string", Extract: hierarchicalSource}, }}) seen := map[string]infer.Inference{} @@ -1637,6 +1639,9 @@ func TestSchemaProposeIncrementalSectionConversionPreservesSource(t *testing.T) if inf := seen["summary"]; inf.Type != "optional_section" || inf.SourceDirective != `body.section["## Summary"]` { t.Fatalf("optional section conversion dropped source: %+v (all %+v)", inf, got) } + if inf := seen["detail"]; inf.Type != "optional_section" || inf.SourceDirective != hierarchicalSource { + t.Fatalf("hierarchical section conversion dropped source: %+v (all %+v)", inf, got) + } } func TestSchemaProposeIncrementalSectionConversionFallsThroughNoncanonicalSources(t *testing.T) { diff --git a/internal/extract/body.go b/internal/extract/body.go index 9611973a..9f1d5747 100644 --- a/internal/extract/body.go +++ b/internal/extract/body.go @@ -341,6 +341,17 @@ func parseSetextUnderline(line string) (int, bool) { func parseFenceLine(line string) (byte, int, string, bool) { line = strings.TrimRight(line, "\r") + if char, count, trailing, ok := parseBareFenceLine(line); ok { + return char, count, trailing, true + } + containerContent, ok := fenceContainerContent(line) + if !ok { + return 0, 0, "", false + } + return parseBareFenceLine(containerContent) +} + +func parseBareFenceLine(line string) (byte, int, string, bool) { indent := len(line) - len(strings.TrimLeft(line, " ")) if indent > 3 || indent >= len(line) || (line[indent] != '`' && line[indent] != '~') { return 0, 0, "", false @@ -357,6 +368,44 @@ func parseFenceLine(line string) (byte, int, string, bool) { return char, count, trailing, true } +func fenceContainerContent(line string) (string, bool) { + indent := len(line) - len(strings.TrimLeft(line, " ")) + if indent > 3 { + return "", false + } + rest := line[indent:] + for len(rest) > 0 && rest[0] == '>' { + rest = rest[1:] + if len(rest) > 0 && (rest[0] == ' ' || rest[0] == '\t') { + rest = rest[1:] + } + } + + markerEnd := 0 + if len(rest) > 1 && (rest[0] == '-' || rest[0] == '+' || rest[0] == '*') { + markerEnd = 1 + } else { + for markerEnd < len(rest) && markerEnd < 9 && rest[markerEnd] >= '0' && rest[markerEnd] <= '9' { + markerEnd++ + } + if markerEnd == 0 || markerEnd >= len(rest) || (rest[markerEnd] != '.' && rest[markerEnd] != ')') { + markerEnd = 0 + } else { + markerEnd++ + } + } + if markerEnd == 0 || markerEnd >= len(rest) || (rest[markerEnd] != ' ' && rest[markerEnd] != '\t') { + if rest != line[indent:] { + return rest, true + } + return "", false + } + for markerEnd < len(rest) && (rest[markerEnd] == ' ' || rest[markerEnd] == '\t') { + markerEnd++ + } + return rest[markerEnd:], true +} + // lineOffset returns the byte offset of the start of a 1-based line number. func lineOffset(source []byte, line int) int { current := 1 diff --git a/internal/extract/body_test.go b/internal/extract/body_test.go index 36abe22a..65372637 100644 --- a/internal/extract/body_test.go +++ b/internal/extract/body_test.go @@ -1,6 +1,7 @@ package extract import ( + "reflect" "testing" "github.com/yuin/goldmark" @@ -69,6 +70,27 @@ func TestExtractSections_HeadingInCodeBlock(t *testing.T) { } } +func TestExtractSectionsFromText_ContainerFenceParity(t *testing.T) { + for _, tt := range []struct { + name string + body string + }{ + {name: "backticks in list", body: "- ```\n ## Fake\n ```\n\n## Real\n\nContent\n"}, + {name: "tildes in list", body: "- ~~~\n ## Fake\n ~~~\n\n## Real\n\nContent\n"}, + } { + t.Run(tt.name, func(t *testing.T) { + astSections := parseSections(tt.body) + textSections := ExtractSectionsFromText(tt.body) + if !reflect.DeepEqual(textSections, astSections) { + t.Fatalf("text sections differ from AST sections: text=%+v AST=%+v", textSections, astSections) + } + if len(textSections) != 1 || textSections[0].Heading != "Real" { + t.Fatalf("sections = %+v; want only the real heading after the fence", textSections) + } + }) + } +} + func TestExtractSections_NoHeadings(t *testing.T) { body := "Just a paragraph.\n\nAnother paragraph.\n" sections := parseSections(body) diff --git a/internal/infer/apply.go b/internal/infer/apply.go index 1c1ef804..bee37c0c 100644 --- a/internal/infer/apply.go +++ b/internal/infer/apply.go @@ -356,7 +356,7 @@ func canonicalSectionDirective(directive string) (string, error) { if source.Kind != extract.BodySourceSection { return "", fmt.Errorf("source_directive must be a body section") } - return extract.CanonicalSectionSource(source.Heading) + return extract.CanonicalSectionSelectorSource(source.Selector) } func existingFieldMatchesSectionIntent(sf rules.SchemaField, source string) bool { diff --git a/internal/infer/apply_section_test.go b/internal/infer/apply_section_test.go index 9276adf6..078d85ed 100644 --- a/internal/infer/apply_section_test.go +++ b/internal/infer/apply_section_test.go @@ -112,3 +112,28 @@ func TestApplySchemaInferences_CanonicalizesParseableSectionSource(t *testing.T) t.Fatalf("source = %q, want canonical %q", got, want) } } + +func TestApplySchemaInferences_PreservesHierarchicalSectionSource(t *testing.T) { + stemPath := writeApplyStem(t, "version: 2\nschema: {}\n") + input := "body.section[`# Root`][`## Parent`][`### Notes`]" + want := `body.section["# Root"]["## Parent"]["### Notes"]` + _, err := ApplySchemaInferences(stemPath, []ReportInference{{Type: "required_section", Field: "notes", SourceDirective: input}}, false) + if err != nil { + t.Fatalf("apply hierarchical source: %v", err) + } + if got := readApplyStem(t, stemPath).Schema["notes"].Extract; got != want { + t.Fatalf("source = %q, want %q", got, want) + } +} + +func TestApplySchemaInferences_DistinguishesHierarchicalSectionParents(t *testing.T) { + stemPath := writeApplyStem(t, "version: 2\nschema:\n notes:\n type: string\n source: 'body.section[\"## Parent A\"][\"### Notes\"]'\n") + before := string(mustReadApplyFile(t, stemPath)) + _, err := ApplySchemaInferences(stemPath, []ReportInference{{Type: "required_section", Field: "notes", SourceDirective: `body.section["## Parent B"]["### Notes"]`}}, false) + if err == nil { + t.Fatal("expected different parents to conflict") + } + if after := string(mustReadApplyFile(t, stemPath)); after != before { + t.Fatalf("conflict mutated stem:\nbefore:\n%s\nafter:\n%s", before, after) + } +} diff --git a/internal/infer/delta.go b/internal/infer/delta.go index e6220e58..e7ea2cf6 100644 --- a/internal/infer/delta.go +++ b/internal/infer/delta.go @@ -105,7 +105,15 @@ func isCovered(inf Inference, stem *rules.StemFile) bool { } func sectionInferenceCovered(inf Inference, sf rules.SchemaField) bool { - if sf.Type != "string" || inf.SourceDirective == "" || sf.Extract != inf.SourceDirective { + if sf.Type != "string" || inf.SourceDirective == "" || sf.Extract == "" { + return false + } + inferredSource, err := canonicalSectionDirective(inf.SourceDirective) + if err != nil { + return false + } + existingSource, err := canonicalSectionDirective(sf.Extract) + if err != nil || existingSource != inferredSource { return false } if inf.Type == "required_section" { diff --git a/internal/infer/delta_test.go b/internal/infer/delta_test.go index 2ef9dd03..79bd8c35 100644 --- a/internal/infer/delta_test.go +++ b/internal/infer/delta_test.go @@ -217,6 +217,18 @@ func TestIsCovered_SectionSourceAndRequiredness(t *testing.T) { field: rules.SchemaField{Type: "string", Extract: `body.section["## Notes"]`}, want: false, }, + { + name: "same hierarchical source is covered", + inf: Inference{Type: "optional_section", Field: "notes", SourceDirective: `body.section["## Parent"]["### Notes"]`}, + field: rules.SchemaField{Type: "string", Extract: `body.section["## Parent"]["### Notes"]`}, + want: true, + }, + { + name: "different hierarchical parent is not covered", + inf: Inference{Type: "optional_section", Field: "notes", SourceDirective: `body.section["## Parent B"]["### Notes"]`}, + field: rules.SchemaField{Type: "string", Extract: `body.section["## Parent A"]["### Notes"]`}, + want: false, + }, { name: "missing existing source is not covered", inf: Inference{Type: "optional_section", Field: "notes", SourceDirective: `body.section["## Notes"]`}, From 3c6dea30545d622b733ca75e8c0259d97c94c2b4 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 01:55:24 -0600 Subject: [PATCH 03/18] fix(extract): align markdown text extraction --- internal/extract/body.go | 185 ++++++++++++++++++++++++---------- internal/extract/body_test.go | 36 ++++++- 2 files changed, 163 insertions(+), 58 deletions(-) diff --git a/internal/extract/body.go b/internal/extract/body.go index 9f1d5747..654ce5a8 100644 --- a/internal/extract/body.go +++ b/internal/extract/body.go @@ -50,13 +50,28 @@ func ExtractSections(node ast.Node, source []byte) []Section { lines := h.Lines() startLine := 0 endOffset := 0 + headingText := "" if lines.Len() > 0 { - seg := lines.At(0) - startLine = lineFromOffset(source, seg.Start) + firstSeg := lines.At(0) + startLine = lineFromOffset(source, firstSeg.Start) lastSeg := lines.At(lines.Len() - 1) endOffset = lastSeg.Stop - if _, _, ok := parseATXHeading(string(source[seg.Start:lastSeg.Stop])); !ok { - underlineStart, underlineEnd := lineOffset(source, startLine+1), lineOffset(source, startLine+2) + + lineStart := lineOffset(source, startLine) + lineEnd := lineOffset(source, startLine+1) + if _, text, ok := parseATXHeading(string(source[lineStart:lineEnd])); ok { + headingText = text + } else { + var text strings.Builder + for i := 0; i < lines.Len(); i++ { + segment := lines.At(i) + text.Write(segment.Value(source)) + } + headingText = strings.TrimSpace(text.String()) + + lastLine := lineFromOffset(source, lastSeg.Start) + underlineStart := lineOffset(source, lastLine+1) + underlineEnd := lineOffset(source, lastLine+2) if _, ok := parseSetextUnderline(string(source[underlineStart:underlineEnd])); ok { endOffset = underlineEnd } @@ -64,7 +79,7 @@ func ExtractSections(node ast.Node, source []byte) []Section { } headings = append(headings, headingInfo{ - text: headingTextAtLine(source, startLine), + text: headingText, level: h.Level, startLine: startLine, endOffset: endOffset, @@ -107,19 +122,6 @@ func ExtractSections(node ast.Node, source []byte) []Section { return sectionsWithPaths(sections) } -func headingTextAtLine(source []byte, line int) string { - start := lineOffset(source, line) - end := start - for end < len(source) && source[end] != '\n' { - end++ - } - raw := string(source[start:end]) - if _, text, ok := parseATXHeading(raw); ok { - return text - } - return strings.TrimSpace(strings.TrimRight(raw, "\r")) -} - // CodeBlock represents a fenced code block in a markdown body. type CodeBlock struct { Language string `json:"language"` @@ -224,8 +226,8 @@ func ExtractSectionsFromText(body string) []Section { end int } var headings []heading - prevLine, prevLineNo, prevStart, prevOK := "", 0, 0, false - inFence, fenceChar, fenceLen := false, byte(0), 0 + candidateLine, candidateStart, candidateOK := 0, 0, false + var fence *fenceContext for start, lineNo := 0, 1; start <= len(source); lineNo++ { end := start for end < len(source) && source[end] != '\n' { @@ -236,25 +238,39 @@ func ExtractSectionsFromText(body string) []Section { lineEnd++ } line := string(source[start:end]) - if char, length, trailing, ok := parseFenceLine(line); ok { - prevOK = false - if !inFence { - inFence, fenceChar, fenceLen = true, char, length - } else if char == fenceChar && length >= fenceLen && strings.TrimSpace(trailing) == "" { - inFence = false - } - } else if !inFence { - if level, text, ok := parseATXHeading(line); ok { - headings = append(headings, heading{text: text, level: level, line: lineNo, start: start, end: lineEnd}) - prevOK = false - } else if level, ok := parseSetextUnderline(line); ok && prevOK { - headings = append(headings, heading{text: strings.TrimSpace(strings.TrimRight(prevLine, "\r")), level: level, line: prevLineNo, start: prevStart, end: lineEnd}) - prevOK = false - } else if _, ok := parseSetextUnderline(line); ok { - prevOK = false - } else { - prevLine, prevLineNo, prevStart, prevOK = line, lineNo, start, strings.TrimSpace(line) != "" + + if fence != nil { + content, belongs := fence.content(line) + if belongs { + if char, length, trailing, ok := parseBareFenceLine(content); ok && char == fence.char && length >= fence.length && strings.TrimSpace(trailing) == "" { + fence = nil + } + candidateOK = false + if end >= len(source) { + break + } + start = lineEnd + continue } + fence = nil + } + + if opened, ok := parseFenceOpening(line); ok { + fence = &opened + candidateOK = false + } else if level, text, ok := parseATXHeading(line); ok { + headings = append(headings, heading{text: text, level: level, line: lineNo, start: start, end: lineEnd}) + candidateOK = false + } else if level, ok := parseSetextUnderline(line); ok && candidateOK { + headingText := strings.TrimSpace(string(source[candidateStart:start])) + headings = append(headings, heading{text: headingText, level: level, line: candidateLine, start: candidateStart, end: lineEnd}) + candidateOK = false + } else if _, ok := parseSetextUnderline(line); ok { + candidateOK = false + } else if strings.TrimSpace(line) == "" { + candidateOK = false + } else if !candidateOK { + candidateLine, candidateStart, candidateOK = lineNo, start, true } if end >= len(source) { break @@ -339,16 +355,28 @@ func parseSetextUnderline(line string) (int, bool) { return 2, true } -func parseFenceLine(line string) (byte, int, string, bool) { +type fenceContext struct { + char byte + length int + quoteDepth int + listIndent int +} + +func parseFenceOpening(line string) (fenceContext, bool) { line = strings.TrimRight(line, "\r") - if char, count, trailing, ok := parseBareFenceLine(line); ok { - return char, count, trailing, true + if char, length, _, ok := parseBareFenceLine(line); ok { + return fenceContext{char: char, length: length}, true } - containerContent, ok := fenceContainerContent(line) + + content, quoteDepth, listIndent, ok := fenceContainerContent(line) if !ok { - return 0, 0, "", false + return fenceContext{}, false } - return parseBareFenceLine(containerContent) + char, length, _, ok := parseBareFenceLine(content) + if !ok { + return fenceContext{}, false + } + return fenceContext{char: char, length: length, quoteDepth: quoteDepth, listIndent: listIndent}, true } func parseBareFenceLine(line string) (byte, int, string, bool) { @@ -368,42 +396,87 @@ func parseBareFenceLine(line string) (byte, int, string, bool) { return char, count, trailing, true } -func fenceContainerContent(line string) (string, bool) { +func fenceContainerContent(line string) (content string, quoteDepth, listIndent int, ok bool) { indent := len(line) - len(strings.TrimLeft(line, " ")) if indent > 3 { - return "", false + return "", 0, 0, false } rest := line[indent:] for len(rest) > 0 && rest[0] == '>' { + quoteDepth++ rest = rest[1:] if len(rest) > 0 && (rest[0] == ' ' || rest[0] == '\t') { rest = rest[1:] } } + listStart := len(rest) - len(strings.TrimLeft(rest, " ")) + listRest := rest[listStart:] markerEnd := 0 - if len(rest) > 1 && (rest[0] == '-' || rest[0] == '+' || rest[0] == '*') { + if len(listRest) > 1 && (listRest[0] == '-' || listRest[0] == '+' || listRest[0] == '*') { markerEnd = 1 } else { - for markerEnd < len(rest) && markerEnd < 9 && rest[markerEnd] >= '0' && rest[markerEnd] <= '9' { + for markerEnd < len(listRest) && markerEnd < 9 && listRest[markerEnd] >= '0' && listRest[markerEnd] <= '9' { markerEnd++ } - if markerEnd == 0 || markerEnd >= len(rest) || (rest[markerEnd] != '.' && rest[markerEnd] != ')') { + if markerEnd == 0 || markerEnd >= len(listRest) || (listRest[markerEnd] != '.' && listRest[markerEnd] != ')') { markerEnd = 0 } else { markerEnd++ } } - if markerEnd == 0 || markerEnd >= len(rest) || (rest[markerEnd] != ' ' && rest[markerEnd] != '\t') { - if rest != line[indent:] { - return rest, true + if markerEnd == 0 || markerEnd >= len(listRest) || (listRest[markerEnd] != ' ' && listRest[markerEnd] != '\t') { + if quoteDepth > 0 { + return rest, quoteDepth, 0, true } - return "", false + return "", 0, 0, false + } + contentStart := markerEnd + for contentStart < len(listRest) && (listRest[contentStart] == ' ' || listRest[contentStart] == '\t') { + contentStart++ + } + listIndent = listStart + contentStart + if quoteDepth == 0 { + listIndent += indent + } + return listRest[contentStart:], quoteDepth, listIndent, true +} + +func (f fenceContext) content(line string) (string, bool) { + line = strings.TrimRight(line, "\r") + if strings.TrimSpace(line) == "" { + return line, true + } + if f.quoteDepth == 0 && f.listIndent == 0 { + return line, true } - for markerEnd < len(rest) && (rest[markerEnd] == ' ' || rest[markerEnd] == '\t') { - markerEnd++ + + rest := line + if f.quoteDepth > 0 { + indent := len(line) - len(strings.TrimLeft(line, " ")) + if indent > 3 { + return "", false + } + rest = line[indent:] + for i := 0; i < f.quoteDepth; i++ { + if len(rest) == 0 || rest[0] != '>' { + return "", false + } + rest = rest[1:] + if len(rest) > 0 && (rest[0] == ' ' || rest[0] == '\t') { + rest = rest[1:] + } + } + } + if f.listIndent == 0 { + return rest, true + } + + contentIndent := len(rest) - len(strings.TrimLeft(rest, " ")) + if contentIndent < f.listIndent { + return "", false } - return rest[markerEnd:], true + return rest[f.listIndent:], true } // lineOffset returns the byte offset of the start of a 1-based line number. diff --git a/internal/extract/body_test.go b/internal/extract/body_test.go index 65372637..0e04ca7e 100644 --- a/internal/extract/body_test.go +++ b/internal/extract/body_test.go @@ -75,8 +75,12 @@ func TestExtractSectionsFromText_ContainerFenceParity(t *testing.T) { name string body string }{ - {name: "backticks in list", body: "- ```\n ## Fake\n ```\n\n## Real\n\nContent\n"}, - {name: "tildes in list", body: "- ~~~\n ## Fake\n ~~~\n\n## Real\n\nContent\n"}, + {name: "closed backtick fence in list", body: "- ```\n ## Fake\n ```\n\n## Real\n\nContent\n"}, + {name: "closed tilde fence in list", body: "- ~~~\n ## Fake\n ~~~\n\n## Real\n\nContent\n"}, + {name: "unclosed backtick fence in list", body: "- ```\n ## Fake\n\n## Real\n\nContent\n"}, + {name: "unclosed tilde fence in list", body: "- ~~~\n ## Fake\n\n## Real\n\nContent\n"}, + {name: "unclosed backtick fence in quote", body: "> ```\n> ## Fake\n\n## Real\n\nContent\n"}, + {name: "unclosed tilde fence in quote", body: "> ~~~\n> ## Fake\n\n## Real\n\nContent\n"}, } { t.Run(tt.name, func(t *testing.T) { astSections := parseSections(tt.body) @@ -91,6 +95,34 @@ func TestExtractSectionsFromText_ContainerFenceParity(t *testing.T) { } } +func TestExtractSectionsFromText_SetextParity(t *testing.T) { + for _, tt := range []struct { + name string + body string + wantHeading string + wantLevel int + }{ + {name: "simple", body: "First\n===\n\nContent\n", wantHeading: "First", wantLevel: 1}, + {name: "multiple lines", body: "First\nSecond\n---\n\nContent\n", wantHeading: "First\nSecond", wantLevel: 2}, + } { + t.Run(tt.name, func(t *testing.T) { + astSections := parseSections(tt.body) + textSections := ExtractSectionsFromText(tt.body) + if !reflect.DeepEqual(textSections, astSections) { + t.Fatalf("text sections differ from AST sections: text=%+v AST=%+v", textSections, astSections) + } + if len(textSections) != 1 { + t.Fatalf("section count = %d; want 1", len(textSections)) + } + wantPath := SectionPath{{Level: tt.wantLevel, Text: tt.wantHeading}} + want := Section{Heading: tt.wantHeading, Level: tt.wantLevel, Path: wantPath, Content: "Content", StartLine: 1} + if !reflect.DeepEqual(textSections[0], want) { + t.Fatalf("section = %+v; want %+v", textSections[0], want) + } + }) + } +} + func TestExtractSections_NoHeadings(t *testing.T) { body := "Just a paragraph.\n\nAnother paragraph.\n" sections := parseSections(body) From bd03ebf4b4e7289d887f44975696ae7a3f1fbfd5 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 02:03:02 -0600 Subject: [PATCH 04/18] fix(extract): unify markdown section parsing --- internal/extract/body.go | 202 +--------------------------------- internal/extract/body_test.go | 23 ++++ 2 files changed, 27 insertions(+), 198 deletions(-) diff --git a/internal/extract/body.go b/internal/extract/body.go index 654ce5a8..51c782ce 100644 --- a/internal/extract/body.go +++ b/internal/extract/body.go @@ -3,8 +3,10 @@ package extract import ( "strings" + "github.com/yuin/goldmark" "github.com/yuin/goldmark/ast" east "github.com/yuin/goldmark/extension/ast" + "github.com/yuin/goldmark/text" ) // HeadingKey identifies one markdown heading. @@ -219,80 +221,8 @@ func ExtractTables(node ast.Node, source []byte) []Table { // ExtractSectionsFromText splits markdown text into heading-delimited sections. func ExtractSectionsFromText(body string) []Section { source := []byte(body) - type heading struct { - text string - level int - line, start int - end int - } - var headings []heading - candidateLine, candidateStart, candidateOK := 0, 0, false - var fence *fenceContext - for start, lineNo := 0, 1; start <= len(source); lineNo++ { - end := start - for end < len(source) && source[end] != '\n' { - end++ - } - lineEnd := end - if end < len(source) { - lineEnd++ - } - line := string(source[start:end]) - - if fence != nil { - content, belongs := fence.content(line) - if belongs { - if char, length, trailing, ok := parseBareFenceLine(content); ok && char == fence.char && length >= fence.length && strings.TrimSpace(trailing) == "" { - fence = nil - } - candidateOK = false - if end >= len(source) { - break - } - start = lineEnd - continue - } - fence = nil - } - - if opened, ok := parseFenceOpening(line); ok { - fence = &opened - candidateOK = false - } else if level, text, ok := parseATXHeading(line); ok { - headings = append(headings, heading{text: text, level: level, line: lineNo, start: start, end: lineEnd}) - candidateOK = false - } else if level, ok := parseSetextUnderline(line); ok && candidateOK { - headingText := strings.TrimSpace(string(source[candidateStart:start])) - headings = append(headings, heading{text: headingText, level: level, line: candidateLine, start: candidateStart, end: lineEnd}) - candidateOK = false - } else if _, ok := parseSetextUnderline(line); ok { - candidateOK = false - } else if strings.TrimSpace(line) == "" { - candidateOK = false - } else if !candidateOK { - candidateLine, candidateStart, candidateOK = lineNo, start, true - } - if end >= len(source) { - break - } - start = lineEnd - } - if len(headings) == 0 { - return []Section{{Heading: "", Level: 0, Content: body, StartLine: 1}} - } - sections := make([]Section, 0, len(headings)) - for i, h := range headings { - contentEnd := len(source) - if i+1 < len(headings) { - contentEnd = headings[i+1].start - } - content := "" - if h.end < contentEnd { - content = strings.TrimSpace(string(source[h.end:contentEnd])) - } - sections = append(sections, Section{Heading: h.text, Level: h.level, Content: content, StartLine: h.line}) - } - return sectionsWithPaths(sections) + node := goldmark.DefaultParser().Parse(text.NewReader(source)) + return ExtractSections(node, source) } func sectionsWithPaths(sections []Section) []Section { @@ -355,130 +285,6 @@ func parseSetextUnderline(line string) (int, bool) { return 2, true } -type fenceContext struct { - char byte - length int - quoteDepth int - listIndent int -} - -func parseFenceOpening(line string) (fenceContext, bool) { - line = strings.TrimRight(line, "\r") - if char, length, _, ok := parseBareFenceLine(line); ok { - return fenceContext{char: char, length: length}, true - } - - content, quoteDepth, listIndent, ok := fenceContainerContent(line) - if !ok { - return fenceContext{}, false - } - char, length, _, ok := parseBareFenceLine(content) - if !ok { - return fenceContext{}, false - } - return fenceContext{char: char, length: length, quoteDepth: quoteDepth, listIndent: listIndent}, true -} - -func parseBareFenceLine(line string) (byte, int, string, bool) { - indent := len(line) - len(strings.TrimLeft(line, " ")) - if indent > 3 || indent >= len(line) || (line[indent] != '`' && line[indent] != '~') { - return 0, 0, "", false - } - char, count := line[indent], 0 - i := indent - for ; i < len(line) && line[i] == char; i++ { - count++ - } - trailing := line[i:] - if count < 3 || (char == '`' && strings.Contains(trailing, "`")) { - return 0, 0, "", false - } - return char, count, trailing, true -} - -func fenceContainerContent(line string) (content string, quoteDepth, listIndent int, ok bool) { - indent := len(line) - len(strings.TrimLeft(line, " ")) - if indent > 3 { - return "", 0, 0, false - } - rest := line[indent:] - for len(rest) > 0 && rest[0] == '>' { - quoteDepth++ - rest = rest[1:] - if len(rest) > 0 && (rest[0] == ' ' || rest[0] == '\t') { - rest = rest[1:] - } - } - - listStart := len(rest) - len(strings.TrimLeft(rest, " ")) - listRest := rest[listStart:] - markerEnd := 0 - if len(listRest) > 1 && (listRest[0] == '-' || listRest[0] == '+' || listRest[0] == '*') { - markerEnd = 1 - } else { - for markerEnd < len(listRest) && markerEnd < 9 && listRest[markerEnd] >= '0' && listRest[markerEnd] <= '9' { - markerEnd++ - } - if markerEnd == 0 || markerEnd >= len(listRest) || (listRest[markerEnd] != '.' && listRest[markerEnd] != ')') { - markerEnd = 0 - } else { - markerEnd++ - } - } - if markerEnd == 0 || markerEnd >= len(listRest) || (listRest[markerEnd] != ' ' && listRest[markerEnd] != '\t') { - if quoteDepth > 0 { - return rest, quoteDepth, 0, true - } - return "", 0, 0, false - } - contentStart := markerEnd - for contentStart < len(listRest) && (listRest[contentStart] == ' ' || listRest[contentStart] == '\t') { - contentStart++ - } - listIndent = listStart + contentStart - if quoteDepth == 0 { - listIndent += indent - } - return listRest[contentStart:], quoteDepth, listIndent, true -} - -func (f fenceContext) content(line string) (string, bool) { - line = strings.TrimRight(line, "\r") - if strings.TrimSpace(line) == "" { - return line, true - } - if f.quoteDepth == 0 && f.listIndent == 0 { - return line, true - } - - rest := line - if f.quoteDepth > 0 { - indent := len(line) - len(strings.TrimLeft(line, " ")) - if indent > 3 { - return "", false - } - rest = line[indent:] - for i := 0; i < f.quoteDepth; i++ { - if len(rest) == 0 || rest[0] != '>' { - return "", false - } - rest = rest[1:] - if len(rest) > 0 && (rest[0] == ' ' || rest[0] == '\t') { - rest = rest[1:] - } - } - } - if f.listIndent == 0 { - return rest, true - } - - contentIndent := len(rest) - len(strings.TrimLeft(rest, " ")) - if contentIndent < f.listIndent { - return "", false - } - return rest[f.listIndent:], true -} - // lineOffset returns the byte offset of the start of a 1-based line number. func lineOffset(source []byte, line int) int { current := 1 diff --git a/internal/extract/body_test.go b/internal/extract/body_test.go index 0e04ca7e..47011cfb 100644 --- a/internal/extract/body_test.go +++ b/internal/extract/body_test.go @@ -70,6 +70,29 @@ func TestExtractSections_HeadingInCodeBlock(t *testing.T) { } } +func TestExtractSectionsFromText_BlockContainerParity(t *testing.T) { + for _, tt := range []struct { + name string + body string + }{ + {name: "ATX heading in list", body: "- item\n\n ## Nested\n\n## Real"}, + {name: "Setext underline in list", body: "- First\n Second\n ---\n\n## Real"}, + {name: "heading in HTML block", body: "
\n## Fake\n
\n\n## Real"}, + {name: "ATX heading in blockquote", body: "> ## Quoted\n\n## Real"}, + } { + t.Run(tt.name, func(t *testing.T) { + astSections := parseSections(tt.body) + textSections := ExtractSectionsFromText(tt.body) + if !reflect.DeepEqual(textSections, astSections) { + t.Fatalf("text sections differ from AST sections: text=%+v AST=%+v", textSections, astSections) + } + if len(textSections) != 1 || textSections[0].Heading != "Real" { + t.Fatalf("sections = %+v; want only the real heading", textSections) + } + }) + } +} + func TestExtractSectionsFromText_ContainerFenceParity(t *testing.T) { for _, tt := range []struct { name string From 5cc08dae77ec2eed47248ea02afc3e3eb0dbe529 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 02:11:27 -0600 Subject: [PATCH 05/18] perf(extract): reuse parsed markdown AST --- internal/extract/body_source.go | 5 ++- internal/extract/body_source_test.go | 59 ++++++++++++++++++++++++++++ internal/extract/extract.go | 3 +- internal/extract/registry.go | 4 +- 4 files changed, 67 insertions(+), 4 deletions(-) diff --git a/internal/extract/body_source.go b/internal/extract/body_source.go index f011c8ad..20b88ff4 100644 --- a/internal/extract/body_source.go +++ b/internal/extract/body_source.go @@ -192,10 +192,13 @@ func SectionSelectorMatches(path SectionPath, selector SectionSelector) bool { } func sectionsForRecord(record *Record) []Section { - if len(record.BodySections) > 0 { + if record.BodySections != nil { sections := append([]Section(nil), record.BodySections...) return sectionsWithPaths(sections) } + if record.AST != nil { + return ExtractSections(record.AST, []byte(record.Body)) + } if record.Body == "" { return nil } diff --git a/internal/extract/body_source_test.go b/internal/extract/body_source_test.go index e10a005b..c29d4e58 100644 --- a/internal/extract/body_source_test.go +++ b/internal/extract/body_source_test.go @@ -4,6 +4,9 @@ import ( "reflect" "strings" "testing" + + "github.com/yuin/goldmark" + gmtext "github.com/yuin/goldmark/text" ) func TestParseBodySource(t *testing.T) { @@ -99,6 +102,62 @@ func TestResolveBodyValue_PresentEmptySection(t *testing.T) { } } +func TestResolveBodyValue_ReusesAvailableAST(t *testing.T) { + astSource := []byte("# AST\n\nAST content\n") + node := goldmark.DefaultParser().Parse(gmtext.NewReader(astSource)) + body := "plain\n\n# Text\n\ntext content\n" + + // The alternate AST makes AST reuse observable without parser instrumentation. + wantSections := ExtractSections(node, []byte(body)) + if reflect.DeepEqual(wantSections, ExtractSectionsFromText(body)) { + t.Fatal("test inputs do not distinguish AST reuse from transient parsing") + } + if len(wantSections) != 1 || wantSections[0].Level != 1 { + t.Fatalf("AST-backed sections = %+v; want one H1 section", wantSections) + } + + rec := &Record{Body: body, AST: node} + gotSections := sectionsForRecord(rec) + if !reflect.DeepEqual(gotSections, wantSections) { + t.Fatalf("record sections = %+v; want AST-backed sections %+v", gotSections, wantSections) + } + got, present, err := ResolveBodyValue(rec, "body.h1") + if err != nil || !present || got != wantSections[0].Heading { + t.Fatalf("value=%q present=%v error=%v; want AST-backed H1 %q", got, present, err, wantSections[0].Heading) + } +} + +func TestResolveBodyValue_ParsesBodyWithoutAST(t *testing.T) { + rec := &Record{Body: "# Text\n\ntext content\n"} + got, present, err := ResolveBodyValue(rec, `body.section["# Text"]`) + if err != nil || !present || got != "text content" { + t.Fatalf("value=%q present=%v error=%v; want transient parsing result", got, present, err) + } +} + +func TestResolveBodyValue_ASTWithoutHeadingsHasNoHeadingResult(t *testing.T) { + body := "paragraph only\n" + node := goldmark.DefaultParser().Parse(gmtext.NewReader([]byte(body))) + rec := &Record{Body: body, AST: node} + + if got, present, err := ResolveBodyValue(rec, "body.h1"); err != nil || present || got != "" { + t.Fatalf("value=%q present=%v error=%v; want no H1", got, present, err) + } + if got, present, err := ResolveBodyValue(rec, `body.section["# Missing"]`); err != nil || present || got != "" { + t.Fatalf("value=%q present=%v error=%v; want no section", got, present, err) + } +} + +func TestResolveBodyValue_PreservesInitializedEmptySections(t *testing.T) { + body := "# Text\n\ntext content\n" + node := goldmark.DefaultParser().Parse(gmtext.NewReader([]byte(body))) + rec := &Record{Body: body, BodySections: []Section{}, AST: node} + + if got, present, err := ResolveBodyValue(rec, "body.h1"); err != nil || present || got != "" { + t.Fatalf("value=%q present=%v error=%v; want initialized empty sections", got, present, err) + } +} + func TestResolveBodyValue_SelectorMatching(t *testing.T) { rec := &Record{Body: "# Root\n\n## Parent A\n\n### Notes\n\nfirst\n\n## Parent B\n\n### Notes\n\nsecond\n\n#### Detail\n\ndetail\n\n# Other Root\n\n## Parent B\n\n### Notes\n\nthird\n"} tests := []struct { diff --git a/internal/extract/extract.go b/internal/extract/extract.go index 9cb85619..8c04d3a2 100644 --- a/internal/extract/extract.go +++ b/internal/extract/extract.go @@ -58,7 +58,8 @@ type ExtractionError struct { } // MarkdownExtractor extracts YAML frontmatter from Markdown files. -// Set ParseAST to false to skip goldmark AST parsing (default: true). +// ParseAST controls whether Extract retains the Goldmark AST in Record.AST. +// Section extraction can use a transient AST when ParseAST is false. type MarkdownExtractor struct { ParseAST *bool } diff --git a/internal/extract/registry.go b/internal/extract/registry.go index 30829453..791498ef 100644 --- a/internal/extract/registry.go +++ b/internal/extract/registry.go @@ -21,8 +21,8 @@ func NewRegistry() *Registry { return r } -// NewASTRegistry creates a registry with AST parsing enabled. -// Use this when body structure (sections, headings) needs to be inspected. +// NewASTRegistry creates a registry with AST retention enabled. +// Use this when callers need Record.AST after extraction. func NewASTRegistry() *Registry { r := &Registry{ byName: make(map[string]Extractor), From c74fea2a86d15980355119c8f7946ce09d3c4dcd Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 02:22:13 -0600 Subject: [PATCH 06/18] fix(extract): handle ATX boundaries efficiently --- internal/extract/body.go | 119 +++++++++++++++++++++------------- internal/extract/body_test.go | 51 +++++++++++++++ 2 files changed, 125 insertions(+), 45 deletions(-) diff --git a/internal/extract/body.go b/internal/extract/body.go index 51c782ce..beb230a5 100644 --- a/internal/extract/body.go +++ b/internal/extract/body.go @@ -1,6 +1,8 @@ package extract import ( + "bytes" + "sort" "strings" "github.com/yuin/goldmark" @@ -35,12 +37,14 @@ type Section struct { // If the document has no headings, a single section with the entire body is returned. func ExtractSections(node ast.Node, source []byte) []Section { type headingInfo struct { - text string - level int - startLine int - endOffset int // byte offset after the heading line + text string + level int + startLine int + startOffset int + endOffset int } + lineIndex := newSourceLineIndex(source) var headings []headingInfo // Collect headings from the AST (top-level block children only). @@ -50,41 +54,42 @@ func ExtractSections(node ast.Node, source []byte) []Section { } h := child.(*ast.Heading) lines := h.Lines() - startLine := 0 - endOffset := 0 + position := h.Pos() + if position < 0 && lines.Len() > 0 { + position = lines.At(0).Start + } + startLine := lineIndex.lineForOffset(position) + startOffset := lineIndex.offset(startLine) + lineEnd := lineIndex.offset(startLine + 1) + endOffset := lineEnd headingText := "" - if lines.Len() > 0 { - firstSeg := lines.At(0) - startLine = lineFromOffset(source, firstSeg.Start) - lastSeg := lines.At(lines.Len() - 1) - endOffset = lastSeg.Stop - lineStart := lineOffset(source, startLine) - lineEnd := lineOffset(source, startLine+1) - if _, text, ok := parseATXHeading(string(source[lineStart:lineEnd])); ok { - headingText = text - } else { - var text strings.Builder - for i := 0; i < lines.Len(); i++ { - segment := lines.At(i) - text.Write(segment.Value(source)) - } - headingText = strings.TrimSpace(text.String()) + if _, text, ok := parseATXHeading(string(source[startOffset:lineEnd])); ok { + headingText = text + } else if lines.Len() > 0 { + var text strings.Builder + for i := 0; i < lines.Len(); i++ { + segment := lines.At(i) + text.Write(segment.Value(source)) + } + headingText = strings.TrimSpace(text.String()) - lastLine := lineFromOffset(source, lastSeg.Start) - underlineStart := lineOffset(source, lastLine+1) - underlineEnd := lineOffset(source, lastLine+2) - if _, ok := parseSetextUnderline(string(source[underlineStart:underlineEnd])); ok { - endOffset = underlineEnd - } + lastSeg := lines.At(lines.Len() - 1) + endOffset = lastSeg.Stop + lastLine := lineIndex.lineForOffset(lastSeg.Start) + underlineStart := lineIndex.offset(lastLine + 1) + underlineEnd := lineIndex.offset(lastLine + 2) + if _, ok := parseSetextUnderline(string(source[underlineStart:underlineEnd])); ok { + endOffset = underlineEnd } } headings = append(headings, headingInfo{ - text: headingText, - level: h.Level, - startLine: startLine, - endOffset: endOffset, + text: headingText, + level: h.Level, + startLine: startLine, + startOffset: startOffset, + endOffset: endOffset, }) } @@ -103,7 +108,7 @@ func ExtractSections(node ast.Node, source []byte) []Section { var contentEnd int if i+1 < len(headings) { // Content ends where the next heading's line starts. - contentEnd = lineOffset(source, headings[i+1].startLine) + contentEnd = headings[i+1].startOffset } else { contentEnd = len(source) } @@ -253,11 +258,13 @@ func parseATXHeading(line string) (int, string, bool) { return 0, "", false } text := strings.TrimSpace(rest[level:]) - if i := len(text) - 1; i > 0 && text[i] == '#' { + if i := len(text) - 1; i >= 0 && text[i] == '#' { for i >= 0 && text[i] == '#' { i-- } - if i >= 0 && (text[i] == ' ' || text[i] == '\t') { + if i < 0 { + text = "" + } else if text[i] == ' ' || text[i] == '\t' { text = strings.TrimSpace(text[:i]) } } @@ -285,18 +292,40 @@ func parseSetextUnderline(line string) (int, bool) { return 2, true } -// lineOffset returns the byte offset of the start of a 1-based line number. -func lineOffset(source []byte, line int) int { - current := 1 - for i := 0; i < len(source); i++ { - if current == line { - return i - } - if source[i] == '\n' { - current++ +type sourceLineIndex struct { + starts []int + sourceLength int +} + +func newSourceLineIndex(source []byte) sourceLineIndex { + starts := make([]int, 1, bytes.Count(source, []byte{'\n'})+1) + for offset, value := range source { + if value == '\n' { + starts = append(starts, offset+1) } } - return len(source) + return sourceLineIndex{starts: starts, sourceLength: len(source)} +} + +func (index sourceLineIndex) lineForOffset(offset int) int { + if offset < 0 { + offset = 0 + } else if offset > index.sourceLength { + offset = index.sourceLength + } + return sort.Search(len(index.starts), func(i int) bool { + return index.starts[i] > offset + }) +} + +func (index sourceLineIndex) offset(line int) int { + if line <= 1 { + return 0 + } + if line > len(index.starts) { + return index.sourceLength + } + return index.starts[line-1] } // ExtractBodyH1 returns the text of the first H1 heading in the body, diff --git a/internal/extract/body_test.go b/internal/extract/body_test.go index 47011cfb..06e8b92a 100644 --- a/internal/extract/body_test.go +++ b/internal/extract/body_test.go @@ -2,6 +2,8 @@ package extract import ( "reflect" + "strconv" + "strings" "testing" "github.com/yuin/goldmark" @@ -55,6 +57,55 @@ func TestExtractSections_MultipleHeadings(t *testing.T) { } } +func TestExtractSections_ATXBoundaries(t *testing.T) { + body := "## Notes ##\n\nFirst\n\n##\n\nSecond\n\n### ###\n\nThird\n\nTitle\n=====\n\nFourth\n\n## Literal##\n\nFifth\n" + want := []Section{ + {Heading: "Notes", Level: 2, Path: SectionPath{{Level: 2, Text: "Notes"}}, Content: "First", StartLine: 1}, + {Heading: "", Level: 2, Path: SectionPath{{Level: 2, Text: ""}}, Content: "Second", StartLine: 5}, + {Heading: "", Level: 3, Path: SectionPath{{Level: 2, Text: ""}, {Level: 3, Text: ""}}, Content: "Third", StartLine: 9}, + {Heading: "Title", Level: 1, Path: SectionPath{{Level: 1, Text: "Title"}}, Content: "Fourth", StartLine: 13}, + {Heading: "Literal##", Level: 2, Path: SectionPath{{Level: 1, Text: "Title"}, {Level: 2, Text: "Literal##"}}, Content: "Fifth", StartLine: 18}, + } + + for name, sections := range map[string][]Section{ + "AST": parseSections(body), + "text": ExtractSectionsFromText(body), + } { + if !reflect.DeepEqual(sections, want) { + t.Errorf("%s sections = %+v; want %+v", name, sections, want) + } + } +} + +func TestExtractSections_AdjacentATXHeadings(t *testing.T) { + body := "# First #\n##\n### Third ###\nContent\n" + want := []Section{ + {Heading: "First", Level: 1, Path: SectionPath{{Level: 1, Text: "First"}}, Content: "", StartLine: 1}, + {Heading: "", Level: 2, Path: SectionPath{{Level: 1, Text: "First"}, {Level: 2, Text: ""}}, Content: "", StartLine: 2}, + {Heading: "Third", Level: 3, Path: SectionPath{{Level: 1, Text: "First"}, {Level: 2, Text: ""}, {Level: 3, Text: "Third"}}, Content: "Content", StartLine: 3}, + } + if sections := parseSections(body); !reflect.DeepEqual(sections, want) { + t.Fatalf("sections = %+v; want %+v", sections, want) + } +} + +func BenchmarkExtractSectionsManyHeadings(b *testing.B) { + for _, count := range []int{100, 1000} { + b.Run(strconv.Itoa(count), func(b *testing.B) { + source := []byte(strings.Repeat("## Heading ##\n\nContent\n\n", count)) + node := goldmark.DefaultParser().Parse(text.NewReader(source)) + b.ReportAllocs() + b.SetBytes(int64(len(source))) + b.ResetTimer() + for i := 0; i < b.N; i++ { + benchmarkSections = ExtractSections(node, source) + } + }) + } +} + +var benchmarkSections []Section + func TestExtractSections_HeadingInCodeBlock(t *testing.T) { body := "## Real\n\nSome text\n\n```\n## Fake\n```\n\n## Also Real\n\nMore text\n" sections := parseSections(body) From 24888f75ff4478af6d2cb7b6700f9d033ffa884d Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 01:35:48 -0600 Subject: [PATCH 07/18] feat(infer): infer hierarchical body section selectors --- internal/infer/body_sections.go | 310 +++++++++++++++++++++++---- internal/infer/body_sections_test.go | 116 +++++++++- 2 files changed, 381 insertions(+), 45 deletions(-) diff --git a/internal/infer/body_sections.go b/internal/infer/body_sections.go index 9cabd889..b38fd4d0 100644 --- a/internal/infer/body_sections.go +++ b/internal/infer/body_sections.go @@ -11,9 +11,8 @@ import ( ) // DetectSectionPatterns analyzes heading structure across records in a directory. -// Every record contributes to the denominator; empty bodies count as records where -// headings are absent. Headings present in at least threshold records are inferred; -// universal headings are required, while non-universal threshold matches are optional. +// Every record contributes to the denominator. A record contributes once to each +// final-heading family, but selector resolution retains all section occurrences. func DetectSectionPatterns(records []*extract.Record, threshold float64) ([]Inference, error) { if math.IsNaN(threshold) || math.IsInf(threshold, 0) || threshold < 0 || threshold > 1 { return nil, fmt.Errorf("section threshold must be finite and in [0,1], got %v", threshold) @@ -23,67 +22,71 @@ func DetectSectionPatterns(records []*extract.Record, threshold float64) ([]Infe } total := len(records) - headingCount := make(map[string]int) - headingExact := make(map[string]string) - - for _, rec := range records { + families := make(map[extract.HeadingKey]*sectionFamily) + for recordOrdinal, rec := range records { if rec == nil || rec.Body == "" { continue } - seen := make(map[string]bool) - for _, sec := range sectionsForInference(rec) { + sections := sectionsForInference(rec) + paths := sectionPathsForInference(sections) + for sectionOrdinal, sec := range sections { if sec.Level <= 0 || sec.Heading == "" { continue } - exact := strings.Repeat("#", sec.Level) + " " + sec.Heading - source, err := extract.CanonicalSectionSource(exact) - if err != nil { - return nil, err - } - if seen[source] { - continue + key := extract.HeadingKey{Level: sec.Level, Text: sec.Heading} + family := families[key] + if family == nil { + family = §ionFamily{key: key, contributors: make(map[int]bool)} + families[key] = family } - headingCount[source]++ - headingExact[source] = exact - seen[source] = true + family.contributors[recordOrdinal] = true + family.occurrences = append(family.occurrences, sectionOccurrence{ + recordPath: rec.Path, + recordOrdinal: recordOrdinal, + sectionOrdinal: sectionOrdinal, + path: paths[sectionOrdinal], + }) } } - allHeadings := make([]sectionCandidate, 0, len(headingCount)) - for source, count := range headingCount { + candidates := make([]sectionCandidate, 0, len(families)) + for _, family := range families { + count := len(family.contributors) freq := float64(count) / float64(total) - exact := headingExact[source] - field := sectionFieldName(strings.TrimSpace(exact[strings.IndexByte(exact, ' ')+1:])) - candidateType := "optional_section" + if freq < threshold { + continue + } + exact := exactSectionHeading(family.key) + typeName := "optional_section" if count == total { - candidateType = "required_section" + typeName = "required_section" } - allHeadings = append(allHeadings, sectionCandidate{ - field: field, + candidates = append(candidates, sectionCandidate{ + family: family, + field: sectionFieldName(family.key.Text), exact: exact, - source: source, count: count, freq: freq, - typeName: candidateType, + typeName: typeName, }) } - sort.Slice(allHeadings, func(i, j int) bool { - if allHeadings[i].field != allHeadings[j].field { - return allHeadings[i].field < allHeadings[j].field + sort.Slice(candidates, func(i, j int) bool { + if candidates[i].field != candidates[j].field { + return candidates[i].field < candidates[j].field } - return allHeadings[i].source < allHeadings[j].source + return candidates[i].exact < candidates[j].exact }) - if err := rejectSectionFieldCollisions(allHeadings); err != nil { - return nil, err - } - - candidates := make([]sectionCandidate, 0, len(allHeadings)) - for _, candidate := range allHeadings { - if candidate.freq >= threshold { - candidates = append(candidates, candidate) + for i := range candidates { + source, err := resolveSectionFamily(candidates[i].family) + if err != nil { + return nil, err } + candidates[i].source = source + } + if err := rejectSectionFieldCollisions(candidates); err != nil { + return nil, err } inferences := make([]Inference, 0, len(candidates)) @@ -101,7 +104,22 @@ func DetectSectionPatterns(records []*extract.Record, threshold float64) ([]Infe return inferences, nil } +type sectionFamily struct { + key extract.HeadingKey + contributors map[int]bool + occurrences []sectionOccurrence +} + +type sectionOccurrence struct { + recordPath string + recordOrdinal int + sectionOrdinal int + path extract.SectionPath + fullSource string +} + type sectionCandidate struct { + family *sectionFamily field string exact string source string @@ -110,6 +128,16 @@ type sectionCandidate struct { typeName string } +type selectorCandidate struct { + source string + selector extract.SectionSelector +} + +type stableOccurrenceGroup struct { + identity string + selectors []selectorCandidate +} + func sectionsForInference(rec *extract.Record) []extract.Section { if len(rec.BodySections) > 0 { return rec.BodySections @@ -120,6 +148,206 @@ func sectionsForInference(rec *extract.Record) []extract.Section { return extract.ExtractSectionsFromText(rec.Body) } +func sectionPathsForInference(sections []extract.Section) []extract.SectionPath { + paths := make([]extract.SectionPath, len(sections)) + path := make(extract.SectionPath, 0, 6) + for i, sec := range sections { + if sec.Level <= 0 { + continue + } + for len(path) > 0 && path[len(path)-1].Level >= sec.Level { + path = path[:len(path)-1] + } + path = append(path, extract.HeadingKey{Level: sec.Level, Text: sec.Heading}) + paths[i] = append(extract.SectionPath(nil), path...) + } + return paths +} + +func resolveSectionFamily(family *sectionFamily) (string, error) { + occurrences := append([]sectionOccurrence(nil), family.occurrences...) + for i := range occurrences { + source, err := extract.CanonicalSectionSelectorSource(extract.SectionSelector(occurrences[i].path)) + if err != nil { + return "", err + } + occurrences[i].fullSource = source + } + sortSectionOccurrences(occurrences) + + if err := rejectDuplicateSectionPaths(family.key, occurrences); err != nil { + return "", err + } + + selectorBySource := make(map[string]selectorCandidate) + for _, occurrence := range occurrences { + for start := len(occurrence.path) - 1; start >= 0; start-- { + source, err := extract.CanonicalSectionSelectorSource(extract.SectionSelector(occurrence.path[start:])) + if err != nil { + return "", err + } + parsed, err := extract.ParseBodySource(source) + if err != nil { + return "", err + } + selectorBySource[source] = selectorCandidate{source: source, selector: parsed.Selector} + } + } + + selectors := make([]selectorCandidate, 0, len(selectorBySource)) + for _, selector := range selectorBySource { + selectors = append(selectors, selector) + } + sort.Slice(selectors, func(i, j int) bool { + if len(selectors[i].selector) != len(selectors[j].selector) { + return len(selectors[i].selector) < len(selectors[j].selector) + } + return selectors[i].source < selectors[j].source + }) + + contributorOrdinals := make([]int, 0, len(family.contributors)) + for recordOrdinal := range family.contributors { + contributorOrdinals = append(contributorOrdinals, recordOrdinal) + } + sort.Slice(contributorOrdinals, func(i, j int) bool { + left, right := firstOccurrenceForRecord(occurrences, contributorOrdinals[i]), firstOccurrenceForRecord(occurrences, contributorOrdinals[j]) + if left.recordPath != right.recordPath { + return left.recordPath < right.recordPath + } + return contributorOrdinals[i] < contributorOrdinals[j] + }) + + groupsByIdentity := make(map[string]*stableOccurrenceGroup) + for _, selector := range selectors { + selected := make([]sectionOccurrence, 0, len(contributorOrdinals)) + common := true + for _, recordOrdinal := range contributorOrdinals { + matches := matchingOccurrences(occurrences, recordOrdinal, selector.selector) + if len(matches) != 1 { + common = false + break + } + selected = append(selected, matches[0]) + } + if !common { + continue + } + identity := occurrenceGroupIdentity(selected) + group := groupsByIdentity[identity] + if group == nil { + group = &stableOccurrenceGroup{identity: identity} + groupsByIdentity[identity] = group + } + group.selectors = append(group.selectors, selector) + } + + exact := exactSectionHeading(family.key) + if len(groupsByIdentity) == 0 { + return "", fmt.Errorf("section family %q has no common selector", exact) + } + + groups := make([]*stableOccurrenceGroup, 0, len(groupsByIdentity)) + for _, group := range groupsByIdentity { + sort.Slice(group.selectors, func(i, j int) bool { + if len(group.selectors[i].selector) != len(group.selectors[j].selector) { + return len(group.selectors[i].selector) < len(group.selectors[j].selector) + } + return group.selectors[i].source < group.selectors[j].source + }) + groups = append(groups, group) + } + sort.Slice(groups, func(i, j int) bool { + left, right := groups[i].selectors[0], groups[j].selectors[0] + if left.source != right.source { + return left.source < right.source + } + return groups[i].identity < groups[j].identity + }) + if len(groups) > 1 { + sources := make([]string, 0, len(groups)) + for _, group := range groups { + sources = append(sources, group.selectors[0].source) + } + return "", fmt.Errorf("section family %q has multiple stable occurrence groups: %s", exact, strings.Join(sources, ", ")) + } + return groups[0].selectors[0].source, nil +} + +func sortSectionOccurrences(occurrences []sectionOccurrence) { + sort.Slice(occurrences, func(i, j int) bool { + if occurrences[i].recordPath != occurrences[j].recordPath { + return occurrences[i].recordPath < occurrences[j].recordPath + } + if occurrences[i].recordOrdinal != occurrences[j].recordOrdinal { + return occurrences[i].recordOrdinal < occurrences[j].recordOrdinal + } + if occurrences[i].sectionOrdinal != occurrences[j].sectionOrdinal { + return occurrences[i].sectionOrdinal < occurrences[j].sectionOrdinal + } + return occurrences[i].fullSource < occurrences[j].fullSource + }) +} + +func rejectDuplicateSectionPaths(key extract.HeadingKey, occurrences []sectionOccurrence) error { + type duplicateKey struct { + recordOrdinal int + fullSource string + } + byPath := make(map[duplicateKey][]sectionOccurrence) + for _, occurrence := range occurrences { + key := duplicateKey{recordOrdinal: occurrence.recordOrdinal, fullSource: occurrence.fullSource} + byPath[key] = append(byPath[key], occurrence) + } + + var duplicates []string + for _, matches := range byPath { + if len(matches) < 2 { + continue + } + ordinals := make([]string, 0, len(matches)) + for _, match := range matches { + ordinals = append(ordinals, strconv.Itoa(match.sectionOrdinal)) + } + duplicates = append(duplicates, fmt.Sprintf("%q (record %d), %s (section ordinals %s)", matches[0].recordPath, matches[0].recordOrdinal, matches[0].fullSource, strings.Join(ordinals, ", "))) + } + if len(duplicates) == 0 { + return nil + } + sort.Strings(duplicates) + return fmt.Errorf("duplicate body section path for family %q: %s", exactSectionHeading(key), strings.Join(duplicates, "; ")) +} + +func firstOccurrenceForRecord(occurrences []sectionOccurrence, recordOrdinal int) sectionOccurrence { + for _, occurrence := range occurrences { + if occurrence.recordOrdinal == recordOrdinal { + return occurrence + } + } + return sectionOccurrence{recordOrdinal: recordOrdinal} +} + +func matchingOccurrences(occurrences []sectionOccurrence, recordOrdinal int, selector extract.SectionSelector) []sectionOccurrence { + var matches []sectionOccurrence + for _, occurrence := range occurrences { + if occurrence.recordOrdinal == recordOrdinal && extract.SectionSelectorMatches(occurrence.path, selector) { + matches = append(matches, occurrence) + } + } + return matches +} + +func occurrenceGroupIdentity(occurrences []sectionOccurrence) string { + var identity strings.Builder + for _, occurrence := range occurrences { + fmt.Fprintf(&identity, "%s\x00%d\x00%d\x00%s\x00", occurrence.recordPath, occurrence.recordOrdinal, occurrence.sectionOrdinal, occurrence.fullSource) + } + return identity.String() +} + +func exactSectionHeading(key extract.HeadingKey) string { + return strings.Repeat("#", key.Level) + " " + key.Text +} + func rejectSectionFieldCollisions(candidates []sectionCandidate) error { var collisions []string for i := 0; i < len(candidates); { diff --git a/internal/infer/body_sections_test.go b/internal/infer/body_sections_test.go index 592de550..9fa2eea0 100644 --- a/internal/infer/body_sections_test.go +++ b/internal/infer/body_sections_test.go @@ -46,6 +46,85 @@ func TestDetectSectionPatterns_UniversalSectionRequiredWithCanonicalSource(t *te } } +func TestDetectSectionPatterns_VariableH1UsesSimpleRequiredSelector(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## Overview\nFirst.\n"), + makeRecord("# Beta\n\n## Overview\nSecond.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + inf, ok := findSectionInference(inferences, "overview") + if !ok { + t.Fatalf("overview inference is missing: %+v", inferences) + } + if inf.Type != "required_section" || inf.SourceDirective != `body.section["## Overview"]` { + t.Fatalf("overview inference = %+v", inf) + } +} + +func TestDetectSectionPatterns_VariableParentsUseSimpleSelectorForSingleOccurrence(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## First Parent\n\n### Notes\nFirst.\n"), + makeRecord("# Beta\n\n## Second Parent\n\n### Notes\nSecond.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + inf, ok := findSectionInference(inferences, "notes") + if !ok || inf.SourceDirective != `body.section["### Notes"]` { + t.Fatalf("notes inference = %+v, present = %v", inf, ok) + } +} + +func TestDetectSectionPatterns_SelectsShortestCommonQualifiedSelector(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Root\n\n## First Parent\n\n### Notes\nVariable.\n\n## Shared\n\n### Notes\nStable.\n"), + makeRecord("# Root\n\n## Second Parent\n\n### Notes\nVariable.\n\n## Shared\n\n### Notes\nStable.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + inf, ok := findSectionInference(inferences, "notes") + if !ok || inf.SourceDirective != `body.section["## Shared"]["### Notes"]` { + t.Fatalf("notes inference = %+v, present = %v", inf, ok) + } +} + +func TestDetectSectionPatterns_IncompatibleParentsHaveNoCommonSelector(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## First A\n\n### Notes\nOne.\n\n## Second A\n\n### Notes\nTwo.\n"), + makeRecord("# Beta\n\n## First B\n\n### Notes\nOne.\n\n## Second B\n\n### Notes\nTwo.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + if err == nil || err.Error() != `section family "### Notes" has no common selector` { + t.Fatalf("error = %v", err) + } + if inferences != nil { + t.Fatalf("inferences = %+v, want nil", inferences) + } +} + +func TestDetectSectionPatterns_MultipleStableOccurrenceGroupsFail(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## First\n\n### Notes\nOne.\n\n## Second\n\n### Notes\nTwo.\n"), + makeRecord("# Beta\n\n## First\n\n### Notes\nOne.\n\n## Second\n\n### Notes\nTwo.\n"), + } + + _, err := DetectSectionPatterns(records, 1) + want := `section family "### Notes" has multiple stable occurrence groups: body.section["## First"]["### Notes"], body.section["## Second"]["### Notes"]` + if err == nil || err.Error() != want { + t.Fatalf("error = %v, want %q", err, want) + } +} + func TestDetectSectionPatterns_ThresholdCandidateOptionalUntilUniversal(t *testing.T) { records := []*extract.Record{ makeRecord("## Notes\nA\n"), makeRecord("## Notes\nB\n"), @@ -189,15 +268,44 @@ func TestDetectSectionPatterns_NameCollisionFails(t *testing.T) { } } -func TestDetectSectionPatterns_NameCollisionFailsBeforeThreshold(t *testing.T) { +func TestDetectSectionPatterns_BelowThresholdFamiliesDoNotCollide(t *testing.T) { records := []*extract.Record{ makeRecord("## Notes\nA\n\n### Notes\nB\n"), makeRecord("# Other\n"), makeRecord("# Other\n"), makeRecord("# Other\n"), makeRecord("# Other\n"), } - _, err := DetectSectionPatterns(records, 0.8) - if err == nil || !strings.Contains(err.Error(), "## Notes") || !strings.Contains(err.Error(), "### Notes") { - t.Fatalf("expected below-threshold colliding headings, got %v", err) + inferences, err := DetectSectionPatterns(records, 0.8) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + if _, ok := findSectionInference(inferences, "notes"); ok { + t.Fatalf("notes inference is above the threshold: %+v", inferences) + } +} + +func TestDetectSectionPatterns_DuplicateFullPathFailsOnlyAtThreshold(t *testing.T) { + records := []*extract.Record{ + makeRecord("## Parent\n\n### Notes\nFirst.\n\n### Notes\nSecond.\n"), + makeRecord("# Other\n"), + } + records[0].Path = "a.md" + records[1].Path = "b.md" + + inferences, err := DetectSectionPatterns(records, 0.75) + if err != nil { + t.Fatalf("below-threshold family returned an error: %v", err) + } + if _, ok := findSectionInference(inferences, "notes"); ok { + t.Fatalf("notes inference is above the threshold: %+v", inferences) + } + + inferences, err = DetectSectionPatterns(records, 0.5) + want := `duplicate body section path for family "### Notes": "a.md" (record 0), body.section["## Parent"]["### Notes"] (section ordinals 1, 2)` + if err == nil || err.Error() != want { + t.Fatalf("error = %v, want %q", err, want) + } + if inferences != nil { + t.Fatalf("inferences = %+v, want nil", inferences) } } From 33d965b1374159f63d9ff8aac08763dfbe798a79 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 01:47:08 -0600 Subject: [PATCH 08/18] fix(infer): expose section inference error codes --- internal/infer/body_sections.go | 6 +++--- internal/infer/body_sections_test.go | 9 +++++---- 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/internal/infer/body_sections.go b/internal/infer/body_sections.go index b38fd4d0..602d595c 100644 --- a/internal/infer/body_sections.go +++ b/internal/infer/body_sections.go @@ -243,7 +243,7 @@ func resolveSectionFamily(family *sectionFamily) (string, error) { exact := exactSectionHeading(family.key) if len(groupsByIdentity) == 0 { - return "", fmt.Errorf("section family %q has no common selector", exact) + return "", fmt.Errorf("no_common_selector: section family %q has no common selector", exact) } groups := make([]*stableOccurrenceGroup, 0, len(groupsByIdentity)) @@ -268,7 +268,7 @@ func resolveSectionFamily(family *sectionFamily) (string, error) { for _, group := range groups { sources = append(sources, group.selectors[0].source) } - return "", fmt.Errorf("section family %q has multiple stable occurrence groups: %s", exact, strings.Join(sources, ", ")) + return "", fmt.Errorf("multiple_stable_groups: section family %q has multiple stable occurrence groups: %s", exact, strings.Join(sources, ", ")) } return groups[0].selectors[0].source, nil } @@ -314,7 +314,7 @@ func rejectDuplicateSectionPaths(key extract.HeadingKey, occurrences []sectionOc return nil } sort.Strings(duplicates) - return fmt.Errorf("duplicate body section path for family %q: %s", exactSectionHeading(key), strings.Join(duplicates, "; ")) + return fmt.Errorf("duplicate_full_path: duplicate body section path for family %q: %s", exactSectionHeading(key), strings.Join(duplicates, "; ")) } func firstOccurrenceForRecord(occurrences []sectionOccurrence, recordOrdinal int) sectionOccurrence { diff --git a/internal/infer/body_sections_test.go b/internal/infer/body_sections_test.go index 9fa2eea0..c2925bd3 100644 --- a/internal/infer/body_sections_test.go +++ b/internal/infer/body_sections_test.go @@ -104,8 +104,9 @@ func TestDetectSectionPatterns_IncompatibleParentsHaveNoCommonSelector(t *testin } inferences, err := DetectSectionPatterns(records, 1) - if err == nil || err.Error() != `section family "### Notes" has no common selector` { - t.Fatalf("error = %v", err) + want := `no_common_selector: section family "### Notes" has no common selector` + if err == nil || err.Error() != want { + t.Fatalf("error = %v, want %q", err, want) } if inferences != nil { t.Fatalf("inferences = %+v, want nil", inferences) @@ -119,7 +120,7 @@ func TestDetectSectionPatterns_MultipleStableOccurrenceGroupsFail(t *testing.T) } _, err := DetectSectionPatterns(records, 1) - want := `section family "### Notes" has multiple stable occurrence groups: body.section["## First"]["### Notes"], body.section["## Second"]["### Notes"]` + want := `multiple_stable_groups: section family "### Notes" has multiple stable occurrence groups: body.section["## First"]["### Notes"], body.section["## Second"]["### Notes"]` if err == nil || err.Error() != want { t.Fatalf("error = %v, want %q", err, want) } @@ -300,7 +301,7 @@ func TestDetectSectionPatterns_DuplicateFullPathFailsOnlyAtThreshold(t *testing. } inferences, err = DetectSectionPatterns(records, 0.5) - want := `duplicate body section path for family "### Notes": "a.md" (record 0), body.section["## Parent"]["### Notes"] (section ordinals 1, 2)` + want := `duplicate_full_path: duplicate body section path for family "### Notes": "a.md" (record 0), body.section["## Parent"]["### Notes"] (section ordinals 1, 2)` if err == nil || err.Error() != want { t.Fatalf("error = %v, want %q", err, want) } From 544b0e737c61710b705117f21bed778fbf082c36 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 01:33:46 -0600 Subject: [PATCH 09/18] fix(scaffold): reject missing qualified body sections --- cmd/rootline/migrate_scaffold_source_test.go | 60 ++++++++++++++++ cmd/rootline/new_source_test.go | 33 +++++++++ internal/rules/section_materialization.go | 20 +++--- .../rules/section_materialization_test.go | 71 +++++++++++++++++++ 4 files changed, 176 insertions(+), 8 deletions(-) diff --git a/cmd/rootline/migrate_scaffold_source_test.go b/cmd/rootline/migrate_scaffold_source_test.go index bacb5d6c..cac82069 100644 --- a/cmd/rootline/migrate_scaffold_source_test.go +++ b/cmd/rootline/migrate_scaffold_source_test.go @@ -131,6 +131,66 @@ func TestMigrateScaffoldRequiredSectionSourceDryRunParityAndNoWrite(t *testing.T } } +func TestMigrateScaffoldRejectsMissingQualifiedSectionWithoutWriting(t *testing.T) { + schema := `schema: + notes: + type: string + required: true + source: 'body.section["## Parent"]["### Notes"]' +` + selector := `body.section["## Parent"]["### Notes"]` + original := "---\ntitle: Task\n---\n# Task\n" + + writeDir := newMigrateScaffoldSectionSourceProject(t, schema, original) + writeOut, writeErr := runCmd(t, "migrate", "--scaffold", writeDir) + if writeErr == nil { + t.Fatalf("expected scaffold to reject the qualified selector, output: %s", writeOut) + } + if !strings.Contains(writeErr.Error(), `field "notes"`) || !strings.Contains(writeErr.Error(), selector) { + t.Fatalf("write error = %q, want field and canonical selector", writeErr) + } + writeContent, err := os.ReadFile(filepath.Join(writeDir, "T001-task.md")) + if err != nil { + t.Fatal(err) + } + if string(writeContent) != original { + t.Fatalf("affected file changed after materialization error\ngot: %q\nwant: %q", writeContent, original) + } + + dryDir := newMigrateScaffoldSectionSourceProject(t, schema, original) + dryOut, dryErr := runCmd(t, "migrate", "--scaffold", "--dry-run", dryDir) + if dryErr == nil { + t.Fatalf("expected dry-run to reject the qualified selector, output: %s", dryOut) + } + if dryErr.Error() != writeErr.Error() { + t.Fatalf("dry-run error = %q, want write error %q", dryErr, writeErr) + } + dryContent, err := os.ReadFile(filepath.Join(dryDir, "T001-task.md")) + if err != nil { + t.Fatal(err) + } + if string(dryContent) != original { + t.Fatalf("dry-run changed the affected file\ngot: %q\nwant: %q", dryContent, original) + } +} + +func TestMigrateScaffoldQualifiedSectionFrontmatterPrecedence(t *testing.T) { + dir := newMigrateScaffoldSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Parent"]["### Notes"]' +`, "---\nnotes: override\n---\n# Task\n") + + result, out, err := runMigrateScaffoldJSON(t, dir) + if err != nil { + t.Fatalf("frontmatter must satisfy the qualified field: %v\noutput: %s", err, out) + } + if result.SectionsAdded != 0 || result.FilesScaffolded != 0 { + t.Fatalf("result = %+v, want no materialization", result) + } +} + func TestMigrateScaffoldRequiredSectionSourcePreservesModeAndValidatesAfterWrite(t *testing.T) { dir := newMigrateScaffoldSectionSourceProject(t, `schema: anchor: diff --git a/cmd/rootline/new_source_test.go b/cmd/rootline/new_source_test.go index 717a47fe..9554b304 100644 --- a/cmd/rootline/new_source_test.go +++ b/cmd/rootline/new_source_test.go @@ -169,6 +169,39 @@ func TestNewSectionSourceNoWriteOnSchemaMaterializationOrValidationFailure(t *te } } +func TestNewRejectsMissingQualifiedSectionBeforeWriting(t *testing.T) { + dir := newSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Parent"]["### Notes"]' +`) + selector := `body.section["## Parent"]["### Notes"]` + + target := filepath.Join(dir, "T001-task.md") + out, err := runCmd(t, "new", target) + if err == nil { + t.Fatalf("expected new to reject the qualified selector, output: %s", out) + } + if !strings.Contains(err.Error(), `field "notes"`) || !strings.Contains(err.Error(), selector) { + t.Fatalf("error = %q, want field and canonical selector", err) + } + if _, statErr := os.Stat(target); !os.IsNotExist(statErr) { + t.Fatalf("target must remain absent, stat error = %v", statErr) + } + + existing := filepath.Join(dir, "existing.md") + original := []byte("---\ntitle: Keep\n---\n# Existing\n") + mustWriteFile(t, existing, original, 0o640) + out, err = runCmd(t, "new", existing, "--force") + if err == nil { + t.Fatalf("expected forced new to reject the qualified selector, output: %s", out) + } + if got := mustReadFile(t, existing); string(got) != string(original) { + t.Fatalf("forced target changed after materialization error\ngot: %q\nwant: %q", got, original) + } +} + func mustResolveNewEffective(t *testing.T, dir, target string) *rules.StemFile { t.Helper() effective, err := rules.ResolveForRecord(dir, target) diff --git a/internal/rules/section_materialization.go b/internal/rules/section_materialization.go index bf7e0123..223004c6 100644 --- a/internal/rules/section_materialization.go +++ b/internal/rules/section_materialization.go @@ -51,13 +51,7 @@ func RequiredSectionMaterializations(record *extract.Record, effective *StemFile out := make([]SectionMaterialization, 0) for _, name := range fields { field := local.Schema[name] - if !field.Required || field.Extract == "" { - continue - } - if !requiredCheckApplies(record, &local, name, field) { - continue - } - if field.Type != "string" { + if field.Extract == "" || field.Type != "string" { continue } source, err := extract.ParseBodySource(field.Extract) @@ -71,9 +65,19 @@ func RequiredSectionMaterializations(record *extract.Record, effective *StemFile if err != nil { return nil, err } - if present { + if present || !field.Required { continue } + if !requiredCheckApplies(record, &local, name, field) { + continue + } + if len(source.Selector) > 1 { + canonical, err := extract.CanonicalSectionSelectorSource(source.Selector) + if err != nil { + return nil, err + } + return nil, fmt.Errorf("cannot materialize required field %q: qualified section selector %s is missing", name, canonical) + } content := field.Default if content == "" { content = "" diff --git a/internal/rules/section_materialization_test.go b/internal/rules/section_materialization_test.go index b836df65..9104f31a 100644 --- a/internal/rules/section_materialization_test.go +++ b/internal/rules/section_materialization_test.go @@ -79,6 +79,77 @@ func TestRequiredSectionMaterializations_TableContract(t *testing.T) { } } +func TestRequiredSectionMaterializations_QualifiedSelectorPolicy(t *testing.T) { + const qualified = `body.section["## Parent"]["### Notes"]` + field := SchemaField{Type: "string", Required: true, Extract: qualified} + + t.Run("simple required absence keeps materialization", func(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{ + "notes": {Type: "string", Required: true, Extract: `body.section["## Notes"]`}, + }} + got, err := RequiredSectionMaterializations(emptyRecord(), stem) + if err != nil { + t.Fatal(err) + } + if len(got) != 1 || got[0].Heading != "## Notes" { + t.Fatalf("materializations = %+v, want the simple section", got) + } + }) + + t.Run("qualified required presence needs no materialization", func(t *testing.T) { + record := &extract.Record{Frontmatter: map[string]any{}, Body: "## Parent\n\n### Notes\n\npresent\n"} + got, err := RequiredSectionMaterializations(record, &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want no materialization", got, err) + } + }) + + t.Run("qualified optional absence needs no materialization", func(t *testing.T) { + field.Required = false + got, err := RequiredSectionMaterializations(emptyRecord(), &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want no materialization", got, err) + } + }) + + t.Run("qualified required absence returns field and canonical selector", func(t *testing.T) { + field.Required = true + got, err := RequiredSectionMaterializations(emptyRecord(), &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err == nil { + t.Fatalf("expected an error, got materializations %+v", got) + } + if !strings.Contains(err.Error(), `field "notes"`) || !strings.Contains(err.Error(), qualified) { + t.Fatalf("error = %q, want field and canonical selector", err) + } + }) + + t.Run("frontmatter presence has precedence over an ambiguous selector", func(t *testing.T) { + record := &extract.Record{ + Frontmatter: map[string]any{"notes": "override"}, + Body: "## Parent\n\n### Notes\n\nfirst\n\n## Parent\n\n### Notes\n\nsecond\n", + } + got, err := RequiredSectionMaterializations(record, &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want frontmatter to satisfy the field", got, err) + } + }) +} + +func TestRequiredSectionMaterializations_OptionalSelectorAmbiguityReturnsError(t *testing.T) { + record := &extract.Record{ + Frontmatter: map[string]any{}, + Body: "## Parent A\n\n### Notes\n\nfirst\n\n## Parent B\n\n### Notes\n\nsecond\n", + } + stem := &StemFile{Schema: map[string]SchemaField{ + "notes": {Type: "string", Extract: `body.section["### Notes"]`}, + }} + + got, err := RequiredSectionMaterializations(record, stem) + if err == nil || !strings.Contains(err.Error(), "ambiguous") { + t.Fatalf("materializations = %+v, error = %v; want ambiguity", got, err) + } +} + func TestRequiredSectionMaterializations_OrdersByHeadingThenField(t *testing.T) { stem := &StemFile{Schema: map[string]SchemaField{ "gamma": {Type: "string", Required: true, Extract: `body.section["## Shared"]`, Default: "same"}, From 08bab6dd6f5bb32f5cc77aad706aa7b0644dec49 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 01:38:42 -0600 Subject: [PATCH 10/18] fix(scaffold): preserve materialization eligibility --- internal/rules/section_materialization.go | 8 +++---- .../rules/section_materialization_test.go | 23 +++++++++++++------ 2 files changed, 20 insertions(+), 11 deletions(-) diff --git a/internal/rules/section_materialization.go b/internal/rules/section_materialization.go index 223004c6..2ff75b0e 100644 --- a/internal/rules/section_materialization.go +++ b/internal/rules/section_materialization.go @@ -51,6 +51,9 @@ func RequiredSectionMaterializations(record *extract.Record, effective *StemFile out := make([]SectionMaterialization, 0) for _, name := range fields { field := local.Schema[name] + if !field.Required || !requiredCheckApplies(record, &local, name, field) { + continue + } if field.Extract == "" || field.Type != "string" { continue } @@ -65,10 +68,7 @@ func RequiredSectionMaterializations(record *extract.Record, effective *StemFile if err != nil { return nil, err } - if present || !field.Required { - continue - } - if !requiredCheckApplies(record, &local, name, field) { + if present { continue } if len(source.Selector) > 1 { diff --git a/internal/rules/section_materialization_test.go b/internal/rules/section_materialization_test.go index 9104f31a..53dee0c7 100644 --- a/internal/rules/section_materialization_test.go +++ b/internal/rules/section_materialization_test.go @@ -135,18 +135,27 @@ func TestRequiredSectionMaterializations_QualifiedSelectorPolicy(t *testing.T) { }) } -func TestRequiredSectionMaterializations_OptionalSelectorAmbiguityReturnsError(t *testing.T) { +func TestRequiredSectionMaterializations_IneligibleSelectorAmbiguityNeedsNoResolution(t *testing.T) { record := &extract.Record{ Frontmatter: map[string]any{}, Body: "## Parent A\n\n### Notes\n\nfirst\n\n## Parent B\n\n### Notes\n\nsecond\n", } - stem := &StemFile{Schema: map[string]SchemaField{ - "notes": {Type: "string", Extract: `body.section["### Notes"]`}, - }} + tests := []struct { + name string + field SchemaField + }{ + {"optional field", SchemaField{Type: "string", Extract: `body.section["### Notes"]`}}, + {"severity off", SchemaField{Type: "string", Required: true, Severity: "off", Extract: `body.section["### Notes"]`}}, + } - got, err := RequiredSectionMaterializations(record, stem) - if err == nil || !strings.Contains(err.Error(), "ambiguous") { - t.Fatalf("materializations = %+v, error = %v; want ambiguity", got, err) + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{"notes": tt.field}} + got, err := RequiredSectionMaterializations(record, stem) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want no materialization", got, err) + } + }) } } From 1e2aad4fa192fe6d42ecaffc56d1bd979b95a363 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 02:34:32 -0600 Subject: [PATCH 11/18] docs: document hierarchical body section selectors --- .claude/skills/rootline/SKILL.md | 4 +++- .claude/skills/rootline/ref-advanced.md | 4 +++- .claude/skills/rootline/ref-schema.md | 2 +- .claude/skills/rootline/ref-validate.md | 7 +++++-- CLAUDE.md | 4 ++-- docs/analyze.md | 2 +- docs/extensibility.md | 4 ++-- docs/init.md | 2 +- docs/levels.md | 2 +- docs/migrate.md | 4 +++- docs/new.md | 4 +++- docs/validate.md | 2 +- 12 files changed, 26 insertions(+), 15 deletions(-) diff --git a/.claude/skills/rootline/SKILL.md b/.claude/skills/rootline/SKILL.md index 3686a3ce..13e4e5fb 100644 --- a/.claude/skills/rootline/SKILL.md +++ b/.claude/skills/rootline/SKILL.md @@ -206,7 +206,9 @@ Do not apply results automatically. For `apply`, inspect the report first and tr ## Canonical Field Contract -Use only `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`; do not coerce YAML values. A body section pairs a real type with `source: body.section["## Heading"]`. Frontmatter is an explicit override; an empty section is present, duplicate matching headings fail, and child omission inherits a stable source binding. `new` and `migrate --scaffold` add missing required sections in lexical heading order with a non-empty default or ``. Validation error paths are governance-root-relative while symbolic sources remain symbolic. Ancestor-qualified selectors are deferred to #190. +Use only `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Do not coerce YAML values. A body section pairs a real type with a source such as `body.section["## Heading"]` or `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous. The selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. Frontmatter is an explicit override. An empty section is present. Child omission inherits a stable source binding. + +`new` and `migrate --scaffold` add missing required simple sections in lexical heading order with a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Inference can emit the shortest common selector. It does not always emit a qualified selector. Validation error paths are governance-root-relative while symbolic sources remain symbolic. ## Reference Files diff --git a/.claude/skills/rootline/ref-advanced.md b/.claude/skills/rootline/ref-advanced.md index a7883444..49357a95 100644 --- a/.claude/skills/rootline/ref-advanced.md +++ b/.claude/skills/rootline/ref-advanced.md @@ -133,7 +133,9 @@ Use `--field summary` only after confirming the analyze JSON contains that path. ## Canonical schema transport -`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus `source: body.section["## Heading"]`. Inference preserves exact headings, makes partial-frequency candidates optional, and fails logical-name collisions. `new` and `migrate --scaffold` materialize missing required sections in lexical heading order with a non-empty default or ``; frontmatter overrides are never written as empty shadow keys. +`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus a source such as `body.section["## Heading"]` or `body.section["## Parent"]["### Notes"]`. Each selector component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. + +Inference preserves exact headings. It can emit the shortest common selector. The shortest common selector can be simple, so inference does not always emit a qualified selector. Inference makes partial-frequency candidates optional and fails logical-name collisions. `new` and `migrate --scaffold` materialize missing required simple sections in lexical heading order with a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Frontmatter overrides are never written as empty shadow keys. ## apply (Removed) diff --git a/.claude/skills/rootline/ref-schema.md b/.claude/skills/rootline/ref-schema.md index c4afcc62..27992d8f 100644 --- a/.claude/skills/rootline/ref-schema.md +++ b/.claude/skills/rootline/ref-schema.md @@ -149,7 +149,7 @@ Use these instead of hand-rolling `WalkUp` + `entries[0]` indexing in new comman - `source` — a logical body extraction directive when the field has one - `defined_in` — the physical `.stem` that declares the field -A source-backed field uses a real type plus `source: body.section["## Heading"]`; frontmatter is an override. Child omission inherits the stable source binding. `new` and `migrate --scaffold` materialize missing required sections in lexical heading order using a non-empty default or ``. +A source-backed field uses a real type plus a source such as `body.section["## Heading"]` or `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. Frontmatter is an override. Child omission inherits the stable source binding. `new` and `migrate --scaffold` materialize missing required simple sections in lexical heading order using a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Author required section-backed fields in `.stem` like this; `defined_in` appears only in command output, not in authored declarations: diff --git a/.claude/skills/rootline/ref-validate.md b/.claude/skills/rootline/ref-validate.md index 69de6a80..53af2805 100644 --- a/.claude/skills/rootline/ref-validate.md +++ b/.claude/skills/rootline/ref-validate.md @@ -113,8 +113,11 @@ Link-check rules (emitted when the effective `.stem` sets `links.checks`): `link Schema fields with `source:` directives (body-extracted fields) now participate in validation. The `required` and `enum` constraints apply to values extracted from the document body: **Extraction directives**: -- `source: body.h1` — extracts the text of the first H1 heading (e.g., `# My Document` → `"My Document"`) -- `source: body.section["## Heading"]` — extracts content under the named section (e.g., `## Notes` with content below it) +- `source: body.h1` extracts the text of the first H1 heading, such as `"My Document"` from `# My Document`. +- `source: body.section["## Heading"]` extracts content under one exact heading. +- `source: body.section["## Parent"]["### Notes"]` extracts content from a hierarchical section. Each component identifies one exact heading. The components must be contiguous. The selector matches a contiguous suffix of the heading path. + +A simple section selector keeps its existing behavior. More than one match is ambiguous. Rootline does not select the first or last match. A required qualified selector with no match fails validation. **Precedence**: Frontmatter takes absolute precedence. If a field key exists in the record's YAML frontmatter, that value is used and body extraction is skipped. diff --git a/CLAUDE.md b/CLAUDE.md index 9b96011a..35b58709 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -184,9 +184,9 @@ Schema fields with `source:` directives (e.g., `source: body.h1` or `source: bod ## Canonical Field Contract -Use exactly `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`; validation does not coerce YAML representations. A body section is a source binding on a real type, such as `source: body.section["## Summary"]`. Frontmatter is an explicit override, duplicate matching headings fail, and an inherited source binding is stable: a child omission inherits it, while a change or removal conflicts. +Use exactly `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Validation does not coerce YAML representations. A body section is a source binding on a real type. A simple binding can use `source: body.section["## Summary"]`. A qualified binding can use `source: body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous. The selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. Frontmatter is an explicit override. An inherited source binding is stable. A child omission inherits it, while a change or removal conflicts. -`new` and `migrate --scaffold` append missing required sections in lexical heading order, using a non-empty default or ``. `query`, `set`, `describe`, and `explain` resolve the same canonical binding; `set` writes frontmatter only. In public inspection output, logical `source` is separate from physical `defined_in`. Path-like validation errors are relative to the governance root; symbolic sources remain symbolic. Ancestor-qualified selectors are deferred to #190. +`new` and `migrate --scaffold` append missing required simple sections in lexical heading order. They use a non-empty default or ``. They do not invent ancestor headings for qualified selectors. A missing required qualified section stops the command before it writes the affected file. Inference can emit the shortest common selector. The shortest common selector can be simple, so inference does not always emit a qualified selector. `query`, `set`, `describe`, and `explain` resolve the same canonical binding. `set` writes frontmatter only. In public inspection output, logical `source` is separate from physical `defined_in`. Path-like validation errors are relative to the governance root. Symbolic sources remain symbolic. ## Module Path diff --git a/docs/analyze.md b/docs/analyze.md index d8662416..32312729 100644 --- a/docs/analyze.md +++ b/docs/analyze.md @@ -190,7 +190,7 @@ notes: source: body.section["## Notes"] ``` -Every record contributes to the denominator. A candidate is optional unless the heading occurs in every record. Distinct exact headings that normalize to one logical name are a logical-name collision: analysis fails with each colliding heading and requires explicit names. Analyze, schema proposal, and schema application preserve the canonical source identity. +Every record contributes to the denominator. A candidate is optional unless the heading occurs in every record. For a hierarchical section, analysis can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so analysis does not always emit a qualified selector. Distinct exact headings that normalize to one logical name are a logical-name collision. Analysis fails with each colliding heading and requires explicit names. Analyze, schema proposal, and schema application preserve the canonical source identity. ## Filtering with --incremental diff --git a/docs/extensibility.md b/docs/extensibility.md index 3e2ab04b..d18058c5 100644 --- a/docs/extensibility.md +++ b/docs/extensibility.md @@ -50,9 +50,9 @@ schema: default: "" ``` -`source: body.h1` and exact `body.section[...]` directives are supported. Frontmatter is an explicit override. An empty section is present with value `""`; duplicate matching headings fail rather than selecting one. Extractors preserve source identity so validation, inference, schema proposal, and schema application use the same canonical directive. +`source: body.h1` and exact `body.section[...]` directives are supported. The syntax `body.section["## Parent"]["### Notes"]` selects a hierarchical section. Each component identifies one exact heading, including its level and text. The components must be contiguous. The complete selector matches a contiguous suffix of the heading path. A simple selector such as `body.section["## Summary"]` keeps its existing behavior. More than one matching path is ambiguous. Rootline does not select the first or last path. -Source-backed fields participate in validation, querying, describe/explain output, and scaffolding. `rootline set` writes a frontmatter override for such a field; it does not edit the body section. +Frontmatter is an explicit override. An empty section is present with value `""`. Extractors preserve source identity so validation, inference, schema proposal, and schema application use the same canonical directive. Source-backed fields participate in validation, querying, describe/explain output, and scaffolding. `rootline set` writes a frontmatter override for such a field. It does not edit the body section. > LSP integration has been considered but carries very high complexity. > It is not in scope. diff --git a/docs/init.md b/docs/init.md index d431a14b..eec3c670 100644 --- a/docs/init.md +++ b/docs/init.md @@ -107,7 +107,7 @@ schema: default: "" ``` -Headings below the 0.80 threshold are omitted from the generated `.stem`. The source preserves exact heading level and text. Frontmatter remains an explicit override, and generated schemas validate the source corpus; distinct exact headings that normalize to one logical field fail with a collision instead of receiving an invented suffix. +Headings below the 0.80 threshold are omitted from the generated `.stem`. The source preserves each exact heading level and text. For a hierarchical section, inference can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies an exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so inference does not always emit a qualified selector. Frontmatter remains an explicit override. Generated schemas validate the source corpus. Distinct exact headings that normalize to one logical field fail with a collision instead of receiving an invented suffix. ### Source-Backed Field Properties diff --git a/docs/levels.md b/docs/levels.md index 3c7a2fc8..9193e209 100644 --- a/docs/levels.md +++ b/docs/levels.md @@ -92,7 +92,7 @@ For example: - Child: `estado: { type: enum, values: [draft, active] }` ✓ Valid narrowing - Child: `estado: { type: string, required: false }` ✗ Invalid (widening if parent required) -When a child `.stem` violates monotonic constraints, `rootline validate --all` detects this in the **monotonic-violations** stemhealth check. Ancestor-qualified section selectors remain deferred to #190; current `body.section[...]` directives always identify one exact heading. +When a child `.stem` violates monotonic constraints, `rootline validate --all` detects this in the **monotonic-violations** stemhealth check. A hierarchical source binding can use `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. A child cannot change an inherited simple or qualified source binding. ## Benefits diff --git a/docs/migrate.md b/docs/migrate.md index 12275c7f..8ea23ef3 100644 --- a/docs/migrate.md +++ b/docs/migrate.md @@ -194,7 +194,9 @@ Running `rootline migrate --scaffold` on a file missing both sections produces: ``` -Sections are inserted at the end of the document body. Multiple additions are appended in lexical heading order. A frontmatter override or an empty-present section needs no materialization. Ambiguous source resolution and invalid declarations fail rather than becoming a successful no-op. Scaffold validates prospective bytes before its atomic write. Use `--dry-run` to review insertions before applying. +Simple sections are inserted at the end of the document body. Multiple simple additions are appended in lexical heading order. A frontmatter override or an empty-present section needs no materialization. Ambiguous source resolution and invalid declarations fail rather than becoming a successful no-op. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `migrate --scaffold` does not invent ancestor headings. If a required qualified section is absent, scaffold stops before it writes the affected file. Scaffold validates prospective bytes before its atomic write. Use `--dry-run` to review insertions before applying. ## Schema Evolution diff --git a/docs/new.md b/docs/new.md index 8c7c6f0b..cf61fa25 100644 --- a/docs/new.md +++ b/docs/new.md @@ -66,4 +66,6 @@ summary: required: true ``` -`new` adds a missing exact section with its non-empty `default:` or ``. Multiple missing sections use lexical heading order. A frontmatter override or an empty-present section already satisfies presence, and duplicate matching headings fail before a file is written. Generated bytes are prospectively validated against the same effective schema. +`new` adds a missing simple section with its non-empty `default:` or ``. Multiple missing simple sections use lexical heading order. A frontmatter override or an empty-present section already satisfies presence. More than one selector match is ambiguous and stops the command before a file is written. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `new` does not invent ancestor headings. If a required qualified section is absent, `new` stops before it writes the affected file. Generated bytes are prospectively validated against the same effective schema. diff --git a/docs/validate.md b/docs/validate.md index 49949396..b350145b 100644 --- a/docs/validate.md +++ b/docs/validate.md @@ -340,7 +340,7 @@ summary: `required means presence`: an empty section is present with value `""`; use `non_empty` when content is required, and `exists` for an effective field including derived values. Frontmatter has precedence and is a deliberate override. -A `body.section[...]` directive matches its exact heading level and text, and duplicate matching headings fail as ambiguous; Rootline does not select the first or last. The same source resolver serves single-file and batch validation, query, describe, and explain. +A component in a `body.section[...]` directive matches one exact heading level and text. `body.section["## Parent"]["### Notes"]` matches a contiguous suffix of the heading path. All selector components must be contiguous. A simple selector keeps its existing behavior. More than one match is ambiguous. Rootline does not select the first or last match. A required qualified selector with no match fails validation. The same source resolver serves single-file and batch validation, query, describe, and explain. Public path-like error `source` values are governance-root-relative with `/` separators. Symbolic sources remain symbolic, so results are stable across working directories and `--all` scan roots. From 5bfe17e1b34b470bace2a6d86b84d37689b8d153 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 02:39:14 -0600 Subject: [PATCH 12/18] docs: preserve section documentation contracts --- .claude/skills/rootline/ref-advanced.md | 2 +- .claude/skills/rootline/ref-validate.md | 8 ++++---- docs/migrate.md | 2 +- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/.claude/skills/rootline/ref-advanced.md b/.claude/skills/rootline/ref-advanced.md index 49357a95..3941fa1b 100644 --- a/.claude/skills/rootline/ref-advanced.md +++ b/.claude/skills/rootline/ref-advanced.md @@ -133,7 +133,7 @@ Use `--field summary` only after confirming the analyze JSON contains that path. ## Canonical schema transport -`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus a source such as `body.section["## Heading"]` or `body.section["## Parent"]["### Notes"]`. Each selector component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. +`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus a source such as `source: body.section["## Heading"]` or `source: body.section["## Parent"]["### Notes"]`. A simple selector identifies the final heading by exact level and text. A qualified selector matches a contiguous suffix of the hierarchical heading path. More than one matching path is ambiguous. Inference preserves exact headings. It can emit the shortest common selector. The shortest common selector can be simple, so inference does not always emit a qualified selector. Inference makes partial-frequency candidates optional and fails logical-name collisions. `new` and `migrate --scaffold` materialize missing required simple sections in lexical heading order with a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Frontmatter overrides are never written as empty shadow keys. diff --git a/.claude/skills/rootline/ref-validate.md b/.claude/skills/rootline/ref-validate.md index 53af2805..68274b48 100644 --- a/.claude/skills/rootline/ref-validate.md +++ b/.claude/skills/rootline/ref-validate.md @@ -114,10 +114,10 @@ Schema fields with `source:` directives (body-extracted fields) now participate **Extraction directives**: - `source: body.h1` extracts the text of the first H1 heading, such as `"My Document"` from `# My Document`. -- `source: body.section["## Heading"]` extracts content under one exact heading. -- `source: body.section["## Parent"]["### Notes"]` extracts content from a hierarchical section. Each component identifies one exact heading. The components must be contiguous. The selector matches a contiguous suffix of the heading path. +- `source: body.section["## Heading"]` identifies the final heading by exact level and text. It extracts the content under that heading. +- `source: body.section["## Parent"]["### Notes"]` matches a contiguous suffix of the hierarchical heading path. It extracts content from the matched section. -A simple section selector keeps its existing behavior. More than one match is ambiguous. Rootline does not select the first or last match. A required qualified selector with no match fails validation. +Only multiple paths that match the complete selector produce ambiguity. Rootline does not select the first or last matching path. A required qualified selector with no match fails validation. **Precedence**: Frontmatter takes absolute precedence. If a field key exists in the record's YAML frontmatter, that value is used and body extraction is skipped. @@ -156,7 +156,7 @@ In validation: ### Source and provenance -`body.section[...]` matches an exact heading level and text. Duplicate matching headings fail rather than choosing an occurrence; frontmatter remains an override. Path-like validation error sources are governance-root-relative, while symbolic sources stay symbolic. +A simple `body.section[...]` selector identifies the final heading by exact level and text. A qualified selector matches a contiguous suffix of the hierarchical heading path. Only multiple paths that match the complete selector produce ambiguity. Frontmatter remains an override. Path-like validation error sources are governance-root-relative, while symbolic sources stay symbolic. ### Reporting Format diff --git a/docs/migrate.md b/docs/migrate.md index 8ea23ef3..adf989db 100644 --- a/docs/migrate.md +++ b/docs/migrate.md @@ -194,7 +194,7 @@ Running `rootline migrate --scaffold` on a file missing both sections produces: ``` -Simple sections are inserted at the end of the document body. Multiple simple additions are appended in lexical heading order. A frontmatter override or an empty-present section needs no materialization. Ambiguous source resolution and invalid declarations fail rather than becoming a successful no-op. +Sections are inserted at the end of the document body only when they use simple selectors and need materialization. Multiple simple additions are appended in lexical heading order. A frontmatter override or an empty-present section needs no materialization. Ambiguous source resolution and invalid declarations fail rather than becoming a successful no-op. A qualified selector can use `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `migrate --scaffold` does not invent ancestor headings. If a required qualified section is absent, scaffold stops before it writes the affected file. Scaffold validates prospective bytes before its atomic write. Use `--dry-run` to review insertions before applying. From c6f2fe14a3be892b20c5fdefc8d187f55a85f913 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 02:51:57 -0600 Subject: [PATCH 13/18] fix(infer): support multiline setext selectors --- internal/extract/body_source.go | 3 - internal/extract/body_source_test.go | 66 +++++++++++++++++ internal/infer/setext_multiline_test.go | 96 +++++++++++++++++++++++++ 3 files changed, 162 insertions(+), 3 deletions(-) create mode 100644 internal/infer/setext_multiline_test.go diff --git a/internal/extract/body_source.go b/internal/extract/body_source.go index 20b88ff4..0c4d58ca 100644 --- a/internal/extract/body_source.go +++ b/internal/extract/body_source.go @@ -217,9 +217,6 @@ func exactHeading(key HeadingKey) string { } func parseHeadingKey(heading string) (HeadingKey, error) { - if strings.ContainsAny(heading, "\r\n") { - return HeadingKey{}, fmt.Errorf("section source heading must be an exact markdown heading, got %q", heading) - } level, text, ok := parseATXHeading(heading) key := HeadingKey{Level: level, Text: text} if !ok || level == 0 || exactHeading(key) != heading { diff --git a/internal/extract/body_source_test.go b/internal/extract/body_source_test.go index c29d4e58..b89a4dba 100644 --- a/internal/extract/body_source_test.go +++ b/internal/extract/body_source_test.go @@ -53,6 +53,7 @@ func TestParseBodySourceRejectsInvalidSelectors(t *testing.T) { {`body.section["Notes"]`, "heading"}, {`body.section["## Parent"]["## Notes"]`, "increase"}, {`body.section["### Parent"]["## Notes"]`, "increase"}, + {"body.section[\"## First\nSecond\"]", "malformed"}, } { _, err := ParseBodySource(tt.directive) if err == nil || !strings.Contains(err.Error(), tt.wantErr) { @@ -94,6 +95,71 @@ func TestCanonicalSectionSource(t *testing.T) { } } +func TestCanonicalSectionSelectorSource_MultilineHeadingRoundTrip(t *testing.T) { + wantSelector := SectionSelector{ + {Level: 1, Text: "Root"}, + {Level: 2, Text: "First\nSecond"}, + } + got, err := CanonicalSectionSelectorSource(wantSelector) + if err != nil { + t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) + } + wantSource := `body.section["# Root"]["## First\nSecond"]` + if got != wantSource { + t.Fatalf("source = %q, want %q", got, wantSource) + } + + parsed, err := ParseBodySource(got) + if err != nil { + t.Fatalf("ParseBodySource returned an error: %v", err) + } + if !reflect.DeepEqual(parsed.Selector, wantSelector) { + t.Fatalf("selector = %+v, want %+v", parsed.Selector, wantSelector) + } + if parsed.Heading != "## First\nSecond" { + t.Fatalf("heading = %q, want the complete multiline heading", parsed.Heading) + } +} + +func TestResolveBodyValue_HeadingSyntaxCompatibility(t *testing.T) { + for _, tt := range []struct { + name string + body string + selector SectionSelector + wantValue string + }{ + { + name: "ATX", + body: "## Notes\n\nATX content\n", + selector: SectionSelector{{Level: 2, Text: "Notes"}}, + wantValue: "ATX content", + }, + { + name: "simple Setext", + body: "Notes\n---\n\nSetext content\n", + selector: SectionSelector{{Level: 2, Text: "Notes"}}, + wantValue: "Setext content", + }, + { + name: "multiline Setext hierarchy", + body: "# Root\n\nFirst\nSecond\n---\n\nMultiline content\n", + selector: SectionSelector{{Level: 1, Text: "Root"}, {Level: 2, Text: "First\nSecond"}}, + wantValue: "Multiline content", + }, + } { + t.Run(tt.name, func(t *testing.T) { + source, err := CanonicalSectionSelectorSource(tt.selector) + if err != nil { + t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) + } + got, present, err := ResolveBodyValue(&Record{Body: tt.body}, source) + if err != nil || !present || got != tt.wantValue { + t.Fatalf("value = %q, present = %v, error = %v; want %q", got, present, err, tt.wantValue) + } + }) + } +} + func TestResolveBodyValue_PresentEmptySection(t *testing.T) { rec := &Record{BodySections: []Section{{Heading: "Notes", Level: 2, Content: ""}}} got, present, err := ResolveBodyValue(rec, `body.section["## Notes"]`) diff --git a/internal/infer/setext_multiline_test.go b/internal/infer/setext_multiline_test.go new file mode 100644 index 00000000..6d519fad --- /dev/null +++ b/internal/infer/setext_multiline_test.go @@ -0,0 +1,96 @@ +package infer + +import ( + "context" + "reflect" + "testing" + + "github.com/pablontiv/rootline/internal/extract" + "github.com/pablontiv/rootline/internal/rules" +) + +const multilineSetextSource = `body.section["## First\nSecond"]` + +func TestDetectSectionPatterns_MultilineSetext(t *testing.T) { + record := makeRecord("First\nSecond\n---\n\nContent\n") + sections := extract.ExtractSections(record.AST, []byte(record.Body)) + if len(sections) != 1 || len(sections[0].Path) != 1 { + t.Fatalf("sections = %+v, want one section path", sections) + } + wantPath := extract.SectionPath{{Level: 2, Text: "First\nSecond"}} + if !reflect.DeepEqual(sections[0].Path, wantPath) { + t.Fatalf("section path = %+v, want %+v", sections[0].Path, wantPath) + } + + inferences, err := DetectSectionPatterns([]*extract.Record{record}, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns rejected a valid multiline Setext heading: %v", err) + } + inf, ok := findSectionInference(inferences, "first_second") + if !ok || inf.Type != "required_section" || inf.SourceDirective != multilineSetextSource { + t.Fatalf("inference = %+v, present = %v", inf, ok) + } + + parsed, err := extract.ParseBodySource(inf.SourceDirective) + if err != nil { + t.Fatalf("ParseBodySource rejected the inferred selector: %v", err) + } + if !reflect.DeepEqual(extract.SectionPath(parsed.Selector), sections[0].Path) { + t.Fatalf("parsed selector = %+v, want section path %+v", parsed.Selector, sections[0].Path) + } + value, present, err := extract.ResolveBodyValue(record, inf.SourceDirective) + if err != nil || !present || value != "Content" { + t.Fatalf("resolved value = %q, present = %v, error = %v", value, present, err) + } +} + +func TestGenerateFlatSchema_MultilineSetext(t *testing.T) { + record := makeRecord("First\nSecond\n---\n\nContent\n") + opts := DefaultInferOptions() + opts.IncludeStructural = false + opts.SectionThreshold = 1 + stem, err := GenerateFlatSchema(context.Background(), ".", []*extract.Record{record}, opts) + if err != nil { + t.Fatalf("GenerateFlatSchema rejected a valid multiline Setext heading: %v", err) + } + field, ok := stem.Schema["first_second"] + if !ok || field.Type != "string" || !field.Required || field.Extract != multilineSetextSource { + t.Fatalf("generated field = %+v, present = %v", field, ok) + } + if errs := rules.Validate(context.Background(), record, stem); len(errs) != 0 { + t.Fatalf("generated schema rejected its source record: %+v", errs) + } +} + +func TestApplySchemaInferences_MultilineSetext(t *testing.T) { + record := makeRecord("First\nSecond\n---\n\nContent\n") + inferences, err := DetectSectionPatterns([]*extract.Record{record}, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns returned an error: %v", err) + } + inf, ok := findSectionInference(inferences, "first_second") + if !ok { + t.Fatalf("multiline Setext inference is missing: %+v", inferences) + } + + stemPath := writeApplyStem(t, "version: 2\nschema: {}\n") + result, err := ApplySchemaInferences(stemPath, []ReportInference{{ + Type: inf.Type, + Field: inf.Field, + SourceDirective: inf.SourceDirective, + }}, false) + if err != nil { + t.Fatalf("ApplySchemaInferences rejected the inferred selector: %v", err) + } + if len(result.Applied) != 1 { + t.Fatalf("applied actions = %v, want one action", result.Applied) + } + field := readApplyStem(t, stemPath).Schema["first_second"] + if field.Type != "string" || !field.Required || field.Extract != multilineSetextSource { + t.Fatalf("applied field = %+v", field) + } + value, present, err := extract.ResolveBodyValue(record, field.Extract) + if err != nil || !present || value != "Content" { + t.Fatalf("applied selector resolved value = %q, present = %v, error = %v", value, present, err) + } +} From 7da6f8546023e9c5c68415e875e5368c160bad1b Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 02:52:17 -0600 Subject: [PATCH 14/18] docs: clarify section inference threshold --- .claude/skills/rootline/ref-advanced.md | 2 +- docs/analyze.md | 2 +- docs/init.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/.claude/skills/rootline/ref-advanced.md b/.claude/skills/rootline/ref-advanced.md index 3941fa1b..510c7505 100644 --- a/.claude/skills/rootline/ref-advanced.md +++ b/.claude/skills/rootline/ref-advanced.md @@ -135,7 +135,7 @@ Use `--field summary` only after confirming the analyze JSON contains that path. `init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus a source such as `source: body.section["## Heading"]` or `source: body.section["## Parent"]["### Notes"]`. A simple selector identifies the final heading by exact level and text. A qualified selector matches a contiguous suffix of the hierarchical heading path. More than one matching path is ambiguous. -Inference preserves exact headings. It can emit the shortest common selector. The shortest common selector can be simple, so inference does not always emit a qualified selector. Inference makes partial-frequency candidates optional and fails logical-name collisions. `new` and `migrate --scaffold` materialize missing required simple sections in lexical heading order with a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Frontmatter overrides are never written as empty shadow keys. +Inference preserves exact headings. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. Rootline can emit the shortest common selector. The shortest common selector can be simple, so inference does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference. Rootline does not invent logical names. Rootline makes a partial-frequency family optional when that family reaches the threshold. `new` and `migrate --scaffold` materialize missing required simple sections in lexical heading order with a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Frontmatter overrides are never written as empty shadow keys. ## apply (Removed) diff --git a/docs/analyze.md b/docs/analyze.md index 32312729..f4f7cca1 100644 --- a/docs/analyze.md +++ b/docs/analyze.md @@ -190,7 +190,7 @@ notes: source: body.section["## Notes"] ``` -Every record contributes to the denominator. A candidate is optional unless the heading occurs in every record. For a hierarchical section, analysis can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so analysis does not always emit a qualified selector. Distinct exact headings that normalize to one logical name are a logical-name collision. Analysis fails with each colliding heading and requires explicit names. Analyze, schema proposal, and schema application preserve the canonical source identity. +Every record contributes to the denominator. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. A family that reaches the threshold is optional unless the heading occurs in every record. For a hierarchical section, analysis can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so analysis does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference and reports each colliding heading. Rootline requires explicit logical names and does not invent logical names. Analyze, schema proposal, and schema application preserve the canonical source identity. ## Filtering with --incremental diff --git a/docs/init.md b/docs/init.md index eec3c670..80beafd0 100644 --- a/docs/init.md +++ b/docs/init.md @@ -107,7 +107,7 @@ schema: default: "" ``` -Headings below the 0.80 threshold are omitted from the generated `.stem`. The source preserves each exact heading level and text. For a hierarchical section, inference can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies an exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so inference does not always emit a qualified selector. Frontmatter remains an explicit override. Generated schemas validate the source corpus. Distinct exact headings that normalize to one logical field fail with a collision instead of receiving an invented suffix. +Rootline calculates the frequency of each exact heading family. Headings below the 0.80 threshold are omitted from the generated `.stem`. Rootline discards each family that does not reach the threshold. The source preserves each exact heading level and text. For a hierarchical section, inference can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies an exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so inference does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference. Rootline does not invent logical names. Frontmatter remains an explicit override. Generated schemas validate the source corpus. ### Source-Backed Field Properties From 82772fe99c1eb776b20c1d443110daa8b06fa3d4 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 03:07:01 -0600 Subject: [PATCH 15/18] fix(extract): preserve multiline heading hashes --- internal/extract/body_source.go | 21 ++++++++- internal/extract/body_source_test.go | 62 ++++++++++++++++--------- internal/extract/body_test.go | 24 ++++++++-- internal/infer/setext_multiline_test.go | 10 ++-- 4 files changed, 87 insertions(+), 30 deletions(-) diff --git a/internal/extract/body_source.go b/internal/extract/body_source.go index 0c4d58ca..b8c463e3 100644 --- a/internal/extract/body_source.go +++ b/internal/extract/body_source.go @@ -217,7 +217,14 @@ func exactHeading(key HeadingKey) string { } func parseHeadingKey(heading string) (HeadingKey, error) { - level, text, ok := parseATXHeading(heading) + var level int + var text string + var ok bool + if strings.Contains(heading, "\n") { + level, text, ok = parseMultilineHeadingKey(heading) + } else { + level, text, ok = parseATXHeading(heading) + } key := HeadingKey{Level: level, Text: text} if !ok || level == 0 || exactHeading(key) != heading { return HeadingKey{}, fmt.Errorf("section source heading must be an exact markdown heading, got %q", heading) @@ -225,6 +232,18 @@ func parseHeadingKey(heading string) (HeadingKey, error) { return key, nil } +func parseMultilineHeadingKey(heading string) (int, string, bool) { + firstLineEnd := strings.IndexByte(heading, '\n') + level := 0 + for level < firstLineEnd && heading[level] == '#' { + level++ + } + if level == 0 || level > 6 || level >= firstLineEnd || heading[level] != ' ' { + return 0, "", false + } + return level, heading[level+1:], true +} + func validateSectionSelector(selector SectionSelector) error { if len(selector) == 0 { return fmt.Errorf("section selector must contain one heading") diff --git a/internal/extract/body_source_test.go b/internal/extract/body_source_test.go index b89a4dba..dec3721c 100644 --- a/internal/extract/body_source_test.go +++ b/internal/extract/body_source_test.go @@ -53,6 +53,10 @@ func TestParseBodySourceRejectsInvalidSelectors(t *testing.T) { {`body.section["Notes"]`, "heading"}, {`body.section["## Parent"]["## Notes"]`, "increase"}, {`body.section["### Parent"]["## Notes"]`, "increase"}, + {`body.section["####### First\nSecond"]`, "heading"}, + {`body.section["##First\nSecond"]`, "heading"}, + {`body.section[" ## First\nSecond"]`, "heading"}, + {`body.section["##\nSecond"]`, "heading"}, {"body.section[\"## First\nSecond\"]", "malformed"}, } { _, err := ParseBodySource(tt.directive) @@ -96,28 +100,38 @@ func TestCanonicalSectionSource(t *testing.T) { } func TestCanonicalSectionSelectorSource_MultilineHeadingRoundTrip(t *testing.T) { - wantSelector := SectionSelector{ - {Level: 1, Text: "Root"}, - {Level: 2, Text: "First\nSecond"}, - } - got, err := CanonicalSectionSelectorSource(wantSelector) - if err != nil { - t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) - } - wantSource := `body.section["# Root"]["## First\nSecond"]` - if got != wantSource { - t.Fatalf("source = %q, want %q", got, wantSource) - } + for _, tt := range []struct { + name string + text string + }{ + {name: "plain", text: "First\nSecond"}, + {name: "trailing hash", text: "First\nSecond #"}, + {name: "trailing hashes", text: "First\nSecond ##"}, + {name: "trailing space", text: "First\nSecond "}, + {name: "trailing backslash", text: "First\nSecond \\"}, + {name: "intermediate hash", text: "First\nMiddle #\nSecond"}, + } { + t.Run(tt.name, func(t *testing.T) { + wantSelector := SectionSelector{ + {Level: 1, Text: "Root"}, + {Level: 2, Text: tt.text}, + } + got, err := CanonicalSectionSelectorSource(wantSelector) + if err != nil { + t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) + } - parsed, err := ParseBodySource(got) - if err != nil { - t.Fatalf("ParseBodySource returned an error: %v", err) - } - if !reflect.DeepEqual(parsed.Selector, wantSelector) { - t.Fatalf("selector = %+v, want %+v", parsed.Selector, wantSelector) - } - if parsed.Heading != "## First\nSecond" { - t.Fatalf("heading = %q, want the complete multiline heading", parsed.Heading) + parsed, err := ParseBodySource(got) + if err != nil { + t.Fatalf("ParseBodySource returned an error for %q: %v", got, err) + } + if !reflect.DeepEqual(parsed.Selector, wantSelector) { + t.Fatalf("selector = %+v, want %+v", parsed.Selector, wantSelector) + } + if parsed.Heading != exactHeading(wantSelector[1]) { + t.Fatalf("heading = %q, want %q", parsed.Heading, exactHeading(wantSelector[1])) + } + }) } } @@ -146,6 +160,12 @@ func TestResolveBodyValue_HeadingSyntaxCompatibility(t *testing.T) { selector: SectionSelector{{Level: 1, Text: "Root"}, {Level: 2, Text: "First\nSecond"}}, wantValue: "Multiline content", }, + { + name: "multiline Setext trailing hash", + body: "First\nSecond #\n---\n\nMultiline content\n", + selector: SectionSelector{{Level: 2, Text: "First\nSecond #"}}, + wantValue: "Multiline content", + }, } { t.Run(tt.name, func(t *testing.T) { source, err := CanonicalSectionSelectorSource(tt.selector) diff --git a/internal/extract/body_test.go b/internal/extract/body_test.go index 06e8b92a..43cb3a2b 100644 --- a/internal/extract/body_test.go +++ b/internal/extract/body_test.go @@ -174,10 +174,12 @@ func TestExtractSectionsFromText_SetextParity(t *testing.T) { name string body string wantHeading string + wantContent string wantLevel int }{ - {name: "simple", body: "First\n===\n\nContent\n", wantHeading: "First", wantLevel: 1}, - {name: "multiple lines", body: "First\nSecond\n---\n\nContent\n", wantHeading: "First\nSecond", wantLevel: 2}, + {name: "simple", body: "First\n===\n\nContent\n", wantHeading: "First", wantContent: "Content", wantLevel: 1}, + {name: "multiple lines", body: "First\nSecond\n---\n\nContent\n", wantHeading: "First\nSecond", wantContent: "Content", wantLevel: 2}, + {name: "multiple lines with trailing hash", body: "First\nSecond #\n---\n", wantHeading: "First\nSecond #", wantLevel: 2}, } { t.Run(tt.name, func(t *testing.T) { astSections := parseSections(tt.body) @@ -189,7 +191,7 @@ func TestExtractSectionsFromText_SetextParity(t *testing.T) { t.Fatalf("section count = %d; want 1", len(textSections)) } wantPath := SectionPath{{Level: tt.wantLevel, Text: tt.wantHeading}} - want := Section{Heading: tt.wantHeading, Level: tt.wantLevel, Path: wantPath, Content: "Content", StartLine: 1} + want := Section{Heading: tt.wantHeading, Level: tt.wantLevel, Path: wantPath, Content: tt.wantContent, StartLine: 1} if !reflect.DeepEqual(textSections[0], want) { t.Fatalf("section = %+v; want %+v", textSections[0], want) } @@ -197,6 +199,22 @@ func TestExtractSectionsFromText_SetextParity(t *testing.T) { } } +func TestParseATXHeading_ClosingSequence(t *testing.T) { + for _, tt := range []struct { + line string + wantText string + }{ + {line: "## Heading ##", wantText: "Heading"}, + {line: "## Heading # ", wantText: "Heading"}, + {line: "## Heading#", wantText: "Heading#"}, + } { + level, text, ok := parseATXHeading(tt.line) + if !ok || level != 2 || text != tt.wantText { + t.Fatalf("parseATXHeading(%q) = %d, %q, %v; want 2, %q, true", tt.line, level, text, ok, tt.wantText) + } + } +} + func TestExtractSections_NoHeadings(t *testing.T) { body := "Just a paragraph.\n\nAnother paragraph.\n" sections := parseSections(body) diff --git a/internal/infer/setext_multiline_test.go b/internal/infer/setext_multiline_test.go index 6d519fad..05d0ab84 100644 --- a/internal/infer/setext_multiline_test.go +++ b/internal/infer/setext_multiline_test.go @@ -9,15 +9,15 @@ import ( "github.com/pablontiv/rootline/internal/rules" ) -const multilineSetextSource = `body.section["## First\nSecond"]` +const multilineSetextSource = `body.section["## First\nSecond #"]` func TestDetectSectionPatterns_MultilineSetext(t *testing.T) { - record := makeRecord("First\nSecond\n---\n\nContent\n") + record := makeRecord("First\nSecond #\n---\n\nContent\n") sections := extract.ExtractSections(record.AST, []byte(record.Body)) if len(sections) != 1 || len(sections[0].Path) != 1 { t.Fatalf("sections = %+v, want one section path", sections) } - wantPath := extract.SectionPath{{Level: 2, Text: "First\nSecond"}} + wantPath := extract.SectionPath{{Level: 2, Text: "First\nSecond #"}} if !reflect.DeepEqual(sections[0].Path, wantPath) { t.Fatalf("section path = %+v, want %+v", sections[0].Path, wantPath) } @@ -45,7 +45,7 @@ func TestDetectSectionPatterns_MultilineSetext(t *testing.T) { } func TestGenerateFlatSchema_MultilineSetext(t *testing.T) { - record := makeRecord("First\nSecond\n---\n\nContent\n") + record := makeRecord("First\nSecond #\n---\n\nContent\n") opts := DefaultInferOptions() opts.IncludeStructural = false opts.SectionThreshold = 1 @@ -63,7 +63,7 @@ func TestGenerateFlatSchema_MultilineSetext(t *testing.T) { } func TestApplySchemaInferences_MultilineSetext(t *testing.T) { - record := makeRecord("First\nSecond\n---\n\nContent\n") + record := makeRecord("First\nSecond #\n---\n\nContent\n") inferences, err := DetectSectionPatterns([]*extract.Record{record}, 1) if err != nil { t.Fatalf("DetectSectionPatterns returned an error: %v", err) From d943d3723406533478fe6a838e9045aa8f083537 Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 03:25:18 -0600 Subject: [PATCH 16/18] fix(extract): preserve canonical heading keys --- cmd/rootline/migrate_scaffold_source_test.go | 52 ++++++++ cmd/rootline/new.go | 2 +- cmd/rootline/new_source_test.go | 33 +++++ internal/extract/body_source.go | 58 ++++++--- internal/extract/body_source_test.go | 122 ++++++++++++++++++ internal/migrate/scaffold.go | 2 +- internal/rules/section_materialization.go | 24 +++- .../section_materialization_edge_test.go | 2 +- .../rules/section_materialization_test.go | 42 ++++++ 9 files changed, 310 insertions(+), 27 deletions(-) diff --git a/cmd/rootline/migrate_scaffold_source_test.go b/cmd/rootline/migrate_scaffold_source_test.go index cac82069..f55c11e9 100644 --- a/cmd/rootline/migrate_scaffold_source_test.go +++ b/cmd/rootline/migrate_scaffold_source_test.go @@ -7,6 +7,7 @@ import ( "strings" "testing" + "github.com/pablontiv/rootline/internal/extract" "github.com/pablontiv/rootline/internal/migrate" ) @@ -69,6 +70,57 @@ func TestMigrateScaffoldRequiredSectionSourceWarningsDoNotBlockWrite(t *testing. } } +func TestMigrateScaffoldMaterializesLosslessSetextHeading(t *testing.T) { + dir := newMigrateScaffoldSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Notes #"]' +`, "") + + result, out, err := runMigrateScaffoldJSON(t, dir) + if err != nil { + t.Fatalf("scaffold failed: %v\noutput: %s", err, out) + } + if result.SectionsAdded != 1 || result.Details[0].Heading != "## Notes #" { + t.Fatalf("result = %+v", result) + } + target := filepath.Join(dir, "T001-task.md") + content := string(mustReadFile(t, target)) + if !strings.Contains(content, "\nNotes #\n---\n\n\n") { + t.Fatalf("scaffolded content does not contain the Setext heading:\n%s", content) + } + record, err := extractProspectiveRecord(target, target, []byte(content)) + if err != nil { + t.Fatal(err) + } + got, present, err := extract.ResolveBodyValue(record, `body.section["## Notes #"]`) + if err != nil || !present || got != "" { + t.Fatalf("extracted value = %q, present = %v, error = %v", got, present, err) + } +} + +func TestMigrateScaffoldRejectsLossyHeadingWithoutWrite(t *testing.T) { + original := "---\ntitle: Task\n---\n# Task\n" + dir := newMigrateScaffoldSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["### Notes #"]' +`, original) + + out, err := runCmd(t, "migrate", "--scaffold", dir) + if err == nil { + t.Fatalf("scaffold output = %s, want an error", out) + } + if !strings.Contains(err.Error(), "without data loss") { + t.Fatalf("error = %q", err) + } + if got := string(mustReadFile(t, filepath.Join(dir, "T001-task.md"))); got != original { + t.Fatalf("affected file changed\ngot: %q\nwant: %q", got, original) + } +} + func TestMigrateScaffoldRequiredSectionSourceRejectsBrokenProspectiveLinkWithoutWrite(t *testing.T) { dir := newMigrateScaffoldSectionSourceProject(t, `schema: notes: diff --git a/cmd/rootline/new.go b/cmd/rootline/new.go index 5cdceb61..8b8eadf8 100644 --- a/cmd/rootline/new.go +++ b/cmd/rootline/new.go @@ -156,7 +156,7 @@ func generateMarkdown(absPath string, effective *rules.StemFile) (string, error) } for _, section := range sections { b.WriteString("\n") - b.WriteString(section.Heading) + b.WriteString(section.MarkdownHeading()) b.WriteString("\n\n") b.WriteString(section.Content) b.WriteString("\n") diff --git a/cmd/rootline/new_source_test.go b/cmd/rootline/new_source_test.go index 9554b304..ee0d5145 100644 --- a/cmd/rootline/new_source_test.go +++ b/cmd/rootline/new_source_test.go @@ -6,6 +6,7 @@ import ( "strings" "testing" + "github.com/pablontiv/rootline/internal/extract" "github.com/pablontiv/rootline/internal/rules" ) @@ -54,6 +55,32 @@ func TestNewSectionSourceMaterializesRequiredSectionsAndValidates(t *testing.T) } } +func TestNewSectionSourceMaterializesLosslessSetextHeading(t *testing.T) { + dir := newSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Notes #"]' +`) + target := filepath.Join(dir, "T001-task.md") + + if out, err := runCmd(t, "new", target); err != nil { + t.Fatalf("new failed: %v\noutput: %s", err, out) + } + content := string(mustReadFile(t, target)) + if !strings.Contains(content, "\nNotes #\n---\n\n\n") { + t.Fatalf("generated content does not contain the Setext heading:\n%s", content) + } + record, err := extractProspectiveNewRecord(target, content) + if err != nil { + t.Fatal(err) + } + got, present, err := extract.ResolveBodyValue(record, `body.section["## Notes #"]`) + if err != nil || !present || got != "" { + t.Fatalf("extracted value = %q, present = %v, error = %v", got, present, err) + } +} + func TestNewSectionSourceUsesRecordPathAwareSchema(t *testing.T) { dir := newSectionSourceProject(t, `schema: task_notes: @@ -133,6 +160,12 @@ func TestNewSectionSourceNoWriteOnSchemaMaterializationOrValidationFailure(t *te type: string required: true source: 'body.section["Notes"]' +`}, + {"heading cannot be materialized without data loss", `schema: + notes: + type: string + required: true + source: 'body.section["### Notes #"]' `}, {"invalid prospective ordinary frontmatter", `schema: status: diff --git a/internal/extract/body_source.go b/internal/extract/body_source.go index b8c463e3..de17f13c 100644 --- a/internal/extract/body_source.go +++ b/internal/extract/body_source.go @@ -217,31 +217,49 @@ func exactHeading(key HeadingKey) string { } func parseHeadingKey(heading string) (HeadingKey, error) { - var level int - var text string - var ok bool - if strings.Contains(heading, "\n") { - level, text, ok = parseMultilineHeadingKey(heading) - } else { - level, text, ok = parseATXHeading(heading) + level := 0 + for level < len(heading) && heading[level] == '#' { + level++ } - key := HeadingKey{Level: level, Text: text} - if !ok || level == 0 || exactHeading(key) != heading { - return HeadingKey{}, fmt.Errorf("section source heading must be an exact markdown heading, got %q", heading) + if level < 1 || level > 6 || level >= len(heading) || heading[level] != ' ' { + return HeadingKey{}, fmt.Errorf("section source heading must contain 1 to 6 hashes and one space, got %q", heading) } - return key, nil + return HeadingKey{Level: level, Text: heading[level+1:]}, nil } -func parseMultilineHeadingKey(heading string) (int, string, bool) { - firstLineEnd := strings.IndexByte(heading, '\n') - level := 0 - for level < firstLineEnd && heading[level] == '#' { - level++ +// MaterializeHeading returns markdown that extracts as key without data loss. +func MaterializeHeading(key HeadingKey) (string, error) { + if err := validateHeadingKey(key); err != nil { + return "", err } - if level == 0 || level > 6 || level >= firstLineEnd || heading[level] != ' ' { - return 0, "", false + + atx := exactHeading(key) + if materializedHeadingMatches(atx, key) { + return atx, nil } - return level, heading[level+1:], true + if key.Level <= 2 { + underline := "---" + if key.Level == 1 { + underline = "===" + } + setext := key.Text + "\n" + underline + if materializedHeadingMatches(setext, key) { + return setext, nil + } + } + return "", fmt.Errorf("heading level %d with text %q cannot be materialized without data loss", key.Level, key.Text) +} + +func materializedHeadingMatches(markdown string, key HeadingKey) bool { + sections := ExtractSectionsFromText(markdown) + return len(sections) == 1 && sections[0].Level == key.Level && sections[0].Heading == key.Text +} + +func validateHeadingKey(key HeadingKey) error { + if key.Level < 1 || key.Level > 6 { + return fmt.Errorf("heading level must be from 1 to 6, got %d", key.Level) + } + return nil } func validateSectionSelector(selector SectionSelector) error { @@ -249,7 +267,7 @@ func validateSectionSelector(selector SectionSelector) error { return fmt.Errorf("section selector must contain one heading") } for i, key := range selector { - if key.Level < 1 || key.Level > 6 || exactHeading(key) == "" { + if err := validateHeadingKey(key); err != nil { return fmt.Errorf("section selector has an invalid heading at index %d", i) } parsed, err := parseHeadingKey(exactHeading(key)) diff --git a/internal/extract/body_source_test.go b/internal/extract/body_source_test.go index dec3721c..a93f2111 100644 --- a/internal/extract/body_source_test.go +++ b/internal/extract/body_source_test.go @@ -2,6 +2,7 @@ package extract import ( "reflect" + "strconv" "strings" "testing" @@ -74,6 +75,56 @@ func TestParseBodySourceAllowsLevelJump(t *testing.T) { } } +func TestParseBodySourceHeadingKeyIsInjective(t *testing.T) { + for level := 1; level <= 6; level++ { + component := strings.Repeat("#", level) + " " + directive := "body.section[" + strconv.Quote(component) + "]" + got, err := ParseBodySource(directive) + if err != nil { + t.Fatalf("level %d returned an error: %v", level, err) + } + want := SectionSelector{{Level: level, Text: ""}} + if !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("level %d selector = %+v, want %+v", level, got.Selector, want) + } + } + + got, err := ParseBodySource("body.section[`## Backslash \\ and spaces `]") + if err != nil { + t.Fatal(err) + } + want := SectionSelector{{Level: 2, Text: "Backslash \\ and spaces "}} + if !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("selector = %+v, want %+v", got.Selector, want) + } + canonical, err := CanonicalSectionSelectorSource(got.Selector) + if err != nil { + t.Fatal(err) + } + if canonical != `body.section["## Backslash \\ and spaces "]` { + t.Fatalf("canonical source = %q", canonical) + } +} + +func TestParseBodySourcePreservesEveryHierarchicalKey(t *testing.T) { + want := SectionSelector{ + {Level: 1, Text: "Root #"}, + {Level: 3, Text: "Middle\\ "}, + {Level: 6, Text: "Leaf ##"}, + } + source, err := CanonicalSectionSelectorSource(want) + if err != nil { + t.Fatal(err) + } + got, err := ParseBodySource(source) + if err != nil { + t.Fatal(err) + } + if !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("selector = %+v, want %+v", got.Selector, want) + } +} + func TestCanonicalSectionSource(t *testing.T) { got, err := CanonicalSectionSource("## Notes") if err != nil { @@ -135,6 +186,77 @@ func TestCanonicalSectionSelectorSource_MultilineHeadingRoundTrip(t *testing.T) } } +func TestHeadingKeyRoundTripFromSetextAndATX(t *testing.T) { + for _, tt := range []struct { + name string + body string + want HeadingKey + wantSource string + }{ + {name: "simple Setext trailing hash", body: "Notes #\n---\n", want: HeadingKey{Level: 2, Text: "Notes #"}, wantSource: `body.section["## Notes #"]`}, + {name: "multiline Setext trailing hash", body: "First\nSecond #\n---\n", want: HeadingKey{Level: 2, Text: "First\nSecond #"}, wantSource: `body.section["## First\nSecond #"]`}, + {name: "ATX closing sequence", body: "## Notes ##\n", want: HeadingKey{Level: 2, Text: "Notes"}, wantSource: `body.section["## Notes"]`}, + } { + t.Run(tt.name, func(t *testing.T) { + sections := ExtractSectionsFromText(tt.body) + if len(sections) != 1 || len(sections[0].Path) != 1 || sections[0].Path[0] != tt.want { + t.Fatalf("sections = %+v, want key %+v", sections, tt.want) + } + source, err := CanonicalSectionSelectorSource(SectionSelector{tt.want}) + if err != nil { + t.Fatal(err) + } + if source != tt.wantSource { + t.Fatalf("source = %q, want %q", source, tt.wantSource) + } + parsed, err := ParseBodySource(source) + if err != nil || !reflect.DeepEqual(parsed.Selector, SectionSelector{tt.want}) { + t.Fatalf("parsed source = %+v, error = %v", parsed, err) + } + }) + } +} + +func TestLiteralTrailingHashesDoNotMatchATXClosingSequence(t *testing.T) { + record := &Record{Body: "## Notes ##\n\nATX content\n"} + got, present, err := ResolveBodyValue(record, `body.section["## Notes ##"]`) + if err != nil || present || got != "" { + t.Fatalf("literal selector result = %q, %v, %v; want no match", got, present, err) + } + got, present, err = ResolveBodyValue(record, `body.section["## Notes"]`) + if err != nil || !present || got != "ATX content" { + t.Fatalf("canonical selector result = %q, %v, %v", got, present, err) + } +} + +func TestMaterializeHeadingPreservesExtractedKey(t *testing.T) { + keys := []HeadingKey{ + {Level: 1, Text: "Notes #"}, + {Level: 2, Text: "First\nSecond #"}, + {Level: 2, Text: "Backslash \\ and spaces"}, + } + for level := 1; level <= 6; level++ { + keys = append(keys, HeadingKey{Level: level, Text: ""}) + } + for _, key := range keys { + markdown, err := MaterializeHeading(key) + if err != nil { + t.Fatalf("MaterializeHeading(%+v) returned an error: %v", key, err) + } + sections := ExtractSectionsFromText(markdown) + if len(sections) != 1 || len(sections[0].Path) != 1 || sections[0].Path[0] != key { + t.Fatalf("MaterializeHeading(%+v) = %q, extracted %+v", key, markdown, sections) + } + } +} + +func TestMaterializeHeadingRejectsDataLoss(t *testing.T) { + key := HeadingKey{Level: 3, Text: "Notes #"} + if got, err := MaterializeHeading(key); err == nil { + t.Fatalf("MaterializeHeading(%+v) = %q, want an error", key, got) + } +} + func TestResolveBodyValue_HeadingSyntaxCompatibility(t *testing.T) { for _, tt := range []struct { name string diff --git a/internal/migrate/scaffold.go b/internal/migrate/scaffold.go index b3554e32..60907da7 100644 --- a/internal/migrate/scaffold.go +++ b/internal/migrate/scaffold.go @@ -159,7 +159,7 @@ func renderScaffoldedContentFromBytes(data []byte, sections []rules.SectionMater for _, section := range sections { body := strings.TrimRight(section.Content, "\n") sb.WriteString("\n") - sb.WriteString(section.Heading) + sb.WriteString(section.MarkdownHeading()) sb.WriteString("\n\n") sb.WriteString(body) sb.WriteString("\n") diff --git a/internal/rules/section_materialization.go b/internal/rules/section_materialization.go index 2ff75b0e..80960af1 100644 --- a/internal/rules/section_materialization.go +++ b/internal/rules/section_materialization.go @@ -9,9 +9,17 @@ import ( ) type SectionMaterialization struct { - Field string - Heading string - Content string + Field string + Heading string + RenderedHeading string + Content string +} + +func (m SectionMaterialization) MarkdownHeading() string { + if m.RenderedHeading != "" { + return m.RenderedHeading + } + return m.Heading } func RequiredSectionMaterializations(record *extract.Record, effective *StemFile) ([]SectionMaterialization, error) { @@ -78,11 +86,19 @@ func RequiredSectionMaterializations(record *extract.Record, effective *StemFile } return nil, fmt.Errorf("cannot materialize required field %q: qualified section selector %s is missing", name, canonical) } + renderedHeading, err := extract.MaterializeHeading(source.Selector[0]) + if err != nil { + return nil, fmt.Errorf("cannot materialize required field %q: %w", name, err) + } content := field.Default if content == "" { content = "" } - out = append(out, SectionMaterialization{Field: name, Heading: source.Heading, Content: content}) + materialization := SectionMaterialization{Field: name, Heading: source.Heading, Content: content} + if renderedHeading != source.Heading { + materialization.RenderedHeading = renderedHeading + } + out = append(out, materialization) } sort.Slice(out, func(i, j int) bool { diff --git a/internal/rules/section_materialization_edge_test.go b/internal/rules/section_materialization_edge_test.go index a5bca9bb..ab310079 100644 --- a/internal/rules/section_materialization_edge_test.go +++ b/internal/rules/section_materialization_edge_test.go @@ -38,7 +38,7 @@ func TestRequiredSectionMaterializations_InvalidDeclarationsUseStableFieldOrder( if err == nil { t.Fatalf("expected declaration error, got %+v", got) } - want := `field "alpha": field "alpha" has unsupported source: section source heading must be an exact markdown heading, got "Alpha"` + want := `field "alpha": field "alpha" has unsupported source: section source heading must contain 1 to 6 hashes and one space, got "Alpha"` if err.Error() != want { t.Fatalf("error = %q, want %q", err.Error(), want) } diff --git a/internal/rules/section_materialization_test.go b/internal/rules/section_materialization_test.go index 53dee0c7..fca9206c 100644 --- a/internal/rules/section_materialization_test.go +++ b/internal/rules/section_materialization_test.go @@ -79,6 +79,48 @@ func TestRequiredSectionMaterializations_TableContract(t *testing.T) { } } +func TestRequiredSectionMaterializations_UsesLosslessHeadingMarkdown(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{ + "notes": { + Type: "string", + Required: true, + Extract: `body.section["## Notes #"]`, + }, + }} + got, err := RequiredSectionMaterializations(emptyRecord(), stem) + if err != nil { + t.Fatal(err) + } + if len(got) != 1 { + t.Fatalf("materializations = %+v", got) + } + if got[0].Heading != "## Notes #" || got[0].MarkdownHeading() != "Notes #\n---" { + t.Fatalf("materialization = %+v", got[0]) + } + sections := extract.ExtractSectionsFromText(got[0].MarkdownHeading()) + want := extract.HeadingKey{Level: 2, Text: "Notes #"} + if len(sections) != 1 || len(sections[0].Path) != 1 || sections[0].Path[0] != want { + t.Fatalf("rendered heading extracted as %+v, want %+v", sections, want) + } +} + +func TestRequiredSectionMaterializations_RejectsLossyHeading(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{ + "notes": { + Type: "string", + Required: true, + Extract: `body.section["### Notes #"]`, + }, + }} + got, err := RequiredSectionMaterializations(emptyRecord(), stem) + if err == nil { + t.Fatalf("materializations = %+v, want an error", got) + } + if !strings.Contains(err.Error(), `field "notes"`) || !strings.Contains(err.Error(), "without data loss") { + t.Fatalf("error = %q", err) + } +} + func TestRequiredSectionMaterializations_QualifiedSelectorPolicy(t *testing.T) { const qualified = `body.section["## Parent"]["### Notes"]` field := SchemaField{Type: "string", Required: true, Extract: qualified} From c00253f9b2178e5b33ec4e074b895419b308947d Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 03:23:00 -0600 Subject: [PATCH 17/18] perf(infer): index hierarchical section occurrences --- internal/infer/body_sections.go | 26 ++++++------ internal/infer/body_sections_bench_test.go | 47 ++++++++++++++++++++++ internal/infer/body_sections_test.go | 17 ++++++++ 3 files changed, 77 insertions(+), 13 deletions(-) create mode 100644 internal/infer/body_sections_bench_test.go diff --git a/internal/infer/body_sections.go b/internal/infer/body_sections.go index 602d595c..1cd6362d 100644 --- a/internal/infer/body_sections.go +++ b/internal/infer/body_sections.go @@ -175,6 +175,11 @@ func resolveSectionFamily(family *sectionFamily) (string, error) { } sortSectionOccurrences(occurrences) + occurrencesByRecord := make(map[int][]sectionOccurrence, len(family.contributors)) + for _, occurrence := range occurrences { + occurrencesByRecord[occurrence.recordOrdinal] = append(occurrencesByRecord[occurrence.recordOrdinal], occurrence) + } + if err := rejectDuplicateSectionPaths(family.key, occurrences); err != nil { return "", err } @@ -210,7 +215,8 @@ func resolveSectionFamily(family *sectionFamily) (string, error) { contributorOrdinals = append(contributorOrdinals, recordOrdinal) } sort.Slice(contributorOrdinals, func(i, j int) bool { - left, right := firstOccurrenceForRecord(occurrences, contributorOrdinals[i]), firstOccurrenceForRecord(occurrences, contributorOrdinals[j]) + left := occurrencesByRecord[contributorOrdinals[i]][0] + right := occurrencesByRecord[contributorOrdinals[j]][0] if left.recordPath != right.recordPath { return left.recordPath < right.recordPath } @@ -222,7 +228,7 @@ func resolveSectionFamily(family *sectionFamily) (string, error) { selected := make([]sectionOccurrence, 0, len(contributorOrdinals)) common := true for _, recordOrdinal := range contributorOrdinals { - matches := matchingOccurrences(occurrences, recordOrdinal, selector.selector) + matches := matchingOccurrences(occurrencesByRecord[recordOrdinal], selector.selector) if len(matches) != 1 { common = false break @@ -317,20 +323,14 @@ func rejectDuplicateSectionPaths(key extract.HeadingKey, occurrences []sectionOc return fmt.Errorf("duplicate_full_path: duplicate body section path for family %q: %s", exactSectionHeading(key), strings.Join(duplicates, "; ")) } -func firstOccurrenceForRecord(occurrences []sectionOccurrence, recordOrdinal int) sectionOccurrence { - for _, occurrence := range occurrences { - if occurrence.recordOrdinal == recordOrdinal { - return occurrence - } - } - return sectionOccurrence{recordOrdinal: recordOrdinal} -} - -func matchingOccurrences(occurrences []sectionOccurrence, recordOrdinal int, selector extract.SectionSelector) []sectionOccurrence { +func matchingOccurrences(occurrences []sectionOccurrence, selector extract.SectionSelector) []sectionOccurrence { var matches []sectionOccurrence for _, occurrence := range occurrences { - if occurrence.recordOrdinal == recordOrdinal && extract.SectionSelectorMatches(occurrence.path, selector) { + if extract.SectionSelectorMatches(occurrence.path, selector) { matches = append(matches, occurrence) + if len(matches) == 2 { + break + } } } return matches diff --git a/internal/infer/body_sections_bench_test.go b/internal/infer/body_sections_bench_test.go new file mode 100644 index 00000000..1a653ac0 --- /dev/null +++ b/internal/infer/body_sections_bench_test.go @@ -0,0 +1,47 @@ +package infer + +import ( + "fmt" + "strconv" + "testing" + + "github.com/pablontiv/rootline/internal/extract" +) + +func BenchmarkDetectSectionPatternsHierarchical(b *testing.B) { + for _, count := range []int{100, 400, 800} { + b.Run(strconv.Itoa(count), func(b *testing.B) { + records := makeHierarchicalSectionRecords(count) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + b.Fatal(err) + } + if len(inferences) != 1 || inferences[0].SourceDirective != `body.section["### Notes"]` { + b.Fatalf("unexpected inferences: %+v", inferences) + } + } + }) + } +} + +func makeHierarchicalSectionRecords(count int) []*extract.Record { + roots := []string{"Product", "Operations"} + parents := []string{"Planning", "Delivery"} + records := make([]*extract.Record, count) + for i := range records { + route := i % 4 + records[i] = &extract.Record{ + Path: fmt.Sprintf("group-%d/record-%04d.md", route, i), + Body: "section fixture", + BodySections: []extract.Section{ + {Level: 1, Heading: roots[route/2]}, + {Level: 2, Heading: parents[route%2]}, + {Level: 3, Heading: "Notes"}, + }, + } + } + return records +} diff --git a/internal/infer/body_sections_test.go b/internal/infer/body_sections_test.go index c2925bd3..cbf8b746 100644 --- a/internal/infer/body_sections_test.go +++ b/internal/infer/body_sections_test.go @@ -216,6 +216,23 @@ func TestDetectSectionPatterns_DuplicateExactHeadingCountsOncePerRecord(t *testi } } +func TestMatchingOccurrencesStopsAfterSecondMatch(t *testing.T) { + selector := extract.SectionSelector{{Level: 2, Text: "Notes"}} + occurrences := []sectionOccurrence{ + {sectionOrdinal: 1, path: extract.SectionPath{{Level: 2, Text: "Notes"}}}, + {sectionOrdinal: 2, path: extract.SectionPath{{Level: 2, Text: "Notes"}}}, + {sectionOrdinal: 3, path: extract.SectionPath{{Level: 2, Text: "Notes"}}}, + } + + matches := matchingOccurrences(occurrences, selector) + if len(matches) != 2 { + t.Fatalf("match count = %d, want 2", len(matches)) + } + if matches[0].sectionOrdinal != 1 || matches[1].sectionOrdinal != 2 { + t.Fatalf("match ordinals = %d, %d, want 1, 2", matches[0].sectionOrdinal, matches[1].sectionOrdinal) + } +} + func TestDetectSectionPatterns_PreservesQuotedBackslashBracketDirective(t *testing.T) { heading := `Need "quotes" \ and [brackets]` inferences, err := DetectSectionPatterns([]*extract.Record{ From f910b87246d1e590e0a7b10f3c4d56c73459243c Mon Sep 17 00:00:00 2001 From: Pablo Ontiveros Date: Wed, 7 Oct 2026 03:38:08 -0600 Subject: [PATCH 18/18] docs: explain canonical section heading keys --- .claude/skills/rootline/SKILL.md | 6 ++++-- .claude/skills/rootline/ref-advanced.md | 6 ++++-- .claude/skills/rootline/ref-schema.md | 6 +++++- CLAUDE.md | 8 ++++++-- docs/analyze.md | 6 ++++-- docs/extensibility.md | 6 +++++- docs/init.md | 8 +++++--- docs/migrate.md | 4 +++- docs/new.md | 4 +++- 9 files changed, 39 insertions(+), 15 deletions(-) diff --git a/.claude/skills/rootline/SKILL.md b/.claude/skills/rootline/SKILL.md index 13e4e5fb..91391711 100644 --- a/.claude/skills/rootline/SKILL.md +++ b/.claude/skills/rootline/SKILL.md @@ -206,9 +206,11 @@ Do not apply results automatically. For `apply`, inspect the report first and tr ## Canonical Field Contract -Use only `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Do not coerce YAML values. A body section pairs a real type with a source such as `body.section["## Heading"]` or `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous. The selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. Frontmatter is an explicit override. An empty section is present. Child omission inherits a stable source binding. +Use only `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Do not coerce YAML values. A body section pairs a real type with a source such as `body.section["## Heading"]`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. -`new` and `migrate --scaffold` add missing required simple sections in lexical heading order with a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Inference can emit the shortest common selector. It does not always emit a qualified selector. Validation error paths are governance-root-relative while symbolic sources remain symbolic. +A qualified selector can use `body.section["## Parent"]["### Notes"]`. Its components must be contiguous. The selector matches a contiguous suffix of the heading path. More than one match is ambiguous. A frontmatter override takes precedence. An empty section is present. Child omission inherits a stable source binding. + +Inference preserves the parsed level and text. It does not preserve the original ATX or Setext form. `new` and `migrate --scaffold` materialize a missing required simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. Validation error paths are governance-root-relative while symbolic sources remain symbolic. ## Reference Files diff --git a/.claude/skills/rootline/ref-advanced.md b/.claude/skills/rootline/ref-advanced.md index 510c7505..ab90b12c 100644 --- a/.claude/skills/rootline/ref-advanced.md +++ b/.claude/skills/rootline/ref-advanced.md @@ -133,9 +133,11 @@ Use `--field summary` only after confirming the analyze JSON contains that path. ## Canonical schema transport -`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus a source such as `source: body.section["## Heading"]` or `source: body.section["## Parent"]["### Notes"]`. A simple selector identifies the final heading by exact level and text. A qualified selector matches a contiguous suffix of the hierarchical heading path. More than one matching path is ambiguous. +`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus a source such as `source: body.section["## Heading"]`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. -Inference preserves exact headings. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. Rootline can emit the shortest common selector. The shortest common selector can be simple, so inference does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference. Rootline does not invent logical names. Rootline makes a partial-frequency family optional when that family reaches the threshold. `new` and `migrate --scaffold` materialize missing required simple sections in lexical heading order with a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. Frontmatter overrides are never written as empty shadow keys. +A qualified selector can use `source: body.section["## Parent"]["### Notes"]`. It matches a contiguous suffix of the heading path. More than one matching path is ambiguous. Inference preserves the parsed heading level and text. It does not preserve the original ATX or Setext form. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. Rootline can emit the shortest common selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. If retained families collide, Rootline fails inference. Rootline does not invent logical names. Rootline makes a retained partial-frequency family optional. + +`new` and `migrate --scaffold` materialize missing required simple selectors in lexical heading order. They use ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. Frontmatter overrides are never written as empty shadow keys. ## apply (Removed) diff --git a/.claude/skills/rootline/ref-schema.md b/.claude/skills/rootline/ref-schema.md index 27992d8f..9d77275b 100644 --- a/.claude/skills/rootline/ref-schema.md +++ b/.claude/skills/rootline/ref-schema.md @@ -149,7 +149,11 @@ Use these instead of hand-rolling `WalkUp` + `entries[0]` indexing in new comman - `source` — a logical body extraction directive when the field has one - `defined_in` — the physical `.stem` that declares the field -A source-backed field uses a real type plus a source such as `body.section["## Heading"]` or `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. Frontmatter is an override. Child omission inherits the stable source binding. `new` and `migrate --scaffold` materialize missing required simple sections in lexical heading order using a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section stops the command before it writes the affected file. +A source-backed field uses a real type plus a source such as `body.section["## Heading"]`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. Its components must be contiguous. The selector matches a contiguous suffix of the heading path. More than one match is ambiguous. Frontmatter is an override. Child omission inherits the stable source binding. Inference preserves the parsed level and text, not the original ATX or Setext form. + +`new` and `migrate --scaffold` materialize missing required simple selectors in lexical heading order. They use ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. Author required section-backed fields in `.stem` like this; `defined_in` appears only in command output, not in authored declarations: diff --git a/CLAUDE.md b/CLAUDE.md index 35b58709..cd912b0a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -184,9 +184,13 @@ Schema fields with `source:` directives (e.g., `source: body.h1` or `source: bod ## Canonical Field Contract -Use exactly `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Validation does not coerce YAML representations. A body section is a source binding on a real type. A simple binding can use `source: body.section["## Summary"]`. A qualified binding can use `source: body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous. The selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. Frontmatter is an explicit override. An inherited source binding is stable. A child omission inherits it, while a change or removal conflicts. +Use exactly `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Validation does not coerce YAML representations. A body section is a source binding on a real type. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. -`new` and `migrate --scaffold` append missing required simple sections in lexical heading order. They use a non-empty default or ``. They do not invent ancestor headings for qualified selectors. A missing required qualified section stops the command before it writes the affected file. Inference can emit the shortest common selector. The shortest common selector can be simple, so inference does not always emit a qualified selector. `query`, `set`, `describe`, and `explain` resolve the same canonical binding. `set` writes frontmatter only. In public inspection output, logical `source` is separate from physical `defined_in`. Path-like validation errors are relative to the governance root. Symbolic sources remain symbolic. +A qualified binding can use `source: body.section["## Parent"]["### Notes"]`. Its components must be contiguous. The selector matches a contiguous suffix of the heading path. More than one match is ambiguous. Frontmatter is an explicit override. A child omission inherits a stable source binding. + +Inference preserves the parsed level and text. It does not preserve the original ATX or Setext form. `new` and `migrate --scaffold` materialize a missing required simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. + +`query`, `set`, `describe`, and `explain` resolve the same canonical binding. `set` writes frontmatter only. In public inspection output, logical `source` is separate from physical `defined_in`. Path-like validation errors are relative to the governance root. Symbolic sources remain symbolic. ## Module Path diff --git a/docs/analyze.md b/docs/analyze.md index f4f7cca1..22561f36 100644 --- a/docs/analyze.md +++ b/docs/analyze.md @@ -182,7 +182,7 @@ actionable for future tooling. ## Canonical Section Inferences -Section candidates preserve the exact heading and carry a real type plus a canonical source binding: +Section candidates carry a real type plus a canonical source binding: ```yaml notes: @@ -190,7 +190,9 @@ notes: source: body.section["## Notes"] ``` -Every record contributes to the denominator. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. A family that reaches the threshold is optional unless the heading occurs in every record. For a hierarchical section, analysis can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so analysis does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference and reports each colliding heading. Rootline requires explicit logical names and does not invent logical names. Analyze, schema proposal, and schema application preserve the canonical source identity. +Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. Thus, an ATX heading `## Notes ##` produces `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text, such as `body.section["## First\nSecond"]`. Inference preserves the parsed level and text. It does not preserve whether the source used ATX or Setext syntax. + +Every record contributes to the denominator. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. A family that reaches the threshold is optional unless the heading occurs in every record. For a hierarchical section, analysis can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so analysis does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference and reports each colliding heading. Rootline requires explicit logical names and does not invent logical names. Analyze, schema proposal, and schema application preserve the canonical source identity. ## Filtering with --incremental diff --git a/docs/extensibility.md b/docs/extensibility.md index d18058c5..ff166730 100644 --- a/docs/extensibility.md +++ b/docs/extensibility.md @@ -50,7 +50,11 @@ schema: default: "" ``` -`source: body.h1` and exact `body.section[...]` directives are supported. The syntax `body.section["## Parent"]["### Notes"]` selects a hierarchical section. Each component identifies one exact heading, including its level and text. The components must be contiguous. The complete selector matches a contiguous suffix of the heading path. A simple selector such as `body.section["## Summary"]` keeps its existing behavior. More than one matching path is ambiguous. Rootline does not select the first or last path. +`source: body.h1` and exact `body.section[...]` directives are supported. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]` because its parsed text is `Notes`. The selector `body.section["## Notes ##"]` instead identifies the literal parsed text `Notes ##`. Setext headings use the same form with `#` characters. A multiline Setext heading uses an escaped `\n` inside the quoted text, such as `body.section["## First\nSecond"]`. + +The syntax `body.section["## Parent"]["### Notes"]` selects a hierarchical section. The components must be contiguous. The complete selector matches a contiguous suffix of the heading path. More than one matching path is ambiguous. Rootline does not select the first or last path. + +Inference preserves the parsed heading level and text. It does not preserve the original ATX or Setext form. Rootline materializes a missing simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. Frontmatter is an explicit override. An empty section is present with value `""`. Extractors preserve source identity so validation, inference, schema proposal, and schema application use the same canonical directive. Source-backed fields participate in validation, querying, describe/explain output, and scaffolding. `rootline set` writes a frontmatter override for such a field. It does not edit the body section. diff --git a/docs/init.md b/docs/init.md index 80beafd0..96b692f8 100644 --- a/docs/init.md +++ b/docs/init.md @@ -79,9 +79,9 @@ Fields present at all levels stay global; fields present at specific levels get ## Section Source Inference -When AST extraction is enabled, `rootline init` scans document bodies and infers source-backed fields for Markdown headings that appear frequently across files. +`rootline init` scans document bodies and infers source-backed fields for Markdown headings that appear frequently across files. -Init uses a **0.80 threshold** — a heading must appear in at least 80% of files to be emitted. This is stricter than `analyze` default threshold of 0.60, to avoid generating spurious required sections from document-specific headings. +Init uses a **0.80 threshold**. A heading must appear in at least 80% of files to be emitted. This threshold is stricter than `analyze` default threshold of 0.60. It prevents document-specific headings from becoming required sections. ### Example @@ -107,7 +107,9 @@ schema: default: "" ``` -Rootline calculates the frequency of each exact heading family. Headings below the 0.80 threshold are omitted from the generated `.stem`. Rootline discards each family that does not reach the threshold. The source preserves each exact heading level and text. For a hierarchical section, inference can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. Each component identifies an exact heading. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so inference does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference. Rootline does not invent logical names. Frontmatter remains an explicit override. Generated schemas validate the source corpus. +Rootline calculates the frequency of each exact heading family. Headings below the 0.80 threshold are omitted from the generated `.stem`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. ATX and Setext headings use the same selector form. A multiline Setext heading uses an escaped `\n` in the quoted text. Inference preserves the parsed level and text. It does not preserve the original ATX or Setext form. + +For a hierarchical section, inference can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so inference does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference. Rootline does not invent logical names. Frontmatter remains an explicit override. Generated schemas validate the source corpus. ### Source-Backed Field Properties diff --git a/docs/migrate.md b/docs/migrate.md index adf989db..bc41d2c2 100644 --- a/docs/migrate.md +++ b/docs/migrate.md @@ -196,7 +196,9 @@ Running `rootline migrate --scaffold` on a file missing both sections produces: Sections are inserted at the end of the document body only when they use simple selectors and need materialization. Multiple simple additions are appended in lexical heading order. A frontmatter override or an empty-present section needs no materialization. Ambiguous source resolution and invalid declarations fail rather than becoming a successful no-op. -A qualified selector can use `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `migrate --scaffold` does not invent ancestor headings. If a required qualified section is absent, scaffold stops before it writes the affected file. Scaffold validates prospective bytes before its atomic write. Use `--dry-run` to review insertions before applying. +Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. Scaffold materializes a simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, scaffold fails before it writes the affected file. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `migrate --scaffold` does not invent ancestor headings. If a required qualified section is absent, scaffold stops before it writes the affected file. Scaffold validates prospective bytes before each affected-file write. It does not guarantee global atomicity across all affected files. Use `--dry-run` to review insertions before applying. ## Schema Evolution diff --git a/docs/new.md b/docs/new.md index cf61fa25..0d2f1e6d 100644 --- a/docs/new.md +++ b/docs/new.md @@ -68,4 +68,6 @@ summary: `new` adds a missing simple section with its non-empty `default:` or ``. Multiple missing simple sections use lexical heading order. A frontmatter override or an empty-present section already satisfies presence. More than one selector match is ambiguous and stops the command before a file is written. -A qualified selector can use `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `new` does not invent ancestor headings. If a required qualified section is absent, `new` stops before it writes the affected file. Generated bytes are prospectively validated against the same effective schema. +Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. Rootline materializes a simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, `new` fails before it writes the affected file. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `new` does not invent ancestor headings. If a required qualified section is absent, `new` stops before it writes the affected file. Generated bytes are prospectively validated against the same effective schema.