diff --git a/.claude/skills/rootline/SKILL.md b/.claude/skills/rootline/SKILL.md index 3686a3ce..91391711 100644 --- a/.claude/skills/rootline/SKILL.md +++ b/.claude/skills/rootline/SKILL.md @@ -206,7 +206,11 @@ Do not apply results automatically. For `apply`, inspect the report first and tr ## Canonical Field Contract -Use only `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`; do not coerce YAML values. A body section pairs a real type with `source: body.section["## Heading"]`. Frontmatter is an explicit override; an empty section is present, duplicate matching headings fail, and child omission inherits a stable source binding. `new` and `migrate --scaffold` add missing required sections in lexical heading order with a non-empty default or ``. Validation error paths are governance-root-relative while symbolic sources remain symbolic. Ancestor-qualified selectors are deferred to #190. +Use only `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Do not coerce YAML values. A body section pairs a real type with a source such as `body.section["## Heading"]`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. Its components must be contiguous. The selector matches a contiguous suffix of the heading path. More than one match is ambiguous. A frontmatter override takes precedence. An empty section is present. Child omission inherits a stable source binding. + +Inference preserves the parsed level and text. It does not preserve the original ATX or Setext form. `new` and `migrate --scaffold` materialize a missing required simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. Validation error paths are governance-root-relative while symbolic sources remain symbolic. ## Reference Files diff --git a/.claude/skills/rootline/ref-advanced.md b/.claude/skills/rootline/ref-advanced.md index a7883444..ab90b12c 100644 --- a/.claude/skills/rootline/ref-advanced.md +++ b/.claude/skills/rootline/ref-advanced.md @@ -133,7 +133,11 @@ Use `--field summary` only after confirming the analyze JSON contains that path. ## Canonical schema transport -`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus `source: body.section["## Heading"]`. Inference preserves exact headings, makes partial-frequency candidates optional, and fails logical-name collisions. `new` and `migrate --scaffold` materialize missing required sections in lexical heading order with a non-empty default or ``; frontmatter overrides are never written as empty shadow keys. +`init`, `analyze`, `schema apply`, and `migrate --split` preserve a section as a real type plus a source such as `source: body.section["## Heading"]`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. + +A qualified selector can use `source: body.section["## Parent"]["### Notes"]`. It matches a contiguous suffix of the heading path. More than one matching path is ambiguous. Inference preserves the parsed heading level and text. It does not preserve the original ATX or Setext form. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. Rootline can emit the shortest common selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. If retained families collide, Rootline fails inference. Rootline does not invent logical names. Rootline makes a retained partial-frequency family optional. + +`new` and `migrate --scaffold` materialize missing required simple selectors in lexical heading order. They use ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. Frontmatter overrides are never written as empty shadow keys. ## apply (Removed) diff --git a/.claude/skills/rootline/ref-schema.md b/.claude/skills/rootline/ref-schema.md index c4afcc62..9d77275b 100644 --- a/.claude/skills/rootline/ref-schema.md +++ b/.claude/skills/rootline/ref-schema.md @@ -149,7 +149,11 @@ Use these instead of hand-rolling `WalkUp` + `entries[0]` indexing in new comman - `source` — a logical body extraction directive when the field has one - `defined_in` — the physical `.stem` that declares the field -A source-backed field uses a real type plus `source: body.section["## Heading"]`; frontmatter is an override. Child omission inherits the stable source binding. `new` and `migrate --scaffold` materialize missing required sections in lexical heading order using a non-empty default or ``. +A source-backed field uses a real type plus a source such as `body.section["## Heading"]`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. Its components must be contiguous. The selector matches a contiguous suffix of the heading path. More than one match is ambiguous. Frontmatter is an override. Child omission inherits the stable source binding. Inference preserves the parsed level and text, not the original ATX or Setext form. + +`new` and `migrate --scaffold` materialize missing required simple selectors in lexical heading order. They use ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. Author required section-backed fields in `.stem` like this; `defined_in` appears only in command output, not in authored declarations: diff --git a/.claude/skills/rootline/ref-validate.md b/.claude/skills/rootline/ref-validate.md index 69de6a80..68274b48 100644 --- a/.claude/skills/rootline/ref-validate.md +++ b/.claude/skills/rootline/ref-validate.md @@ -113,8 +113,11 @@ Link-check rules (emitted when the effective `.stem` sets `links.checks`): `link Schema fields with `source:` directives (body-extracted fields) now participate in validation. The `required` and `enum` constraints apply to values extracted from the document body: **Extraction directives**: -- `source: body.h1` — extracts the text of the first H1 heading (e.g., `# My Document` → `"My Document"`) -- `source: body.section["## Heading"]` — extracts content under the named section (e.g., `## Notes` with content below it) +- `source: body.h1` extracts the text of the first H1 heading, such as `"My Document"` from `# My Document`. +- `source: body.section["## Heading"]` identifies the final heading by exact level and text. It extracts the content under that heading. +- `source: body.section["## Parent"]["### Notes"]` matches a contiguous suffix of the hierarchical heading path. It extracts content from the matched section. + +Only multiple paths that match the complete selector produce ambiguity. Rootline does not select the first or last matching path. A required qualified selector with no match fails validation. **Precedence**: Frontmatter takes absolute precedence. If a field key exists in the record's YAML frontmatter, that value is used and body extraction is skipped. @@ -153,7 +156,7 @@ In validation: ### Source and provenance -`body.section[...]` matches an exact heading level and text. Duplicate matching headings fail rather than choosing an occurrence; frontmatter remains an override. Path-like validation error sources are governance-root-relative, while symbolic sources stay symbolic. +A simple `body.section[...]` selector identifies the final heading by exact level and text. A qualified selector matches a contiguous suffix of the hierarchical heading path. Only multiple paths that match the complete selector produce ambiguity. Frontmatter remains an override. Path-like validation error sources are governance-root-relative, while symbolic sources stay symbolic. ### Reporting Format diff --git a/CLAUDE.md b/CLAUDE.md index 9b96011a..cd912b0a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -184,9 +184,13 @@ Schema fields with `source:` directives (e.g., `source: body.h1` or `source: bod ## Canonical Field Contract -Use exactly `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`; validation does not coerce YAML representations. A body section is a source binding on a real type, such as `source: body.section["## Summary"]`. Frontmatter is an explicit override, duplicate matching headings fail, and an inherited source binding is stable: a child omission inherits it, while a change or removal conflicts. +Use exactly `string`, `list`, `enum`, `sequence`, `link`, `boolean`, and `integer`. Validation does not coerce YAML representations. A body section is a source binding on a real type. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text. -`new` and `migrate --scaffold` append missing required sections in lexical heading order, using a non-empty default or ``. `query`, `set`, `describe`, and `explain` resolve the same canonical binding; `set` writes frontmatter only. In public inspection output, logical `source` is separate from physical `defined_in`. Path-like validation errors are relative to the governance root; symbolic sources remain symbolic. Ancestor-qualified selectors are deferred to #190. +A qualified binding can use `source: body.section["## Parent"]["### Notes"]`. Its components must be contiguous. The selector matches a contiguous suffix of the heading path. More than one match is ambiguous. Frontmatter is an explicit override. A child omission inherits a stable source binding. + +Inference preserves the parsed level and text. It does not preserve the original ATX or Setext form. `new` and `migrate --scaffold` materialize a missing required simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. The commands use a non-empty default or ``. They do not invent ancestor headings. A missing required qualified section also stops the command before it writes the affected file. + +`query`, `set`, `describe`, and `explain` resolve the same canonical binding. `set` writes frontmatter only. In public inspection output, logical `source` is separate from physical `defined_in`. Path-like validation errors are relative to the governance root. Symbolic sources remain symbolic. ## Module Path diff --git a/cmd/rootline/migrate_scaffold_source_test.go b/cmd/rootline/migrate_scaffold_source_test.go index bacb5d6c..f55c11e9 100644 --- a/cmd/rootline/migrate_scaffold_source_test.go +++ b/cmd/rootline/migrate_scaffold_source_test.go @@ -7,6 +7,7 @@ import ( "strings" "testing" + "github.com/pablontiv/rootline/internal/extract" "github.com/pablontiv/rootline/internal/migrate" ) @@ -69,6 +70,57 @@ func TestMigrateScaffoldRequiredSectionSourceWarningsDoNotBlockWrite(t *testing. } } +func TestMigrateScaffoldMaterializesLosslessSetextHeading(t *testing.T) { + dir := newMigrateScaffoldSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Notes #"]' +`, "") + + result, out, err := runMigrateScaffoldJSON(t, dir) + if err != nil { + t.Fatalf("scaffold failed: %v\noutput: %s", err, out) + } + if result.SectionsAdded != 1 || result.Details[0].Heading != "## Notes #" { + t.Fatalf("result = %+v", result) + } + target := filepath.Join(dir, "T001-task.md") + content := string(mustReadFile(t, target)) + if !strings.Contains(content, "\nNotes #\n---\n\n\n") { + t.Fatalf("scaffolded content does not contain the Setext heading:\n%s", content) + } + record, err := extractProspectiveRecord(target, target, []byte(content)) + if err != nil { + t.Fatal(err) + } + got, present, err := extract.ResolveBodyValue(record, `body.section["## Notes #"]`) + if err != nil || !present || got != "" { + t.Fatalf("extracted value = %q, present = %v, error = %v", got, present, err) + } +} + +func TestMigrateScaffoldRejectsLossyHeadingWithoutWrite(t *testing.T) { + original := "---\ntitle: Task\n---\n# Task\n" + dir := newMigrateScaffoldSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["### Notes #"]' +`, original) + + out, err := runCmd(t, "migrate", "--scaffold", dir) + if err == nil { + t.Fatalf("scaffold output = %s, want an error", out) + } + if !strings.Contains(err.Error(), "without data loss") { + t.Fatalf("error = %q", err) + } + if got := string(mustReadFile(t, filepath.Join(dir, "T001-task.md"))); got != original { + t.Fatalf("affected file changed\ngot: %q\nwant: %q", got, original) + } +} + func TestMigrateScaffoldRequiredSectionSourceRejectsBrokenProspectiveLinkWithoutWrite(t *testing.T) { dir := newMigrateScaffoldSectionSourceProject(t, `schema: notes: @@ -131,6 +183,66 @@ func TestMigrateScaffoldRequiredSectionSourceDryRunParityAndNoWrite(t *testing.T } } +func TestMigrateScaffoldRejectsMissingQualifiedSectionWithoutWriting(t *testing.T) { + schema := `schema: + notes: + type: string + required: true + source: 'body.section["## Parent"]["### Notes"]' +` + selector := `body.section["## Parent"]["### Notes"]` + original := "---\ntitle: Task\n---\n# Task\n" + + writeDir := newMigrateScaffoldSectionSourceProject(t, schema, original) + writeOut, writeErr := runCmd(t, "migrate", "--scaffold", writeDir) + if writeErr == nil { + t.Fatalf("expected scaffold to reject the qualified selector, output: %s", writeOut) + } + if !strings.Contains(writeErr.Error(), `field "notes"`) || !strings.Contains(writeErr.Error(), selector) { + t.Fatalf("write error = %q, want field and canonical selector", writeErr) + } + writeContent, err := os.ReadFile(filepath.Join(writeDir, "T001-task.md")) + if err != nil { + t.Fatal(err) + } + if string(writeContent) != original { + t.Fatalf("affected file changed after materialization error\ngot: %q\nwant: %q", writeContent, original) + } + + dryDir := newMigrateScaffoldSectionSourceProject(t, schema, original) + dryOut, dryErr := runCmd(t, "migrate", "--scaffold", "--dry-run", dryDir) + if dryErr == nil { + t.Fatalf("expected dry-run to reject the qualified selector, output: %s", dryOut) + } + if dryErr.Error() != writeErr.Error() { + t.Fatalf("dry-run error = %q, want write error %q", dryErr, writeErr) + } + dryContent, err := os.ReadFile(filepath.Join(dryDir, "T001-task.md")) + if err != nil { + t.Fatal(err) + } + if string(dryContent) != original { + t.Fatalf("dry-run changed the affected file\ngot: %q\nwant: %q", dryContent, original) + } +} + +func TestMigrateScaffoldQualifiedSectionFrontmatterPrecedence(t *testing.T) { + dir := newMigrateScaffoldSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Parent"]["### Notes"]' +`, "---\nnotes: override\n---\n# Task\n") + + result, out, err := runMigrateScaffoldJSON(t, dir) + if err != nil { + t.Fatalf("frontmatter must satisfy the qualified field: %v\noutput: %s", err, out) + } + if result.SectionsAdded != 0 || result.FilesScaffolded != 0 { + t.Fatalf("result = %+v, want no materialization", result) + } +} + func TestMigrateScaffoldRequiredSectionSourcePreservesModeAndValidatesAfterWrite(t *testing.T) { dir := newMigrateScaffoldSectionSourceProject(t, `schema: anchor: diff --git a/cmd/rootline/new.go b/cmd/rootline/new.go index 5cdceb61..8b8eadf8 100644 --- a/cmd/rootline/new.go +++ b/cmd/rootline/new.go @@ -156,7 +156,7 @@ func generateMarkdown(absPath string, effective *rules.StemFile) (string, error) } for _, section := range sections { b.WriteString("\n") - b.WriteString(section.Heading) + b.WriteString(section.MarkdownHeading()) b.WriteString("\n\n") b.WriteString(section.Content) b.WriteString("\n") diff --git a/cmd/rootline/new_source_test.go b/cmd/rootline/new_source_test.go index 717a47fe..ee0d5145 100644 --- a/cmd/rootline/new_source_test.go +++ b/cmd/rootline/new_source_test.go @@ -6,6 +6,7 @@ import ( "strings" "testing" + "github.com/pablontiv/rootline/internal/extract" "github.com/pablontiv/rootline/internal/rules" ) @@ -54,6 +55,32 @@ func TestNewSectionSourceMaterializesRequiredSectionsAndValidates(t *testing.T) } } +func TestNewSectionSourceMaterializesLosslessSetextHeading(t *testing.T) { + dir := newSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Notes #"]' +`) + target := filepath.Join(dir, "T001-task.md") + + if out, err := runCmd(t, "new", target); err != nil { + t.Fatalf("new failed: %v\noutput: %s", err, out) + } + content := string(mustReadFile(t, target)) + if !strings.Contains(content, "\nNotes #\n---\n\n\n") { + t.Fatalf("generated content does not contain the Setext heading:\n%s", content) + } + record, err := extractProspectiveNewRecord(target, content) + if err != nil { + t.Fatal(err) + } + got, present, err := extract.ResolveBodyValue(record, `body.section["## Notes #"]`) + if err != nil || !present || got != "" { + t.Fatalf("extracted value = %q, present = %v, error = %v", got, present, err) + } +} + func TestNewSectionSourceUsesRecordPathAwareSchema(t *testing.T) { dir := newSectionSourceProject(t, `schema: task_notes: @@ -133,6 +160,12 @@ func TestNewSectionSourceNoWriteOnSchemaMaterializationOrValidationFailure(t *te type: string required: true source: 'body.section["Notes"]' +`}, + {"heading cannot be materialized without data loss", `schema: + notes: + type: string + required: true + source: 'body.section["### Notes #"]' `}, {"invalid prospective ordinary frontmatter", `schema: status: @@ -169,6 +202,39 @@ func TestNewSectionSourceNoWriteOnSchemaMaterializationOrValidationFailure(t *te } } +func TestNewRejectsMissingQualifiedSectionBeforeWriting(t *testing.T) { + dir := newSectionSourceProject(t, `schema: + notes: + type: string + required: true + source: 'body.section["## Parent"]["### Notes"]' +`) + selector := `body.section["## Parent"]["### Notes"]` + + target := filepath.Join(dir, "T001-task.md") + out, err := runCmd(t, "new", target) + if err == nil { + t.Fatalf("expected new to reject the qualified selector, output: %s", out) + } + if !strings.Contains(err.Error(), `field "notes"`) || !strings.Contains(err.Error(), selector) { + t.Fatalf("error = %q, want field and canonical selector", err) + } + if _, statErr := os.Stat(target); !os.IsNotExist(statErr) { + t.Fatalf("target must remain absent, stat error = %v", statErr) + } + + existing := filepath.Join(dir, "existing.md") + original := []byte("---\ntitle: Keep\n---\n# Existing\n") + mustWriteFile(t, existing, original, 0o640) + out, err = runCmd(t, "new", existing, "--force") + if err == nil { + t.Fatalf("expected forced new to reject the qualified selector, output: %s", out) + } + if got := mustReadFile(t, existing); string(got) != string(original) { + t.Fatalf("forced target changed after materialization error\ngot: %q\nwant: %q", got, original) + } +} + func mustResolveNewEffective(t *testing.T, dir, target string) *rules.StemFile { t.Helper() effective, err := rules.ResolveForRecord(dir, target) diff --git a/cmd/rootline/schema.go b/cmd/rootline/schema.go index 9a929fd3..f08d2b24 100644 --- a/cmd/rootline/schema.go +++ b/cmd/rootline/schema.go @@ -1034,7 +1034,7 @@ func schemaToInferences(stem *rules.StemFile) []infer.Inference { for fieldName, sf := range stem.Schema { if sf.Extract != "" && sf.Type == "string" { if source, err := extract.ParseBodySource(sf.Extract); err == nil && source.Kind == extract.BodySourceSection { - if canonical, err := extract.CanonicalSectionSource(source.Heading); err == nil && canonical == sf.Extract { + if canonical, err := extract.CanonicalSectionSelectorSource(source.Selector); err == nil && canonical == sf.Extract { infType := "optional_section" if sf.Required { infType = "required_section" diff --git a/cmd/rootline/schema_test.go b/cmd/rootline/schema_test.go index 04a6ba8b..aa2c942c 100644 --- a/cmd/rootline/schema_test.go +++ b/cmd/rootline/schema_test.go @@ -1622,9 +1622,11 @@ func TestSchemaApplyWritesProposedPatchVerbatim(t *testing.T) { func TestSchemaProposeIncrementalSectionConversionPreservesSource(t *testing.T) { source := `body.section["## Notes"]` + hierarchicalSource := `body.section["# Root"]["## Parent"]["### Detail"]` got := schemaToInferences(&rules.StemFile{Schema: map[string]rules.SchemaField{ "notes": {Type: "string", Required: true, Extract: source}, "summary": {Type: "string", Extract: `body.section["## Summary"]`}, + "detail": {Type: "string", Extract: hierarchicalSource}, }}) seen := map[string]infer.Inference{} @@ -1637,6 +1639,9 @@ func TestSchemaProposeIncrementalSectionConversionPreservesSource(t *testing.T) if inf := seen["summary"]; inf.Type != "optional_section" || inf.SourceDirective != `body.section["## Summary"]` { t.Fatalf("optional section conversion dropped source: %+v (all %+v)", inf, got) } + if inf := seen["detail"]; inf.Type != "optional_section" || inf.SourceDirective != hierarchicalSource { + t.Fatalf("hierarchical section conversion dropped source: %+v (all %+v)", inf, got) + } } func TestSchemaProposeIncrementalSectionConversionFallsThroughNoncanonicalSources(t *testing.T) { diff --git a/docs/analyze.md b/docs/analyze.md index d8662416..22561f36 100644 --- a/docs/analyze.md +++ b/docs/analyze.md @@ -182,7 +182,7 @@ actionable for future tooling. ## Canonical Section Inferences -Section candidates preserve the exact heading and carry a real type plus a canonical source binding: +Section candidates carry a real type plus a canonical source binding: ```yaml notes: @@ -190,7 +190,9 @@ notes: source: body.section["## Notes"] ``` -Every record contributes to the denominator. A candidate is optional unless the heading occurs in every record. Distinct exact headings that normalize to one logical name are a logical-name collision: analysis fails with each colliding heading and requires explicit names. Analyze, schema proposal, and schema application preserve the canonical source identity. +Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. Thus, an ATX heading `## Notes ##` produces `body.section["## Notes"]`. The selector `body.section["## Notes ##"]` identifies literal parsed text `Notes ##`. Setext headings use the same form. A multiline Setext heading uses an escaped `\n` in the quoted text, such as `body.section["## First\nSecond"]`. Inference preserves the parsed level and text. It does not preserve whether the source used ATX or Setext syntax. + +Every record contributes to the denominator. Rootline calculates the frequency of each exact heading family. Rootline discards each family that does not reach the threshold. A family that reaches the threshold is optional unless the heading occurs in every record. For a hierarchical section, analysis can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so analysis does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference and reports each colliding heading. Rootline requires explicit logical names and does not invent logical names. Analyze, schema proposal, and schema application preserve the canonical source identity. ## Filtering with --incremental diff --git a/docs/extensibility.md b/docs/extensibility.md index 3e2ab04b..ff166730 100644 --- a/docs/extensibility.md +++ b/docs/extensibility.md @@ -50,9 +50,13 @@ schema: default: "" ``` -`source: body.h1` and exact `body.section[...]` directives are supported. Frontmatter is an explicit override. An empty section is present with value `""`; duplicate matching headings fail rather than selecting one. Extractors preserve source identity so validation, inference, schema proposal, and schema application use the same canonical directive. +`source: body.h1` and exact `body.section[...]` directives are supported. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode only the heading level. They do not encode the original Markdown form. An ATX heading `## Notes ##` uses `body.section["## Notes"]` because its parsed text is `Notes`. The selector `body.section["## Notes ##"]` instead identifies the literal parsed text `Notes ##`. Setext headings use the same form with `#` characters. A multiline Setext heading uses an escaped `\n` inside the quoted text, such as `body.section["## First\nSecond"]`. -Source-backed fields participate in validation, querying, describe/explain output, and scaffolding. `rootline set` writes a frontmatter override for such a field; it does not edit the body section. +The syntax `body.section["## Parent"]["### Notes"]` selects a hierarchical section. The components must be contiguous. The complete selector matches a contiguous suffix of the heading path. More than one matching path is ambiguous. Rootline does not select the first or last path. + +Inference preserves the parsed heading level and text. It does not preserve the original ATX or Setext form. Rootline materializes a missing simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, Rootline fails before it writes the affected file. + +Frontmatter is an explicit override. An empty section is present with value `""`. Extractors preserve source identity so validation, inference, schema proposal, and schema application use the same canonical directive. Source-backed fields participate in validation, querying, describe/explain output, and scaffolding. `rootline set` writes a frontmatter override for such a field. It does not edit the body section. > LSP integration has been considered but carries very high complexity. > It is not in scope. diff --git a/docs/init.md b/docs/init.md index d431a14b..96b692f8 100644 --- a/docs/init.md +++ b/docs/init.md @@ -79,9 +79,9 @@ Fields present at all levels stay global; fields present at specific levels get ## Section Source Inference -When AST extraction is enabled, `rootline init` scans document bodies and infers source-backed fields for Markdown headings that appear frequently across files. +`rootline init` scans document bodies and infers source-backed fields for Markdown headings that appear frequently across files. -Init uses a **0.80 threshold** — a heading must appear in at least 80% of files to be emitted. This is stricter than `analyze` default threshold of 0.60, to avoid generating spurious required sections from document-specific headings. +Init uses a **0.80 threshold**. A heading must appear in at least 80% of files to be emitted. This threshold is stricter than `analyze` default threshold of 0.60. It prevents document-specific headings from becoming required sections. ### Example @@ -107,7 +107,9 @@ schema: default: "" ``` -Headings below the 0.80 threshold are omitted from the generated `.stem`. The source preserves exact heading level and text. Frontmatter remains an explicit override, and generated schemas validate the source corpus; distinct exact headings that normalize to one logical field fail with a collision instead of receiving an invented suffix. +Rootline calculates the frequency of each exact heading family. Headings below the 0.80 threshold are omitted from the generated `.stem`. Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. ATX and Setext headings use the same selector form. A multiline Setext heading uses an escaped `\n` in the quoted text. Inference preserves the parsed level and text. It does not preserve the original ATX or Setext form. + +For a hierarchical section, inference can emit the shortest common selector, such as `body.section["## Parent"]["### Notes"]`. The components are contiguous, and the selector matches a contiguous suffix of the heading path. The shortest common selector can be simple, so inference does not always emit a qualified selector. After the threshold filter, Rootline resolves selectors and checks logical-name collisions only among families that reach the threshold. A discarded family does not produce a collision. If families that reach the threshold collide, Rootline fails inference. Rootline does not invent logical names. Frontmatter remains an explicit override. Generated schemas validate the source corpus. ### Source-Backed Field Properties diff --git a/docs/levels.md b/docs/levels.md index 3c7a2fc8..9193e209 100644 --- a/docs/levels.md +++ b/docs/levels.md @@ -92,7 +92,7 @@ For example: - Child: `estado: { type: enum, values: [draft, active] }` ✓ Valid narrowing - Child: `estado: { type: string, required: false }` ✗ Invalid (widening if parent required) -When a child `.stem` violates monotonic constraints, `rootline validate --all` detects this in the **monotonic-violations** stemhealth check. Ancestor-qualified section selectors remain deferred to #190; current `body.section[...]` directives always identify one exact heading. +When a child `.stem` violates monotonic constraints, `rootline validate --all` detects this in the **monotonic-violations** stemhealth check. A hierarchical source binding can use `body.section["## Parent"]["### Notes"]`. Each component identifies one exact heading. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. A simple selector keeps its existing behavior. More than one match is ambiguous. A child cannot change an inherited simple or qualified source binding. ## Benefits diff --git a/docs/migrate.md b/docs/migrate.md index 12275c7f..bc41d2c2 100644 --- a/docs/migrate.md +++ b/docs/migrate.md @@ -194,7 +194,11 @@ Running `rootline migrate --scaffold` on a file missing both sections produces: ``` -Sections are inserted at the end of the document body. Multiple additions are appended in lexical heading order. A frontmatter override or an empty-present section needs no materialization. Ambiguous source resolution and invalid declarations fail rather than becoming a successful no-op. Scaffold validates prospective bytes before its atomic write. Use `--dry-run` to review insertions before applying. +Sections are inserted at the end of the document body only when they use simple selectors and need materialization. Multiple simple additions are appended in lexical heading order. A frontmatter override or an empty-present section needs no materialization. Ambiguous source resolution and invalid declarations fail rather than becoming a successful no-op. + +Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. Scaffold materializes a simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, scaffold fails before it writes the affected file. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `migrate --scaffold` does not invent ancestor headings. If a required qualified section is absent, scaffold stops before it writes the affected file. Scaffold validates prospective bytes before each affected-file write. It does not guarantee global atomicity across all affected files. Use `--dry-run` to review insertions before applying. ## Schema Evolution diff --git a/docs/new.md b/docs/new.md index 8c7c6f0b..0d2f1e6d 100644 --- a/docs/new.md +++ b/docs/new.md @@ -66,4 +66,8 @@ summary: required: true ``` -`new` adds a missing exact section with its non-empty `default:` or ``. Multiple missing sections use lexical heading order. A frontmatter override or an empty-present section already satisfies presence, and duplicate matching headings fail before a file is written. Generated bytes are prospectively validated against the same effective schema. +`new` adds a missing simple section with its non-empty `default:` or ``. Multiple missing simple sections use lexical heading order. A frontmatter override or an empty-present section already satisfies presence. More than one selector match is ambiguous and stops the command before a file is written. + +Each selector component contains one to six `#` characters, one space, and the exact parsed heading text. The `#` characters encode the heading level, not the original Markdown form. Rootline materializes a simple selector with ATX or Setext syntax that preserves the level and text. Setext syntax is available only for levels 1 and 2. If neither form preserves the selector, `new` fails before it writes the affected file. + +A qualified selector can use `body.section["## Parent"]["### Notes"]`. The components must be contiguous, and the selector matches a contiguous suffix of the heading path. `new` does not invent ancestor headings. If a required qualified section is absent, `new` stops before it writes the affected file. Generated bytes are prospectively validated against the same effective schema. diff --git a/docs/validate.md b/docs/validate.md index 49949396..b350145b 100644 --- a/docs/validate.md +++ b/docs/validate.md @@ -340,7 +340,7 @@ summary: `required means presence`: an empty section is present with value `""`; use `non_empty` when content is required, and `exists` for an effective field including derived values. Frontmatter has precedence and is a deliberate override. -A `body.section[...]` directive matches its exact heading level and text, and duplicate matching headings fail as ambiguous; Rootline does not select the first or last. The same source resolver serves single-file and batch validation, query, describe, and explain. +A component in a `body.section[...]` directive matches one exact heading level and text. `body.section["## Parent"]["### Notes"]` matches a contiguous suffix of the heading path. All selector components must be contiguous. A simple selector keeps its existing behavior. More than one match is ambiguous. Rootline does not select the first or last match. A required qualified selector with no match fails validation. The same source resolver serves single-file and batch validation, query, describe, and explain. Public path-like error `source` values are governance-root-relative with `/` separators. Symbolic sources remain symbolic, so results are stable across working directories and `--all` scan roots. diff --git a/internal/extract/body.go b/internal/extract/body.go index 0eaf01e1..beb230a5 100644 --- a/internal/extract/body.go +++ b/internal/extract/body.go @@ -1,18 +1,35 @@ package extract import ( + "bytes" + "sort" "strings" + "github.com/yuin/goldmark" "github.com/yuin/goldmark/ast" east "github.com/yuin/goldmark/extension/ast" + "github.com/yuin/goldmark/text" ) +// HeadingKey identifies one markdown heading. +type HeadingKey struct { + Level int `json:"level"` + Text string `json:"text"` +} + +// SectionPath identifies a section from its root heading to its own heading. +type SectionPath []HeadingKey + +// SectionSelector identifies a section by a contiguous suffix of its path. +type SectionSelector []HeadingKey + // Section represents a heading-delimited section in a markdown body. type Section struct { - Heading string `json:"heading"` - Level int `json:"level"` - Content string `json:"content"` - StartLine int `json:"start_line"` + Heading string `json:"heading"` + Level int `json:"level"` + Path SectionPath `json:"path,omitempty"` + Content string `json:"content"` + StartLine int `json:"start_line"` } // ExtractSections splits a markdown body into sections delimited by headings. @@ -20,12 +37,14 @@ type Section struct { // If the document has no headings, a single section with the entire body is returned. func ExtractSections(node ast.Node, source []byte) []Section { type headingInfo struct { - text string - level int - startLine int - endOffset int // byte offset after the heading line + text string + level int + startLine int + startOffset int + endOffset int } + lineIndex := newSourceLineIndex(source) var headings []headingInfo // Collect headings from the AST (top-level block children only). @@ -34,36 +53,43 @@ func ExtractSections(node ast.Node, source []byte) []Section { continue } h := child.(*ast.Heading) - // Extract heading text from child text nodes. - var text strings.Builder - for c := h.FirstChild(); c != nil; c = c.NextSibling() { - if c.Kind() == ast.KindText { - seg := c.(*ast.Text).Segment - text.Write(seg.Value(source)) - } + lines := h.Lines() + position := h.Pos() + if position < 0 && lines.Len() > 0 { + position = lines.At(0).Start } + startLine := lineIndex.lineForOffset(position) + startOffset := lineIndex.offset(startLine) + lineEnd := lineIndex.offset(startLine + 1) + endOffset := lineEnd + headingText := "" + + if _, text, ok := parseATXHeading(string(source[startOffset:lineEnd])); ok { + headingText = text + } else if lines.Len() > 0 { + var text strings.Builder + for i := 0; i < lines.Len(); i++ { + segment := lines.At(i) + text.Write(segment.Value(source)) + } + headingText = strings.TrimSpace(text.String()) - lines := h.Lines() - startLine := 0 - endOffset := 0 - if lines.Len() > 0 { - seg := lines.At(0) - startLine = lineFromOffset(source, seg.Start) lastSeg := lines.At(lines.Len() - 1) endOffset = lastSeg.Stop - if _, _, ok := parseATXHeading(string(source[seg.Start:lastSeg.Stop])); !ok { - underlineStart, underlineEnd := lineOffset(source, startLine+1), lineOffset(source, startLine+2) - if _, ok := parseSetextUnderline(string(source[underlineStart:underlineEnd])); ok { - endOffset = underlineEnd - } + lastLine := lineIndex.lineForOffset(lastSeg.Start) + underlineStart := lineIndex.offset(lastLine + 1) + underlineEnd := lineIndex.offset(lastLine + 2) + if _, ok := parseSetextUnderline(string(source[underlineStart:underlineEnd])); ok { + endOffset = underlineEnd } } headings = append(headings, headingInfo{ - text: text.String(), - level: h.Level, - startLine: startLine, - endOffset: endOffset, + text: headingText, + level: h.Level, + startLine: startLine, + startOffset: startOffset, + endOffset: endOffset, }) } @@ -82,7 +108,7 @@ func ExtractSections(node ast.Node, source []byte) []Section { var contentEnd int if i+1 < len(headings) { // Content ends where the next heading's line starts. - contentEnd = lineOffset(source, headings[i+1].startLine) + contentEnd = headings[i+1].startOffset } else { contentEnd = len(source) } @@ -100,7 +126,7 @@ func ExtractSections(node ast.Node, source []byte) []Section { }) } - return sections + return sectionsWithPaths(sections) } // CodeBlock represents a fenced code block in a markdown body. @@ -200,64 +226,22 @@ func ExtractTables(node ast.Node, source []byte) []Table { // ExtractSectionsFromText splits markdown text into heading-delimited sections. func ExtractSectionsFromText(body string) []Section { source := []byte(body) - type heading struct { - text string - level int - line, start int - end int - } - var headings []heading - prevLine, prevLineNo, prevStart, prevOK := "", 0, 0, false - inFence, fenceChar, fenceLen := false, byte(0), 0 - for start, lineNo := 0, 1; start <= len(source); lineNo++ { - end := start - for end < len(source) && source[end] != '\n' { - end++ - } - lineEnd := end - if end < len(source) { - lineEnd++ - } - line := string(source[start:end]) - if char, length, ok := parseFenceLine(line); ok { - prevOK = false - if !inFence { - inFence, fenceChar, fenceLen = true, char, length - } else if char == fenceChar && length >= fenceLen { - inFence = false - } - } else if !inFence { - if level, text, ok := parseATXHeading(line); ok { - headings = append(headings, heading{text: text, level: level, line: lineNo, start: start, end: lineEnd}) - prevOK = false - } else if level, ok := parseSetextUnderline(line); ok && prevOK { - headings = append(headings, heading{text: strings.TrimSpace(strings.TrimRight(prevLine, "\r")), level: level, line: prevLineNo, start: prevStart, end: lineEnd}) - prevOK = false - } else if _, ok := parseSetextUnderline(line); ok { - prevOK = false - } else { - prevLine, prevLineNo, prevStart, prevOK = line, lineNo, start, strings.TrimSpace(line) != "" - } - } - if end >= len(source) { - break - } - start = lineEnd - } - if len(headings) == 0 { - return []Section{{Heading: "", Level: 0, Content: body, StartLine: 1}} - } - sections := make([]Section, 0, len(headings)) - for i, h := range headings { - contentEnd := len(source) - if i+1 < len(headings) { - contentEnd = headings[i+1].start + node := goldmark.DefaultParser().Parse(text.NewReader(source)) + return ExtractSections(node, source) +} + +func sectionsWithPaths(sections []Section) []Section { + path := make(SectionPath, 0, 6) + for i := range sections { + if sections[i].Level <= 0 { + sections[i].Path = nil + continue } - content := "" - if h.end < contentEnd { - content = strings.TrimSpace(string(source[h.end:contentEnd])) + for len(path) > 0 && path[len(path)-1].Level >= sections[i].Level { + path = path[:len(path)-1] } - sections = append(sections, Section{Heading: h.text, Level: h.level, Content: content, StartLine: h.line}) + path = append(path, HeadingKey{Level: sections[i].Level, Text: sections[i].Heading}) + sections[i].Path = append(SectionPath(nil), path...) } return sections } @@ -274,11 +258,13 @@ func parseATXHeading(line string) (int, string, bool) { return 0, "", false } text := strings.TrimSpace(rest[level:]) - if i := len(text) - 1; i > 0 && text[i] == '#' { + if i := len(text) - 1; i >= 0 && text[i] == '#' { for i >= 0 && text[i] == '#' { i-- } - if i >= 0 && (text[i] == ' ' || text[i] == '\t') { + if i < 0 { + text = "" + } else if text[i] == ' ' || text[i] == '\t' { text = strings.TrimSpace(text[:i]) } } @@ -306,31 +292,40 @@ func parseSetextUnderline(line string) (int, bool) { return 2, true } -func parseFenceLine(line string) (byte, int, bool) { - line = strings.TrimRight(line, "\r") - indent := len(line) - len(strings.TrimLeft(line, " ")) - if indent > 3 || indent >= len(line) || (line[indent] != '`' && line[indent] != '~') { - return 0, 0, false +type sourceLineIndex struct { + starts []int + sourceLength int +} + +func newSourceLineIndex(source []byte) sourceLineIndex { + starts := make([]int, 1, bytes.Count(source, []byte{'\n'})+1) + for offset, value := range source { + if value == '\n' { + starts = append(starts, offset+1) + } } - char, count := line[indent], 0 - for i := indent; i < len(line) && line[i] == char; i++ { - count++ + return sourceLineIndex{starts: starts, sourceLength: len(source)} +} + +func (index sourceLineIndex) lineForOffset(offset int) int { + if offset < 0 { + offset = 0 + } else if offset > index.sourceLength { + offset = index.sourceLength } - return char, count, count >= 3 + return sort.Search(len(index.starts), func(i int) bool { + return index.starts[i] > offset + }) } -// lineOffset returns the byte offset of the start of a 1-based line number. -func lineOffset(source []byte, line int) int { - current := 1 - for i := 0; i < len(source); i++ { - if current == line { - return i - } - if source[i] == '\n' { - current++ - } +func (index sourceLineIndex) offset(line int) int { + if line <= 1 { + return 0 + } + if line > len(index.starts) { + return index.sourceLength } - return len(source) + return index.starts[line-1] } // ExtractBodyH1 returns the text of the first H1 heading in the body, diff --git a/internal/extract/body_source.go b/internal/extract/body_source.go index f3925b5b..de17f13c 100644 --- a/internal/extract/body_source.go +++ b/internal/extract/body_source.go @@ -14,37 +14,118 @@ const ( ) type BodySource struct { - Kind BodySourceKind - Heading string + Kind BodySourceKind + + // Heading retains the final exact heading for callers that use simple selectors. + Heading string + Selector SectionSelector } func ParseBodySource(directive string) (BodySource, error) { switch { case directive == "body.h1": return BodySource{Kind: BodySourceH1}, nil - case strings.HasPrefix(directive, "body.section["): - if !strings.HasSuffix(directive, "]") { - return BodySource{}, fmt.Errorf("malformed body section source %q", directive) - } - quoted := strings.TrimSuffix(strings.TrimPrefix(directive, "body.section["), "]") - heading, err := strconv.Unquote(quoted) + case strings.HasPrefix(directive, "body.section"): + selector, err := parseSectionSelector(directive) if err != nil { - return BodySource{}, fmt.Errorf("malformed body section source %q", directive) - } - if err := validateExactHeading(heading); err != nil { return BodySource{}, err } - return BodySource{Kind: BodySourceSection, Heading: heading}, nil + return BodySource{ + Kind: BodySourceSection, + Heading: exactHeading(selector[len(selector)-1]), + Selector: selector, + }, nil default: return BodySource{}, fmt.Errorf("unsupported body source %q", directive) } } +func parseSectionSelector(directive string) (SectionSelector, error) { + const prefix = "body.section" + if !strings.HasPrefix(directive, prefix) { + return nil, fmt.Errorf("unsupported body source %q", directive) + } + + rest := strings.TrimPrefix(directive, prefix) + selector := make(SectionSelector, 0, 1) + for rest != "" { + if len(rest) < 4 || rest[0] != '[' || (rest[1] != '"' && rest[1] != '`') { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + quoteEnd := quotedStringEnd(rest[1:]) + if quoteEnd < 0 || quoteEnd+2 >= len(rest) || rest[quoteEnd+2] != ']' { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + quoted := rest[1 : quoteEnd+2] + heading, err := strconv.Unquote(quoted) + if err != nil { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + key, err := parseHeadingKey(heading) + if err != nil { + return nil, err + } + if len(selector) > 0 && key.Level <= selector[len(selector)-1].Level { + return nil, fmt.Errorf("section source heading levels must increase in %q", directive) + } + selector = append(selector, key) + rest = rest[quoteEnd+3:] + } + if len(selector) == 0 { + return nil, fmt.Errorf("malformed body section source %q", directive) + } + return selector, nil +} + +// quotedStringEnd returns the index of the closing quote in a quoted string. +func quotedStringEnd(quoted string) int { + if len(quoted) == 0 || (quoted[0] != '"' && quoted[0] != '`') { + return -1 + } + delimiter := quoted[0] + escaped := false + for i := 1; i < len(quoted); i++ { + if delimiter == '`' { + if quoted[i] == delimiter { + return i + } + continue + } + if escaped { + escaped = false + continue + } + switch quoted[i] { + case '\\': + escaped = true + case delimiter: + return i + } + } + return -1 +} + func CanonicalSectionSource(exactHeading string) (string, error) { - if err := validateExactHeading(exactHeading); err != nil { + key, err := parseHeadingKey(exactHeading) + if err != nil { + return "", err + } + return CanonicalSectionSelectorSource(SectionSelector{key}) +} + +// CanonicalSectionSelectorSource serializes a section selector. +func CanonicalSectionSelectorSource(selector SectionSelector) (string, error) { + if err := validateSectionSelector(selector); err != nil { return "", err } - return "body.section[" + strconv.Quote(exactHeading) + "]", nil + var source strings.Builder + source.WriteString("body.section") + for _, key := range selector { + source.WriteByte('[') + source.WriteString(strconv.Quote(exactHeading(key))) + source.WriteByte(']') + } + return source.String(), nil } func ResolveBodyValue(record *Record, directive string) (string, bool, error) { @@ -60,7 +141,7 @@ func ResolveBodyValue(record *Record, directive string) (string, bool, error) { h1 := resolveH1(record) return h1, h1 != "", nil case BodySourceSection: - return resolveUniqueSection(sectionsForRecord(record), source.Heading) + return resolveUniqueSection(sectionsForRecord(record), source.Selector) default: return "", false, fmt.Errorf("unsupported body source %q", directive) } @@ -75,26 +156,48 @@ func resolveH1(record *Record) string { return "" } -func resolveUniqueSection(sections []Section, exactHeading string) (string, bool, error) { - var match *Section +func resolveUniqueSection(sections []Section, selector SectionSelector) (string, bool, error) { + match := -1 for i := range sections { - if sectionExactHeading(sections[i]) != exactHeading { + if !SectionSelectorMatches(sections[i].Path, selector) { continue } - if match != nil { - return "", false, fmt.Errorf("ambiguous body section source %q", exactHeading) + if match >= 0 { + description := exactHeading(selector[0]) + if len(selector) > 1 { + description, _ = CanonicalSectionSelectorSource(selector) + } + return "", false, fmt.Errorf("ambiguous body section source %q", description) } - match = §ions[i] + match = i } - if match == nil { + if match < 0 { return "", false, nil } - return match.Content, true, nil + return sections[match].Content, true, nil +} + +// SectionSelectorMatches reports whether selector is a contiguous path suffix. +func SectionSelectorMatches(path SectionPath, selector SectionSelector) bool { + if len(selector) == 0 || len(selector) > len(path) { + return false + } + offset := len(path) - len(selector) + for i := range selector { + if path[offset+i] != selector[i] { + return false + } + } + return true } func sectionsForRecord(record *Record) []Section { - if len(record.BodySections) > 0 { - return record.BodySections + if record.BodySections != nil { + sections := append([]Section(nil), record.BodySections...) + return sectionsWithPaths(sections) + } + if record.AST != nil { + return ExtractSections(record.AST, []byte(record.Body)) } if record.Body == "" { return nil @@ -106,13 +209,74 @@ func sectionExactHeading(sec Section) string { if sec.Level <= 0 { return sec.Heading } - return strings.Repeat("#", sec.Level) + " " + sec.Heading + return exactHeading(HeadingKey{Level: sec.Level, Text: sec.Heading}) +} + +func exactHeading(key HeadingKey) string { + return strings.Repeat("#", key.Level) + " " + key.Text +} + +func parseHeadingKey(heading string) (HeadingKey, error) { + level := 0 + for level < len(heading) && heading[level] == '#' { + level++ + } + if level < 1 || level > 6 || level >= len(heading) || heading[level] != ' ' { + return HeadingKey{}, fmt.Errorf("section source heading must contain 1 to 6 hashes and one space, got %q", heading) + } + return HeadingKey{Level: level, Text: heading[level+1:]}, nil +} + +// MaterializeHeading returns markdown that extracts as key without data loss. +func MaterializeHeading(key HeadingKey) (string, error) { + if err := validateHeadingKey(key); err != nil { + return "", err + } + + atx := exactHeading(key) + if materializedHeadingMatches(atx, key) { + return atx, nil + } + if key.Level <= 2 { + underline := "---" + if key.Level == 1 { + underline = "===" + } + setext := key.Text + "\n" + underline + if materializedHeadingMatches(setext, key) { + return setext, nil + } + } + return "", fmt.Errorf("heading level %d with text %q cannot be materialized without data loss", key.Level, key.Text) +} + +func materializedHeadingMatches(markdown string, key HeadingKey) bool { + sections := ExtractSectionsFromText(markdown) + return len(sections) == 1 && sections[0].Level == key.Level && sections[0].Heading == key.Text +} + +func validateHeadingKey(key HeadingKey) error { + if key.Level < 1 || key.Level > 6 { + return fmt.Errorf("heading level must be from 1 to 6, got %d", key.Level) + } + return nil } -func validateExactHeading(heading string) error { - level, text, ok := parseATXHeading(heading) - if !ok || level == 0 || sectionExactHeading(Section{Heading: text, Level: level}) != heading { - return fmt.Errorf("section source heading must be an exact markdown heading, got %q", heading) +func validateSectionSelector(selector SectionSelector) error { + if len(selector) == 0 { + return fmt.Errorf("section selector must contain one heading") + } + for i, key := range selector { + if err := validateHeadingKey(key); err != nil { + return fmt.Errorf("section selector has an invalid heading at index %d", i) + } + parsed, err := parseHeadingKey(exactHeading(key)) + if err != nil || parsed != key { + return fmt.Errorf("section selector has an invalid heading at index %d", i) + } + if i > 0 && key.Level <= selector[i-1].Level { + return fmt.Errorf("section selector heading levels must increase") + } } return nil } diff --git a/internal/extract/body_source_test.go b/internal/extract/body_source_test.go index 4d1a9e26..a93f2111 100644 --- a/internal/extract/body_source_test.go +++ b/internal/extract/body_source_test.go @@ -2,62 +2,424 @@ package extract import ( "reflect" + "strconv" "strings" "testing" + + "github.com/yuin/goldmark" + gmtext "github.com/yuin/goldmark/text" ) func TestParseBodySource(t *testing.T) { - got, err := ParseBodySource("body.h1") - if err != nil || got != (BodySource{Kind: BodySourceH1}) { - t.Fatalf("body.h1 parsed as %+v err=%v", got, err) + tests := []struct { + directive string + want BodySource + }{ + { + directive: "body.h1", + want: BodySource{Kind: BodySourceH1}, + }, + { + directive: `body.section["## Notes"]`, + want: BodySource{ + Kind: BodySourceSection, + Heading: "## Notes", + Selector: SectionSelector{{Level: 2, Text: "Notes"}}, + }, + }, + { + directive: `body.section["## Parent"]["### Notes"]`, + want: BodySource{ + Kind: BodySourceSection, + Heading: "### Notes", + Selector: SectionSelector{ + {Level: 2, Text: "Parent"}, + {Level: 3, Text: "Notes"}, + }, + }, + }, } - got, err = ParseBodySource(`body.section["## Notes"]`) - if err != nil || got != (BodySource{Kind: BodySourceSection, Heading: "## Notes"}) { - t.Fatalf("section parsed as %+v err=%v", got, err) + for _, tt := range tests { + got, err := ParseBodySource(tt.directive) + if err != nil || !reflect.DeepEqual(got, tt.want) { + t.Fatalf("ParseBodySource(%q) = %+v, %v; want %+v", tt.directive, got, err, tt.want) + } } +} + +func TestParseBodySourceRejectsInvalidSelectors(t *testing.T) { for _, tt := range []struct{ directive, wantErr string }{ {"body.title", "unsupported"}, {`body.section[## Notes]`, "malformed"}, {`body.section["Notes"]`, "heading"}, + {`body.section["## Parent"]["## Notes"]`, "increase"}, + {`body.section["### Parent"]["## Notes"]`, "increase"}, + {`body.section["####### First\nSecond"]`, "heading"}, + {`body.section["##First\nSecond"]`, "heading"}, + {`body.section[" ## First\nSecond"]`, "heading"}, + {`body.section["##\nSecond"]`, "heading"}, + {"body.section[\"## First\nSecond\"]", "malformed"}, } { _, err := ParseBodySource(tt.directive) if err == nil || !strings.Contains(err.Error(), tt.wantErr) { - t.Fatalf("%s: expected %q error, got %v", tt.directive, tt.wantErr, err) + t.Fatalf("ParseBodySource(%q) error = %v; want text %q", tt.directive, err, tt.wantErr) } } } +func TestParseBodySourceAllowsLevelJump(t *testing.T) { + got, err := ParseBodySource(`body.section["# Root"]["#### Detail"]`) + want := SectionSelector{{Level: 1, Text: "Root"}, {Level: 4, Text: "Detail"}} + if err != nil || !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("selector = %+v, %v; want %+v", got.Selector, err, want) + } +} + +func TestParseBodySourceHeadingKeyIsInjective(t *testing.T) { + for level := 1; level <= 6; level++ { + component := strings.Repeat("#", level) + " " + directive := "body.section[" + strconv.Quote(component) + "]" + got, err := ParseBodySource(directive) + if err != nil { + t.Fatalf("level %d returned an error: %v", level, err) + } + want := SectionSelector{{Level: level, Text: ""}} + if !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("level %d selector = %+v, want %+v", level, got.Selector, want) + } + } + + got, err := ParseBodySource("body.section[`## Backslash \\ and spaces `]") + if err != nil { + t.Fatal(err) + } + want := SectionSelector{{Level: 2, Text: "Backslash \\ and spaces "}} + if !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("selector = %+v, want %+v", got.Selector, want) + } + canonical, err := CanonicalSectionSelectorSource(got.Selector) + if err != nil { + t.Fatal(err) + } + if canonical != `body.section["## Backslash \\ and spaces "]` { + t.Fatalf("canonical source = %q", canonical) + } +} + +func TestParseBodySourcePreservesEveryHierarchicalKey(t *testing.T) { + want := SectionSelector{ + {Level: 1, Text: "Root #"}, + {Level: 3, Text: "Middle\\ "}, + {Level: 6, Text: "Leaf ##"}, + } + source, err := CanonicalSectionSelectorSource(want) + if err != nil { + t.Fatal(err) + } + got, err := ParseBodySource(source) + if err != nil { + t.Fatal(err) + } + if !reflect.DeepEqual(got.Selector, want) { + t.Fatalf("selector = %+v, want %+v", got.Selector, want) + } +} + func TestCanonicalSectionSource(t *testing.T) { got, err := CanonicalSectionSource("## Notes") if err != nil { - t.Fatalf("CanonicalSectionSource: %v", err) + t.Fatalf("CanonicalSectionSource returned an error: %v", err) } if got != `body.section["## Notes"]` { t.Fatalf("CanonicalSectionSource() = %q", got) } + + got, err = CanonicalSectionSelectorSource(SectionSelector{ + {Level: 2, Text: "Parent"}, + {Level: 4, Text: `Quote "value"`}, + }) + if err != nil { + t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) + } + if got != `body.section["## Parent"]["#### Quote \"value\""]` { + t.Fatalf("CanonicalSectionSelectorSource() = %q", got) + } + parsed, err := ParseBodySource(got) + if err != nil || len(parsed.Selector) != 2 || parsed.Selector[1].Text != `Quote "value"` { + t.Fatalf("serialized selector parsed as %+v, %v", parsed, err) + } +} + +func TestCanonicalSectionSelectorSource_MultilineHeadingRoundTrip(t *testing.T) { + for _, tt := range []struct { + name string + text string + }{ + {name: "plain", text: "First\nSecond"}, + {name: "trailing hash", text: "First\nSecond #"}, + {name: "trailing hashes", text: "First\nSecond ##"}, + {name: "trailing space", text: "First\nSecond "}, + {name: "trailing backslash", text: "First\nSecond \\"}, + {name: "intermediate hash", text: "First\nMiddle #\nSecond"}, + } { + t.Run(tt.name, func(t *testing.T) { + wantSelector := SectionSelector{ + {Level: 1, Text: "Root"}, + {Level: 2, Text: tt.text}, + } + got, err := CanonicalSectionSelectorSource(wantSelector) + if err != nil { + t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) + } + + parsed, err := ParseBodySource(got) + if err != nil { + t.Fatalf("ParseBodySource returned an error for %q: %v", got, err) + } + if !reflect.DeepEqual(parsed.Selector, wantSelector) { + t.Fatalf("selector = %+v, want %+v", parsed.Selector, wantSelector) + } + if parsed.Heading != exactHeading(wantSelector[1]) { + t.Fatalf("heading = %q, want %q", parsed.Heading, exactHeading(wantSelector[1])) + } + }) + } +} + +func TestHeadingKeyRoundTripFromSetextAndATX(t *testing.T) { + for _, tt := range []struct { + name string + body string + want HeadingKey + wantSource string + }{ + {name: "simple Setext trailing hash", body: "Notes #\n---\n", want: HeadingKey{Level: 2, Text: "Notes #"}, wantSource: `body.section["## Notes #"]`}, + {name: "multiline Setext trailing hash", body: "First\nSecond #\n---\n", want: HeadingKey{Level: 2, Text: "First\nSecond #"}, wantSource: `body.section["## First\nSecond #"]`}, + {name: "ATX closing sequence", body: "## Notes ##\n", want: HeadingKey{Level: 2, Text: "Notes"}, wantSource: `body.section["## Notes"]`}, + } { + t.Run(tt.name, func(t *testing.T) { + sections := ExtractSectionsFromText(tt.body) + if len(sections) != 1 || len(sections[0].Path) != 1 || sections[0].Path[0] != tt.want { + t.Fatalf("sections = %+v, want key %+v", sections, tt.want) + } + source, err := CanonicalSectionSelectorSource(SectionSelector{tt.want}) + if err != nil { + t.Fatal(err) + } + if source != tt.wantSource { + t.Fatalf("source = %q, want %q", source, tt.wantSource) + } + parsed, err := ParseBodySource(source) + if err != nil || !reflect.DeepEqual(parsed.Selector, SectionSelector{tt.want}) { + t.Fatalf("parsed source = %+v, error = %v", parsed, err) + } + }) + } +} + +func TestLiteralTrailingHashesDoNotMatchATXClosingSequence(t *testing.T) { + record := &Record{Body: "## Notes ##\n\nATX content\n"} + got, present, err := ResolveBodyValue(record, `body.section["## Notes ##"]`) + if err != nil || present || got != "" { + t.Fatalf("literal selector result = %q, %v, %v; want no match", got, present, err) + } + got, present, err = ResolveBodyValue(record, `body.section["## Notes"]`) + if err != nil || !present || got != "ATX content" { + t.Fatalf("canonical selector result = %q, %v, %v", got, present, err) + } +} + +func TestMaterializeHeadingPreservesExtractedKey(t *testing.T) { + keys := []HeadingKey{ + {Level: 1, Text: "Notes #"}, + {Level: 2, Text: "First\nSecond #"}, + {Level: 2, Text: "Backslash \\ and spaces"}, + } + for level := 1; level <= 6; level++ { + keys = append(keys, HeadingKey{Level: level, Text: ""}) + } + for _, key := range keys { + markdown, err := MaterializeHeading(key) + if err != nil { + t.Fatalf("MaterializeHeading(%+v) returned an error: %v", key, err) + } + sections := ExtractSectionsFromText(markdown) + if len(sections) != 1 || len(sections[0].Path) != 1 || sections[0].Path[0] != key { + t.Fatalf("MaterializeHeading(%+v) = %q, extracted %+v", key, markdown, sections) + } + } +} + +func TestMaterializeHeadingRejectsDataLoss(t *testing.T) { + key := HeadingKey{Level: 3, Text: "Notes #"} + if got, err := MaterializeHeading(key); err == nil { + t.Fatalf("MaterializeHeading(%+v) = %q, want an error", key, got) + } +} + +func TestResolveBodyValue_HeadingSyntaxCompatibility(t *testing.T) { + for _, tt := range []struct { + name string + body string + selector SectionSelector + wantValue string + }{ + { + name: "ATX", + body: "## Notes\n\nATX content\n", + selector: SectionSelector{{Level: 2, Text: "Notes"}}, + wantValue: "ATX content", + }, + { + name: "simple Setext", + body: "Notes\n---\n\nSetext content\n", + selector: SectionSelector{{Level: 2, Text: "Notes"}}, + wantValue: "Setext content", + }, + { + name: "multiline Setext hierarchy", + body: "# Root\n\nFirst\nSecond\n---\n\nMultiline content\n", + selector: SectionSelector{{Level: 1, Text: "Root"}, {Level: 2, Text: "First\nSecond"}}, + wantValue: "Multiline content", + }, + { + name: "multiline Setext trailing hash", + body: "First\nSecond #\n---\n\nMultiline content\n", + selector: SectionSelector{{Level: 2, Text: "First\nSecond #"}}, + wantValue: "Multiline content", + }, + } { + t.Run(tt.name, func(t *testing.T) { + source, err := CanonicalSectionSelectorSource(tt.selector) + if err != nil { + t.Fatalf("CanonicalSectionSelectorSource returned an error: %v", err) + } + got, present, err := ResolveBodyValue(&Record{Body: tt.body}, source) + if err != nil || !present || got != tt.wantValue { + t.Fatalf("value = %q, present = %v, error = %v; want %q", got, present, err, tt.wantValue) + } + }) + } } func TestResolveBodyValue_PresentEmptySection(t *testing.T) { rec := &Record{BodySections: []Section{{Heading: "Notes", Level: 2, Content: ""}}} got, present, err := ResolveBodyValue(rec, `body.section["## Notes"]`) if err != nil || !present || got != "" { - t.Fatalf("got value=%q present=%v err=%v", got, present, err) + t.Fatalf("value=%q present=%v error=%v; want a present empty value", got, present, err) + } +} + +func TestResolveBodyValue_ReusesAvailableAST(t *testing.T) { + astSource := []byte("# AST\n\nAST content\n") + node := goldmark.DefaultParser().Parse(gmtext.NewReader(astSource)) + body := "plain\n\n# Text\n\ntext content\n" + + // The alternate AST makes AST reuse observable without parser instrumentation. + wantSections := ExtractSections(node, []byte(body)) + if reflect.DeepEqual(wantSections, ExtractSectionsFromText(body)) { + t.Fatal("test inputs do not distinguish AST reuse from transient parsing") + } + if len(wantSections) != 1 || wantSections[0].Level != 1 { + t.Fatalf("AST-backed sections = %+v; want one H1 section", wantSections) + } + + rec := &Record{Body: body, AST: node} + gotSections := sectionsForRecord(rec) + if !reflect.DeepEqual(gotSections, wantSections) { + t.Fatalf("record sections = %+v; want AST-backed sections %+v", gotSections, wantSections) + } + got, present, err := ResolveBodyValue(rec, "body.h1") + if err != nil || !present || got != wantSections[0].Heading { + t.Fatalf("value=%q present=%v error=%v; want AST-backed H1 %q", got, present, err, wantSections[0].Heading) + } +} + +func TestResolveBodyValue_ParsesBodyWithoutAST(t *testing.T) { + rec := &Record{Body: "# Text\n\ntext content\n"} + got, present, err := ResolveBodyValue(rec, `body.section["# Text"]`) + if err != nil || !present || got != "text content" { + t.Fatalf("value=%q present=%v error=%v; want transient parsing result", got, present, err) + } +} + +func TestResolveBodyValue_ASTWithoutHeadingsHasNoHeadingResult(t *testing.T) { + body := "paragraph only\n" + node := goldmark.DefaultParser().Parse(gmtext.NewReader([]byte(body))) + rec := &Record{Body: body, AST: node} + + if got, present, err := ResolveBodyValue(rec, "body.h1"); err != nil || present || got != "" { + t.Fatalf("value=%q present=%v error=%v; want no H1", got, present, err) + } + if got, present, err := ResolveBodyValue(rec, `body.section["# Missing"]`); err != nil || present || got != "" { + t.Fatalf("value=%q present=%v error=%v; want no section", got, present, err) + } +} + +func TestResolveBodyValue_PreservesInitializedEmptySections(t *testing.T) { + body := "# Text\n\ntext content\n" + node := goldmark.DefaultParser().Parse(gmtext.NewReader([]byte(body))) + rec := &Record{Body: body, BodySections: []Section{}, AST: node} + + if got, present, err := ResolveBodyValue(rec, "body.h1"); err != nil || present || got != "" { + t.Fatalf("value=%q present=%v error=%v; want initialized empty sections", got, present, err) + } +} + +func TestResolveBodyValue_SelectorMatching(t *testing.T) { + rec := &Record{Body: "# Root\n\n## Parent A\n\n### Notes\n\nfirst\n\n## Parent B\n\n### Notes\n\nsecond\n\n#### Detail\n\ndetail\n\n# Other Root\n\n## Parent B\n\n### Notes\n\nthird\n"} + tests := []struct { + name string + directive string + want string + present bool + ambiguous bool + }{ + {name: "simple selector is ambiguous", directive: `body.section["### Notes"]`, ambiguous: true}, + {name: "qualified selector distinguishes parents", directive: `body.section["## Parent A"]["### Notes"]`, want: "first", present: true}, + {name: "partial suffix has one match", directive: `body.section["### Notes"]["#### Detail"]`, want: "detail", present: true}, + {name: "partial suffix has no match", directive: `body.section["## Missing"]["### Notes"]`}, + {name: "partial suffix has multiple matches", directive: `body.section["## Parent B"]["### Notes"]`, ambiguous: true}, + {name: "full suffix has one match", directive: `body.section["# Root"]["## Parent B"]["### Notes"]`, want: "second", present: true}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, present, err := ResolveBodyValue(rec, tt.directive) + if tt.ambiguous { + if err == nil || !strings.Contains(err.Error(), "ambiguous") { + t.Fatalf("error = %v; want ambiguity", err) + } + return + } + if err != nil || present != tt.present || got != tt.want { + t.Fatalf("value=%q present=%v error=%v; want value=%q present=%v", got, present, err, tt.want, tt.present) + } + }) + } +} + +func TestResolveBodyValue_SelectorCannotOmitDirectParent(t *testing.T) { + rec := &Record{Body: "# Root\n\n## Parent\n\n### Intermediate\n\n#### Notes\n\nvalue\n"} + _, present, err := ResolveBodyValue(rec, `body.section["## Parent"]["#### Notes"]`) + if err != nil || present { + t.Fatalf("present=%v error=%v; want no match", present, err) + } + got, present, err := ResolveBodyValue(rec, `body.section["### Intermediate"]["#### Notes"]`) + if err != nil || !present || got != "value" { + t.Fatalf("value=%q present=%v error=%v; want the direct-parent match", got, present, err) } } -func TestResolveBodyValue_DuplicateSectionIsAmbiguous(t *testing.T) { - rec := &Record{BodySections: []Section{ - {Heading: "Notes", Level: 2, Content: "first"}, - {Heading: "Notes", Level: 2, Content: "second"}, - }} - _, _, err := ResolveBodyValue(rec, `body.section["## Notes"]`) +func TestResolveBodyValue_DuplicateFullPathIsAmbiguous(t *testing.T) { + rec := &Record{Body: "# Root\n\n## Parent\n\n### Notes\n\nfirst\n\n### Notes\n\nsecond\n"} + _, _, err := ResolveBodyValue(rec, `body.section["# Root"]["## Parent"]["### Notes"]`) if err == nil || !strings.Contains(err.Error(), "ambiguous") { - t.Fatalf("expected ambiguity error, got %v", err) + t.Fatalf("error = %v; want ambiguity", err) } } -func TestMarkdownExtractor_PreservesDuplicateSectionsAndFencedCodeExclusion(t *testing.T) { - content := []byte("---\ntitle: Fixture\n---\nTitle\n=====\n\n```\n## Fake\n```\n\nEmpty\n-----\n\n## Duplicate\n\nfirst\n\n## Duplicate\n\nsecond\n") +func TestMarkdownExtractor_PreservesPathsAndFencedCodeExclusion(t *testing.T) { + content := []byte("---\ntitle: Fixture\n---\nRoot\n====\n\n```\n## Fake\n```\n\n~~~~\n## Fake Two\n~~~~ not a closing fence\n## Fake Three\n~~~~\n\n## Empty\n\n## Parent *A*\n\n### Duplicate\n\nfirst\n\n## Parent B\n\n### Duplicate\n\nsecond\n") parseAST := true extractors := map[string]*MarkdownExtractor{ "ast": {ParseAST: &parseAST}, @@ -68,32 +430,51 @@ func TestMarkdownExtractor_PreservesDuplicateSectionsAndFencedCodeExclusion(t *t for name, ext := range extractors { rec, err := ext.Extract(name+".md", content) if err != nil { - t.Fatalf("%s Extract: %v", name, err) + t.Fatalf("%s extraction failed: %v", name, err) } - if got := len(rec.BodySections); got != 4 { - t.Fatalf("%s BodySections len = %d, want 4: %+v", name, got, rec.BodySections) + if got := len(rec.BodySections); got != 6 { + t.Fatalf("%s section count = %d; want 6: %+v", name, got, rec.BodySections) } - if rec.BodySections[1].Heading != "Empty" || rec.BodySections[1].Level != 2 || rec.BodySections[1].Content != "" { + if rec.BodySections[1].Heading != "Empty" || rec.BodySections[1].Content != "" { t.Fatalf("%s empty section = %+v", name, rec.BodySections[1]) } - for _, sec := range rec.BodySections { - if sec.Heading == "Fake" { - t.Fatalf("%s included fenced code heading: %+v", name, rec.BodySections) - } - } - if got, present, err := ResolveBodyValue(rec, "body.h1"); err != nil || !present || got != "Title" { - t.Fatalf("%s h1 resolution value=%q present=%v err=%v", name, got, present, err) + wantLastPath := SectionPath{ + {Level: 1, Text: "Root"}, + {Level: 2, Text: "Parent B"}, + {Level: 3, Text: "Duplicate"}, } - if _, present, err := ResolveBodyValue(rec, `body.section["## Empty"]`); err != nil || !present { - t.Fatalf("%s empty resolution present=%v err=%v", name, present, err) + if !reflect.DeepEqual(rec.BodySections[5].Path, wantLastPath) { + t.Fatalf("%s final path = %+v; want %+v", name, rec.BodySections[5].Path, wantLastPath) } - if _, _, err := ResolveBodyValue(rec, `body.section["## Duplicate"]`); err == nil || !strings.Contains(err.Error(), "ambiguous") { - t.Fatalf("%s expected duplicate ambiguity, got %v", name, err) + for _, sec := range rec.BodySections { + if strings.HasPrefix(sec.Heading, "Fake") { + t.Fatalf("%s included a fenced heading: %+v", name, rec.BodySections) + } } if baseline == nil { baseline = rec.BodySections } else if !reflect.DeepEqual(baseline, rec.BodySections) { - t.Fatalf("text-backed sections differ from AST: AST=%+v got=%+v", baseline, rec.BodySections) + t.Fatalf("text sections differ from AST sections: AST=%+v text=%+v", baseline, rec.BodySections) + } + } +} + +func TestSectionPathsCloseBranches(t *testing.T) { + sections := ExtractSectionsFromText("# First\n\n### Jump\n\n#### Leaf\n\n## Sibling\n\n# Second\n\n### Other\n") + want := []SectionPath{ + {{Level: 1, Text: "First"}}, + {{Level: 1, Text: "First"}, {Level: 3, Text: "Jump"}}, + {{Level: 1, Text: "First"}, {Level: 3, Text: "Jump"}, {Level: 4, Text: "Leaf"}}, + {{Level: 1, Text: "First"}, {Level: 2, Text: "Sibling"}}, + {{Level: 1, Text: "Second"}}, + {{Level: 1, Text: "Second"}, {Level: 3, Text: "Other"}}, + } + if len(sections) != len(want) { + t.Fatalf("section count = %d; want %d", len(sections), len(want)) + } + for i := range sections { + if !reflect.DeepEqual(sections[i].Path, want[i]) { + t.Fatalf("path %d = %+v; want %+v", i, sections[i].Path, want[i]) } } } diff --git a/internal/extract/body_test.go b/internal/extract/body_test.go index 36abe22a..43cb3a2b 100644 --- a/internal/extract/body_test.go +++ b/internal/extract/body_test.go @@ -1,6 +1,9 @@ package extract import ( + "reflect" + "strconv" + "strings" "testing" "github.com/yuin/goldmark" @@ -54,6 +57,55 @@ func TestExtractSections_MultipleHeadings(t *testing.T) { } } +func TestExtractSections_ATXBoundaries(t *testing.T) { + body := "## Notes ##\n\nFirst\n\n##\n\nSecond\n\n### ###\n\nThird\n\nTitle\n=====\n\nFourth\n\n## Literal##\n\nFifth\n" + want := []Section{ + {Heading: "Notes", Level: 2, Path: SectionPath{{Level: 2, Text: "Notes"}}, Content: "First", StartLine: 1}, + {Heading: "", Level: 2, Path: SectionPath{{Level: 2, Text: ""}}, Content: "Second", StartLine: 5}, + {Heading: "", Level: 3, Path: SectionPath{{Level: 2, Text: ""}, {Level: 3, Text: ""}}, Content: "Third", StartLine: 9}, + {Heading: "Title", Level: 1, Path: SectionPath{{Level: 1, Text: "Title"}}, Content: "Fourth", StartLine: 13}, + {Heading: "Literal##", Level: 2, Path: SectionPath{{Level: 1, Text: "Title"}, {Level: 2, Text: "Literal##"}}, Content: "Fifth", StartLine: 18}, + } + + for name, sections := range map[string][]Section{ + "AST": parseSections(body), + "text": ExtractSectionsFromText(body), + } { + if !reflect.DeepEqual(sections, want) { + t.Errorf("%s sections = %+v; want %+v", name, sections, want) + } + } +} + +func TestExtractSections_AdjacentATXHeadings(t *testing.T) { + body := "# First #\n##\n### Third ###\nContent\n" + want := []Section{ + {Heading: "First", Level: 1, Path: SectionPath{{Level: 1, Text: "First"}}, Content: "", StartLine: 1}, + {Heading: "", Level: 2, Path: SectionPath{{Level: 1, Text: "First"}, {Level: 2, Text: ""}}, Content: "", StartLine: 2}, + {Heading: "Third", Level: 3, Path: SectionPath{{Level: 1, Text: "First"}, {Level: 2, Text: ""}, {Level: 3, Text: "Third"}}, Content: "Content", StartLine: 3}, + } + if sections := parseSections(body); !reflect.DeepEqual(sections, want) { + t.Fatalf("sections = %+v; want %+v", sections, want) + } +} + +func BenchmarkExtractSectionsManyHeadings(b *testing.B) { + for _, count := range []int{100, 1000} { + b.Run(strconv.Itoa(count), func(b *testing.B) { + source := []byte(strings.Repeat("## Heading ##\n\nContent\n\n", count)) + node := goldmark.DefaultParser().Parse(text.NewReader(source)) + b.ReportAllocs() + b.SetBytes(int64(len(source))) + b.ResetTimer() + for i := 0; i < b.N; i++ { + benchmarkSections = ExtractSections(node, source) + } + }) + } +} + +var benchmarkSections []Section + func TestExtractSections_HeadingInCodeBlock(t *testing.T) { body := "## Real\n\nSome text\n\n```\n## Fake\n```\n\n## Also Real\n\nMore text\n" sections := parseSections(body) @@ -69,6 +121,100 @@ func TestExtractSections_HeadingInCodeBlock(t *testing.T) { } } +func TestExtractSectionsFromText_BlockContainerParity(t *testing.T) { + for _, tt := range []struct { + name string + body string + }{ + {name: "ATX heading in list", body: "- item\n\n ## Nested\n\n## Real"}, + {name: "Setext underline in list", body: "- First\n Second\n ---\n\n## Real"}, + {name: "heading in HTML block", body: "
\n## Fake\n
\n\n## Real"}, + {name: "ATX heading in blockquote", body: "> ## Quoted\n\n## Real"}, + } { + t.Run(tt.name, func(t *testing.T) { + astSections := parseSections(tt.body) + textSections := ExtractSectionsFromText(tt.body) + if !reflect.DeepEqual(textSections, astSections) { + t.Fatalf("text sections differ from AST sections: text=%+v AST=%+v", textSections, astSections) + } + if len(textSections) != 1 || textSections[0].Heading != "Real" { + t.Fatalf("sections = %+v; want only the real heading", textSections) + } + }) + } +} + +func TestExtractSectionsFromText_ContainerFenceParity(t *testing.T) { + for _, tt := range []struct { + name string + body string + }{ + {name: "closed backtick fence in list", body: "- ```\n ## Fake\n ```\n\n## Real\n\nContent\n"}, + {name: "closed tilde fence in list", body: "- ~~~\n ## Fake\n ~~~\n\n## Real\n\nContent\n"}, + {name: "unclosed backtick fence in list", body: "- ```\n ## Fake\n\n## Real\n\nContent\n"}, + {name: "unclosed tilde fence in list", body: "- ~~~\n ## Fake\n\n## Real\n\nContent\n"}, + {name: "unclosed backtick fence in quote", body: "> ```\n> ## Fake\n\n## Real\n\nContent\n"}, + {name: "unclosed tilde fence in quote", body: "> ~~~\n> ## Fake\n\n## Real\n\nContent\n"}, + } { + t.Run(tt.name, func(t *testing.T) { + astSections := parseSections(tt.body) + textSections := ExtractSectionsFromText(tt.body) + if !reflect.DeepEqual(textSections, astSections) { + t.Fatalf("text sections differ from AST sections: text=%+v AST=%+v", textSections, astSections) + } + if len(textSections) != 1 || textSections[0].Heading != "Real" { + t.Fatalf("sections = %+v; want only the real heading after the fence", textSections) + } + }) + } +} + +func TestExtractSectionsFromText_SetextParity(t *testing.T) { + for _, tt := range []struct { + name string + body string + wantHeading string + wantContent string + wantLevel int + }{ + {name: "simple", body: "First\n===\n\nContent\n", wantHeading: "First", wantContent: "Content", wantLevel: 1}, + {name: "multiple lines", body: "First\nSecond\n---\n\nContent\n", wantHeading: "First\nSecond", wantContent: "Content", wantLevel: 2}, + {name: "multiple lines with trailing hash", body: "First\nSecond #\n---\n", wantHeading: "First\nSecond #", wantLevel: 2}, + } { + t.Run(tt.name, func(t *testing.T) { + astSections := parseSections(tt.body) + textSections := ExtractSectionsFromText(tt.body) + if !reflect.DeepEqual(textSections, astSections) { + t.Fatalf("text sections differ from AST sections: text=%+v AST=%+v", textSections, astSections) + } + if len(textSections) != 1 { + t.Fatalf("section count = %d; want 1", len(textSections)) + } + wantPath := SectionPath{{Level: tt.wantLevel, Text: tt.wantHeading}} + want := Section{Heading: tt.wantHeading, Level: tt.wantLevel, Path: wantPath, Content: tt.wantContent, StartLine: 1} + if !reflect.DeepEqual(textSections[0], want) { + t.Fatalf("section = %+v; want %+v", textSections[0], want) + } + }) + } +} + +func TestParseATXHeading_ClosingSequence(t *testing.T) { + for _, tt := range []struct { + line string + wantText string + }{ + {line: "## Heading ##", wantText: "Heading"}, + {line: "## Heading # ", wantText: "Heading"}, + {line: "## Heading#", wantText: "Heading#"}, + } { + level, text, ok := parseATXHeading(tt.line) + if !ok || level != 2 || text != tt.wantText { + t.Fatalf("parseATXHeading(%q) = %d, %q, %v; want 2, %q, true", tt.line, level, text, ok, tt.wantText) + } + } +} + func TestExtractSections_NoHeadings(t *testing.T) { body := "Just a paragraph.\n\nAnother paragraph.\n" sections := parseSections(body) diff --git a/internal/extract/extract.go b/internal/extract/extract.go index 9cb85619..8c04d3a2 100644 --- a/internal/extract/extract.go +++ b/internal/extract/extract.go @@ -58,7 +58,8 @@ type ExtractionError struct { } // MarkdownExtractor extracts YAML frontmatter from Markdown files. -// Set ParseAST to false to skip goldmark AST parsing (default: true). +// ParseAST controls whether Extract retains the Goldmark AST in Record.AST. +// Section extraction can use a transient AST when ParseAST is false. type MarkdownExtractor struct { ParseAST *bool } diff --git a/internal/extract/registry.go b/internal/extract/registry.go index 30829453..791498ef 100644 --- a/internal/extract/registry.go +++ b/internal/extract/registry.go @@ -21,8 +21,8 @@ func NewRegistry() *Registry { return r } -// NewASTRegistry creates a registry with AST parsing enabled. -// Use this when body structure (sections, headings) needs to be inspected. +// NewASTRegistry creates a registry with AST retention enabled. +// Use this when callers need Record.AST after extraction. func NewASTRegistry() *Registry { r := &Registry{ byName: make(map[string]Extractor), diff --git a/internal/infer/apply.go b/internal/infer/apply.go index 1c1ef804..bee37c0c 100644 --- a/internal/infer/apply.go +++ b/internal/infer/apply.go @@ -356,7 +356,7 @@ func canonicalSectionDirective(directive string) (string, error) { if source.Kind != extract.BodySourceSection { return "", fmt.Errorf("source_directive must be a body section") } - return extract.CanonicalSectionSource(source.Heading) + return extract.CanonicalSectionSelectorSource(source.Selector) } func existingFieldMatchesSectionIntent(sf rules.SchemaField, source string) bool { diff --git a/internal/infer/apply_section_test.go b/internal/infer/apply_section_test.go index 9276adf6..078d85ed 100644 --- a/internal/infer/apply_section_test.go +++ b/internal/infer/apply_section_test.go @@ -112,3 +112,28 @@ func TestApplySchemaInferences_CanonicalizesParseableSectionSource(t *testing.T) t.Fatalf("source = %q, want canonical %q", got, want) } } + +func TestApplySchemaInferences_PreservesHierarchicalSectionSource(t *testing.T) { + stemPath := writeApplyStem(t, "version: 2\nschema: {}\n") + input := "body.section[`# Root`][`## Parent`][`### Notes`]" + want := `body.section["# Root"]["## Parent"]["### Notes"]` + _, err := ApplySchemaInferences(stemPath, []ReportInference{{Type: "required_section", Field: "notes", SourceDirective: input}}, false) + if err != nil { + t.Fatalf("apply hierarchical source: %v", err) + } + if got := readApplyStem(t, stemPath).Schema["notes"].Extract; got != want { + t.Fatalf("source = %q, want %q", got, want) + } +} + +func TestApplySchemaInferences_DistinguishesHierarchicalSectionParents(t *testing.T) { + stemPath := writeApplyStem(t, "version: 2\nschema:\n notes:\n type: string\n source: 'body.section[\"## Parent A\"][\"### Notes\"]'\n") + before := string(mustReadApplyFile(t, stemPath)) + _, err := ApplySchemaInferences(stemPath, []ReportInference{{Type: "required_section", Field: "notes", SourceDirective: `body.section["## Parent B"]["### Notes"]`}}, false) + if err == nil { + t.Fatal("expected different parents to conflict") + } + if after := string(mustReadApplyFile(t, stemPath)); after != before { + t.Fatalf("conflict mutated stem:\nbefore:\n%s\nafter:\n%s", before, after) + } +} diff --git a/internal/infer/body_sections.go b/internal/infer/body_sections.go index 9cabd889..1cd6362d 100644 --- a/internal/infer/body_sections.go +++ b/internal/infer/body_sections.go @@ -11,9 +11,8 @@ import ( ) // DetectSectionPatterns analyzes heading structure across records in a directory. -// Every record contributes to the denominator; empty bodies count as records where -// headings are absent. Headings present in at least threshold records are inferred; -// universal headings are required, while non-universal threshold matches are optional. +// Every record contributes to the denominator. A record contributes once to each +// final-heading family, but selector resolution retains all section occurrences. func DetectSectionPatterns(records []*extract.Record, threshold float64) ([]Inference, error) { if math.IsNaN(threshold) || math.IsInf(threshold, 0) || threshold < 0 || threshold > 1 { return nil, fmt.Errorf("section threshold must be finite and in [0,1], got %v", threshold) @@ -23,67 +22,71 @@ func DetectSectionPatterns(records []*extract.Record, threshold float64) ([]Infe } total := len(records) - headingCount := make(map[string]int) - headingExact := make(map[string]string) - - for _, rec := range records { + families := make(map[extract.HeadingKey]*sectionFamily) + for recordOrdinal, rec := range records { if rec == nil || rec.Body == "" { continue } - seen := make(map[string]bool) - for _, sec := range sectionsForInference(rec) { + sections := sectionsForInference(rec) + paths := sectionPathsForInference(sections) + for sectionOrdinal, sec := range sections { if sec.Level <= 0 || sec.Heading == "" { continue } - exact := strings.Repeat("#", sec.Level) + " " + sec.Heading - source, err := extract.CanonicalSectionSource(exact) - if err != nil { - return nil, err - } - if seen[source] { - continue + key := extract.HeadingKey{Level: sec.Level, Text: sec.Heading} + family := families[key] + if family == nil { + family = §ionFamily{key: key, contributors: make(map[int]bool)} + families[key] = family } - headingCount[source]++ - headingExact[source] = exact - seen[source] = true + family.contributors[recordOrdinal] = true + family.occurrences = append(family.occurrences, sectionOccurrence{ + recordPath: rec.Path, + recordOrdinal: recordOrdinal, + sectionOrdinal: sectionOrdinal, + path: paths[sectionOrdinal], + }) } } - allHeadings := make([]sectionCandidate, 0, len(headingCount)) - for source, count := range headingCount { + candidates := make([]sectionCandidate, 0, len(families)) + for _, family := range families { + count := len(family.contributors) freq := float64(count) / float64(total) - exact := headingExact[source] - field := sectionFieldName(strings.TrimSpace(exact[strings.IndexByte(exact, ' ')+1:])) - candidateType := "optional_section" + if freq < threshold { + continue + } + exact := exactSectionHeading(family.key) + typeName := "optional_section" if count == total { - candidateType = "required_section" + typeName = "required_section" } - allHeadings = append(allHeadings, sectionCandidate{ - field: field, + candidates = append(candidates, sectionCandidate{ + family: family, + field: sectionFieldName(family.key.Text), exact: exact, - source: source, count: count, freq: freq, - typeName: candidateType, + typeName: typeName, }) } - sort.Slice(allHeadings, func(i, j int) bool { - if allHeadings[i].field != allHeadings[j].field { - return allHeadings[i].field < allHeadings[j].field + sort.Slice(candidates, func(i, j int) bool { + if candidates[i].field != candidates[j].field { + return candidates[i].field < candidates[j].field } - return allHeadings[i].source < allHeadings[j].source + return candidates[i].exact < candidates[j].exact }) - if err := rejectSectionFieldCollisions(allHeadings); err != nil { - return nil, err - } - - candidates := make([]sectionCandidate, 0, len(allHeadings)) - for _, candidate := range allHeadings { - if candidate.freq >= threshold { - candidates = append(candidates, candidate) + for i := range candidates { + source, err := resolveSectionFamily(candidates[i].family) + if err != nil { + return nil, err } + candidates[i].source = source + } + if err := rejectSectionFieldCollisions(candidates); err != nil { + return nil, err } inferences := make([]Inference, 0, len(candidates)) @@ -101,7 +104,22 @@ func DetectSectionPatterns(records []*extract.Record, threshold float64) ([]Infe return inferences, nil } +type sectionFamily struct { + key extract.HeadingKey + contributors map[int]bool + occurrences []sectionOccurrence +} + +type sectionOccurrence struct { + recordPath string + recordOrdinal int + sectionOrdinal int + path extract.SectionPath + fullSource string +} + type sectionCandidate struct { + family *sectionFamily field string exact string source string @@ -110,6 +128,16 @@ type sectionCandidate struct { typeName string } +type selectorCandidate struct { + source string + selector extract.SectionSelector +} + +type stableOccurrenceGroup struct { + identity string + selectors []selectorCandidate +} + func sectionsForInference(rec *extract.Record) []extract.Section { if len(rec.BodySections) > 0 { return rec.BodySections @@ -120,6 +148,206 @@ func sectionsForInference(rec *extract.Record) []extract.Section { return extract.ExtractSectionsFromText(rec.Body) } +func sectionPathsForInference(sections []extract.Section) []extract.SectionPath { + paths := make([]extract.SectionPath, len(sections)) + path := make(extract.SectionPath, 0, 6) + for i, sec := range sections { + if sec.Level <= 0 { + continue + } + for len(path) > 0 && path[len(path)-1].Level >= sec.Level { + path = path[:len(path)-1] + } + path = append(path, extract.HeadingKey{Level: sec.Level, Text: sec.Heading}) + paths[i] = append(extract.SectionPath(nil), path...) + } + return paths +} + +func resolveSectionFamily(family *sectionFamily) (string, error) { + occurrences := append([]sectionOccurrence(nil), family.occurrences...) + for i := range occurrences { + source, err := extract.CanonicalSectionSelectorSource(extract.SectionSelector(occurrences[i].path)) + if err != nil { + return "", err + } + occurrences[i].fullSource = source + } + sortSectionOccurrences(occurrences) + + occurrencesByRecord := make(map[int][]sectionOccurrence, len(family.contributors)) + for _, occurrence := range occurrences { + occurrencesByRecord[occurrence.recordOrdinal] = append(occurrencesByRecord[occurrence.recordOrdinal], occurrence) + } + + if err := rejectDuplicateSectionPaths(family.key, occurrences); err != nil { + return "", err + } + + selectorBySource := make(map[string]selectorCandidate) + for _, occurrence := range occurrences { + for start := len(occurrence.path) - 1; start >= 0; start-- { + source, err := extract.CanonicalSectionSelectorSource(extract.SectionSelector(occurrence.path[start:])) + if err != nil { + return "", err + } + parsed, err := extract.ParseBodySource(source) + if err != nil { + return "", err + } + selectorBySource[source] = selectorCandidate{source: source, selector: parsed.Selector} + } + } + + selectors := make([]selectorCandidate, 0, len(selectorBySource)) + for _, selector := range selectorBySource { + selectors = append(selectors, selector) + } + sort.Slice(selectors, func(i, j int) bool { + if len(selectors[i].selector) != len(selectors[j].selector) { + return len(selectors[i].selector) < len(selectors[j].selector) + } + return selectors[i].source < selectors[j].source + }) + + contributorOrdinals := make([]int, 0, len(family.contributors)) + for recordOrdinal := range family.contributors { + contributorOrdinals = append(contributorOrdinals, recordOrdinal) + } + sort.Slice(contributorOrdinals, func(i, j int) bool { + left := occurrencesByRecord[contributorOrdinals[i]][0] + right := occurrencesByRecord[contributorOrdinals[j]][0] + if left.recordPath != right.recordPath { + return left.recordPath < right.recordPath + } + return contributorOrdinals[i] < contributorOrdinals[j] + }) + + groupsByIdentity := make(map[string]*stableOccurrenceGroup) + for _, selector := range selectors { + selected := make([]sectionOccurrence, 0, len(contributorOrdinals)) + common := true + for _, recordOrdinal := range contributorOrdinals { + matches := matchingOccurrences(occurrencesByRecord[recordOrdinal], selector.selector) + if len(matches) != 1 { + common = false + break + } + selected = append(selected, matches[0]) + } + if !common { + continue + } + identity := occurrenceGroupIdentity(selected) + group := groupsByIdentity[identity] + if group == nil { + group = &stableOccurrenceGroup{identity: identity} + groupsByIdentity[identity] = group + } + group.selectors = append(group.selectors, selector) + } + + exact := exactSectionHeading(family.key) + if len(groupsByIdentity) == 0 { + return "", fmt.Errorf("no_common_selector: section family %q has no common selector", exact) + } + + groups := make([]*stableOccurrenceGroup, 0, len(groupsByIdentity)) + for _, group := range groupsByIdentity { + sort.Slice(group.selectors, func(i, j int) bool { + if len(group.selectors[i].selector) != len(group.selectors[j].selector) { + return len(group.selectors[i].selector) < len(group.selectors[j].selector) + } + return group.selectors[i].source < group.selectors[j].source + }) + groups = append(groups, group) + } + sort.Slice(groups, func(i, j int) bool { + left, right := groups[i].selectors[0], groups[j].selectors[0] + if left.source != right.source { + return left.source < right.source + } + return groups[i].identity < groups[j].identity + }) + if len(groups) > 1 { + sources := make([]string, 0, len(groups)) + for _, group := range groups { + sources = append(sources, group.selectors[0].source) + } + return "", fmt.Errorf("multiple_stable_groups: section family %q has multiple stable occurrence groups: %s", exact, strings.Join(sources, ", ")) + } + return groups[0].selectors[0].source, nil +} + +func sortSectionOccurrences(occurrences []sectionOccurrence) { + sort.Slice(occurrences, func(i, j int) bool { + if occurrences[i].recordPath != occurrences[j].recordPath { + return occurrences[i].recordPath < occurrences[j].recordPath + } + if occurrences[i].recordOrdinal != occurrences[j].recordOrdinal { + return occurrences[i].recordOrdinal < occurrences[j].recordOrdinal + } + if occurrences[i].sectionOrdinal != occurrences[j].sectionOrdinal { + return occurrences[i].sectionOrdinal < occurrences[j].sectionOrdinal + } + return occurrences[i].fullSource < occurrences[j].fullSource + }) +} + +func rejectDuplicateSectionPaths(key extract.HeadingKey, occurrences []sectionOccurrence) error { + type duplicateKey struct { + recordOrdinal int + fullSource string + } + byPath := make(map[duplicateKey][]sectionOccurrence) + for _, occurrence := range occurrences { + key := duplicateKey{recordOrdinal: occurrence.recordOrdinal, fullSource: occurrence.fullSource} + byPath[key] = append(byPath[key], occurrence) + } + + var duplicates []string + for _, matches := range byPath { + if len(matches) < 2 { + continue + } + ordinals := make([]string, 0, len(matches)) + for _, match := range matches { + ordinals = append(ordinals, strconv.Itoa(match.sectionOrdinal)) + } + duplicates = append(duplicates, fmt.Sprintf("%q (record %d), %s (section ordinals %s)", matches[0].recordPath, matches[0].recordOrdinal, matches[0].fullSource, strings.Join(ordinals, ", "))) + } + if len(duplicates) == 0 { + return nil + } + sort.Strings(duplicates) + return fmt.Errorf("duplicate_full_path: duplicate body section path for family %q: %s", exactSectionHeading(key), strings.Join(duplicates, "; ")) +} + +func matchingOccurrences(occurrences []sectionOccurrence, selector extract.SectionSelector) []sectionOccurrence { + var matches []sectionOccurrence + for _, occurrence := range occurrences { + if extract.SectionSelectorMatches(occurrence.path, selector) { + matches = append(matches, occurrence) + if len(matches) == 2 { + break + } + } + } + return matches +} + +func occurrenceGroupIdentity(occurrences []sectionOccurrence) string { + var identity strings.Builder + for _, occurrence := range occurrences { + fmt.Fprintf(&identity, "%s\x00%d\x00%d\x00%s\x00", occurrence.recordPath, occurrence.recordOrdinal, occurrence.sectionOrdinal, occurrence.fullSource) + } + return identity.String() +} + +func exactSectionHeading(key extract.HeadingKey) string { + return strings.Repeat("#", key.Level) + " " + key.Text +} + func rejectSectionFieldCollisions(candidates []sectionCandidate) error { var collisions []string for i := 0; i < len(candidates); { diff --git a/internal/infer/body_sections_bench_test.go b/internal/infer/body_sections_bench_test.go new file mode 100644 index 00000000..1a653ac0 --- /dev/null +++ b/internal/infer/body_sections_bench_test.go @@ -0,0 +1,47 @@ +package infer + +import ( + "fmt" + "strconv" + "testing" + + "github.com/pablontiv/rootline/internal/extract" +) + +func BenchmarkDetectSectionPatternsHierarchical(b *testing.B) { + for _, count := range []int{100, 400, 800} { + b.Run(strconv.Itoa(count), func(b *testing.B) { + records := makeHierarchicalSectionRecords(count) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + b.Fatal(err) + } + if len(inferences) != 1 || inferences[0].SourceDirective != `body.section["### Notes"]` { + b.Fatalf("unexpected inferences: %+v", inferences) + } + } + }) + } +} + +func makeHierarchicalSectionRecords(count int) []*extract.Record { + roots := []string{"Product", "Operations"} + parents := []string{"Planning", "Delivery"} + records := make([]*extract.Record, count) + for i := range records { + route := i % 4 + records[i] = &extract.Record{ + Path: fmt.Sprintf("group-%d/record-%04d.md", route, i), + Body: "section fixture", + BodySections: []extract.Section{ + {Level: 1, Heading: roots[route/2]}, + {Level: 2, Heading: parents[route%2]}, + {Level: 3, Heading: "Notes"}, + }, + } + } + return records +} diff --git a/internal/infer/body_sections_test.go b/internal/infer/body_sections_test.go index 592de550..cbf8b746 100644 --- a/internal/infer/body_sections_test.go +++ b/internal/infer/body_sections_test.go @@ -46,6 +46,86 @@ func TestDetectSectionPatterns_UniversalSectionRequiredWithCanonicalSource(t *te } } +func TestDetectSectionPatterns_VariableH1UsesSimpleRequiredSelector(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## Overview\nFirst.\n"), + makeRecord("# Beta\n\n## Overview\nSecond.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + inf, ok := findSectionInference(inferences, "overview") + if !ok { + t.Fatalf("overview inference is missing: %+v", inferences) + } + if inf.Type != "required_section" || inf.SourceDirective != `body.section["## Overview"]` { + t.Fatalf("overview inference = %+v", inf) + } +} + +func TestDetectSectionPatterns_VariableParentsUseSimpleSelectorForSingleOccurrence(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## First Parent\n\n### Notes\nFirst.\n"), + makeRecord("# Beta\n\n## Second Parent\n\n### Notes\nSecond.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + inf, ok := findSectionInference(inferences, "notes") + if !ok || inf.SourceDirective != `body.section["### Notes"]` { + t.Fatalf("notes inference = %+v, present = %v", inf, ok) + } +} + +func TestDetectSectionPatterns_SelectsShortestCommonQualifiedSelector(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Root\n\n## First Parent\n\n### Notes\nVariable.\n\n## Shared\n\n### Notes\nStable.\n"), + makeRecord("# Root\n\n## Second Parent\n\n### Notes\nVariable.\n\n## Shared\n\n### Notes\nStable.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + inf, ok := findSectionInference(inferences, "notes") + if !ok || inf.SourceDirective != `body.section["## Shared"]["### Notes"]` { + t.Fatalf("notes inference = %+v, present = %v", inf, ok) + } +} + +func TestDetectSectionPatterns_IncompatibleParentsHaveNoCommonSelector(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## First A\n\n### Notes\nOne.\n\n## Second A\n\n### Notes\nTwo.\n"), + makeRecord("# Beta\n\n## First B\n\n### Notes\nOne.\n\n## Second B\n\n### Notes\nTwo.\n"), + } + + inferences, err := DetectSectionPatterns(records, 1) + want := `no_common_selector: section family "### Notes" has no common selector` + if err == nil || err.Error() != want { + t.Fatalf("error = %v, want %q", err, want) + } + if inferences != nil { + t.Fatalf("inferences = %+v, want nil", inferences) + } +} + +func TestDetectSectionPatterns_MultipleStableOccurrenceGroupsFail(t *testing.T) { + records := []*extract.Record{ + makeRecord("# Alpha\n\n## First\n\n### Notes\nOne.\n\n## Second\n\n### Notes\nTwo.\n"), + makeRecord("# Beta\n\n## First\n\n### Notes\nOne.\n\n## Second\n\n### Notes\nTwo.\n"), + } + + _, err := DetectSectionPatterns(records, 1) + want := `multiple_stable_groups: section family "### Notes" has multiple stable occurrence groups: body.section["## First"]["### Notes"], body.section["## Second"]["### Notes"]` + if err == nil || err.Error() != want { + t.Fatalf("error = %v, want %q", err, want) + } +} + func TestDetectSectionPatterns_ThresholdCandidateOptionalUntilUniversal(t *testing.T) { records := []*extract.Record{ makeRecord("## Notes\nA\n"), makeRecord("## Notes\nB\n"), @@ -136,6 +216,23 @@ func TestDetectSectionPatterns_DuplicateExactHeadingCountsOncePerRecord(t *testi } } +func TestMatchingOccurrencesStopsAfterSecondMatch(t *testing.T) { + selector := extract.SectionSelector{{Level: 2, Text: "Notes"}} + occurrences := []sectionOccurrence{ + {sectionOrdinal: 1, path: extract.SectionPath{{Level: 2, Text: "Notes"}}}, + {sectionOrdinal: 2, path: extract.SectionPath{{Level: 2, Text: "Notes"}}}, + {sectionOrdinal: 3, path: extract.SectionPath{{Level: 2, Text: "Notes"}}}, + } + + matches := matchingOccurrences(occurrences, selector) + if len(matches) != 2 { + t.Fatalf("match count = %d, want 2", len(matches)) + } + if matches[0].sectionOrdinal != 1 || matches[1].sectionOrdinal != 2 { + t.Fatalf("match ordinals = %d, %d, want 1, 2", matches[0].sectionOrdinal, matches[1].sectionOrdinal) + } +} + func TestDetectSectionPatterns_PreservesQuotedBackslashBracketDirective(t *testing.T) { heading := `Need "quotes" \ and [brackets]` inferences, err := DetectSectionPatterns([]*extract.Record{ @@ -189,15 +286,44 @@ func TestDetectSectionPatterns_NameCollisionFails(t *testing.T) { } } -func TestDetectSectionPatterns_NameCollisionFailsBeforeThreshold(t *testing.T) { +func TestDetectSectionPatterns_BelowThresholdFamiliesDoNotCollide(t *testing.T) { records := []*extract.Record{ makeRecord("## Notes\nA\n\n### Notes\nB\n"), makeRecord("# Other\n"), makeRecord("# Other\n"), makeRecord("# Other\n"), makeRecord("# Other\n"), } - _, err := DetectSectionPatterns(records, 0.8) - if err == nil || !strings.Contains(err.Error(), "## Notes") || !strings.Contains(err.Error(), "### Notes") { - t.Fatalf("expected below-threshold colliding headings, got %v", err) + inferences, err := DetectSectionPatterns(records, 0.8) + if err != nil { + t.Fatalf("DetectSectionPatterns: %v", err) + } + if _, ok := findSectionInference(inferences, "notes"); ok { + t.Fatalf("notes inference is above the threshold: %+v", inferences) + } +} + +func TestDetectSectionPatterns_DuplicateFullPathFailsOnlyAtThreshold(t *testing.T) { + records := []*extract.Record{ + makeRecord("## Parent\n\n### Notes\nFirst.\n\n### Notes\nSecond.\n"), + makeRecord("# Other\n"), + } + records[0].Path = "a.md" + records[1].Path = "b.md" + + inferences, err := DetectSectionPatterns(records, 0.75) + if err != nil { + t.Fatalf("below-threshold family returned an error: %v", err) + } + if _, ok := findSectionInference(inferences, "notes"); ok { + t.Fatalf("notes inference is above the threshold: %+v", inferences) + } + + inferences, err = DetectSectionPatterns(records, 0.5) + want := `duplicate_full_path: duplicate body section path for family "### Notes": "a.md" (record 0), body.section["## Parent"]["### Notes"] (section ordinals 1, 2)` + if err == nil || err.Error() != want { + t.Fatalf("error = %v, want %q", err, want) + } + if inferences != nil { + t.Fatalf("inferences = %+v, want nil", inferences) } } diff --git a/internal/infer/delta.go b/internal/infer/delta.go index e6220e58..e7ea2cf6 100644 --- a/internal/infer/delta.go +++ b/internal/infer/delta.go @@ -105,7 +105,15 @@ func isCovered(inf Inference, stem *rules.StemFile) bool { } func sectionInferenceCovered(inf Inference, sf rules.SchemaField) bool { - if sf.Type != "string" || inf.SourceDirective == "" || sf.Extract != inf.SourceDirective { + if sf.Type != "string" || inf.SourceDirective == "" || sf.Extract == "" { + return false + } + inferredSource, err := canonicalSectionDirective(inf.SourceDirective) + if err != nil { + return false + } + existingSource, err := canonicalSectionDirective(sf.Extract) + if err != nil || existingSource != inferredSource { return false } if inf.Type == "required_section" { diff --git a/internal/infer/delta_test.go b/internal/infer/delta_test.go index 2ef9dd03..79bd8c35 100644 --- a/internal/infer/delta_test.go +++ b/internal/infer/delta_test.go @@ -217,6 +217,18 @@ func TestIsCovered_SectionSourceAndRequiredness(t *testing.T) { field: rules.SchemaField{Type: "string", Extract: `body.section["## Notes"]`}, want: false, }, + { + name: "same hierarchical source is covered", + inf: Inference{Type: "optional_section", Field: "notes", SourceDirective: `body.section["## Parent"]["### Notes"]`}, + field: rules.SchemaField{Type: "string", Extract: `body.section["## Parent"]["### Notes"]`}, + want: true, + }, + { + name: "different hierarchical parent is not covered", + inf: Inference{Type: "optional_section", Field: "notes", SourceDirective: `body.section["## Parent B"]["### Notes"]`}, + field: rules.SchemaField{Type: "string", Extract: `body.section["## Parent A"]["### Notes"]`}, + want: false, + }, { name: "missing existing source is not covered", inf: Inference{Type: "optional_section", Field: "notes", SourceDirective: `body.section["## Notes"]`}, diff --git a/internal/infer/setext_multiline_test.go b/internal/infer/setext_multiline_test.go new file mode 100644 index 00000000..05d0ab84 --- /dev/null +++ b/internal/infer/setext_multiline_test.go @@ -0,0 +1,96 @@ +package infer + +import ( + "context" + "reflect" + "testing" + + "github.com/pablontiv/rootline/internal/extract" + "github.com/pablontiv/rootline/internal/rules" +) + +const multilineSetextSource = `body.section["## First\nSecond #"]` + +func TestDetectSectionPatterns_MultilineSetext(t *testing.T) { + record := makeRecord("First\nSecond #\n---\n\nContent\n") + sections := extract.ExtractSections(record.AST, []byte(record.Body)) + if len(sections) != 1 || len(sections[0].Path) != 1 { + t.Fatalf("sections = %+v, want one section path", sections) + } + wantPath := extract.SectionPath{{Level: 2, Text: "First\nSecond #"}} + if !reflect.DeepEqual(sections[0].Path, wantPath) { + t.Fatalf("section path = %+v, want %+v", sections[0].Path, wantPath) + } + + inferences, err := DetectSectionPatterns([]*extract.Record{record}, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns rejected a valid multiline Setext heading: %v", err) + } + inf, ok := findSectionInference(inferences, "first_second") + if !ok || inf.Type != "required_section" || inf.SourceDirective != multilineSetextSource { + t.Fatalf("inference = %+v, present = %v", inf, ok) + } + + parsed, err := extract.ParseBodySource(inf.SourceDirective) + if err != nil { + t.Fatalf("ParseBodySource rejected the inferred selector: %v", err) + } + if !reflect.DeepEqual(extract.SectionPath(parsed.Selector), sections[0].Path) { + t.Fatalf("parsed selector = %+v, want section path %+v", parsed.Selector, sections[0].Path) + } + value, present, err := extract.ResolveBodyValue(record, inf.SourceDirective) + if err != nil || !present || value != "Content" { + t.Fatalf("resolved value = %q, present = %v, error = %v", value, present, err) + } +} + +func TestGenerateFlatSchema_MultilineSetext(t *testing.T) { + record := makeRecord("First\nSecond #\n---\n\nContent\n") + opts := DefaultInferOptions() + opts.IncludeStructural = false + opts.SectionThreshold = 1 + stem, err := GenerateFlatSchema(context.Background(), ".", []*extract.Record{record}, opts) + if err != nil { + t.Fatalf("GenerateFlatSchema rejected a valid multiline Setext heading: %v", err) + } + field, ok := stem.Schema["first_second"] + if !ok || field.Type != "string" || !field.Required || field.Extract != multilineSetextSource { + t.Fatalf("generated field = %+v, present = %v", field, ok) + } + if errs := rules.Validate(context.Background(), record, stem); len(errs) != 0 { + t.Fatalf("generated schema rejected its source record: %+v", errs) + } +} + +func TestApplySchemaInferences_MultilineSetext(t *testing.T) { + record := makeRecord("First\nSecond #\n---\n\nContent\n") + inferences, err := DetectSectionPatterns([]*extract.Record{record}, 1) + if err != nil { + t.Fatalf("DetectSectionPatterns returned an error: %v", err) + } + inf, ok := findSectionInference(inferences, "first_second") + if !ok { + t.Fatalf("multiline Setext inference is missing: %+v", inferences) + } + + stemPath := writeApplyStem(t, "version: 2\nschema: {}\n") + result, err := ApplySchemaInferences(stemPath, []ReportInference{{ + Type: inf.Type, + Field: inf.Field, + SourceDirective: inf.SourceDirective, + }}, false) + if err != nil { + t.Fatalf("ApplySchemaInferences rejected the inferred selector: %v", err) + } + if len(result.Applied) != 1 { + t.Fatalf("applied actions = %v, want one action", result.Applied) + } + field := readApplyStem(t, stemPath).Schema["first_second"] + if field.Type != "string" || !field.Required || field.Extract != multilineSetextSource { + t.Fatalf("applied field = %+v", field) + } + value, present, err := extract.ResolveBodyValue(record, field.Extract) + if err != nil || !present || value != "Content" { + t.Fatalf("applied selector resolved value = %q, present = %v, error = %v", value, present, err) + } +} diff --git a/internal/migrate/scaffold.go b/internal/migrate/scaffold.go index b3554e32..60907da7 100644 --- a/internal/migrate/scaffold.go +++ b/internal/migrate/scaffold.go @@ -159,7 +159,7 @@ func renderScaffoldedContentFromBytes(data []byte, sections []rules.SectionMater for _, section := range sections { body := strings.TrimRight(section.Content, "\n") sb.WriteString("\n") - sb.WriteString(section.Heading) + sb.WriteString(section.MarkdownHeading()) sb.WriteString("\n\n") sb.WriteString(body) sb.WriteString("\n") diff --git a/internal/rules/field_source_test.go b/internal/rules/field_source_test.go index 1575c70f..89e1cbf7 100644 --- a/internal/rules/field_source_test.go +++ b/internal/rules/field_source_test.go @@ -11,15 +11,12 @@ func TestResolveFieldValue_FrontmatterPresenceOverridesBodySource(t *testing.T) for _, value := range []any{"", nil} { rec := &extract.Record{ Frontmatter: map[string]any{"notes": value}, - BodySections: []extract.Section{ - {Heading: "Notes", Level: 2, Content: "first"}, - {Heading: "Notes", Level: 2, Content: "second"}, - }, + Body: "# Root\n\n## Parent\n\n### Notes\n\nfirst\n\n### Notes\n\nsecond\n", } - got, present, err := ResolveFieldValue(rec, "notes", SchemaField{Extract: `body.section["## Notes"]`}) + got, present, err := ResolveFieldValue(rec, "notes", SchemaField{Extract: `body.section["# Root"]["## Parent"]["### Notes"]`}) if err != nil || !present || got != value { - t.Fatalf("got value=%#v present=%v err=%v; want frontmatter value %#v", got, present, err, value) + t.Fatalf("value=%#v present=%v error=%v; want frontmatter value %#v", got, present, err, value) } } } diff --git a/internal/rules/section_materialization.go b/internal/rules/section_materialization.go index bf7e0123..80960af1 100644 --- a/internal/rules/section_materialization.go +++ b/internal/rules/section_materialization.go @@ -9,9 +9,17 @@ import ( ) type SectionMaterialization struct { - Field string - Heading string - Content string + Field string + Heading string + RenderedHeading string + Content string +} + +func (m SectionMaterialization) MarkdownHeading() string { + if m.RenderedHeading != "" { + return m.RenderedHeading + } + return m.Heading } func RequiredSectionMaterializations(record *extract.Record, effective *StemFile) ([]SectionMaterialization, error) { @@ -51,13 +59,10 @@ func RequiredSectionMaterializations(record *extract.Record, effective *StemFile out := make([]SectionMaterialization, 0) for _, name := range fields { field := local.Schema[name] - if !field.Required || field.Extract == "" { - continue - } - if !requiredCheckApplies(record, &local, name, field) { + if !field.Required || !requiredCheckApplies(record, &local, name, field) { continue } - if field.Type != "string" { + if field.Extract == "" || field.Type != "string" { continue } source, err := extract.ParseBodySource(field.Extract) @@ -74,11 +79,26 @@ func RequiredSectionMaterializations(record *extract.Record, effective *StemFile if present { continue } + if len(source.Selector) > 1 { + canonical, err := extract.CanonicalSectionSelectorSource(source.Selector) + if err != nil { + return nil, err + } + return nil, fmt.Errorf("cannot materialize required field %q: qualified section selector %s is missing", name, canonical) + } + renderedHeading, err := extract.MaterializeHeading(source.Selector[0]) + if err != nil { + return nil, fmt.Errorf("cannot materialize required field %q: %w", name, err) + } content := field.Default if content == "" { content = "" } - out = append(out, SectionMaterialization{Field: name, Heading: source.Heading, Content: content}) + materialization := SectionMaterialization{Field: name, Heading: source.Heading, Content: content} + if renderedHeading != source.Heading { + materialization.RenderedHeading = renderedHeading + } + out = append(out, materialization) } sort.Slice(out, func(i, j int) bool { diff --git a/internal/rules/section_materialization_edge_test.go b/internal/rules/section_materialization_edge_test.go index a5bca9bb..ab310079 100644 --- a/internal/rules/section_materialization_edge_test.go +++ b/internal/rules/section_materialization_edge_test.go @@ -38,7 +38,7 @@ func TestRequiredSectionMaterializations_InvalidDeclarationsUseStableFieldOrder( if err == nil { t.Fatalf("expected declaration error, got %+v", got) } - want := `field "alpha": field "alpha" has unsupported source: section source heading must be an exact markdown heading, got "Alpha"` + want := `field "alpha": field "alpha" has unsupported source: section source heading must contain 1 to 6 hashes and one space, got "Alpha"` if err.Error() != want { t.Fatalf("error = %q, want %q", err.Error(), want) } diff --git a/internal/rules/section_materialization_test.go b/internal/rules/section_materialization_test.go index b836df65..fca9206c 100644 --- a/internal/rules/section_materialization_test.go +++ b/internal/rules/section_materialization_test.go @@ -79,6 +79,128 @@ func TestRequiredSectionMaterializations_TableContract(t *testing.T) { } } +func TestRequiredSectionMaterializations_UsesLosslessHeadingMarkdown(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{ + "notes": { + Type: "string", + Required: true, + Extract: `body.section["## Notes #"]`, + }, + }} + got, err := RequiredSectionMaterializations(emptyRecord(), stem) + if err != nil { + t.Fatal(err) + } + if len(got) != 1 { + t.Fatalf("materializations = %+v", got) + } + if got[0].Heading != "## Notes #" || got[0].MarkdownHeading() != "Notes #\n---" { + t.Fatalf("materialization = %+v", got[0]) + } + sections := extract.ExtractSectionsFromText(got[0].MarkdownHeading()) + want := extract.HeadingKey{Level: 2, Text: "Notes #"} + if len(sections) != 1 || len(sections[0].Path) != 1 || sections[0].Path[0] != want { + t.Fatalf("rendered heading extracted as %+v, want %+v", sections, want) + } +} + +func TestRequiredSectionMaterializations_RejectsLossyHeading(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{ + "notes": { + Type: "string", + Required: true, + Extract: `body.section["### Notes #"]`, + }, + }} + got, err := RequiredSectionMaterializations(emptyRecord(), stem) + if err == nil { + t.Fatalf("materializations = %+v, want an error", got) + } + if !strings.Contains(err.Error(), `field "notes"`) || !strings.Contains(err.Error(), "without data loss") { + t.Fatalf("error = %q", err) + } +} + +func TestRequiredSectionMaterializations_QualifiedSelectorPolicy(t *testing.T) { + const qualified = `body.section["## Parent"]["### Notes"]` + field := SchemaField{Type: "string", Required: true, Extract: qualified} + + t.Run("simple required absence keeps materialization", func(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{ + "notes": {Type: "string", Required: true, Extract: `body.section["## Notes"]`}, + }} + got, err := RequiredSectionMaterializations(emptyRecord(), stem) + if err != nil { + t.Fatal(err) + } + if len(got) != 1 || got[0].Heading != "## Notes" { + t.Fatalf("materializations = %+v, want the simple section", got) + } + }) + + t.Run("qualified required presence needs no materialization", func(t *testing.T) { + record := &extract.Record{Frontmatter: map[string]any{}, Body: "## Parent\n\n### Notes\n\npresent\n"} + got, err := RequiredSectionMaterializations(record, &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want no materialization", got, err) + } + }) + + t.Run("qualified optional absence needs no materialization", func(t *testing.T) { + field.Required = false + got, err := RequiredSectionMaterializations(emptyRecord(), &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want no materialization", got, err) + } + }) + + t.Run("qualified required absence returns field and canonical selector", func(t *testing.T) { + field.Required = true + got, err := RequiredSectionMaterializations(emptyRecord(), &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err == nil { + t.Fatalf("expected an error, got materializations %+v", got) + } + if !strings.Contains(err.Error(), `field "notes"`) || !strings.Contains(err.Error(), qualified) { + t.Fatalf("error = %q, want field and canonical selector", err) + } + }) + + t.Run("frontmatter presence has precedence over an ambiguous selector", func(t *testing.T) { + record := &extract.Record{ + Frontmatter: map[string]any{"notes": "override"}, + Body: "## Parent\n\n### Notes\n\nfirst\n\n## Parent\n\n### Notes\n\nsecond\n", + } + got, err := RequiredSectionMaterializations(record, &StemFile{Schema: map[string]SchemaField{"notes": field}}) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want frontmatter to satisfy the field", got, err) + } + }) +} + +func TestRequiredSectionMaterializations_IneligibleSelectorAmbiguityNeedsNoResolution(t *testing.T) { + record := &extract.Record{ + Frontmatter: map[string]any{}, + Body: "## Parent A\n\n### Notes\n\nfirst\n\n## Parent B\n\n### Notes\n\nsecond\n", + } + tests := []struct { + name string + field SchemaField + }{ + {"optional field", SchemaField{Type: "string", Extract: `body.section["### Notes"]`}}, + {"severity off", SchemaField{Type: "string", Required: true, Severity: "off", Extract: `body.section["### Notes"]`}}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + stem := &StemFile{Schema: map[string]SchemaField{"notes": tt.field}} + got, err := RequiredSectionMaterializations(record, stem) + if err != nil || len(got) != 0 { + t.Fatalf("materializations = %+v, error = %v; want no materialization", got, err) + } + }) + } +} + func TestRequiredSectionMaterializations_OrdersByHeadingThenField(t *testing.T) { stem := &StemFile{Schema: map[string]SchemaField{ "gamma": {Type: "string", Required: true, Extract: `body.section["## Shared"]`, Default: "same"},