Files
datascape/sections.go
T
2026-08-23 14:25:12 +02:00

101 lines
3.0 KiB
Go

package main
import (
"bytes"
"regexp"
"strings"
"github.com/yuin/goldmark/ast"
"github.com/yuin/goldmark/text"
)
var sectionHeadingRe = regexp.MustCompile(`(?m)^#{1,6} `)
// headingTextRe matches an ATX heading at the start of a section. The
// heading text is everything after the `#`s and the required space, on the
// first line.
var headingTextRe = regexp.MustCompile(`^(#{1,6})\s+([^\n]*)`)
// splitSections splits raw markdown into sections.
// Section 0 is any content before the first heading.
// Each subsequent section begins at a heading line and runs to the next.
func splitSections(raw []byte) [][]byte {
locs := sectionHeadingRe.FindAllIndex(raw, -1)
if len(locs) == 0 {
return [][]byte{raw}
}
sections := make([][]byte, 0, len(locs)+1)
prev := 0
for _, loc := range locs {
sections = append(sections, raw[prev:loc[0]])
prev = loc[0]
}
sections = append(sections, raw[prev:])
return sections
}
// headingIDs returns the auto-generated id of every heading in raw markdown,
// in document order. The kth heading (1-indexed) corresponds to section k from
// splitSections. Uses the package-level goldmark parser so duplicate-id
// numbering matches what the renderer emits.
func headingIDs(raw []byte) []string {
doc := md.Parser().Parse(text.NewReader(raw))
var ids []string
ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) {
if !entering {
return ast.WalkContinue, nil
}
if _, ok := n.(*ast.Heading); ok {
if v, ok := n.AttributeString("id"); ok {
if b, ok := v.([]byte); ok {
ids = append(ids, string(b))
}
}
}
return ast.WalkContinue, nil
})
return ids
}
// joinSections reassembles sections produced by splitSections.
// Inserts a newline between sections when a non-empty section lacks a
// trailing newline, so an edited section cannot inline the next heading.
func joinSections(sections [][]byte) []byte {
var buf bytes.Buffer
for i, s := range sections {
buf.Write(s)
if i < len(sections)-1 && len(s) > 0 && s[len(s)-1] != '\n' {
buf.WriteByte('\n')
}
}
return buf.Bytes()
}
// sectionHeading returns the heading level (1..6) and trimmed text of a
// section produced by splitSections. Returns level=0 for the pre-heading
// section (index 0).
func sectionHeading(section []byte) (level int, text string) {
m := headingTextRe.FindSubmatch(section)
if m == nil {
return 0, ""
}
return len(m[1]), strings.TrimSpace(string(m[2]))
}
// sectionSpanEnd returns the exclusive end index of the section span starting
// at start: the first following section whose heading level is <= start's.
// Sections nested deeper than start (its subsections) belong to the span, so
// editing a `##` covers every `###`+ under it up to the next `##` or `#`.
func secionSpanEnd(sections [][]byte, start int) int {
level, _ := sectionHeading(sections[start])
if level == 0 {
return start + 1
}
for i := start + 1; i <= len(sections); i++ {
if l, _ := sectionHeading(sections[i]); l > 0 && l <= level {
return i
}
}
return len(sections)
}