package main import ( "bytes" "path" "regexp" "strings" "github.com/yuin/goldmark" "github.com/yuin/goldmark/ast" "github.com/yuin/goldmark/parser" "github.com/yuin/goldmark/renderer" "github.com/yuin/goldmark/text" "github.com/yuin/goldmark/util" ) // wikiEmbedRe matches an ![[...]] token anchored at the current inline reader. // The inner blob forbids brackets and newlines but allows ':' so the whole // target::caption::align payload travels as one capture — the caption and // alignment tail is split apart in Go (parseEmbedFields). This mirrors // wikiLinkPattern's shape so the move rewriter needs no embed-specific changes. var wikiEmbedRe = regexp.MustCompile(`^!\[\[([^\[\]\n]+)\]\]`) const ( alignLeft = "left" alignRight = "right" alignCenter = "center" ) // wikiEmbedNode is the AST node produced by wikiEmbedParser. type wikiEmbedNode struct { ast.BaseInline Target string Caption string Align string } var kindWikiEmbed = ast.NewNodeKind("WikiEmbed") func (n *wikiEmbedNode) Kind() ast.NodeKind { return kindWikiEmbed } func (n *wikiEmbedNode) Dump(source []byte, level int) { ast.DumpHelper(n, source, level, map[string]string{ "Target": n.Target, "Caption": n.Caption, "Align": n.Align, }, nil) } // alignKeyword reports whether s (trimmed) is one of the alignment keywords. func alignKeyword(s string) (string, bool) { switch strings.TrimSpace(s) { case alignLeft: return alignLeft, true case alignRight: return alignRight, true case alignCenter: return alignCenter, true } return "", false } // parseEmbedFields splits the inner ![[...]] blob into target, caption, and // alignment per the embed syntax rules. The target is the field before the // first "::". The trailing "::"-fields are interpreted as: // - none: no caption, default alignment (right) // - one keyword field: that alignment, no caption // - one non-keyword field: that caption, default alignment // - two or more: the last field is the alignment slot (default when it is not // a keyword) and the earlier fields re-join with "::" as the caption, so a // caption may contain a literal "::" as long as an alignment field trails it. func parseEmbedFields(inner string) (target, caption, align string) { fields := strings.Split(inner, "::") target = strings.TrimSpace(fields[0]) tail := fields[1:] align = alignRight switch len(tail) { case 0: // target only case 1: if kw, ok := alignKeyword(tail[0]); ok { align = kw } else { caption = strings.TrimSpace(tail[0]) } default: if kw, ok := alignKeyword(tail[len(tail)-1]); ok { align = kw } caption = strings.TrimSpace(strings.Join(tail[:len(tail)-1], "::")) } return target, caption, align } type wikiEmbedParser struct{} func (p *wikiEmbedParser) Trigger() []byte { return []byte{'!'} } func (p *wikiEmbedParser) Parse(parent ast.Node, block text.Reader, pc parser.Context) ast.Node { line, _ := block.PeekLine() // Require the exact "") // falls through to goldmark's default image parser at priority 200. if len(line) < 5 || line[0] != '!' || line[1] != '[' || line[2] != '[' { return nil } m := wikiEmbedRe.FindSubmatchIndex(line) if m == nil { return nil } target, caption, align := parseEmbedFields(string(line[m[2]:m[3]])) if !isValidWikiTarget([]byte(target)) { return nil } block.Advance(m[1]) return &wikiEmbedNode{Target: target, Caption: caption, Align: align} } // embedBlockTransformer lifts an embed that sits alone in a paragraph up to // block level. Goldmark wraps inline content in
, but the embed renders a
// is auto-closed by the browser —
// which strands empty elements (they still carry .content paragraph margins)
// and breaks the float layout once several embeds share a page. Dissolving the
// wrapping paragraph makes each embed render as a clean block sibling with no
// stray .
type embedBlockTransformer struct{ root string }
func (t embedBlockTransformer) Transform(doc *ast.Document, reader text.Reader, pc parser.Context) {
source := reader.Source()
var paras []*ast.Paragraph
ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) {
if entering {
if p, ok := n.(*ast.Paragraph); ok && t.paragraphFiguresOnly(p, source) {
paras = append(paras, p)
}
}
return ast.WalkContinue, nil
})
for _, p := range paras {
parent := p.Parent()
if parent == nil {
continue
}
// InsertBefore isolates the embed from p first, so hoist each embed to a
// block sibling ahead of the (soon removed) paragraph, then drop p.
for c := p.FirstChild(); c != nil; {
next := c.NextSibling()
if _, ok := c.(*wikiEmbedNode); ok {
parent.InsertBefore(parent, p, c)
}
c = next
}
parent.RemoveChild(parent, p)
}
}
// paragraphFiguresOnly reports whether p holds at least one figure-rendering
// embed and nothing else visible — only such embeds and whitespace/line-break
// text. Only these are safe to dissolve: a paragraph carrying prose, or an embed
// that degrades to an inline link (missing / non-image target), stays wrapped so
// the fallback anchor keeps its paragraph.
func (t embedBlockTransformer) paragraphFiguresOnly(p *ast.Paragraph, source []byte) bool {
hasFigure := false
for c := p.FirstChild(); c != nil; c = c.NextSibling() {
switch n := c.(type) {
case *wikiEmbedNode:
if !embedIsImage(t.root, n.Target) {
return false
}
hasFigure = true
case *ast.Text:
if len(bytes.TrimSpace(n.Segment.Value(source))) != 0 {
return false
}
default:
return false
}
}
return hasFigure
}
// embedIsImage reports whether target resolves to an existing image file, i.e.
// the embed will render as a `)
if n.Caption != "" {
w.WriteString(`