package docsite import ( "bytes" "cmp" "fmt" "html" "path" "slices" "strings" "unicode" "unicode/utf8" "github.com/yuin/goldmark" "github.com/yuin/goldmark/ast" "github.com/yuin/goldmark/extension" "github.com/yuin/goldmark/parser" "github.com/yuin/goldmark/renderer" "github.com/yuin/goldmark/text" "github.com/yuin/goldmark/util" ) // newMarkdown returns the one goldmark pipeline every page goes through: // CommonMark plus GFM, heading IDs, and the page transformers. The html // renderer keeps its default safe mode, so raw HTML in Markdown is never // passed through. func newMarkdown() goldmark.Markdown { return goldmark.New( goldmark.WithExtensions(extension.GFM), goldmark.WithParserOptions( parser.WithAutoHeadingID(), parser.WithASTTransformers( util.Prioritized(h1Stripper{}, 100), util.Prioritized(linkRewriter{}, 200), util.Prioritized(fenceAnnotator{}, 300), ), ), // goldmark registers lower priority values last, so 100 overrides the // default html renderer (1000) for fenced code blocks. goldmark.WithRendererOptions(renderer.WithNodeRenderers(util.Prioritized(codeRenderer{}, 100))), ) } // fenceAnnotator marks fences that carry src= with the reference and its // source_url link, for codeRenderer's caption. type fenceAnnotator struct{} func (fenceAnnotator) Transform(doc *ast.Document, reader text.Reader, pc parser.Context) { pctx, _ := pc.Get(pageKey).(*pageContext) src := reader.Source() _ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) { fc, ok := n.(*ast.FencedCodeBlock) if !entering || !ok || fc.Info == nil { return ast.WalkContinue, nil } ref, ok := ParseSrc(string(fc.Info.Segment.Value(src))) if !ok { return ast.WalkContinue, nil } fc.SetAttributeString("data-src", []byte(ref.String())) if pctx != nil && pctx.site.cfg.SourceURL != "" { fc.SetAttributeString("data-href", []byte(strings.ReplaceAll(pctx.site.cfg.SourceURL, "{path}", ref.Path))) } return ast.WalkSkipChildren, nil }) } // codeRenderer renders every fenced code block inside
; // a src= fence gets a
naming its source, linked to source_url. type codeRenderer struct{} func (codeRenderer) RegisterFuncs(r renderer.NodeRendererFuncRegisterer) { r.Register(ast.KindFencedCodeBlock, renderFence) } func attrString(n ast.Node, name string) string { v, ok := n.AttributeString(name) if !ok { return "" } b, _ := v.([]byte) return string(b) } func renderFence(w util.BufWriter, src []byte, node ast.Node, entering bool) (ast.WalkStatus, error) { if !entering { return ast.WalkContinue, nil } n := node.(*ast.FencedCodeBlock) _, _ = w.WriteString(`
`) if ref := attrString(n, "data-src"); ref != "" { _, _ = w.WriteString("
") if href := attrString(n, "data-href"); href != "" { _, _ = w.WriteString(`` + html.EscapeString(ref) + "") } else { _, _ = w.WriteString(html.EscapeString(ref)) } _, _ = w.WriteString("
") } _, _ = w.WriteString("
')
	for i := 0; i < n.Lines().Len(); i++ {
		line := n.Lines().At(i)
		_, _ = w.WriteString(html.EscapeString(string(line.Value(src))))
	}
	_, _ = w.WriteString("
\n") return ast.WalkSkipChildren, nil } // h1Stripper removes the page's leading "# Title" heading; the template // renders the title once. type h1Stripper struct{} func (h1Stripper) Transform(doc *ast.Document, _ text.Reader, _ parser.Context) { if h, ok := doc.FirstChild().(*ast.Heading); ok && h.Level == 1 { doc.RemoveChild(doc, h) } } // slugIDs is the one heading-ID algorithm of the site, shared by the // renderer, the search index and the link checker. It is GitHub-compatible: // lowercase, Unicode letters, digits, "_" and "-" kept, spaces mapped to // "-", other punctuation dropped, and duplicates suffixed -1, -2. type slugIDs struct { used map[string]bool } func newSlugIDs() *slugIDs { return &slugIDs{used: map[string]bool{}} } // Generate implements parser.IDs. func (s *slugIDs) Generate(value []byte, _ ast.NodeKind) []byte { base := slugify(string(value)) if base == "" { base = "section" } id := base for i := 1; s.used[id]; i++ { id = fmt.Sprintf("%s-%d", base, i) } s.used[id] = true return []byte(id) } // Put implements parser.IDs. func (s *slugIDs) Put(value []byte) { s.used[string(value)] = true } func slugify(v string) string { var b strings.Builder for _, r := range strings.ToLower(strings.TrimSpace(v)) { switch { case unicode.IsLetter(r) || unicode.IsDigit(r) || r == '_' || r == '-': b.WriteRune(r) case unicode.IsSpace(r): b.WriteByte('-') } } return b.String() } var pageKey = parser.NewContextKey() // pageContext carries per-page state into the AST transformers. type pageContext struct { site *site page *Page // mdLinks maps a rewritten link destination to its .md form, for the // raw Markdown output. mdLinks map[string]string } // linkRewriter points relative links at other pages to their site URLs: // a guide's link to another page's .md, or to modules//README.md, and a // README's ..//README.md all become the target page. Fragments are kept; // external and mailto links are untouched. type linkRewriter struct{} func (linkRewriter) Transform(doc *ast.Document, _ text.Reader, pc parser.Context) { pctx, _ := pc.Get(pageKey).(*pageContext) if pctx == nil { return } _ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) { link, ok := n.(*ast.Link) if !entering || !ok { return ast.WalkContinue, nil } dest := string(link.Destination) target, frag, ok := pctx.site.resolveLink(pctx.page, dest) if !ok { return ast.WalkContinue, nil } link.Destination = []byte(pctx.site.url(target.URL+".html") + frag) pctx.mdLinks[dest] = pctx.site.url(target.URL+".md") + frag return ast.WalkContinue, nil }) } // resolveLink maps a relative link destination in page from to the page it // names, with its "#fragment" (or ""). func (s *site) resolveLink(from *Page, dest string) (*Page, string, bool) { if dest == "" || strings.HasPrefix(dest, "#") || strings.HasPrefix(dest, "/") || hasScheme(dest) { return nil, "", false } target, frag, _ := strings.Cut(dest, "#") if frag != "" { frag = "#" + frag } if !strings.HasSuffix(target, ".md") { return nil, "", false } p, ok := s.bySource[path.Clean(path.Join(path.Dir(from.Source), target))] if !ok { return nil, "", false } return p, frag, true } func hasScheme(dest string) bool { i := strings.IndexByte(dest, ':') return i > 0 && !strings.ContainsAny(dest[:i], "/?#") } // heading is one heading of a rendered page. type heading struct { Level int ID string Text string node *ast.Heading } // renderedPage is one page after parsing and rendering. type renderedPage struct { html []byte doc ast.Node headings []heading mdLinks map[string]string } // parsePage parses a page body with a fresh slug-ID table and the page's // transformer context. func (s *site) parsePage(md goldmark.Markdown, p *Page) (ast.Node, *pageContext) { pctx := &pageContext{site: s, page: p, mdLinks: map[string]string{}} ctx := parser.NewContext(parser.WithIDs(newSlugIDs())) ctx.Set(pageKey, pctx) doc := md.Parser().Parse(text.NewReader(p.Body), parser.WithContext(ctx)) return doc, pctx } // parseRaw parses a Markdown body with the renderer's parser and slug IDs // but no page context, so link destinations stay as written. func (s *site) parseRaw(md goldmark.Markdown, body []byte) ast.Node { ctx := parser.NewContext(parser.WithIDs(newSlugIDs())) return md.Parser().Parse(text.NewReader(body), parser.WithContext(ctx)) } // renderPage parses and renders a page body to HTML. func (s *site) renderPage(md goldmark.Markdown, p *Page) (renderedPage, error) { doc, pctx := s.parsePage(md, p) var buf bytes.Buffer if err := md.Renderer().Render(&buf, p.Body, doc); err != nil { return renderedPage{}, fmt.Errorf("docsite: render %s: %w", p.Source, err) } return renderedPage{html: buf.Bytes(), doc: doc, headings: pageHeadings(doc, p.Body), mdLinks: pctx.mdLinks}, nil } // headingIDs returns the heading IDs of a page, in document order, from the // same parse and slug algorithm the renderer uses. func (s *site) headingIDs(md goldmark.Markdown, p *Page) []string { doc, _ := s.parsePage(md, p) var ids []string for _, h := range pageHeadings(doc, p.Body) { ids = append(ids, h.ID) } return ids } func pageHeadings(doc ast.Node, src []byte) []heading { var out []heading _ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) { h, ok := n.(*ast.Heading) if !entering || !ok { return ast.WalkContinue, nil } id, _ := h.AttributeString("id") idb, _ := id.([]byte) out = append(out, heading{Level: h.Level, ID: string(idb), Text: plainText(h, src), node: h}) return ast.WalkSkipChildren, nil }) return out } // plainText extracts the text of a node from its text nodes only (never // from rendered HTML), with whitespace collapsed. func plainText(n ast.Node, src []byte) string { var b strings.Builder _ = ast.Walk(n, func(c ast.Node, entering bool) (ast.WalkStatus, error) { if !entering { return ast.WalkContinue, nil } switch t := c.(type) { case *ast.Text: b.Write(t.Segment.Value(src)) if t.SoftLineBreak() || t.HardLineBreak() { b.WriteByte(' ') } case *ast.String: b.Write(t.Value) case *ast.FencedCodeBlock, *ast.CodeBlock, *ast.HTMLBlock, *ast.RawHTML: return ast.WalkSkipChildren, nil case *ast.Paragraph, *ast.Heading, *ast.ListItem, *ast.TextBlock: b.WriteByte(' ') } return ast.WalkContinue, nil }) return strings.Join(strings.Fields(b.String()), " ") } // sectionText is the plain text of the blocks after h up to the next // heading of the same or a higher level, cut to max runes. func sectionText(h *ast.Heading, src []byte, max int) string { var parts []string for n := h.NextSibling(); n != nil; n = n.NextSibling() { if next, ok := n.(*ast.Heading); ok && next.Level <= h.Level { break } if t := plainText(n, src); t != "" { parts = append(parts, t) } } return truncateRunes(strings.Join(parts, " "), max) } func truncateRunes(s string, max int) string { if utf8.RuneCountInString(s) <= max { return s } r := []rune(s) return strings.TrimSpace(string(r[:max])) } // fence is one fenced code block found by scanning Markdown lines. type fence struct { open, close int // 0-based line indexes; close == -1 when unterminated indent int char byte count int info string } // scanFences finds fenced code blocks in Markdown lines, following the // CommonMark opening and closing rules. func scanFences(lines []string) []fence { var out []fence for i := 0; i < len(lines); i++ { f, ok := openFence(lines[i]) if !ok { continue } f.open, f.close = i, -1 for j := i + 1; j < len(lines); j++ { if closesFence(lines[j], f) { f.close = j break } } out = append(out, f) if f.close < 0 { break } i = f.close } return out } func openFence(line string) (fence, bool) { trimmed := strings.TrimLeft(line, " ") if len(trimmed) < 3 || (trimmed[0] != '`' && trimmed[0] != '~') { return fence{}, false } c := trimmed[0] n := 0 for n < len(trimmed) && trimmed[n] == c { n++ } if n < 3 { return fence{}, false } info := strings.TrimSpace(trimmed[n:]) if c == '`' && strings.ContainsRune(info, '`') { return fence{}, false } return fence{indent: len(line) - len(trimmed), char: c, count: n, info: info}, true } func closesFence(line string, f fence) bool { trimmed := strings.TrimLeft(line, " ") n := 0 for n < len(trimmed) && trimmed[n] == f.char { n++ } return n >= f.count && strings.TrimSpace(trimmed[n:]) == "" } // rewriteMarkdown applies the raw-output transforms to a Markdown body: // fence info strings are reduced to their language word (dropping src=), // and link destinations outside code fences are replaced through links. func rewriteMarkdown(body string, links map[string]string) string { lines := strings.Split(body, "\n") inFence := make([]bool, len(lines)) for _, f := range scanFences(lines) { end := f.close if end < 0 { end = len(lines) - 1 } for i := f.open; i <= end; i++ { inFence[i] = true } } origs := make([]string, 0, len(links)) for o := range links { origs = append(origs, o) } slices.SortFunc(origs, func(a, b string) int { return cmp.Or(cmp.Compare(len(b), len(a)), strings.Compare(a, b)) }) for _, f := range scanFences(lines) { if lang, _, _ := strings.Cut(f.info, " "); lang != f.info { lines[f.open] = strings.Repeat(" ", f.indent) + strings.Repeat(string(f.char), f.count) + lang } } for i, line := range lines { if inFence[i] || !strings.Contains(line, "](") { continue } for _, o := range origs { line = strings.ReplaceAll(line, "]("+o+")", "]("+links[o]+")") line = strings.ReplaceAll(line, "]("+o+" ", "]("+links[o]+" ") } lines[i] = line } return strings.Join(lines, "\n") }