- discover modules/<m> with non-test Go files; a missing README is a readme: problem - one GitHub-compatible slug parser.IDs for heading anchors, passed per page - rewrite links to .md pages and module READMEs to site .html and .md URLs - search-index.json gains one entry per H2 with 300-char plain text - add the api section to docs/site.yaml; docsite.Pages exposes reading order - tests: TestSlugIDs, TestReadmeIngestion, TestEveryModuleInSidebar, TestDocsAIOutputsInSync
360 lines
9.8 KiB
Go
360 lines
9.8 KiB
Go
package docsite
|
|
|
|
import (
|
|
"bytes"
|
|
"cmp"
|
|
"fmt"
|
|
"path"
|
|
"slices"
|
|
"strings"
|
|
"unicode"
|
|
"unicode/utf8"
|
|
|
|
"github.com/yuin/goldmark"
|
|
"github.com/yuin/goldmark/ast"
|
|
"github.com/yuin/goldmark/extension"
|
|
"github.com/yuin/goldmark/parser"
|
|
"github.com/yuin/goldmark/text"
|
|
"github.com/yuin/goldmark/util"
|
|
)
|
|
|
|
// newMarkdown returns the one goldmark pipeline every page goes through:
|
|
// CommonMark plus GFM, heading IDs, and the page transformers. The html
|
|
// renderer keeps its default safe mode, so raw HTML in Markdown is never
|
|
// passed through.
|
|
func newMarkdown() goldmark.Markdown {
|
|
return goldmark.New(
|
|
goldmark.WithExtensions(extension.GFM),
|
|
goldmark.WithParserOptions(
|
|
parser.WithAutoHeadingID(),
|
|
parser.WithASTTransformers(
|
|
util.Prioritized(h1Stripper{}, 100),
|
|
util.Prioritized(linkRewriter{}, 200),
|
|
),
|
|
),
|
|
)
|
|
}
|
|
|
|
// h1Stripper removes the page's leading "# Title" heading; the template
|
|
// renders the title once.
|
|
type h1Stripper struct{}
|
|
|
|
func (h1Stripper) Transform(doc *ast.Document, _ text.Reader, _ parser.Context) {
|
|
if h, ok := doc.FirstChild().(*ast.Heading); ok && h.Level == 1 {
|
|
doc.RemoveChild(doc, h)
|
|
}
|
|
}
|
|
|
|
// slugIDs is the one heading-ID algorithm of the site, shared by the
|
|
// renderer, the search index and the link checker. It is GitHub-compatible:
|
|
// lowercase, Unicode letters, digits, "_" and "-" kept, spaces mapped to
|
|
// "-", other punctuation dropped, and duplicates suffixed -1, -2.
|
|
type slugIDs struct {
|
|
used map[string]bool
|
|
}
|
|
|
|
func newSlugIDs() *slugIDs {
|
|
return &slugIDs{used: map[string]bool{}}
|
|
}
|
|
|
|
// Generate implements parser.IDs.
|
|
func (s *slugIDs) Generate(value []byte, _ ast.NodeKind) []byte {
|
|
base := slugify(string(value))
|
|
if base == "" {
|
|
base = "section"
|
|
}
|
|
id := base
|
|
for i := 1; s.used[id]; i++ {
|
|
id = fmt.Sprintf("%s-%d", base, i)
|
|
}
|
|
s.used[id] = true
|
|
return []byte(id)
|
|
}
|
|
|
|
// Put implements parser.IDs.
|
|
func (s *slugIDs) Put(value []byte) {
|
|
s.used[string(value)] = true
|
|
}
|
|
|
|
func slugify(v string) string {
|
|
var b strings.Builder
|
|
for _, r := range strings.ToLower(strings.TrimSpace(v)) {
|
|
switch {
|
|
case unicode.IsLetter(r) || unicode.IsDigit(r) || r == '_' || r == '-':
|
|
b.WriteRune(r)
|
|
case unicode.IsSpace(r):
|
|
b.WriteByte('-')
|
|
}
|
|
}
|
|
return b.String()
|
|
}
|
|
|
|
var pageKey = parser.NewContextKey()
|
|
|
|
// pageContext carries per-page state into the AST transformers.
|
|
type pageContext struct {
|
|
site *site
|
|
page *Page
|
|
// mdLinks maps a rewritten link destination to its .md form, for the
|
|
// raw Markdown output.
|
|
mdLinks map[string]string
|
|
}
|
|
|
|
// linkRewriter points relative links at other pages to their site URLs:
|
|
// a guide's link to another page's .md, or to modules/<m>/README.md, and a
|
|
// README's ../<m>/README.md all become the target page. Fragments are kept;
|
|
// external and mailto links are untouched.
|
|
type linkRewriter struct{}
|
|
|
|
func (linkRewriter) Transform(doc *ast.Document, _ text.Reader, pc parser.Context) {
|
|
pctx, _ := pc.Get(pageKey).(*pageContext)
|
|
if pctx == nil {
|
|
return
|
|
}
|
|
_ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) {
|
|
link, ok := n.(*ast.Link)
|
|
if !entering || !ok {
|
|
return ast.WalkContinue, nil
|
|
}
|
|
dest := string(link.Destination)
|
|
target, frag, ok := pctx.site.resolveLink(pctx.page, dest)
|
|
if !ok {
|
|
return ast.WalkContinue, nil
|
|
}
|
|
link.Destination = []byte(pctx.site.url(target.URL+".html") + frag)
|
|
pctx.mdLinks[dest] = pctx.site.url(target.URL+".md") + frag
|
|
return ast.WalkContinue, nil
|
|
})
|
|
}
|
|
|
|
// resolveLink maps a relative link destination in page from to the page it
|
|
// names, with its "#fragment" (or "").
|
|
func (s *site) resolveLink(from *Page, dest string) (*Page, string, bool) {
|
|
if dest == "" || strings.HasPrefix(dest, "#") || strings.HasPrefix(dest, "/") || hasScheme(dest) {
|
|
return nil, "", false
|
|
}
|
|
target, frag, _ := strings.Cut(dest, "#")
|
|
if frag != "" {
|
|
frag = "#" + frag
|
|
}
|
|
if !strings.HasSuffix(target, ".md") {
|
|
return nil, "", false
|
|
}
|
|
p, ok := s.bySource[path.Clean(path.Join(path.Dir(from.Source), target))]
|
|
if !ok {
|
|
return nil, "", false
|
|
}
|
|
return p, frag, true
|
|
}
|
|
|
|
func hasScheme(dest string) bool {
|
|
i := strings.IndexByte(dest, ':')
|
|
return i > 0 && !strings.ContainsAny(dest[:i], "/?#")
|
|
}
|
|
|
|
// heading is one heading of a rendered page.
|
|
type heading struct {
|
|
Level int
|
|
ID string
|
|
Text string
|
|
node *ast.Heading
|
|
}
|
|
|
|
// renderedPage is one page after parsing and rendering.
|
|
type renderedPage struct {
|
|
html []byte
|
|
doc ast.Node
|
|
headings []heading
|
|
mdLinks map[string]string
|
|
}
|
|
|
|
// parsePage parses a page body with a fresh slug-ID table and the page's
|
|
// transformer context.
|
|
func (s *site) parsePage(md goldmark.Markdown, p *Page) (ast.Node, *pageContext) {
|
|
pctx := &pageContext{site: s, page: p, mdLinks: map[string]string{}}
|
|
ctx := parser.NewContext(parser.WithIDs(newSlugIDs()))
|
|
ctx.Set(pageKey, pctx)
|
|
doc := md.Parser().Parse(text.NewReader(p.Body), parser.WithContext(ctx))
|
|
return doc, pctx
|
|
}
|
|
|
|
// renderPage parses and renders a page body to HTML.
|
|
func (s *site) renderPage(md goldmark.Markdown, p *Page) (renderedPage, error) {
|
|
doc, pctx := s.parsePage(md, p)
|
|
var buf bytes.Buffer
|
|
if err := md.Renderer().Render(&buf, p.Body, doc); err != nil {
|
|
return renderedPage{}, fmt.Errorf("docsite: render %s: %w", p.Source, err)
|
|
}
|
|
return renderedPage{html: buf.Bytes(), doc: doc, headings: pageHeadings(doc, p.Body), mdLinks: pctx.mdLinks}, nil
|
|
}
|
|
|
|
// headingIDs returns the heading IDs of a page, in document order, from the
|
|
// same parse and slug algorithm the renderer uses.
|
|
func (s *site) headingIDs(md goldmark.Markdown, p *Page) []string {
|
|
doc, _ := s.parsePage(md, p)
|
|
var ids []string
|
|
for _, h := range pageHeadings(doc, p.Body) {
|
|
ids = append(ids, h.ID)
|
|
}
|
|
return ids
|
|
}
|
|
|
|
func pageHeadings(doc ast.Node, src []byte) []heading {
|
|
var out []heading
|
|
_ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) {
|
|
h, ok := n.(*ast.Heading)
|
|
if !entering || !ok {
|
|
return ast.WalkContinue, nil
|
|
}
|
|
id, _ := h.AttributeString("id")
|
|
idb, _ := id.([]byte)
|
|
out = append(out, heading{Level: h.Level, ID: string(idb), Text: plainText(h, src), node: h})
|
|
return ast.WalkSkipChildren, nil
|
|
})
|
|
return out
|
|
}
|
|
|
|
// plainText extracts the text of a node from its text nodes only (never
|
|
// from rendered HTML), with whitespace collapsed.
|
|
func plainText(n ast.Node, src []byte) string {
|
|
var b strings.Builder
|
|
_ = ast.Walk(n, func(c ast.Node, entering bool) (ast.WalkStatus, error) {
|
|
if !entering {
|
|
return ast.WalkContinue, nil
|
|
}
|
|
switch t := c.(type) {
|
|
case *ast.Text:
|
|
b.Write(t.Segment.Value(src))
|
|
if t.SoftLineBreak() || t.HardLineBreak() {
|
|
b.WriteByte(' ')
|
|
}
|
|
case *ast.String:
|
|
b.Write(t.Value)
|
|
case *ast.FencedCodeBlock, *ast.CodeBlock, *ast.HTMLBlock, *ast.RawHTML:
|
|
return ast.WalkSkipChildren, nil
|
|
case *ast.Paragraph, *ast.Heading, *ast.ListItem, *ast.TextBlock:
|
|
b.WriteByte(' ')
|
|
}
|
|
return ast.WalkContinue, nil
|
|
})
|
|
return strings.Join(strings.Fields(b.String()), " ")
|
|
}
|
|
|
|
// sectionText is the plain text of the blocks after h up to the next
|
|
// heading of the same or a higher level, cut to max runes.
|
|
func sectionText(h *ast.Heading, src []byte, max int) string {
|
|
var parts []string
|
|
for n := h.NextSibling(); n != nil; n = n.NextSibling() {
|
|
if next, ok := n.(*ast.Heading); ok && next.Level <= h.Level {
|
|
break
|
|
}
|
|
if t := plainText(n, src); t != "" {
|
|
parts = append(parts, t)
|
|
}
|
|
}
|
|
return truncateRunes(strings.Join(parts, " "), max)
|
|
}
|
|
|
|
func truncateRunes(s string, max int) string {
|
|
if utf8.RuneCountInString(s) <= max {
|
|
return s
|
|
}
|
|
r := []rune(s)
|
|
return strings.TrimSpace(string(r[:max]))
|
|
}
|
|
|
|
// fence is one fenced code block found by scanning Markdown lines.
|
|
type fence struct {
|
|
open, close int // 0-based line indexes; close == -1 when unterminated
|
|
indent int
|
|
char byte
|
|
count int
|
|
info string
|
|
}
|
|
|
|
// scanFences finds fenced code blocks in Markdown lines, following the
|
|
// CommonMark opening and closing rules.
|
|
func scanFences(lines []string) []fence {
|
|
var out []fence
|
|
for i := 0; i < len(lines); i++ {
|
|
f, ok := openFence(lines[i])
|
|
if !ok {
|
|
continue
|
|
}
|
|
f.open, f.close = i, -1
|
|
for j := i + 1; j < len(lines); j++ {
|
|
if closesFence(lines[j], f) {
|
|
f.close = j
|
|
break
|
|
}
|
|
}
|
|
out = append(out, f)
|
|
if f.close < 0 {
|
|
break
|
|
}
|
|
i = f.close
|
|
}
|
|
return out
|
|
}
|
|
|
|
func openFence(line string) (fence, bool) {
|
|
trimmed := strings.TrimLeft(line, " ")
|
|
if len(trimmed) < 3 || (trimmed[0] != '`' && trimmed[0] != '~') {
|
|
return fence{}, false
|
|
}
|
|
c := trimmed[0]
|
|
n := 0
|
|
for n < len(trimmed) && trimmed[n] == c {
|
|
n++
|
|
}
|
|
if n < 3 {
|
|
return fence{}, false
|
|
}
|
|
info := strings.TrimSpace(trimmed[n:])
|
|
if c == '`' && strings.ContainsRune(info, '`') {
|
|
return fence{}, false
|
|
}
|
|
return fence{indent: len(line) - len(trimmed), char: c, count: n, info: info}, true
|
|
}
|
|
|
|
func closesFence(line string, f fence) bool {
|
|
trimmed := strings.TrimLeft(line, " ")
|
|
n := 0
|
|
for n < len(trimmed) && trimmed[n] == f.char {
|
|
n++
|
|
}
|
|
return n >= f.count && strings.TrimSpace(trimmed[n:]) == ""
|
|
}
|
|
|
|
// rewriteMarkdown applies the raw-output transforms to a Markdown body:
|
|
// link destinations outside code fences are replaced through links.
|
|
func rewriteMarkdown(body string, links map[string]string) string {
|
|
lines := strings.Split(body, "\n")
|
|
inFence := make([]bool, len(lines))
|
|
for _, f := range scanFences(lines) {
|
|
end := f.close
|
|
if end < 0 {
|
|
end = len(lines) - 1
|
|
}
|
|
for i := f.open; i <= end; i++ {
|
|
inFence[i] = true
|
|
}
|
|
}
|
|
origs := make([]string, 0, len(links))
|
|
for o := range links {
|
|
origs = append(origs, o)
|
|
}
|
|
slices.SortFunc(origs, func(a, b string) int { return cmp.Or(cmp.Compare(len(b), len(a)), strings.Compare(a, b)) })
|
|
for i, line := range lines {
|
|
if inFence[i] || !strings.Contains(line, "](") {
|
|
continue
|
|
}
|
|
for _, o := range origs {
|
|
line = strings.ReplaceAll(line, "]("+o+")", "]("+links[o]+")")
|
|
line = strings.ReplaceAll(line, "]("+o+" ", "]("+links[o]+" ")
|
|
}
|
|
lines[i] = line
|
|
}
|
|
return strings.Join(lines, "\n")
|
|
}
|