From e433dcf0c932bf6aa0e4c7e177b93e2c08fc66c0 Mon Sep 17 00:00:00 2001 From: Jakub Zych Date: Wed, 30 Sep 2026 21:21:56 +0200 Subject: [PATCH] feat(11.1-01): publish every module README as an API reference page - discover modules/ with non-test Go files; a missing README is a readme: problem - one GitHub-compatible slug parser.IDs for heading anchors, passed per page - rewrite links to .md pages and module READMEs to site .html and .md URLs - search-index.json gains one entry per H2 with 300-char plain text - add the api section to docs/site.yaml; docsite.Pages exposes reading order - tests: TestSlugIDs, TestReadmeIngestion, TestEveryModuleInSidebar, TestDocsAIOutputsInSync --- cmd/summer/docs_test.go | 166 ++++++++++++++++ docs/site.yaml | 2 + internal/docsite/docsite_test.go | 149 +++++++++++++++ internal/docsite/emit.go | 55 ++++-- internal/docsite/load.go | 167 +++++++++++++++-- internal/docsite/render.go | 312 ++++++++++++++++++++++++++++++- 6 files changed, 815 insertions(+), 36 deletions(-) create mode 100644 internal/docsite/docsite_test.go diff --git a/cmd/summer/docs_test.go b/cmd/summer/docs_test.go index 70f8e62..956478a 100644 --- a/cmd/summer/docs_test.go +++ b/cmd/summer/docs_test.go @@ -1,10 +1,14 @@ package main import ( + "bufio" "bytes" + "io/fs" "os" "path/filepath" "regexp" + "slices" + "strings" "testing" "git.golem15.com/golem15/summercms/internal/docsite" @@ -57,3 +61,165 @@ func TestDocsBuildRealTree(t *testing.T) { } } } + +// frameworkModules lists modules/ directories that hold a non-test Go +// file, discovered independently of docsite. +func frameworkModules(t *testing.T) []string { + t.Helper() + entries, err := os.ReadDir(filepath.Join(repoRoot, "modules")) + if err != nil { + t.Fatal(err) + } + var names []string + for _, e := range entries { + if !e.IsDir() { + continue + } + goFiles, err := filepath.Glob(filepath.Join(repoRoot, "modules", e.Name(), "*.go")) + if err != nil { + t.Fatal(err) + } + if slices.ContainsFunc(goFiles, func(f string) bool { return !strings.HasSuffix(f, "_test.go") }) { + names = append(names, e.Name()) + } + } + if len(names) == 0 { + t.Fatal("no framework modules found") + } + return names +} + +func TestEveryModuleInSidebar(t *testing.T) { + out, _ := buildRealTree(t) + index, err := os.ReadFile(filepath.Join(out, "index.html")) + if err != nil { + t.Fatal(err) + } + start := bytes.Index(index, []byte(`")) + if start < 0 || end < 0 { + t.Fatal("index.html has no sidebar nav") + } + sidebar := string(index[start : start+end]) + for _, m := range frameworkModules(t) { + if _, err := os.Stat(filepath.Join(out, "api", m+".html")); err != nil { + t.Errorf("module %s has no API page: %v", m, err) + } + if !strings.Contains(sidebar, `href="/api/`+m+`.html"`) { + t.Errorf("sidebar does not link api/%s.html", m) + } + } +} + +// TestDocsAIOutputsInSync asserts that the page tree, the .html pages, the +// .md siblings, llms.txt and llms-full.txt all list the same pages in the +// same reading order. +func TestDocsAIOutputsInSync(t *testing.T) { + pages, problems, err := docsite.Pages(docsite.Options{Root: repoRoot}) + if err != nil || len(problems) > 0 { + t.Fatalf("Pages: %v %v", err, problems) + } + var want []string + for _, p := range pages { + want = append(want, p.URL) + } + out, _ := buildRealTree(t) + + var htmlFiles, mdFiles []string + if err := filepath.WalkDir(out, func(p string, d fs.DirEntry, err error) error { + if err != nil || d.IsDir() { + return err + } + rel, _ := filepath.Rel(out, p) + rel = filepath.ToSlash(rel) + if strings.HasPrefix(rel, "assets/") { + return nil + } + switch filepath.Ext(rel) { + case ".html": + htmlFiles = append(htmlFiles, strings.TrimSuffix(rel, ".html")) + case ".md": + mdFiles = append(mdFiles, strings.TrimSuffix(rel, ".md")) + } + return nil + }); err != nil { + t.Fatal(err) + } + sorted := slices.Sorted(slices.Values(want)) + slices.Sort(htmlFiles) + slices.Sort(mdFiles) + if !slices.Equal(htmlFiles, sorted) { + t.Errorf(".html pages = %v, want %v", htmlFiles, sorted) + } + if !slices.Equal(mdFiles, sorted) { + t.Errorf(".md pages = %v, want %v", mdFiles, sorted) + } + + llms := readLines(t, filepath.Join(out, "llms.txt")) + if len(llms) == 0 || llms[0] != "# SummerCMS" { + t.Fatalf("llms.txt line 1 = %q, want # SummerCMS", first(llms)) + } + if next := nextNonEmpty(llms, 1); next < 0 || !strings.HasPrefix(llms[next], "> ") { + t.Errorf("llms.txt: the line after the H1 must be a > summary") + } + item := regexp.MustCompile(`^- \[[^\]]+\]\(/([^)]+)\.md\): \S`) + var linked []string + for i, line := range llms { + if strings.HasPrefix(line, "## ") { + if next := nextNonEmpty(llms, i+1); next < 0 || !strings.HasPrefix(llms[next], "- [") { + t.Errorf("llms.txt: %q is not followed by a link list", line) + } + } + if m := item.FindStringSubmatch(line); m != nil { + linked = append(linked, m[1]) + } + } + if !slices.Equal(linked, want) { + t.Errorf("llms.txt links = %v, want %v", linked, want) + } + + var sources []string + for _, line := range readLines(t, filepath.Join(out, "llms-full.txt")) { + if rest, ok := strings.CutPrefix(line, "Source: /"); ok { + sources = append(sources, strings.TrimSuffix(rest, ".html")) + } + } + if !slices.Equal(sources, want) { + t.Errorf("llms-full.txt sources = %v, want %v", sources, want) + } +} + +func readLines(t *testing.T, path string) []string { + t.Helper() + f, err := os.Open(path) + if err != nil { + t.Fatal(err) + } + defer f.Close() + var lines []string + sc := bufio.NewScanner(f) + sc.Buffer(make([]byte, 0, 1<<20), 1<<24) + for sc.Scan() { + lines = append(lines, sc.Text()) + } + if err := sc.Err(); err != nil { + t.Fatal(err) + } + return lines +} + +func nextNonEmpty(lines []string, from int) int { + for i := from; i < len(lines); i++ { + if strings.TrimSpace(lines[i]) != "" { + return i + } + } + return -1 +} + +func first(lines []string) string { + if len(lines) == 0 { + return "" + } + return lines[0] +} diff --git a/docs/site.yaml b/docs/site.yaml index e98929b..f9bed85 100644 --- a/docs/site.yaml +++ b/docs/site.yaml @@ -13,3 +13,5 @@ llms_notes: sections: - name: setup title: Setup + - name: api + title: API reference diff --git a/internal/docsite/docsite_test.go b/internal/docsite/docsite_test.go new file mode 100644 index 0000000..7aaeb9f --- /dev/null +++ b/internal/docsite/docsite_test.go @@ -0,0 +1,149 @@ +package docsite + +import ( + "os" + "path/filepath" + "slices" + "strconv" + "strings" + "testing" +) + +// writeTree writes files (path relative to root, forward slashes) under a +// fresh temp dir and returns it. +func writeTree(t *testing.T, files map[string]string) string { + t.Helper() + root := t.TempDir() + for name, body := range files { + p := filepath.Join(root, filepath.FromSlash(name)) + if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(p, []byte(body), 0o644); err != nil { + t.Fatal(err) + } + } + return root +} + +func problemLines(problems []Problem) []string { + var out []string + for _, p := range problems { + out = append(out, p.String()) + } + return out +} + +const fixtureSite = `title: Acme +description: Acme docs. +sections: + - name: setup + title: Setup + - name: api + title: API reference +` + +func page(title, section string, order int, body string) string { + return "---\ntitle: " + title + "\ndescription: " + title + " page.\nsection: " + section + + "\norder: " + strconv.Itoa(order) + "\n---\n# " + title + "\n\n" + body +} + +func TestSlugIDs(t *testing.T) { + ids := newSlugIDs() + for _, tc := range []struct{ in, want string }{ + {"Install the CLI", "install-the-cli"}, + {"Install the CLI", "install-the-cli-1"}, + {"Install the CLI", "install-the-cli-2"}, + {"snake_case and dash-case", "snake_case-and-dash-case"}, + {"What's new? (v2.0)", "whats-new-v20"}, + {"`bonfire.Call` usage", "bonfirecall-usage"}, + {"Zażółć gęślą", "zażółć-gęślą"}, + {"!!!", "section"}, + } { + if got := string(ids.Generate([]byte(tc.in), 0)); got != tc.want { + t.Errorf("Generate(%q) = %q, want %q", tc.in, got, tc.want) + } + } + ids.Put([]byte("custom")) + if got := string(ids.Generate([]byte("Custom"), 0)); got != "custom-1" { + t.Errorf("after Put, Generate(Custom) = %q, want custom-1", got) + } +} + +func TestReadmeIngestion(t *testing.T) { + files := map[string]string{ + "docs/site.yaml": fixtureSite, + "docs/index.md": page("Acme docs", "index", 0, "Read [alpha](../modules/alpha/README.md#usage).\n"), + "docs/setup/start.md": page("Start", "setup", 10, "## First steps\n\nText.\n"), + "modules/alpha/alpha.go": "package alpha\n", + "modules/alpha/README.md": "# alpha\n\nAlpha does one thing well.\n\n## Usage\n\n" + + "See [delta](../delta/README.md#api) and [start](../../docs/setup/start.md).\n", + "modules/beta/beta.go": "package beta\n", + "modules/gamma/gamma_test.go": "package gamma\n", + "modules/delta/delta.go": "package delta\n", + "modules/delta/README.md": "# delta\n\nDelta does another thing.\n\n## API\n\nText.\n", + } + root := writeTree(t, files) + + pages, problems, err := Pages(Options{Root: root}) + if err != nil { + t.Fatal(err) + } + want := "modules/beta: readme: package has Go files but no README.md" + if got := problemLines(problems); !slices.Equal(got, []string{want}) { + t.Fatalf("problems = %q, want [%q]", got, want) + } + var urls []string + for _, p := range pages { + urls = append(urls, p.URL) + } + if !slices.Equal(urls, []string{"index", "setup/start", "api/alpha", "api/delta"}) { + t.Fatalf("reading order = %v", urls) + } + alpha := pages[2] + if alpha.Title != "alpha" || alpha.Description != "Alpha does one thing well." || alpha.Section != "api" || + alpha.Module != "alpha" || alpha.Source != "modules/alpha/README.md" { + t.Fatalf("alpha page = %+v", alpha) + } + + if err := os.WriteFile(filepath.Join(root, "modules/beta/README.md"), []byte("# beta\n\nBeta.\n"), 0o644); err != nil { + t.Fatal(err) + } + out := filepath.Join(t.TempDir(), "site") + res, problems, err := Build(Options{Root: root, Out: out}) + if err != nil || len(problems) > 0 { + t.Fatalf("Build: %v %q", err, problemLines(problems)) + } + if res.Pages != 5 { + t.Fatalf("pages = %d, want 5", res.Pages) + } + read := func(name string) string { + t.Helper() + b, err := os.ReadFile(filepath.Join(out, name)) + if err != nil { + t.Fatal(err) + } + return string(b) + } + html := read("api/alpha.html") + for _, want := range []string{`href="/api/delta.html#api"`, `href="/setup/start.html"`, `

Usage

`} { + if !strings.Contains(html, want) { + t.Errorf("api/alpha.html missing %s", want) + } + } + if strings.Contains(html, "

alpha

\n

Alpha does one thing well.

\n

Alpha does") { + t.Error("summary line rendered twice") + } + md := read("api/alpha.md") + if !strings.HasPrefix(md, "# alpha\n\n> Alpha does one thing well.\n\n## Usage\n") || + !strings.Contains(md, "(/api/delta.md#api)") || !strings.Contains(md, "(/setup/start.md)") { + t.Errorf("api/alpha.md = %q", md) + } + if idx := read("index.html"); !strings.Contains(idx, `href="/api/alpha.html#usage"`) { + t.Error("index.html does not link the rewritten module README") + } + search := read("search-index.json") + if !strings.Contains(search, `{"p":2,"a":"usage","h":"Usage","x":"See delta and start."}`) { + t.Errorf("search index missing the alpha Usage entry: %s", search) + } +} diff --git a/internal/docsite/emit.go b/internal/docsite/emit.go index c72bbdc..807fe28 100644 --- a/internal/docsite/emit.go +++ b/internal/docsite/emit.go @@ -77,11 +77,15 @@ func (s *site) render() ([]Problem, error) { return nil, fmt.Errorf("docsite: render template for %s: %w", p.Source, err) } s.outputs[p.URL+".html"] = buf.Bytes() - s.outputs[p.URL+".md"] = s.pageMarkdown(p) + } + bodies := make([]string, len(s.pages)) + for i, p := range s.pages { + bodies[i] = markdownBody(p, rendered[i].mdLinks) + s.outputs[p.URL+".md"] = pageMarkdown(p, bodies[i]) } s.outputs["llms.txt"] = s.llmsTxt() - s.outputs["llms-full.txt"] = s.llmsFull() - idx, err := s.searchIndex() + s.outputs["llms-full.txt"] = s.llmsFull(bodies) + idx, err := s.searchIndex(rendered) if err != nil { return nil, err } @@ -109,25 +113,27 @@ func (s *site) nav(current *Page) []navSection { } // pageMarkdown returns the clean Markdown sibling of a page: no -// frontmatter, "# Title", "> description", then the body without its H1. -func (s *site) pageMarkdown(p *Page) []byte { +// frontmatter, "# Title", "> description", then the transformed body. +func pageMarkdown(p *Page, body string) []byte { var b bytes.Buffer fmt.Fprintf(&b, "# %s\n\n> %s\n\n", p.Title, p.Description) - b.WriteString(s.markdownBody(p)) + b.WriteString(body) return b.Bytes() } // markdownBody is the page body without its H1 and leading blank lines, -// ending in one newline. -func (s *site) markdownBody(p *Page) string { +// with links to other pages pointing at their .md URLs, ending in one +// newline. +func markdownBody(p *Page, links map[string]string) string { body := string(p.Body) - if first, rest, ok := strings.Cut(body, "\n"); ok && strings.HasPrefix(first, "# ") { - body = rest - } else if !ok && strings.HasPrefix(first, "# ") { + if first, rest, ok := strings.Cut(body, "\n"); strings.HasPrefix(first, "# ") { body = "" + if ok { + body = rest + } } - body = strings.TrimLeft(body, "\n") - return strings.TrimRight(body, "\n") + "\n" + body = strings.Trim(body, "\n") + return rewriteMarkdown(body, links) + "\n" } // llmsTxt writes the llms.txt index (llmstxt.org shape). @@ -161,14 +167,14 @@ func (s *site) llmsTxt() []byte { } // llmsFull concatenates every page in reading order. -func (s *site) llmsFull() []byte { +func (s *site) llmsFull(bodies []string) []byte { var b bytes.Buffer for i, p := range s.pages { if i > 0 { b.WriteString("\n") } fmt.Fprintf(&b, "# %s\nSource: %s\n\n%s\n\n", p.Title, s.url(p.URL+".html"), p.Description) - b.WriteString(s.markdownBody(p)) + b.WriteString(bodies[i]) } return b.Bytes() } @@ -191,14 +197,29 @@ type searchIndex struct { Entries []searchEntry `json:"e"` } -func (s *site) searchIndex() ([]byte, error) { +const searchTextMax = 300 + +// searchIndex lists every page and one entry per H2 heading, so a hit +// deep-links to its anchor. +func (s *site) searchIndex(rendered []renderedPage) ([]byte, error) { idx := searchIndex{Pages: []searchPage{}, Entries: []searchEntry{}} - for _, p := range s.pages { + for i, p := range s.pages { sec := s.cfg.Title if p.Section != indexSection { sec = s.cfg.sectionTitle(p.Section) } idx.Pages = append(idx.Pages, searchPage{URL: s.url(p.URL + ".html"), Title: p.Title, Section: sec}) + for _, h := range rendered[i].headings { + if h.Level != 2 { + continue + } + idx.Entries = append(idx.Entries, searchEntry{ + Page: i, + Anchor: h.ID, + Heading: h.Text, + Text: sectionText(h.node, p.Body, searchTextMax), + }) + } } raw, err := json.Marshal(idx) if err != nil { diff --git a/internal/docsite/load.go b/internal/docsite/load.go index e339f97..5d07b72 100644 --- a/internal/docsite/load.go +++ b/internal/docsite/load.go @@ -3,6 +3,7 @@ package docsite import ( "bytes" "cmp" + "errors" "fmt" "io/fs" "os" @@ -139,12 +140,13 @@ func (s Site) hasSection(name string) bool { // site is one assembled documentation site. type site struct { - opts Options - cfg Site - cfgRaw []byte - base string - pages []*Page // reading order - outputs map[string][]byte + opts Options + cfg Site + cfgRaw []byte + base string + pages []*Page // reading order + bySource map[string]*Page + outputs map[string][]byte } // rel returns the display path of an absolute path: repository-relative @@ -156,13 +158,47 @@ func (s *site) rel(abs string) string { return filepath.ToSlash(abs) } +// Pages loads the docs pages and the module READMEs and returns them in +// reading order (the order of the sidebar, the pager and llms-full.txt), +// with any load problems. It renders nothing. +func Pages(opts Options) ([]Page, []Problem, error) { + s, problems, err := load(opts) + if err != nil || s == nil { + return nil, problems, err + } + out := make([]Page, len(s.pages)) + for i, p := range s.pages { + out[i] = *p + } + return out, problems, nil +} + // assemble loads, verifies and renders a site in memory. func assemble(opts Options) (*site, []Problem, error) { + s, problems, err := load(opts) + if err != nil || s == nil { + return nil, problems, err + } + if len(problems) == 0 { + rp, err := s.render() + if err != nil { + return nil, nil, err + } + problems = append(problems, rp...) + } + sortProblems(problems) + return s, problems, nil +} + +// load reads site.yaml, the docs pages and the module READMEs and puts +// the pages in reading order. A nil site with problems means site.yaml +// itself is invalid. +func load(opts Options) (*site, []Problem, error) { opts, err := opts.normalize() if err != nil { return nil, nil, err } - s := &site{opts: opts, outputs: map[string][]byte{}} + s := &site{opts: opts, outputs: map[string][]byte{}, bySource: map[string]*Page{}} cfgPath := filepath.Join(opts.Src, "site.yaml") raw, err := os.ReadFile(cfgPath) if err != nil { @@ -182,19 +218,120 @@ func assemble(opts Options) (*site, []Problem, error) { if err != nil { return nil, nil, err } - s.order(guides) - problems = append(problems, s.emptySections()...) - if len(problems) == 0 { - rp, err := s.render() - if err != nil { - return nil, nil, err - } - problems = append(problems, rp...) + modules, mp, err := s.loadModules() + if err != nil { + return nil, nil, err } + problems = append(problems, mp...) + if len(modules) > 0 && !cfg.hasSection(apiSection) { + problems = append(problems, Problem{File: s.rel(cfgPath), Line: 1, Rule: "section", + Message: fmt.Sprintf("%q is not listed (every module README is published there)", apiSection)}) + } + s.order(append(guides, modules...)) + for _, p := range s.pages { + s.bySource[p.Source] = p + } + problems = append(problems, s.emptySections()...) sortProblems(problems) return s, problems, nil } +// loadModules turns every modules/ directory that holds a non-test Go +// file into an API reference page built from its README.md. The module +// list is discovered, never hard-coded. +func (s *site) loadModules() ([]*Page, []Problem, error) { + names, err := moduleNames(s.opts.Root) + if err != nil { + return nil, nil, err + } + var pages []*Page + var problems []Problem + for i, name := range names { + dir := filepath.Join(s.opts.Root, "modules", name) + readme := filepath.Join(dir, "README.md") + raw, err := os.ReadFile(readme) + if errors.Is(err, fs.ErrNotExist) { + problems = append(problems, Problem{File: s.rel(dir), Rule: "readme", Message: "package has Go files but no README.md"}) + continue + } + if err != nil { + return nil, nil, fmt.Errorf("docsite: read %s: %w", readme, err) + } + title, desc, body, ok := splitReadme(raw) + if !ok { + problems = append(problems, Problem{File: s.rel(readme), Line: 1, Rule: "readme", Message: `first line must be the "# " title followed by a summary line`}) + continue + } + pages = append(pages, &Page{ + Source: s.rel(readme), + URL: apiSection + "/" + name, + Section: apiSection, + Title: title, + Description: desc, + Order: i, + Module: name, + Body: body, + BodyLine: 1, + abs: readme, + }) + } + return pages, problems, nil +} + +// moduleNames lists the top-level modules/ directories that hold a non-test +// Go file, sorted by name. +func moduleNames(root string) ([]string, error) { + entries, err := os.ReadDir(filepath.Join(root, "modules")) + if errors.Is(err, fs.ErrNotExist) { + return nil, nil + } + if err != nil { + return nil, fmt.Errorf("docsite: read modules: %w", err) + } + var names []string + for _, e := range entries { + if !e.IsDir() || strings.HasPrefix(e.Name(), ".") || strings.HasPrefix(e.Name(), "_") { + continue + } + files, err := os.ReadDir(filepath.Join(root, "modules", e.Name())) + if err != nil { + return nil, fmt.Errorf("docsite: read module %s: %w", e.Name(), err) + } + if slices.ContainsFunc(files, func(f fs.DirEntry) bool { + n := f.Name() + return !f.IsDir() && strings.HasSuffix(n, ".go") && !strings.HasSuffix(n, "_test.go") + }) { + names = append(names, e.Name()) + } + } + slices.Sort(names) + return names, nil +} + +// splitReadme returns a README's H1 text, its summary line and the body +// with the summary line blanked (line numbers are kept so problems still +// cite the README's own lines). +func splitReadme(raw []byte) (title, desc string, body []byte, ok bool) { + lines := strings.Split(string(raw), "\n") + if len(lines) == 0 || !strings.HasPrefix(lines[0], "# ") { + return "", "", nil, false + } + title = strings.TrimSpace(strings.TrimPrefix(lines[0], "# ")) + for i := 1; i < len(lines); i++ { + line := strings.TrimSpace(lines[i]) + if line == "" { + continue + } + if strings.HasPrefix(line, "#") || title == "" { + return "", "", nil, false + } + desc = line + lines[i] = "" + return title, desc, []byte(strings.Join(lines, "\n")), true + } + return "", "", nil, false +} + // loadGuides walks Src and loads every page with its frontmatter. func (s *site) loadGuides() ([]*Page, []Problem, error) { files, err := walkPages(s.opts.Src) diff --git a/internal/docsite/render.go b/internal/docsite/render.go index 822d747..b9f5ac8 100644 --- a/internal/docsite/render.go +++ b/internal/docsite/render.go @@ -2,7 +2,13 @@ package docsite import ( "bytes" + "cmp" "fmt" + "path" + "slices" + "strings" + "unicode" + "unicode/utf8" "github.com/yuin/goldmark" "github.com/yuin/goldmark/ast" @@ -23,6 +29,7 @@ func newMarkdown() goldmark.Markdown { parser.WithAutoHeadingID(), parser.WithASTTransformers( util.Prioritized(h1Stripper{}, 100), + util.Prioritized(linkRewriter{}, 200), ), ), ) @@ -38,18 +45,315 @@ func (h1Stripper) Transform(doc *ast.Document, _ text.Reader, _ parser.Context) } } +// slugIDs is the one heading-ID algorithm of the site, shared by the +// renderer, the search index and the link checker. It is GitHub-compatible: +// lowercase, Unicode letters, digits, "_" and "-" kept, spaces mapped to +// "-", other punctuation dropped, and duplicates suffixed -1, -2. +type slugIDs struct { + used map[string]bool +} + +func newSlugIDs() *slugIDs { + return &slugIDs{used: map[string]bool{}} +} + +// Generate implements parser.IDs. +func (s *slugIDs) Generate(value []byte, _ ast.NodeKind) []byte { + base := slugify(string(value)) + if base == "" { + base = "section" + } + id := base + for i := 1; s.used[id]; i++ { + id = fmt.Sprintf("%s-%d", base, i) + } + s.used[id] = true + return []byte(id) +} + +// Put implements parser.IDs. +func (s *slugIDs) Put(value []byte) { + s.used[string(value)] = true +} + +func slugify(v string) string { + var b strings.Builder + for _, r := range strings.ToLower(strings.TrimSpace(v)) { + switch { + case unicode.IsLetter(r) || unicode.IsDigit(r) || r == '_' || r == '-': + b.WriteRune(r) + case unicode.IsSpace(r): + b.WriteByte('-') + } + } + return b.String() +} + +var pageKey = parser.NewContextKey() + +// pageContext carries per-page state into the AST transformers. +type pageContext struct { + site *site + page *Page + // mdLinks maps a rewritten link destination to its .md form, for the + // raw Markdown output. + mdLinks map[string]string +} + +// linkRewriter points relative links at other pages to their site URLs: +// a guide's link to another page's .md, or to modules//README.md, and a +// README's ..//README.md all become the target page. Fragments are kept; +// external and mailto links are untouched. +type linkRewriter struct{} + +func (linkRewriter) Transform(doc *ast.Document, _ text.Reader, pc parser.Context) { + pctx, _ := pc.Get(pageKey).(*pageContext) + if pctx == nil { + return + } + _ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) { + link, ok := n.(*ast.Link) + if !entering || !ok { + return ast.WalkContinue, nil + } + dest := string(link.Destination) + target, frag, ok := pctx.site.resolveLink(pctx.page, dest) + if !ok { + return ast.WalkContinue, nil + } + link.Destination = []byte(pctx.site.url(target.URL+".html") + frag) + pctx.mdLinks[dest] = pctx.site.url(target.URL+".md") + frag + return ast.WalkContinue, nil + }) +} + +// resolveLink maps a relative link destination in page from to the page it +// names, with its "#fragment" (or ""). +func (s *site) resolveLink(from *Page, dest string) (*Page, string, bool) { + if dest == "" || strings.HasPrefix(dest, "#") || strings.HasPrefix(dest, "/") || hasScheme(dest) { + return nil, "", false + } + target, frag, _ := strings.Cut(dest, "#") + if frag != "" { + frag = "#" + frag + } + if !strings.HasSuffix(target, ".md") { + return nil, "", false + } + p, ok := s.bySource[path.Clean(path.Join(path.Dir(from.Source), target))] + if !ok { + return nil, "", false + } + return p, frag, true +} + +func hasScheme(dest string) bool { + i := strings.IndexByte(dest, ':') + return i > 0 && !strings.ContainsAny(dest[:i], "/?#") +} + +// heading is one heading of a rendered page. +type heading struct { + Level int + ID string + Text string + node *ast.Heading +} + // renderedPage is one page after parsing and rendering. type renderedPage struct { - html []byte + html []byte + doc ast.Node + headings []heading + mdLinks map[string]string +} + +// parsePage parses a page body with a fresh slug-ID table and the page's +// transformer context. +func (s *site) parsePage(md goldmark.Markdown, p *Page) (ast.Node, *pageContext) { + pctx := &pageContext{site: s, page: p, mdLinks: map[string]string{}} + ctx := parser.NewContext(parser.WithIDs(newSlugIDs())) + ctx.Set(pageKey, pctx) + doc := md.Parser().Parse(text.NewReader(p.Body), parser.WithContext(ctx)) + return doc, pctx } // renderPage parses and renders a page body to HTML. func (s *site) renderPage(md goldmark.Markdown, p *Page) (renderedPage, error) { - ctx := parser.NewContext() - doc := md.Parser().Parse(text.NewReader(p.Body), parser.WithContext(ctx)) + doc, pctx := s.parsePage(md, p) var buf bytes.Buffer if err := md.Renderer().Render(&buf, p.Body, doc); err != nil { return renderedPage{}, fmt.Errorf("docsite: render %s: %w", p.Source, err) } - return renderedPage{html: buf.Bytes()}, nil + return renderedPage{html: buf.Bytes(), doc: doc, headings: pageHeadings(doc, p.Body), mdLinks: pctx.mdLinks}, nil +} + +// headingIDs returns the heading IDs of a page, in document order, from the +// same parse and slug algorithm the renderer uses. +func (s *site) headingIDs(md goldmark.Markdown, p *Page) []string { + doc, _ := s.parsePage(md, p) + var ids []string + for _, h := range pageHeadings(doc, p.Body) { + ids = append(ids, h.ID) + } + return ids +} + +func pageHeadings(doc ast.Node, src []byte) []heading { + var out []heading + _ = ast.Walk(doc, func(n ast.Node, entering bool) (ast.WalkStatus, error) { + h, ok := n.(*ast.Heading) + if !entering || !ok { + return ast.WalkContinue, nil + } + id, _ := h.AttributeString("id") + idb, _ := id.([]byte) + out = append(out, heading{Level: h.Level, ID: string(idb), Text: plainText(h, src), node: h}) + return ast.WalkSkipChildren, nil + }) + return out +} + +// plainText extracts the text of a node from its text nodes only (never +// from rendered HTML), with whitespace collapsed. +func plainText(n ast.Node, src []byte) string { + var b strings.Builder + _ = ast.Walk(n, func(c ast.Node, entering bool) (ast.WalkStatus, error) { + if !entering { + return ast.WalkContinue, nil + } + switch t := c.(type) { + case *ast.Text: + b.Write(t.Segment.Value(src)) + if t.SoftLineBreak() || t.HardLineBreak() { + b.WriteByte(' ') + } + case *ast.String: + b.Write(t.Value) + case *ast.FencedCodeBlock, *ast.CodeBlock, *ast.HTMLBlock, *ast.RawHTML: + return ast.WalkSkipChildren, nil + case *ast.Paragraph, *ast.Heading, *ast.ListItem, *ast.TextBlock: + b.WriteByte(' ') + } + return ast.WalkContinue, nil + }) + return strings.Join(strings.Fields(b.String()), " ") +} + +// sectionText is the plain text of the blocks after h up to the next +// heading of the same or a higher level, cut to max runes. +func sectionText(h *ast.Heading, src []byte, max int) string { + var parts []string + for n := h.NextSibling(); n != nil; n = n.NextSibling() { + if next, ok := n.(*ast.Heading); ok && next.Level <= h.Level { + break + } + if t := plainText(n, src); t != "" { + parts = append(parts, t) + } + } + return truncateRunes(strings.Join(parts, " "), max) +} + +func truncateRunes(s string, max int) string { + if utf8.RuneCountInString(s) <= max { + return s + } + r := []rune(s) + return strings.TrimSpace(string(r[:max])) +} + +// fence is one fenced code block found by scanning Markdown lines. +type fence struct { + open, close int // 0-based line indexes; close == -1 when unterminated + indent int + char byte + count int + info string +} + +// scanFences finds fenced code blocks in Markdown lines, following the +// CommonMark opening and closing rules. +func scanFences(lines []string) []fence { + var out []fence + for i := 0; i < len(lines); i++ { + f, ok := openFence(lines[i]) + if !ok { + continue + } + f.open, f.close = i, -1 + for j := i + 1; j < len(lines); j++ { + if closesFence(lines[j], f) { + f.close = j + break + } + } + out = append(out, f) + if f.close < 0 { + break + } + i = f.close + } + return out +} + +func openFence(line string) (fence, bool) { + trimmed := strings.TrimLeft(line, " ") + if len(trimmed) < 3 || (trimmed[0] != '`' && trimmed[0] != '~') { + return fence{}, false + } + c := trimmed[0] + n := 0 + for n < len(trimmed) && trimmed[n] == c { + n++ + } + if n < 3 { + return fence{}, false + } + info := strings.TrimSpace(trimmed[n:]) + if c == '`' && strings.ContainsRune(info, '`') { + return fence{}, false + } + return fence{indent: len(line) - len(trimmed), char: c, count: n, info: info}, true +} + +func closesFence(line string, f fence) bool { + trimmed := strings.TrimLeft(line, " ") + n := 0 + for n < len(trimmed) && trimmed[n] == f.char { + n++ + } + return n >= f.count && strings.TrimSpace(trimmed[n:]) == "" +} + +// rewriteMarkdown applies the raw-output transforms to a Markdown body: +// link destinations outside code fences are replaced through links. +func rewriteMarkdown(body string, links map[string]string) string { + lines := strings.Split(body, "\n") + inFence := make([]bool, len(lines)) + for _, f := range scanFences(lines) { + end := f.close + if end < 0 { + end = len(lines) - 1 + } + for i := f.open; i <= end; i++ { + inFence[i] = true + } + } + origs := make([]string, 0, len(links)) + for o := range links { + origs = append(origs, o) + } + slices.SortFunc(origs, func(a, b string) int { return cmp.Or(cmp.Compare(len(b), len(a)), strings.Compare(a, b)) }) + for i, line := range lines { + if inFence[i] || !strings.Contains(line, "](") { + continue + } + for _, o := range origs { + line = strings.ReplaceAll(line, "]("+o+")", "]("+links[o]+")") + line = strings.ReplaceAll(line, "]("+o+" ", "]("+links[o]+" ") + } + lines[i] = line + } + return strings.Join(lines, "\n") }