Files
summercms/internal/docsite/load.go
Jakub Zych f4605292b5 feat(11.1-02): check module identifiers in docs and READMEs
- go/parser index of every modules/ package and sub-package, with methods,
  fields, interface methods and promoted members
- code spans in docs pages, module READMEs and the root README fail
  Check and docs:build when the named identifier does not exist
- scripts/check-phase11.1.sh with preconditions, deps, docs, forbidden,
  go and a self-test that plants one violation per rule
2026-09-30 21:35:56 +02:00

578 lines
17 KiB
Go

package docsite
import (
"bytes"
"cmp"
"errors"
"fmt"
"io/fs"
"os"
"path"
"path/filepath"
"regexp"
"slices"
"strings"
"unicode/utf8"
"github.com/goccy/go-yaml"
)
// Site is docs/site.yaml, decoded strictly.
type Site struct {
Title string `yaml:"title"`
Description string `yaml:"description"`
BaseURL string `yaml:"base_url"`
// EditURL and SourceURL carry a {path} token replaced with a
// repository-relative path.
EditURL string `yaml:"edit_url"`
SourceURL string `yaml:"source_url"`
LLMSNotes []string `yaml:"llms_notes"`
Sections []Section `yaml:"sections"`
}
// Section is one sidebar group, listed in sidebar order in site.yaml.
type Section struct {
Name string `yaml:"name"`
Title string `yaml:"title"`
}
// Frontmatter is the YAML block at the top of every docs page. Every field
// is required.
type Frontmatter struct {
Title string `yaml:"title"`
Description string `yaml:"description"`
Section string `yaml:"section"`
Order int `yaml:"order"`
}
// Page is one page of the site: a docs/ page or an ingested module README.
type Page struct {
// Source is the repository-relative source path (forward slashes).
Source string
// URL is the extension-less output path, such as "setup/installation",
// "api/lagoon" or "index".
URL string
Section string
Title string
Description string
Order int
// Module is the module name for API reference pages.
Module string
// Body is the Markdown body, starting with the "# Title" line.
Body []byte
// BodyLine is the 1-based line of Body's first line in Source.
BodyLine int
abs string
}
const (
indexSection = "index"
apiSection = "api"
maxDescLen = 160
)
var frontmatterFields = []string{"title", "description", "section", "order"}
// LoadSite reads and parses a site.yaml file.
func LoadSite(path string) (Site, error) {
raw, err := os.ReadFile(path)
if err != nil {
return Site{}, fmt.Errorf("docsite: read site config %s: %w", path, err)
}
return ParseSite(raw)
}
// ParseSite decodes site.yaml, rejecting unknown fields, and validates the
// section list.
func ParseSite(raw []byte) (Site, error) {
var s Site
dec := yaml.NewDecoder(bytes.NewReader(raw), yaml.DisallowUnknownField())
if err := dec.Decode(&s); err != nil {
return Site{}, fmt.Errorf("docsite: parse site config: %s", firstLine(err.Error()))
}
if s.Title == "" {
return Site{}, fmt.Errorf("docsite: site config: title is required")
}
if s.Description == "" {
return Site{}, fmt.Errorf("docsite: site config: description is required")
}
if len(s.Sections) == 0 {
return Site{}, fmt.Errorf("docsite: site config: sections is required")
}
seen := map[string]bool{}
for _, sec := range s.Sections {
switch {
case sec.Name == "" || sec.Title == "":
return Site{}, fmt.Errorf("docsite: site config: every section needs a name and a title")
case sec.Name == indexSection:
return Site{}, fmt.Errorf("docsite: site config: section name %q is reserved for docs/index.md", indexSection)
case !slugName.MatchString(sec.Name):
return Site{}, fmt.Errorf("docsite: site config: section name %q must be lowercase letters, digits and dashes", sec.Name)
case seen[sec.Name]:
return Site{}, fmt.Errorf("docsite: site config: section %q is listed twice", sec.Name)
}
seen[sec.Name] = true
}
return s, nil
}
var slugName = regexp.MustCompile(`^[a-z0-9][a-z0-9-]*$`)
func firstLine(s string) string {
s, _, _ = strings.Cut(strings.TrimSpace(s), "\n")
return strings.TrimSpace(s)
}
// sectionTitle returns the display title of a section.
func (s Site) sectionTitle(name string) string {
for _, sec := range s.Sections {
if sec.Name == name {
return sec.Title
}
}
return name
}
func (s Site) hasSection(name string) bool {
return slices.ContainsFunc(s.Sections, func(sec Section) bool { return sec.Name == name })
}
// site is one assembled documentation site.
type site struct {
opts Options
cfg Site
cfgRaw []byte
base string
pages []*Page // reading order
bySource map[string]*Page
outputs map[string][]byte
}
// rel returns the display path of an absolute path: repository-relative
// with forward slashes when inside Root, otherwise the absolute path.
func (s *site) rel(abs string) string {
if r, err := filepath.Rel(s.opts.Root, abs); err == nil && within(abs, s.opts.Root) {
return filepath.ToSlash(r)
}
return filepath.ToSlash(abs)
}
// Pages loads the docs pages and the module READMEs and returns them in
// reading order (the order of the sidebar, the pager and llms-full.txt),
// with any load problems. It renders nothing.
func Pages(opts Options) ([]Page, []Problem, error) {
s, problems, err := load(opts)
if err != nil || s == nil {
return nil, problems, err
}
out := make([]Page, len(s.pages))
for i, p := range s.pages {
out[i] = *p
}
return out, problems, nil
}
// assemble loads, verifies and renders a site in memory.
func assemble(opts Options) (*site, []Problem, error) {
s, problems, err := load(opts)
if err != nil || s == nil {
return nil, problems, err
}
sp, err := s.checkSnippets()
if err != nil {
return nil, nil, err
}
problems = append(problems, sp...)
cp, err := s.checkContent()
if err != nil {
return nil, nil, err
}
problems = append(problems, cp...)
if len(problems) == 0 {
rp, err := s.render()
if err != nil {
return nil, nil, err
}
problems = append(problems, rp...)
}
sortProblems(problems)
return s, problems, nil
}
// load reads site.yaml, the docs pages and the module READMEs and puts
// the pages in reading order. A nil site with problems means site.yaml
// itself is invalid.
func load(opts Options) (*site, []Problem, error) {
opts, err := opts.normalize()
if err != nil {
return nil, nil, err
}
s := &site{opts: opts, outputs: map[string][]byte{}, bySource: map[string]*Page{}}
cfgPath := filepath.Join(opts.Src, "site.yaml")
raw, err := os.ReadFile(cfgPath)
if err != nil {
return nil, nil, fmt.Errorf("docsite: read site config %s: %w", cfgPath, err)
}
cfg, err := ParseSite(raw)
if err != nil {
return nil, []Problem{{File: s.rel(cfgPath), Line: 1, Rule: "site", Message: strings.TrimPrefix(err.Error(), "docsite: ")}}, nil
}
s.cfg, s.cfgRaw = cfg, raw
s.base = strings.TrimRight(cfg.BaseURL, "/")
if opts.BaseURL != "" {
s.base = strings.TrimRight(opts.BaseURL, "/")
}
guides, problems, err := s.loadGuides()
if err != nil {
return nil, nil, err
}
modules, mp, err := s.loadModules()
if err != nil {
return nil, nil, err
}
problems = append(problems, mp...)
if len(modules) > 0 && !cfg.hasSection(apiSection) {
problems = append(problems, Problem{File: s.rel(cfgPath), Line: 1, Rule: "section",
Message: fmt.Sprintf("%q is not listed (every module README is published there)", apiSection)})
}
s.order(append(guides, modules...))
for _, p := range s.pages {
s.bySource[p.Source] = p
}
problems = append(problems, s.emptySections()...)
sortProblems(problems)
return s, problems, nil
}
// loadModules turns every modules/<name> directory that holds a non-test Go
// file into an API reference page built from its README.md. The module
// list is discovered, never hard-coded.
func (s *site) loadModules() ([]*Page, []Problem, error) {
names, err := moduleNames(s.opts.Root)
if err != nil {
return nil, nil, err
}
var pages []*Page
var problems []Problem
for i, name := range names {
dir := filepath.Join(s.opts.Root, "modules", name)
readme := filepath.Join(dir, "README.md")
raw, err := os.ReadFile(readme)
if errors.Is(err, fs.ErrNotExist) {
problems = append(problems, Problem{File: s.rel(dir), Rule: "readme", Message: "package has Go files but no README.md"})
continue
}
if err != nil {
return nil, nil, fmt.Errorf("docsite: read %s: %w", readme, err)
}
title, desc, body, ok := splitReadme(raw)
if !ok {
problems = append(problems, Problem{File: s.rel(readme), Line: 1, Rule: "readme", Message: `first line must be the "# <module>" title followed by a summary line`})
continue
}
pages = append(pages, &Page{
Source: s.rel(readme),
URL: apiSection + "/" + name,
Section: apiSection,
Title: title,
Description: desc,
Order: i,
Module: name,
Body: body,
BodyLine: 1,
abs: readme,
})
}
return pages, problems, nil
}
// moduleNames lists the top-level modules/ directories that hold a non-test
// Go file, sorted by name.
func moduleNames(root string) ([]string, error) {
entries, err := os.ReadDir(filepath.Join(root, "modules"))
if errors.Is(err, fs.ErrNotExist) {
return nil, nil
}
if err != nil {
return nil, fmt.Errorf("docsite: read modules: %w", err)
}
var names []string
for _, e := range entries {
if !e.IsDir() || strings.HasPrefix(e.Name(), ".") || strings.HasPrefix(e.Name(), "_") {
continue
}
files, err := os.ReadDir(filepath.Join(root, "modules", e.Name()))
if err != nil {
return nil, fmt.Errorf("docsite: read module %s: %w", e.Name(), err)
}
if slices.ContainsFunc(files, func(f fs.DirEntry) bool {
n := f.Name()
return !f.IsDir() && strings.HasSuffix(n, ".go") && !strings.HasSuffix(n, "_test.go")
}) {
names = append(names, e.Name())
}
}
slices.Sort(names)
return names, nil
}
// splitReadme returns a README's H1 text, its summary line and the body
// with the summary line blanked (line numbers are kept so problems still
// cite the README's own lines).
func splitReadme(raw []byte) (title, desc string, body []byte, ok bool) {
lines := strings.Split(string(raw), "\n")
if len(lines) == 0 || !strings.HasPrefix(lines[0], "# ") {
return "", "", nil, false
}
title = strings.TrimSpace(strings.TrimPrefix(lines[0], "# "))
for i := 1; i < len(lines); i++ {
line := strings.TrimSpace(lines[i])
if line == "" {
continue
}
if strings.HasPrefix(line, "#") || title == "" {
return "", "", nil, false
}
desc = line
lines[i] = ""
return title, desc, []byte(strings.Join(lines, "\n")), true
}
return "", "", nil, false
}
// loadGuides walks Src and loads every page with its frontmatter.
func (s *site) loadGuides() ([]*Page, []Problem, error) {
files, err := walkPages(s.opts.Src)
if err != nil {
return nil, nil, err
}
var pages []*Page
var problems []Problem
orders := map[string]map[int]string{}
hasIndex := false
for _, abs := range files {
raw, err := os.ReadFile(abs)
if err != nil {
return nil, nil, fmt.Errorf("docsite: read %s: %w", abs, err)
}
relSrc, err := filepath.Rel(s.opts.Src, abs)
if err != nil {
return nil, nil, fmt.Errorf("docsite: %w", err)
}
relSrc = filepath.ToSlash(relSrc)
file := s.rel(abs)
fail := func(detail string) {
problems = append(problems, Problem{File: file, Line: 1, Rule: "frontmatter", Message: detail})
}
fmRaw, body, bodyLine, ok := splitFrontmatter(raw)
if !ok {
fail(`the file must start with a "---" frontmatter block closed by a "---" line`)
continue
}
fm, details := parseFrontmatter(fmRaw)
if len(details) > 0 {
for _, d := range details {
fail(d)
}
continue
}
dir := path.Dir(relSrc)
switch {
case relSrc == "index.md":
hasIndex = true
if fm.Section != indexSection {
fail(fmt.Sprintf("section %q does not match directory %q (the landing page uses section %q)", fm.Section, ".", indexSection))
}
case dir == ".":
fail(fmt.Sprintf("section %q does not match directory %q (only index.md sits at the docs root)", fm.Section, "."))
case fm.Section != dir:
fail(fmt.Sprintf("section %q does not match directory %q", fm.Section, dir))
case fm.Section == apiSection:
fail(fmt.Sprintf("section %q is reserved for the ingested module READMEs", apiSection))
case !s.cfg.hasSection(fm.Section):
fail(fmt.Sprintf("section %q is not listed in %s", fm.Section, s.rel(filepath.Join(s.opts.Src, "site.yaml"))))
}
if first, _, _ := bytes.Cut(body, []byte("\n")); string(first) != "# "+fm.Title {
fail(fmt.Sprintf("first line must be %q", "# "+fm.Title))
}
if utf8.RuneCountInString(fm.Description) > maxDescLen {
fail(fmt.Sprintf("description is longer than %d characters", maxDescLen))
}
if orders[fm.Section] == nil {
orders[fm.Section] = map[int]string{}
}
if other, dup := orders[fm.Section][fm.Order]; dup {
fail(fmt.Sprintf("order %d already used by %s", fm.Order, other))
} else {
orders[fm.Section][fm.Order] = file
}
pages = append(pages, &Page{
Source: file,
URL: strings.TrimSuffix(relSrc, ".md"),
Section: fm.Section,
Title: fm.Title,
Description: fm.Description,
Order: fm.Order,
Body: body,
BodyLine: bodyLine,
abs: abs,
})
}
if !hasIndex {
problems = append(problems, Problem{File: s.rel(filepath.Join(s.opts.Src, "index.md")), Rule: "page", Message: "the landing page is missing"})
}
return pages, problems, nil
}
// walkPages lists the Markdown pages under src in lexical order, skipping
// examples/, dot and underscore entries.
func walkPages(src string) ([]string, error) {
var files []string
err := filepath.WalkDir(src, func(p string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if p == src {
return nil
}
name := d.Name()
skip := strings.HasPrefix(name, ".") || strings.HasPrefix(name, "_") ||
(d.IsDir() && filepath.Dir(p) == src && name == "examples")
if d.IsDir() {
if skip {
return filepath.SkipDir
}
return nil
}
if !skip && strings.HasSuffix(name, ".md") {
files = append(files, p)
}
return nil
})
if err != nil {
return nil, fmt.Errorf("docsite: walk %s: %w", src, err)
}
return files, nil
}
// splitFrontmatter separates a leading "---" block from the body. bodyLine
// is the 1-based line of the body's first line.
func splitFrontmatter(raw []byte) (fm, body []byte, bodyLine int, ok bool) {
rest, found := bytes.CutPrefix(raw, []byte("---\n"))
if !found {
return nil, nil, 0, false
}
line := 2
for off := 0; off <= len(rest); {
end := bytes.IndexByte(rest[off:], '\n')
var cur []byte
next := len(rest) + 1
if end < 0 {
cur = rest[off:]
} else {
cur = rest[off : off+end]
next = off + end + 1
}
if string(bytes.TrimRight(cur, " \t\r")) == "---" {
if next > len(rest) {
return rest[:off], nil, line + 1, true
}
return rest[:off], rest[next:], line + 1, true
}
off = next
line++
}
return nil, nil, 0, false
}
// parseFrontmatter decodes a frontmatter block strictly. It returns the
// UI-SPEC problem details: unknown and missing fields first, then decode
// errors.
func parseFrontmatter(raw []byte) (Frontmatter, []string) {
var keys map[string]any
if err := yaml.Unmarshal(raw, &keys); err != nil {
return Frontmatter{}, []string{firstLine(err.Error())}
}
var details []string
unknown := make([]string, 0)
for k := range keys {
if !slices.Contains(frontmatterFields, k) {
unknown = append(unknown, k)
}
}
slices.Sort(unknown)
for _, k := range unknown {
details = append(details, fmt.Sprintf("unknown field %q", k))
}
for _, k := range frontmatterFields {
v, ok := keys[k]
if !ok || v == nil || v == "" {
details = append(details, fmt.Sprintf("missing field %q", k))
}
}
if len(details) > 0 {
return Frontmatter{}, details
}
var fm Frontmatter
dec := yaml.NewDecoder(bytes.NewReader(raw), yaml.DisallowUnknownField())
if err := dec.Decode(&fm); err != nil {
return Frontmatter{}, []string{firstLine(err.Error())}
}
return fm, nil
}
// order sorts pages into reading order: index first, then each site.yaml
// section in order with its pages by order (API pages by module name).
func (s *site) order(pages []*Page) {
bySection := map[string][]*Page{}
for _, p := range pages {
bySection[p.Section] = append(bySection[p.Section], p)
}
s.pages = s.pages[:0]
s.pages = append(s.pages, bySection[indexSection]...)
for _, sec := range s.cfg.Sections {
list := bySection[sec.Name]
slices.SortStableFunc(list, func(a, b *Page) int {
if a.Module != "" || b.Module != "" {
return strings.Compare(a.Module, b.Module)
}
return cmp.Or(cmp.Compare(a.Order, b.Order), strings.Compare(a.URL, b.URL))
})
s.pages = append(s.pages, list...)
}
}
// emptySections reports every site.yaml section without pages.
func (s *site) emptySections() []Problem {
count := map[string]int{}
for _, p := range s.pages {
count[p.Section]++
}
var problems []Problem
for _, sec := range s.cfg.Sections {
if count[sec.Name] > 0 {
continue
}
problems = append(problems, Problem{
File: s.rel(filepath.Join(s.opts.Src, "site.yaml")),
Line: sectionLine(s.cfgRaw, sec.Name),
Rule: "section",
Message: fmt.Sprintf("%q has no pages (add the section in the same change as its first page)", sec.Name),
})
}
return problems
}
// sectionLine finds the site.yaml line that names a section.
func sectionLine(raw []byte, name string) int {
re := regexp.MustCompile(`^\s*(-\s*)?name:\s*["']?` + regexp.QuoteMeta(name) + `["']?\s*$`)
for i, line := range strings.Split(string(raw), "\n") {
if re.MatchString(line) {
return i + 1
}
}
return 1
}