The human chose the second option: a route sits beside the key rather than replacing it. So `slug` renames what a bundle is served at, in every language, and identity stays derived from the path — which is exactly what keeps ADR-0033 intact, since series membership is the directory. A series landing page can now be renamed without orphaning its chapters, and there is a test that says so. `Site` resolves routes at index time, because only it can see whether every variant agrees. Disagreement is dropped rather than resolved, as is a slug landing where another bundle already answers — the same rule colliding keys and contested aliases already follow. The key a slug moved away from stops answering, so the old address does not quietly keep working. Two bugs surfaced doing this, both older than this change: An alias naming its own bundle's former key was rejected as "an alias that names a real bundle" — which made rename-plus-alias, the entire point of ADR-0008's alias mechanism, impossible. The check now asks what a request asks: is anything actually served there. Aliases were counted per declaring *file*, so a bundle whose two language variants both listed the same alias looked like two rival claimants and lost the alias. It is a set of keys now. This one only appears with translated content, which is why no fixture had caught it since entry 4 — the real binary did, on the first multilingual rename.
72 lines
2.7 KiB
Go
72 lines
2.7 KiB
Go
package web
|
|
|
|
import (
|
|
"fmt"
|
|
"io/fs"
|
|
"log/slog"
|
|
"net/http"
|
|
"strings"
|
|
|
|
"khosra/internal/content"
|
|
)
|
|
|
|
// robots and sitemap are the two files a crawler looks for by exact name.
|
|
const (
|
|
robotsPath = "/robots.txt"
|
|
sitemapPath = "/sitemap.xml"
|
|
)
|
|
|
|
// serveRobots answers /robots.txt, preferring the site's own file.
|
|
//
|
|
// A site that ships robots.txt has said something deliberate, so it is served verbatim; otherwise the engine
|
|
// emits the minimum that is true — everything is public, and here is the sitemap. The Sitemap line only
|
|
// appears with a declared base, because a relative sitemap reference is not something a crawler accepts.
|
|
func serveRobots(w http.ResponseWriter, req *http.Request, siteFS fs.FS, base string) {
|
|
if siteFS != nil {
|
|
if data, err := fs.ReadFile(siteFS, "robots.txt"); err == nil {
|
|
writeAs(w, "text/plain; charset=utf-8", data, "robots.txt")
|
|
return
|
|
}
|
|
}
|
|
var out strings.Builder
|
|
out.WriteString("User-agent: *\nDisallow:\n")
|
|
if base != "" {
|
|
fmt.Fprintf(&out, "Sitemap: %s\n", content.Absolute(base, sitemapPath))
|
|
}
|
|
writeAs(w, "text/plain; charset=utf-8", []byte(out.String()), "robots.txt")
|
|
}
|
|
|
|
// serveSitemap answers /sitemap.xml with every bundle in every language it exists in.
|
|
//
|
|
// It needs a declared base: the sitemap format has no room for a relative URL, so without one the honest
|
|
// answer is that this file does not exist rather than a file full of paths no crawler can use (ADR-0039).
|
|
// Every URL comes from content.URL, like every other path the engine emits, so a sitemap can never disagree
|
|
// with what is actually served.
|
|
func serveSitemap(w http.ResponseWriter, req *http.Request, site *content.Site, base string) {
|
|
if base == "" {
|
|
slog.Warn("no sitemap: the site declares no base URL", "file", content.SettingsFile)
|
|
http.NotFound(w, req)
|
|
return
|
|
}
|
|
var out strings.Builder
|
|
out.WriteString(`<?xml version="1.0" encoding="utf-8"?>` + "\n")
|
|
out.WriteString(`<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">` + "\n")
|
|
for _, entry := range site.Everything() {
|
|
fmt.Fprintf(&out, "<url><loc>%s</loc>", xmlEscape(content.Absolute(base, content.URL(entry.Route, entry.Lang))))
|
|
if !entry.Date.IsZero() {
|
|
fmt.Fprintf(&out, "<lastmod>%s</lastmod>", entry.Date.Format("2006-01-02"))
|
|
}
|
|
out.WriteString("</url>\n")
|
|
}
|
|
out.WriteString("</urlset>\n")
|
|
writeAs(w, "application/xml; charset=utf-8", []byte(out.String()), "sitemap.xml")
|
|
}
|
|
|
|
// xmlEscape escapes the five characters XML reserves. A URL should contain none of them, and a sitemap that
|
|
// silently breaks on the one that does is worse than a slightly paranoid replacement.
|
|
func xmlEscape(s string) string {
|
|
return strings.NewReplacer(
|
|
"&", "&", "<", "<", ">", ">", `"`, """, "'", "'",
|
|
).Replace(s)
|
|
}
|