package discover import ( "fmt" "io/fs" "log/slog" "net/http" "strings" "khosra/internal/content" ) // The two paths, exported so the wiring can reason about precedence without restating strings. const ( RobotsPath = "/robots.txt" SitemapPath = "/sitemap.xml" ) // Routes returns the two exact paths this feature owns. // // site is a callback rather than a value because a sitemap must list what is served *now*: the index is // swapped whole on every rebuild (ADR-0077), and a captured pointer would serve the site as it was at // startup. settings is copied, since editing `site.yaml` needs a restart anyway (ADR-0055). func Routes(siteFS fs.FS, settings content.Settings, site func() *content.Site) map[string]http.Handler { return map[string]http.Handler{ RobotsPath: http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { robots(w, siteFS, settings.Base) }), SitemapPath: http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { sitemap(w, req, site(), settings.Base) }), } } // robots answers /robots.txt, preferring the site's own file. // // A site that ships robots.txt has said something deliberate, so it is served verbatim; otherwise the engine // emits the minimum that is true — everything is public, and here is the sitemap. The Sitemap line only // appears with a declared base, because a relative sitemap reference is not something a crawler accepts. func robots(w http.ResponseWriter, siteFS fs.FS, base string) { if siteFS != nil { if data, err := fs.ReadFile(siteFS, "robots.txt"); err == nil { writeAs(w, "text/plain; charset=utf-8", data, "robots.txt") return } } var out strings.Builder out.WriteString("User-agent: *\nDisallow:\n") if base != "" { fmt.Fprintf(&out, "Sitemap: %s\n", content.Absolute(base, SitemapPath)) } writeAs(w, "text/plain; charset=utf-8", []byte(out.String()), "robots.txt") } // sitemap answers /sitemap.xml with every bundle in every language it exists in. // // It needs a declared base: the sitemap format has no room for a relative URL, so without one the honest // answer is that this file does not exist rather than a file full of paths no crawler can use (ADR-0039). // Every URL comes from content.URL, like every other path the engine emits, so a sitemap can never disagree // with what is actually served. func sitemap(w http.ResponseWriter, req *http.Request, site *content.Site, base string) { if base == "" { slog.Warn("no sitemap: the site declares no base URL", "file", content.SettingsFile) http.NotFound(w, req) return } var out strings.Builder out.WriteString(`` + "\n") out.WriteString(`` + "\n") for _, entry := range site.Everything() { fmt.Fprintf(&out, "%s", xmlEscape(content.Absolute(base, content.URL(entry.Route, entry.Lang)))) if !entry.Date.IsZero() { fmt.Fprintf(&out, "%s", entry.Date.Format("2006-01-02")) } out.WriteString("\n") } out.WriteString("\n") writeAs(w, "application/xml; charset=utf-8", []byte(out.String()), "sitemap.xml") } // xmlEscape escapes the five characters XML reserves. A URL should contain none of them, and a sitemap that // silently breaks on the one that does is worse than a slightly paranoid replacement. func xmlEscape(s string) string { return strings.NewReplacer( "&", "&", "<", "<", ">", ">", `"`, """, "'", "'", ).Replace(s) } // writeAs sets the type and writes, logging a failed write rather than pretending it succeeded. // // Five lines copied from internal/web rather than shared: a feature may not import web (conventions.md), and // two copies of five obvious lines is cheaper than a package existing to hold them. func writeAs(w http.ResponseWriter, contentType string, out []byte, what string) { w.Header().Set("Content-Type", contentType) if _, err := w.Write(out); err != nil { slog.Warn("write failed", "what", what, "err", err) } }