Files
khosra/internal/web/discover_test.go
T
bdeshi 2b0387e997 serve robots.txt and sitemap.xml
Two exact paths a crawler asks for by name, so they are mux entries rather than
resolver cases — no bundle can collide, since a key always sits under a section.

robots.txt at the site root is served verbatim, because a site that ships one has
said something deliberate; otherwise the engine emits the minimum that is true and
points at the sitemap. The sitemap lists every bundle in every language it exists
in, since each variant is separately reachable, with lastmod only where a bundle
has a date. Every URL comes from content.URL like every other path the engine
emits, so a sitemap cannot disagree with what is actually served.

Both need a declared base. Without one the sitemap answers 404 rather than listing
paths no crawler can resolve, and robots omits the Sitemap line rather than
writing a relative one.

write() was setting text/html for every caller, and headers only go out with the
first byte — so a handler setting its own type would have had it silently replaced,
which is how a sitemap gets served as a web page. It now splits into write and
writeAs, and the tests assert the content types rather than only the bodies.
2026-07-31 02:23:26 +06:00

102 lines
3.7 KiB
Go

package web
import (
"net/http"
"net/http/httptest"
"strings"
"testing"
"testing/fstest"
"khosra/internal/content"
"khosra/internal/render"
)
func crawlerHandler(t *testing.T, settings content.Settings, extra fstest.MapFS) http.Handler {
t.Helper()
fsys := fstest.MapFS{
"content/posts/hello.md": {Data: []byte("---\ntitle: Hello\ndate: 2026-03-08\n---\nx\n")},
"content/posts/hello.bn.md": {Data: []byte("---\ntitle: হ্যালো\ndate: 2026-03-08\n---\nx\n")},
"content/pages/about.md": {Data: []byte("---\ntitle: About\n---\nx\n")},
}
for name, file := range extra {
fsys[name] = file
}
bundles, err := content.Scan(fsys)
if err != nil {
t.Fatal(err)
}
r, err := render.New(nil, settings, nil)
if err != nil {
t.Fatal(err)
}
return Handler(content.NewSite(bundles), r, fsys, settings)
}
func TestSitemapListsEveryVariantAbsolutely(t *testing.T) {
h := crawlerHandler(t, content.Settings{Base: "https://khosra.example"}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/sitemap.xml", nil))
if rec.Code != http.StatusOK {
t.Fatalf("got %d, want 200", rec.Code)
}
if ct := rec.Header().Get("Content-Type"); !strings.HasPrefix(ct, "application/xml") {
t.Errorf("content-type = %q — a sitemap served as HTML is a sitemap nothing reads", ct)
}
body := rec.Body.String()
for _, want := range []string{
"<loc>https://khosra.example/posts/hello/</loc>",
"<loc>https://khosra.example/bn/posts/hello/</loc>", // each language is its own URL
"<loc>https://khosra.example/pages/about/</loc>",
"<lastmod>2026-03-08</lastmod>",
} {
if !strings.Contains(body, want) {
t.Errorf("missing %q:\n%s", want, body)
}
}
if strings.Contains(body, "<lastmod></lastmod>") {
t.Error("an undated bundle should carry no lastmod at all")
}
}
func TestNoBaseMeansNoSitemap(t *testing.T) {
// The format has no room for a relative URL, so the honest answer is that the file does not exist
// (ADR-0039) rather than one full of paths no crawler can resolve.
h := crawlerHandler(t, content.Settings{}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/sitemap.xml", nil))
if rec.Code != http.StatusNotFound {
t.Errorf("got %d, want 404", rec.Code)
}
}
func TestRobotsIsGeneratedOrTheSitesOwn(t *testing.T) {
h := crawlerHandler(t, content.Settings{Base: "https://khosra.example"}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/robots.txt", nil))
body := rec.Body.String()
if !strings.Contains(body, "User-agent: *") || !strings.Contains(body, "Sitemap: https://khosra.example/sitemap.xml") {
t.Errorf("generated robots should point at the sitemap:\n%s", body)
}
if ct := rec.Header().Get("Content-Type"); !strings.HasPrefix(ct, "text/plain") {
t.Errorf("content-type = %q", ct)
}
// A site that ships its own has said something deliberate.
h = crawlerHandler(t, content.Settings{Base: "https://khosra.example"},
fstest.MapFS{"robots.txt": {Data: []byte("User-agent: *\nDisallow: /drafts/\n")}})
rec = httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/robots.txt", nil))
if got := rec.Body.String(); !strings.Contains(got, "Disallow: /drafts/") || strings.Contains(got, "Sitemap:") {
t.Errorf("the site's own robots.txt should be served verbatim:\n%s", got)
}
}
func TestRobotsWithoutABaseOmitsTheSitemapLine(t *testing.T) {
h := crawlerHandler(t, content.Settings{}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/robots.txt", nil))
if got := rec.Body.String(); strings.Contains(got, "Sitemap:") {
t.Errorf("a relative sitemap reference is not something a crawler accepts:\n%s", got)
}
}