package web
import (
"net/http"
"net/http/httptest"
"strings"
"testing"
"testing/fstest"
"khosra/internal/content"
"khosra/internal/render"
)
func crawlerHandler(t *testing.T, settings content.Settings, extra fstest.MapFS) http.Handler {
t.Helper()
fsys := fstest.MapFS{
"content/posts/hello.md": {Data: []byte("---\ntitle: Hello\ndate: 2026-03-08\n---\nx\n")},
"content/posts/hello.bn.md": {Data: []byte("---\ntitle: হ্যালো\ndate: 2026-03-08\n---\nx\n")},
"content/pages/about.md": {Data: []byte("---\ntitle: About\n---\nx\n")},
}
for name, file := range extra {
fsys[name] = file
}
bundles, err := content.Scan(fsys)
if err != nil {
t.Fatal(err)
}
r, err := render.New(nil, settings, nil)
if err != nil {
t.Fatal(err)
}
return Handler(Fixed(content.NewSite(bundles)), r, fsys, nil, settings)
}
func TestSitemapListsEveryVariantAbsolutely(t *testing.T) {
h := crawlerHandler(t, content.Settings{Base: "https://khosra.example"}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/sitemap.xml", nil))
if rec.Code != http.StatusOK {
t.Fatalf("got %d, want 200", rec.Code)
}
if ct := rec.Header().Get("Content-Type"); !strings.HasPrefix(ct, "application/xml") {
t.Errorf("content-type = %q — a sitemap served as HTML is a sitemap nothing reads", ct)
}
body := rec.Body.String()
for _, want := range []string{
"https://khosra.example/posts/hello/",
"https://khosra.example/bn/posts/hello/", // each language is its own URL
"https://khosra.example/pages/about/",
"2026-03-08",
} {
if !strings.Contains(body, want) {
t.Errorf("missing %q:\n%s", want, body)
}
}
if strings.Contains(body, "") {
t.Error("an undated bundle should carry no lastmod at all")
}
}
func TestNoBaseMeansNoSitemap(t *testing.T) {
// The format has no room for a relative URL, so the honest answer is that the file does not exist
// (ADR-0039) rather than one full of paths no crawler can resolve.
h := crawlerHandler(t, content.Settings{}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/sitemap.xml", nil))
if rec.Code != http.StatusNotFound {
t.Errorf("got %d, want 404", rec.Code)
}
}
func TestRobotsIsGeneratedOrTheSitesOwn(t *testing.T) {
h := crawlerHandler(t, content.Settings{Base: "https://khosra.example"}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/robots.txt", nil))
body := rec.Body.String()
if !strings.Contains(body, "User-agent: *") || !strings.Contains(body, "Sitemap: https://khosra.example/sitemap.xml") {
t.Errorf("generated robots should point at the sitemap:\n%s", body)
}
if ct := rec.Header().Get("Content-Type"); !strings.HasPrefix(ct, "text/plain") {
t.Errorf("content-type = %q", ct)
}
// A site that ships its own has said something deliberate.
h = crawlerHandler(t, content.Settings{Base: "https://khosra.example"},
fstest.MapFS{"robots.txt": {Data: []byte("User-agent: *\nDisallow: /drafts/\n")}})
rec = httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/robots.txt", nil))
if got := rec.Body.String(); !strings.Contains(got, "Disallow: /drafts/") || strings.Contains(got, "Sitemap:") {
t.Errorf("the site's own robots.txt should be served verbatim:\n%s", got)
}
}
func TestRobotsWithoutABaseOmitsTheSitemapLine(t *testing.T) {
h := crawlerHandler(t, content.Settings{}, nil)
rec := httptest.NewRecorder()
h.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/robots.txt", nil))
if got := rec.Body.String(); strings.Contains(got, "Sitemap:") {
t.Errorf("a relative sitemap reference is not something a crawler accepts:\n%s", got)
}
}