crawl: stop letting a watch widen on-demand reading, and honour Crawl-delay

The on-demand crawler was built over allow_hosts plus every watched host.
webfetch reads a non-empty allow list as these and nothing else, so a config
with one watch and no allow_hosts at all silently narrowed on-demand reading
to the watched site. Every other url he pasted came back as a flat refusal
with nothing in the log to explain it. The two crawlers now take two host
lists from one crawlHosts helper.

Crawl-delay was parsed into Rules and never read. The only pacing was the
fetcher's flat one request per host per second, which cannot express what a
site asked for, and deploy/README claimed the field was honoured. Page now
waits it out between the robots fetch and the page fetch, and a delay longer
than the turn fails the read instead of hanging it.

A robots.txt that failed was treated as no rules, so a site whose server was
having a bad minute became a site with no restrictions. A 5xx now refuses the
crawl. A 404 still means unrestricted, which is what the standard says.

The refusal check matched substrings of webfetch's message text from a package
that cannot import webfetch, so a reworded error would have silently turned
into a robots verdict. internal/crawl now exports ErrFetchRefused and
ErrFetchStatus and the adapter in cmd/mavend maps the webfetch sentinels onto
them. Robots group selection picks the longest matching agent prefix instead
of the first one in file order.

queryWeb passed a claim it could not serve when no crawler was configured, so
an unconfigured deployment answered a web question with an apology instead of
falling through to the model.

Found in review of #67.
This commit is contained in:
kami
2026-08-01 14:33:26 +04:00
parent 57161fb762
commit 327726a06a
8 changed files with 384 additions and 79 deletions
+63 -9
View File
@@ -2,6 +2,7 @@ package main
import (
"context"
"errors"
"net/http"
"net/http/httptest"
"strings"
@@ -13,6 +14,7 @@ import (
"github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/voice"
"github.com/kami/maven/internal/webfetch"
)
// The default config reads nothing. This is the whole "off unless configured"
@@ -123,16 +125,13 @@ func TestQueryWebPassesWithoutAURL(t *testing.T) {
}
}
// Not configured is said out loud rather than falling through, so a small model
// never invents a page's contents from its URL.
func TestQueryWebSaysWhenNotConfigured(t *testing.T) {
// A daemon where page reading was never turned on — the default — answers the
// question the way it did before the capability existed. Claiming the turn to
// report a configuration status is for something that exists and failed.
func TestQueryWebPassesWhenNotConfigured(t *testing.T) {
h := buildWebHandler(nil)
reply, ok := askWeb(h, "посмотри https://example.org/page")
if !ok {
t.Fatal("the web source did not claim a question with a URL")
}
if !strings.Contains(reply, "не настроено") {
t.Errorf("reply = %q, want the not-configured answer", reply)
if reply, ok := askWeb(h, "посмотри https://example.org/page"); ok {
t.Fatalf("an unconfigured crawler claimed the turn with %q", reply)
}
}
@@ -184,3 +183,58 @@ func (robotsDenyFetcher) Get(_ context.Context, u string) (*crawl.Response, erro
}
return &crawl.Response{URL: u, ContentType: "text/html", Body: []byte("<html>nope</html>")}, nil
}
// TestCrawlHostsKeepsAWatchOutOfTheOnDemandAllowlist — the on-demand crawler
// used to be built over allow_hosts PLUS every watched host. webfetch reads a
// non-empty allow list as "these and nothing else", so one watch on a config
// with no allow_hosts at all turned unrestricted on-demand reading into
// "the watched site only", and every other URL he pasted came back as
// "не получилось прочитать страницу." with nothing in the log to explain it.
func TestCrawlHostsKeepsAWatchOutOfTheOnDemandAllowlist(t *testing.T) {
cc := &config.CrawlConfig{
OnDemand: true,
Watches: []config.CrawlWatchConfig{{Name: "p", URL: "https://watched.example/p"}},
}
if got := crawlHosts(cc, false); len(got) != 0 {
t.Errorf("on-demand allowlist = %v; a watch is not an allowlist entry, and an empty list is what means \"anything public\"", got)
}
if got := crawlHosts(cc, true); len(got) != 1 || got[0] != "watched.example" {
t.Errorf("watch allowlist = %v; want the watched host so a watch needs no hand-written entry", got)
}
// With allow_hosts set, his list is what on-demand gets, unchanged.
cc.AllowHosts = []string{"wiki.example"}
on := crawlHosts(cc, false)
if len(on) != 1 || on[0] != "wiki.example" {
t.Errorf("on-demand allowlist = %v; want exactly his allow_hosts", on)
}
if got := crawlHosts(cc, true); len(got) != 2 {
t.Errorf("watch allowlist = %v; want his hosts plus the watched one", got)
}
}
// TestCrawlFetcherReportsARefusalAsARefusal — internal/crawl cannot import
// webfetch, so it used to recognise a guard refusal by matching substrings of
// webfetch's message text. This adapter owns both packages and is where the
// translation belongs.
func TestCrawlFetcherReportsARefusalAsARefusal(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
http.Error(w, "boom", http.StatusBadGateway)
}))
defer srv.Close()
blocked := &crawlFetcher{f: webfetch.New(webfetch.Config{AllowHosts: []string{"wiki.example"}})}
if _, err := blocked.Get(context.Background(), "https://other.example/a"); !errors.Is(err, crawl.ErrFetchRefused) {
t.Errorf("a host outside allow_hosts = %v; want crawl.ErrFetchRefused", err)
}
if _, err := blocked.Get(context.Background(), "file:///etc/passwd"); !errors.Is(err, crawl.ErrFetchRefused) {
t.Errorf("a non-http scheme = %v; want crawl.ErrFetchRefused", err)
}
// A 5xx is a different thing: the server answered, badly. robots.txt over
// this must refuse the crawl rather than read it as "no rules".
open := &crawlFetcher{f: webfetch.New(webfetch.Config{AllowHosts: []string{"127.0.0.1"}, AllowPrivate: true})}
if _, err := open.Get(context.Background(), srv.URL+"/robots.txt"); !errors.Is(err, crawl.ErrFetchStatus) {
t.Errorf("a 502 = %v; want crawl.ErrFetchStatus", err)
}
}