crawl: stop letting a watch widen on-demand reading, and honour Crawl-delay
The on-demand crawler was built over allow_hosts plus every watched host. webfetch reads a non-empty allow list as these and nothing else, so a config with one watch and no allow_hosts at all silently narrowed on-demand reading to the watched site. Every other url he pasted came back as a flat refusal with nothing in the log to explain it. The two crawlers now take two host lists from one crawlHosts helper. Crawl-delay was parsed into Rules and never read. The only pacing was the fetcher's flat one request per host per second, which cannot express what a site asked for, and deploy/README claimed the field was honoured. Page now waits it out between the robots fetch and the page fetch, and a delay longer than the turn fails the read instead of hanging it. A robots.txt that failed was treated as no rules, so a site whose server was having a bad minute became a site with no restrictions. A 5xx now refuses the crawl. A 404 still means unrestricted, which is what the standard says. The refusal check matched substrings of webfetch's message text from a package that cannot import webfetch, so a reworded error would have silently turned into a robots verdict. internal/crawl now exports ErrFetchRefused and ErrFetchStatus and the adapter in cmd/mavend maps the webfetch sentinels onto them. Robots group selection picks the longest matching agent prefix instead of the first one in file order. queryWeb passed a claim it could not serve when no crawler was configured, so an unconfigured deployment answered a web question with an apology instead of falling through to the model. Found in review of #67.
This commit is contained in:
@@ -2,6 +2,7 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
@@ -13,6 +14,7 @@ import (
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
"github.com/kami/maven/internal/webfetch"
|
||||
)
|
||||
|
||||
// The default config reads nothing. This is the whole "off unless configured"
|
||||
@@ -123,16 +125,13 @@ func TestQueryWebPassesWithoutAURL(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Not configured is said out loud rather than falling through, so a small model
|
||||
// never invents a page's contents from its URL.
|
||||
func TestQueryWebSaysWhenNotConfigured(t *testing.T) {
|
||||
// A daemon where page reading was never turned on — the default — answers the
|
||||
// question the way it did before the capability existed. Claiming the turn to
|
||||
// report a configuration status is for something that exists and failed.
|
||||
func TestQueryWebPassesWhenNotConfigured(t *testing.T) {
|
||||
h := buildWebHandler(nil)
|
||||
reply, ok := askWeb(h, "посмотри https://example.org/page")
|
||||
if !ok {
|
||||
t.Fatal("the web source did not claim a question with a URL")
|
||||
}
|
||||
if !strings.Contains(reply, "не настроено") {
|
||||
t.Errorf("reply = %q, want the not-configured answer", reply)
|
||||
if reply, ok := askWeb(h, "посмотри https://example.org/page"); ok {
|
||||
t.Fatalf("an unconfigured crawler claimed the turn with %q", reply)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -184,3 +183,58 @@ func (robotsDenyFetcher) Get(_ context.Context, u string) (*crawl.Response, erro
|
||||
}
|
||||
return &crawl.Response{URL: u, ContentType: "text/html", Body: []byte("<html>nope</html>")}, nil
|
||||
}
|
||||
|
||||
// TestCrawlHostsKeepsAWatchOutOfTheOnDemandAllowlist — the on-demand crawler
|
||||
// used to be built over allow_hosts PLUS every watched host. webfetch reads a
|
||||
// non-empty allow list as "these and nothing else", so one watch on a config
|
||||
// with no allow_hosts at all turned unrestricted on-demand reading into
|
||||
// "the watched site only", and every other URL he pasted came back as
|
||||
// "не получилось прочитать страницу." with nothing in the log to explain it.
|
||||
func TestCrawlHostsKeepsAWatchOutOfTheOnDemandAllowlist(t *testing.T) {
|
||||
cc := &config.CrawlConfig{
|
||||
OnDemand: true,
|
||||
Watches: []config.CrawlWatchConfig{{Name: "p", URL: "https://watched.example/p"}},
|
||||
}
|
||||
if got := crawlHosts(cc, false); len(got) != 0 {
|
||||
t.Errorf("on-demand allowlist = %v; a watch is not an allowlist entry, and an empty list is what means \"anything public\"", got)
|
||||
}
|
||||
if got := crawlHosts(cc, true); len(got) != 1 || got[0] != "watched.example" {
|
||||
t.Errorf("watch allowlist = %v; want the watched host so a watch needs no hand-written entry", got)
|
||||
}
|
||||
|
||||
// With allow_hosts set, his list is what on-demand gets, unchanged.
|
||||
cc.AllowHosts = []string{"wiki.example"}
|
||||
on := crawlHosts(cc, false)
|
||||
if len(on) != 1 || on[0] != "wiki.example" {
|
||||
t.Errorf("on-demand allowlist = %v; want exactly his allow_hosts", on)
|
||||
}
|
||||
if got := crawlHosts(cc, true); len(got) != 2 {
|
||||
t.Errorf("watch allowlist = %v; want his hosts plus the watched one", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCrawlFetcherReportsARefusalAsARefusal — internal/crawl cannot import
|
||||
// webfetch, so it used to recognise a guard refusal by matching substrings of
|
||||
// webfetch's message text. This adapter owns both packages and is where the
|
||||
// translation belongs.
|
||||
func TestCrawlFetcherReportsARefusalAsARefusal(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, "boom", http.StatusBadGateway)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
blocked := &crawlFetcher{f: webfetch.New(webfetch.Config{AllowHosts: []string{"wiki.example"}})}
|
||||
if _, err := blocked.Get(context.Background(), "https://other.example/a"); !errors.Is(err, crawl.ErrFetchRefused) {
|
||||
t.Errorf("a host outside allow_hosts = %v; want crawl.ErrFetchRefused", err)
|
||||
}
|
||||
if _, err := blocked.Get(context.Background(), "file:///etc/passwd"); !errors.Is(err, crawl.ErrFetchRefused) {
|
||||
t.Errorf("a non-http scheme = %v; want crawl.ErrFetchRefused", err)
|
||||
}
|
||||
|
||||
// A 5xx is a different thing: the server answered, badly. robots.txt over
|
||||
// this must refuse the crawl rather than read it as "no rules".
|
||||
open := &crawlFetcher{f: webfetch.New(webfetch.Config{AllowHosts: []string{"127.0.0.1"}, AllowPrivate: true})}
|
||||
if _, err := open.Get(context.Background(), srv.URL+"/robots.txt"); !errors.Is(err, crawl.ErrFetchStatus) {
|
||||
t.Errorf("a 502 = %v; want crawl.ErrFetchStatus", err)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user