Files
Maven/cmd/mavend/crawls.go
T
kami 327726a06a crawl: stop letting a watch widen on-demand reading, and honour Crawl-delay
The on-demand crawler was built over allow_hosts plus every watched host.
webfetch reads a non-empty allow list as these and nothing else, so a config
with one watch and no allow_hosts at all silently narrowed on-demand reading
to the watched site. Every other url he pasted came back as a flat refusal
with nothing in the log to explain it. The two crawlers now take two host
lists from one crawlHosts helper.

Crawl-delay was parsed into Rules and never read. The only pacing was the
fetcher's flat one request per host per second, which cannot express what a
site asked for, and deploy/README claimed the field was honoured. Page now
waits it out between the robots fetch and the page fetch, and a delay longer
than the turn fails the read instead of hanging it.

A robots.txt that failed was treated as no rules, so a site whose server was
having a bad minute became a site with no restrictions. A 5xx now refuses the
crawl. A 404 still means unrestricted, which is what the standard says.

The refusal check matched substrings of webfetch's message text from a package
that cannot import webfetch, so a reworded error would have silently turned
into a robots verdict. internal/crawl now exports ErrFetchRefused and
ErrFetchStatus and the adapter in cmd/mavend maps the webfetch sentinels onto
them. Robots group selection picks the longest matching agent prefix instead
of the first one in file order.

queryWeb passed a claim it could not serve when no crawler was configured, so
an unconfigured deployment answered a web question with an apology instead of
falling through to the model.

Found in review of #67.
2026-08-01 14:33:26 +04:00

220 lines
8.0 KiB
Go

// mavend/crawls.go — the driver for reading web pages (Vikunja #259,
// docs/plans/14-web-crawler.md). The crawler is pure and lives in
// internal/crawl; this is the impure half: the guarded fetcher, a ticker for the
// scheduled watches, and the fact-backed dedup hashes.
//
// Two paths, one config block, both off unless configured:
//
// - ON DEMAND — he names a URL out loud and she reads it. That is the
// `queryWeb` source in actions_query.go, LAST in the chain: after his
// memory, after the notes, and (once Kiwix is wired into the chain) after
// the local ZIMs. A local read costs nothing and leaks nothing; a fetch puts
// a URL in someone's log, so it goes last.
// - SCHEDULED — a watched page is re-read on its interval, and a page whose
// text changed is written as a note. It does NOT announce itself. Same rule
// as the feed poller: notes, never nudges.
//
// Only the URL goes out. Nothing here reads a note, a fact, the persona block or
// the history, and internal/crawl has no access to the store at all.
package main
import (
"context"
"errors"
"fmt"
"log"
"net/url"
"time"
"github.com/kami/maven/internal/config"
"github.com/kami/maven/internal/crawl"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/webfetch"
)
// newCrawler builds the crawler from the `crawl` block, or returns nil when
// there is none. Every caller checks for nil, and nil means no page is ever
// fetched.
func newCrawler(cfg *config.Config) *crawl.Crawler {
if cfg.Crawl == nil {
return nil
}
cc := cfg.Crawl
// The WATCH crawler, and only it, reaches the watched hosts. webfetch reads
// a non-empty allow list as "these and nothing else", so folding the watch
// hosts in turned a single watch into an allowlist for everything: a config
// with one watch and on_demand true silently refused every other page he
// pasted, with "не получилось прочитать страницу." and no clue why.
return crawlerWithHosts(cc, crawlHosts(cc, true))
}
// crawlHosts — the allowlist for one of the two crawlers. forWatches adds the
// watched pages' own hosts, so a watch does not have to be allowlisted by hand.
//
// The on-demand crawler gets his allow_hosts and nothing else. webfetch reads a
// non-empty list as "these and nothing else", so adding the watch hosts there
// would silently narrow on-demand reading to the watched sites.
func crawlHosts(cc *config.CrawlConfig, forWatches bool) []string {
hosts := append([]string(nil), cc.AllowHosts...)
if !forWatches {
return hosts
}
for _, w := range cc.Watches {
if u, err := url.Parse(w.URL); err == nil && u.Hostname() != "" {
hosts = append(hosts, u.Hostname())
}
}
return hosts
}
// crawlerWithHosts builds a crawler over one allowlist. Two callers, two lists:
// see newCrawler and onDemandCrawler.
func crawlerWithHosts(cc *config.CrawlConfig, hosts []string) *crawl.Crawler {
ua := cc.UserAgent
if ua == "" {
ua = webfetch.DefaultUserAgent
}
fetcher := webfetch.New(webfetch.Config{
AllowHosts: hosts,
DenyHosts: cc.DenyHosts,
Timeout: time.Duration(cc.Timeout),
MaxBytes: cc.MaxBytes,
UserAgent: ua,
})
// The user-agent handed to the crawler is the one the fetcher sends: obeying
// robots rules written for a different name would be a lie.
return crawl.New(&crawlFetcher{f: fetcher}, crawl.Config{
UserAgent: ua,
MaxRunes: cc.MaxRunes,
})
}
// onDemandCrawler returns a crawler for the answer path, or nil when on-demand
// reading is off. The scheduled watches can be on while this is off: reading a
// fixed list of pages on a timer and reading whatever URL is in an utterance are
// different permissions, and the config keeps them separate.
func onDemandCrawler(cfg *config.Config) *crawl.Crawler {
if cfg.Crawl == nil || !cfg.Crawl.OnDemand {
return nil
}
cc := cfg.Crawl
// His own allow_hosts, and nothing added behind his back. Empty means "any
// host that is not denied and not private", which is what on-demand reading
// of a URL he just said out loud has to mean.
if len(cc.AllowHosts) > 0 {
log.Printf("crawl: allow_hosts is set, so on-demand reading is limited to those %d host(s)", len(cc.AllowHosts))
}
return crawlerWithHosts(cc, crawlHosts(cc, false))
}
// crawlWorker — ticker + watcher for the scheduled half.
type crawlWorker struct {
watcher *crawl.Watcher
interval time.Duration
}
// crawlTickInterval — how often the worker asks what is due. Per-watch cadence
// is the watcher's business.
const crawlTickInterval = 15 * time.Minute
// newCrawlWorker wires the scheduled crawls, or nil when nothing is watched.
func newCrawlWorker(c *crawl.Crawler, api ipc.CoreAPI, emb router.Embedder, cfg *config.Config) *crawlWorker {
if c == nil || cfg.Crawl == nil || len(cfg.Crawl.Watches) == 0 {
return nil
}
watches := make([]crawl.WatchConfig, 0, len(cfg.Crawl.Watches))
for _, w := range cfg.Crawl.Watches {
watches = append(watches, crawl.WatchConfig{
Name: w.Name,
URL: w.URL,
Interval: time.Duration(w.Interval),
})
}
watcher := crawl.NewWatcher(c, watches, api, &factHashes{api: api},
crawlEmbedder(emb), time.Duration(cfg.Crawl.Interval))
if watcher == nil {
log.Printf("crawl: configured but nothing watchable — scheduled crawls disabled")
return nil
}
log.Printf("crawl: watching %d page(s), checking what is due every %s", len(watches), crawlTickInterval)
return &crawlWorker{watcher: watcher, interval: crawlTickInterval}
}
// run checks what is due until ctx is canceled. The first round runs
// immediately; it writes notes only, so an early round startles nobody.
func (w *crawlWorker) run(ctx context.Context) {
w.watcher.CheckDue(ctx, time.Now())
t := time.NewTicker(w.interval)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case now := <-t.C:
w.watcher.CheckDue(ctx, now)
}
}
}
// crawlFetcher adapts webfetch to crawl.Fetcher, which is the seam that keeps
// net/http out of the crawler package.
type crawlFetcher struct{ f *webfetch.Fetcher }
// Get maps webfetch's sentinels onto crawl's. This adapter is the one place
// that imports both packages, so the mapping belongs here; the crawler used to
// match on three substrings of a message it could not see the definition of,
// and a reworded error would have quietly turned a blocked host into "there is
// no robots.txt here".
func (a *crawlFetcher) Get(ctx context.Context, u string) (*crawl.Response, error) {
resp, err := a.f.Get(ctx, u)
if err != nil {
switch {
case errors.Is(err, webfetch.ErrBlocked), errors.Is(err, webfetch.ErrPrivate), errors.Is(err, webfetch.ErrScheme):
return nil, fmt.Errorf("%w: %v", crawl.ErrFetchRefused, err)
case errors.Is(err, webfetch.ErrStatus):
return nil, fmt.Errorf("%w: %v", crawl.ErrFetchStatus, err)
}
return nil, err
}
return &crawl.Response{URL: resp.URL, ContentType: resp.ContentType, Body: resp.Body}, nil
}
// factHashes stores each watch's last content hash as a config fact, so a
// restart does not re-note an unchanged page. Same mechanism the feed reader
// uses for its marks, and inspectable on /dash.
type factHashes struct{ api ipc.CoreAPI }
func hashKey(name string) string { return "crawl:hash:" + name }
func (h *factHashes) LastHash(ctx context.Context, name string) (string, error) {
f, err := h.api.LatestFact(ctx, hashKey(name))
if err != nil {
// No hash yet is not an error: the watcher treats "" as "never read".
return "", nil
}
return f.Value, nil
}
func (h *factHashes) SetHash(ctx context.Context, name, hash string) error {
_, err := h.api.WriteFact(ctx, ipc.WriteFactReq{
Ts: time.Now(),
Kind: "config",
Key: hashKey(name),
Value: hash,
Source: "poll:crawl",
Confidence: 1.0,
})
return err
}
// crawlEmbedder adapts router.Embedder for the watcher, embedding with
// EmbedPassage (a page is text being searched FOR, and the e5 embedder is
// asymmetric).
func crawlEmbedder(emb router.Embedder) crawl.Embedder {
if emb == nil {
return nil
}
return passageEmbedder{emb}
}