2c1b0eede0
The network fallback behind the local sources, off unless configured. internal/crawl is pure: a stdlib robots.txt parser (group specificity, wildcards, Crawl-delay, cached per host), HTML-to-plaintext extraction, and a watcher that notes a watched page only when its text changed. It has no store access and no net/http; cmd/mavend/crawls.go is the impure half. Every limit is code and tested: the guarded fetcher from #258 enforces the host allowlist/denylist, refuses private addresses in the dialer Control hook (so DNS rebinding and each redirect hop are covered), caps size and redirects, times out, and spaces requests per host. A robots.txt Disallow is refused with no override. On demand, reading is a query source placed last in the chain, after his memory, his notes, and the local Kiwix ZIMs once those are wired: no URL in the utterance means no fetch, and only the URL ever leaves the box. Scheduled watches write notes and announce nothing. The vendored tree has no x/net/html, goquery or temoto/robotstxt, so the parsers are stdlib. No new dependency.
185 lines
6.3 KiB
Go
185 lines
6.3 KiB
Go
// mavend/crawls.go — the driver for reading web pages (Vikunja #259,
|
|
// docs/plans/14-web-crawler.md). The crawler is pure and lives in
|
|
// internal/crawl; this is the impure half: the guarded fetcher, a ticker for the
|
|
// scheduled watches, and the fact-backed dedup hashes.
|
|
//
|
|
// Two paths, one config block, both off unless configured:
|
|
//
|
|
// - ON DEMAND — he names a URL out loud and she reads it. That is the
|
|
// `queryWeb` source in actions_query.go, LAST in the chain: after his
|
|
// memory, after the notes, and (once Kiwix is wired into the chain) after
|
|
// the local ZIMs. A local read costs nothing and leaks nothing; a fetch puts
|
|
// a URL in someone's log, so it goes last.
|
|
// - SCHEDULED — a watched page is re-read on its interval, and a page whose
|
|
// text changed is written as a note. It does NOT announce itself. Same rule
|
|
// as the feed poller: notes, never nudges.
|
|
//
|
|
// Only the URL goes out. Nothing here reads a note, a fact, the persona block or
|
|
// the history, and internal/crawl has no access to the store at all.
|
|
package main
|
|
|
|
import (
|
|
"context"
|
|
"log"
|
|
"net/url"
|
|
"time"
|
|
|
|
"github.com/kami/maven/internal/config"
|
|
"github.com/kami/maven/internal/crawl"
|
|
"github.com/kami/maven/internal/ipc"
|
|
"github.com/kami/maven/internal/router"
|
|
"github.com/kami/maven/internal/webfetch"
|
|
)
|
|
|
|
// newCrawler builds the crawler from the `crawl` block, or returns nil when
|
|
// there is none. Every caller checks for nil, and nil means no page is ever
|
|
// fetched.
|
|
func newCrawler(cfg *config.Config) *crawl.Crawler {
|
|
if cfg.Crawl == nil {
|
|
return nil
|
|
}
|
|
cc := cfg.Crawl
|
|
|
|
hosts := append([]string(nil), cc.AllowHosts...)
|
|
// A watched page's own host is always reachable; otherwise an allowlist and
|
|
// a watch list would have to be kept in sync by hand.
|
|
for _, w := range cc.Watches {
|
|
if u, err := url.Parse(w.URL); err == nil && u.Hostname() != "" {
|
|
hosts = append(hosts, u.Hostname())
|
|
}
|
|
}
|
|
// An allowlist plus on-demand is a contradiction worth logging rather than
|
|
// silently resolving: he asked for arbitrary pages AND for a fixed list.
|
|
// The allowlist wins, because it is the narrower instruction.
|
|
if len(hosts) > 0 && cc.OnDemand && len(cc.AllowHosts) > 0 {
|
|
log.Printf("crawl: allow_hosts is set, so on-demand reading is limited to those hosts")
|
|
}
|
|
ua := cc.UserAgent
|
|
if ua == "" {
|
|
ua = webfetch.DefaultUserAgent
|
|
}
|
|
fetcher := webfetch.New(webfetch.Config{
|
|
AllowHosts: hosts,
|
|
DenyHosts: cc.DenyHosts,
|
|
Timeout: time.Duration(cc.Timeout),
|
|
MaxBytes: cc.MaxBytes,
|
|
UserAgent: ua,
|
|
})
|
|
// The user-agent handed to the crawler is the one the fetcher sends: obeying
|
|
// robots rules written for a different name would be a lie.
|
|
return crawl.New(&crawlFetcher{f: fetcher}, crawl.Config{
|
|
UserAgent: ua,
|
|
MaxRunes: cc.MaxRunes,
|
|
})
|
|
}
|
|
|
|
// onDemandCrawler returns a crawler for the answer path, or nil when on-demand
|
|
// reading is off. The scheduled watches can be on while this is off: reading a
|
|
// fixed list of pages on a timer and reading whatever URL is in an utterance are
|
|
// different permissions, and the config keeps them separate.
|
|
func onDemandCrawler(cfg *config.Config) *crawl.Crawler {
|
|
if cfg.Crawl == nil || !cfg.Crawl.OnDemand {
|
|
return nil
|
|
}
|
|
return newCrawler(cfg)
|
|
}
|
|
|
|
// crawlWorker — ticker + watcher for the scheduled half.
|
|
type crawlWorker struct {
|
|
watcher *crawl.Watcher
|
|
interval time.Duration
|
|
}
|
|
|
|
// crawlTickInterval — how often the worker asks what is due. Per-watch cadence
|
|
// is the watcher's business.
|
|
const crawlTickInterval = 15 * time.Minute
|
|
|
|
// newCrawlWorker wires the scheduled crawls, or nil when nothing is watched.
|
|
func newCrawlWorker(c *crawl.Crawler, api ipc.CoreAPI, emb router.Embedder, cfg *config.Config) *crawlWorker {
|
|
if c == nil || cfg.Crawl == nil || len(cfg.Crawl.Watches) == 0 {
|
|
return nil
|
|
}
|
|
watches := make([]crawl.WatchConfig, 0, len(cfg.Crawl.Watches))
|
|
for _, w := range cfg.Crawl.Watches {
|
|
watches = append(watches, crawl.WatchConfig{
|
|
Name: w.Name,
|
|
URL: w.URL,
|
|
Interval: time.Duration(w.Interval),
|
|
})
|
|
}
|
|
watcher := crawl.NewWatcher(c, watches, api, &factHashes{api: api},
|
|
crawlEmbedder(emb), time.Duration(cfg.Crawl.Interval))
|
|
if watcher == nil {
|
|
log.Printf("crawl: configured but nothing watchable — scheduled crawls disabled")
|
|
return nil
|
|
}
|
|
log.Printf("crawl: watching %d page(s), checking what is due every %s", len(watches), crawlTickInterval)
|
|
return &crawlWorker{watcher: watcher, interval: crawlTickInterval}
|
|
}
|
|
|
|
// run checks what is due until ctx is canceled. The first round runs
|
|
// immediately; it writes notes only, so an early round startles nobody.
|
|
func (w *crawlWorker) run(ctx context.Context) {
|
|
w.watcher.CheckDue(ctx, time.Now())
|
|
t := time.NewTicker(w.interval)
|
|
defer t.Stop()
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case now := <-t.C:
|
|
w.watcher.CheckDue(ctx, now)
|
|
}
|
|
}
|
|
}
|
|
|
|
// crawlFetcher adapts webfetch to crawl.Fetcher, which is the seam that keeps
|
|
// net/http out of the crawler package.
|
|
type crawlFetcher struct{ f *webfetch.Fetcher }
|
|
|
|
func (a *crawlFetcher) Get(ctx context.Context, u string) (*crawl.Response, error) {
|
|
resp, err := a.f.Get(ctx, u)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return &crawl.Response{URL: resp.URL, ContentType: resp.ContentType, Body: resp.Body}, nil
|
|
}
|
|
|
|
// factHashes stores each watch's last content hash as a config fact, so a
|
|
// restart does not re-note an unchanged page. Same mechanism the feed reader
|
|
// uses for its marks, and inspectable on /dash.
|
|
type factHashes struct{ api ipc.CoreAPI }
|
|
|
|
func hashKey(name string) string { return "crawl:hash:" + name }
|
|
|
|
func (h *factHashes) LastHash(ctx context.Context, name string) (string, error) {
|
|
f, err := h.api.LatestFact(ctx, hashKey(name))
|
|
if err != nil {
|
|
// No hash yet is not an error: the watcher treats "" as "never read".
|
|
return "", nil
|
|
}
|
|
return f.Value, nil
|
|
}
|
|
|
|
func (h *factHashes) SetHash(ctx context.Context, name, hash string) error {
|
|
_, err := h.api.WriteFact(ctx, ipc.WriteFactReq{
|
|
Ts: time.Now(),
|
|
Kind: "config",
|
|
Key: hashKey(name),
|
|
Value: hash,
|
|
Source: "poll:crawl",
|
|
Confidence: 1.0,
|
|
})
|
|
return err
|
|
}
|
|
|
|
// crawlEmbedder adapts router.Embedder for the watcher, embedding with
|
|
// EmbedPassage (a page is text being searched FOR, and the e5 embedder is
|
|
// asymmetric).
|
|
func crawlEmbedder(emb router.Embedder) crawl.Embedder {
|
|
if emb == nil {
|
|
return nil
|
|
}
|
|
return passageEmbedder{emb}
|
|
}
|