Read a web page when he names one, and watch a few on a timer (#259)
The network fallback behind the local sources, off unless configured. internal/crawl is pure: a stdlib robots.txt parser (group specificity, wildcards, Crawl-delay, cached per host), HTML-to-plaintext extraction, and a watcher that notes a watched page only when its text changed. It has no store access and no net/http; cmd/mavend/crawls.go is the impure half. Every limit is code and tested: the guarded fetcher from #258 enforces the host allowlist/denylist, refuses private addresses in the dialer Control hook (so DNS rebinding and each redirect hop are covered), caps size and redirects, times out, and spaces requests per host. A robots.txt Disallow is refused with no override. On demand, reading is a query source placed last in the chain, after his memory, his notes, and the local Kiwix ZIMs once those are wired: no URL in the utterance means no fetch, and only the URL ever leaves the box. Scheduled watches write notes and announce nothing. The vendored tree has no x/net/html, goquery or temoto/robotstxt, so the parsers are stdlib. No new dependency.
This commit is contained in:
@@ -0,0 +1,184 @@
|
||||
// mavend/crawls.go — the driver for reading web pages (Vikunja #259,
|
||||
// docs/plans/14-web-crawler.md). The crawler is pure and lives in
|
||||
// internal/crawl; this is the impure half: the guarded fetcher, a ticker for the
|
||||
// scheduled watches, and the fact-backed dedup hashes.
|
||||
//
|
||||
// Two paths, one config block, both off unless configured:
|
||||
//
|
||||
// - ON DEMAND — he names a URL out loud and she reads it. That is the
|
||||
// `queryWeb` source in actions_query.go, LAST in the chain: after his
|
||||
// memory, after the notes, and (once Kiwix is wired into the chain) after
|
||||
// the local ZIMs. A local read costs nothing and leaks nothing; a fetch puts
|
||||
// a URL in someone's log, so it goes last.
|
||||
// - SCHEDULED — a watched page is re-read on its interval, and a page whose
|
||||
// text changed is written as a note. It does NOT announce itself. Same rule
|
||||
// as the feed poller: notes, never nudges.
|
||||
//
|
||||
// Only the URL goes out. Nothing here reads a note, a fact, the persona block or
|
||||
// the history, and internal/crawl has no access to the store at all.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"net/url"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/crawl"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/webfetch"
|
||||
)
|
||||
|
||||
// newCrawler builds the crawler from the `crawl` block, or returns nil when
|
||||
// there is none. Every caller checks for nil, and nil means no page is ever
|
||||
// fetched.
|
||||
func newCrawler(cfg *config.Config) *crawl.Crawler {
|
||||
if cfg.Crawl == nil {
|
||||
return nil
|
||||
}
|
||||
cc := cfg.Crawl
|
||||
|
||||
hosts := append([]string(nil), cc.AllowHosts...)
|
||||
// A watched page's own host is always reachable; otherwise an allowlist and
|
||||
// a watch list would have to be kept in sync by hand.
|
||||
for _, w := range cc.Watches {
|
||||
if u, err := url.Parse(w.URL); err == nil && u.Hostname() != "" {
|
||||
hosts = append(hosts, u.Hostname())
|
||||
}
|
||||
}
|
||||
// An allowlist plus on-demand is a contradiction worth logging rather than
|
||||
// silently resolving: he asked for arbitrary pages AND for a fixed list.
|
||||
// The allowlist wins, because it is the narrower instruction.
|
||||
if len(hosts) > 0 && cc.OnDemand && len(cc.AllowHosts) > 0 {
|
||||
log.Printf("crawl: allow_hosts is set, so on-demand reading is limited to those hosts")
|
||||
}
|
||||
ua := cc.UserAgent
|
||||
if ua == "" {
|
||||
ua = webfetch.DefaultUserAgent
|
||||
}
|
||||
fetcher := webfetch.New(webfetch.Config{
|
||||
AllowHosts: hosts,
|
||||
DenyHosts: cc.DenyHosts,
|
||||
Timeout: time.Duration(cc.Timeout),
|
||||
MaxBytes: cc.MaxBytes,
|
||||
UserAgent: ua,
|
||||
})
|
||||
// The user-agent handed to the crawler is the one the fetcher sends: obeying
|
||||
// robots rules written for a different name would be a lie.
|
||||
return crawl.New(&crawlFetcher{f: fetcher}, crawl.Config{
|
||||
UserAgent: ua,
|
||||
MaxRunes: cc.MaxRunes,
|
||||
})
|
||||
}
|
||||
|
||||
// onDemandCrawler returns a crawler for the answer path, or nil when on-demand
|
||||
// reading is off. The scheduled watches can be on while this is off: reading a
|
||||
// fixed list of pages on a timer and reading whatever URL is in an utterance are
|
||||
// different permissions, and the config keeps them separate.
|
||||
func onDemandCrawler(cfg *config.Config) *crawl.Crawler {
|
||||
if cfg.Crawl == nil || !cfg.Crawl.OnDemand {
|
||||
return nil
|
||||
}
|
||||
return newCrawler(cfg)
|
||||
}
|
||||
|
||||
// crawlWorker — ticker + watcher for the scheduled half.
|
||||
type crawlWorker struct {
|
||||
watcher *crawl.Watcher
|
||||
interval time.Duration
|
||||
}
|
||||
|
||||
// crawlTickInterval — how often the worker asks what is due. Per-watch cadence
|
||||
// is the watcher's business.
|
||||
const crawlTickInterval = 15 * time.Minute
|
||||
|
||||
// newCrawlWorker wires the scheduled crawls, or nil when nothing is watched.
|
||||
func newCrawlWorker(c *crawl.Crawler, api ipc.CoreAPI, emb router.Embedder, cfg *config.Config) *crawlWorker {
|
||||
if c == nil || cfg.Crawl == nil || len(cfg.Crawl.Watches) == 0 {
|
||||
return nil
|
||||
}
|
||||
watches := make([]crawl.WatchConfig, 0, len(cfg.Crawl.Watches))
|
||||
for _, w := range cfg.Crawl.Watches {
|
||||
watches = append(watches, crawl.WatchConfig{
|
||||
Name: w.Name,
|
||||
URL: w.URL,
|
||||
Interval: time.Duration(w.Interval),
|
||||
})
|
||||
}
|
||||
watcher := crawl.NewWatcher(c, watches, api, &factHashes{api: api},
|
||||
crawlEmbedder(emb), time.Duration(cfg.Crawl.Interval))
|
||||
if watcher == nil {
|
||||
log.Printf("crawl: configured but nothing watchable — scheduled crawls disabled")
|
||||
return nil
|
||||
}
|
||||
log.Printf("crawl: watching %d page(s), checking what is due every %s", len(watches), crawlTickInterval)
|
||||
return &crawlWorker{watcher: watcher, interval: crawlTickInterval}
|
||||
}
|
||||
|
||||
// run checks what is due until ctx is canceled. The first round runs
|
||||
// immediately; it writes notes only, so an early round startles nobody.
|
||||
func (w *crawlWorker) run(ctx context.Context) {
|
||||
w.watcher.CheckDue(ctx, time.Now())
|
||||
t := time.NewTicker(w.interval)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case now := <-t.C:
|
||||
w.watcher.CheckDue(ctx, now)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// crawlFetcher adapts webfetch to crawl.Fetcher, which is the seam that keeps
|
||||
// net/http out of the crawler package.
|
||||
type crawlFetcher struct{ f *webfetch.Fetcher }
|
||||
|
||||
func (a *crawlFetcher) Get(ctx context.Context, u string) (*crawl.Response, error) {
|
||||
resp, err := a.f.Get(ctx, u)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &crawl.Response{URL: resp.URL, ContentType: resp.ContentType, Body: resp.Body}, nil
|
||||
}
|
||||
|
||||
// factHashes stores each watch's last content hash as a config fact, so a
|
||||
// restart does not re-note an unchanged page. Same mechanism the feed reader
|
||||
// uses for its marks, and inspectable on /dash.
|
||||
type factHashes struct{ api ipc.CoreAPI }
|
||||
|
||||
func hashKey(name string) string { return "crawl:hash:" + name }
|
||||
|
||||
func (h *factHashes) LastHash(ctx context.Context, name string) (string, error) {
|
||||
f, err := h.api.LatestFact(ctx, hashKey(name))
|
||||
if err != nil {
|
||||
// No hash yet is not an error: the watcher treats "" as "never read".
|
||||
return "", nil
|
||||
}
|
||||
return f.Value, nil
|
||||
}
|
||||
|
||||
func (h *factHashes) SetHash(ctx context.Context, name, hash string) error {
|
||||
_, err := h.api.WriteFact(ctx, ipc.WriteFactReq{
|
||||
Ts: time.Now(),
|
||||
Kind: "config",
|
||||
Key: hashKey(name),
|
||||
Value: hash,
|
||||
Source: "poll:crawl",
|
||||
Confidence: 1.0,
|
||||
})
|
||||
return err
|
||||
}
|
||||
|
||||
// crawlEmbedder adapts router.Embedder for the watcher, embedding with
|
||||
// EmbedPassage (a page is text being searched FOR, and the e5 embedder is
|
||||
// asymmetric).
|
||||
func crawlEmbedder(emb router.Embedder) crawl.Embedder {
|
||||
if emb == nil {
|
||||
return nil
|
||||
}
|
||||
return passageEmbedder{emb}
|
||||
}
|
||||
Reference in New Issue
Block a user