Read a web page when he names one, and watch a few on a timer (#259)

The network fallback behind the local sources, off unless configured.

internal/crawl is pure: a stdlib robots.txt parser (group specificity,
wildcards, Crawl-delay, cached per host), HTML-to-plaintext extraction, and a
watcher that notes a watched page only when its text changed. It has no store
access and no net/http; cmd/mavend/crawls.go is the impure half.

Every limit is code and tested: the guarded fetcher from #258 enforces the host
allowlist/denylist, refuses private addresses in the dialer Control hook (so DNS
rebinding and each redirect hop are covered), caps size and redirects, times out,
and spaces requests per host. A robots.txt Disallow is refused with no override.

On demand, reading is a query source placed last in the chain, after his memory,
his notes, and the local Kiwix ZIMs once those are wired: no URL in the
utterance means no fetch, and only the URL ever leaves the box. Scheduled
watches write notes and announce nothing.

The vendored tree has no x/net/html, goquery or temoto/robotstxt, so the parsers
are stdlib. No new dependency.
This commit is contained in:
kami
2026-08-01 03:40:21 +04:00
parent cb3641e7bb
commit 2c1b0eede0
18 changed files with 1696 additions and 11 deletions
+59
View File
@@ -8,6 +8,7 @@ import (
"strings"
"time"
"github.com/kami/maven/internal/crawl"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/memory"
"github.com/kami/maven/internal/morning"
@@ -78,6 +79,13 @@ var querySources = []querySource{
{"embed", (*reactiveHandler).queryEmbed},
{"memory", (*reactiveHandler).queryMemory},
{"notes", (*reactiveHandler).queryNotes},
// LAST before the model answers from memory, and that position is the whole
// design (Vikunja #259): local sources first. The model, his own notes and
// facts, and — once internal/kiwix is wired into this chain — the offline
// ZIMs all get their turn before anything touches the network. This source
// only claims a turn where he named a URL out loud, so it never competes
// with a local answer.
{"web", (*reactiveHandler).queryWeb},
{"general-knowledge", (*reactiveHandler).queryGeneral},
}
@@ -371,6 +379,57 @@ func (h *reactiveHandler) queryNotes(ctx context.Context, t *queryTurn) (string,
return reply, true
}
// webPageContextRunes — how much of a fetched page is handed to the phraser.
// Less than the crawler keeps: the rest of the 4096-token window belongs to the
// prompt, the persona block and the reply.
const webPageContextRunes = 1500
// queryWeb — "посмотри https://example.org/x — что там?" (Vikunja #259).
//
// It claims a turn ONLY when he named a URL, which is what keeps a fallback from
// becoming a habit: no URL, no fetch, and the model answers from what is local.
// What leaves the box is the URL and nothing else — no note, no fact, no history
// travels with it.
func (h *reactiveHandler) queryWeb(ctx context.Context, t *queryTurn) (string, bool) {
link, ok := router.FirstURL(t.dec.Utterance)
if !ok {
return "", false
}
if h.crawler == nil {
// Claim rather than fall through: he asked about a specific page, and
// letting the model answer from the URL's spelling alone is how a small
// model invents a page's contents.
return "я не читаю страницы — это не настроено.", true
}
ctxFetch, cancel := context.WithTimeout(ctx, 30*time.Second)
defer cancel()
page, err := h.crawler.Page(ctxFetch, link)
if err != nil {
if errors.Is(err, crawl.ErrRobots) {
return "эта страница закрыта для чтения — robots.txt не разрешает.", true
}
log.Printf("voice: web: %v", err)
return "не получилось прочитать страницу.", true
}
if page.Text == "" {
return "страница открылась, но читать там нечего.", true
}
// The page is handed to the phraser the same way a note is: as context for
// the question he actually asked. She answers the question, she does not
// recite the page.
snippet := page.Title + "\n" + crawl.TrimRunes(page.Text, webPageContextRunes)
reply, perr := h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet})
if perr != nil {
log.Printf("voice: web: phrase: %v", perr)
}
if reply == "" {
// No phraser (or it failed): read back the top of the page rather than
// pretend the fetch did not happen.
return "вот что на странице: " + crawl.TrimRunes(page.Text, 300), true
}
return reply, true
}
// queryGeneral — general knowledge from the phraser, the last source before
// giving up. It always claims: either the model answers or Maven says she
// doesn't know.
+184
View File
@@ -0,0 +1,184 @@
// mavend/crawls.go — the driver for reading web pages (Vikunja #259,
// docs/plans/14-web-crawler.md). The crawler is pure and lives in
// internal/crawl; this is the impure half: the guarded fetcher, a ticker for the
// scheduled watches, and the fact-backed dedup hashes.
//
// Two paths, one config block, both off unless configured:
//
// - ON DEMAND — he names a URL out loud and she reads it. That is the
// `queryWeb` source in actions_query.go, LAST in the chain: after his
// memory, after the notes, and (once Kiwix is wired into the chain) after
// the local ZIMs. A local read costs nothing and leaks nothing; a fetch puts
// a URL in someone's log, so it goes last.
// - SCHEDULED — a watched page is re-read on its interval, and a page whose
// text changed is written as a note. It does NOT announce itself. Same rule
// as the feed poller: notes, never nudges.
//
// Only the URL goes out. Nothing here reads a note, a fact, the persona block or
// the history, and internal/crawl has no access to the store at all.
package main
import (
"context"
"log"
"net/url"
"time"
"github.com/kami/maven/internal/config"
"github.com/kami/maven/internal/crawl"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/webfetch"
)
// newCrawler builds the crawler from the `crawl` block, or returns nil when
// there is none. Every caller checks for nil, and nil means no page is ever
// fetched.
func newCrawler(cfg *config.Config) *crawl.Crawler {
if cfg.Crawl == nil {
return nil
}
cc := cfg.Crawl
hosts := append([]string(nil), cc.AllowHosts...)
// A watched page's own host is always reachable; otherwise an allowlist and
// a watch list would have to be kept in sync by hand.
for _, w := range cc.Watches {
if u, err := url.Parse(w.URL); err == nil && u.Hostname() != "" {
hosts = append(hosts, u.Hostname())
}
}
// An allowlist plus on-demand is a contradiction worth logging rather than
// silently resolving: he asked for arbitrary pages AND for a fixed list.
// The allowlist wins, because it is the narrower instruction.
if len(hosts) > 0 && cc.OnDemand && len(cc.AllowHosts) > 0 {
log.Printf("crawl: allow_hosts is set, so on-demand reading is limited to those hosts")
}
ua := cc.UserAgent
if ua == "" {
ua = webfetch.DefaultUserAgent
}
fetcher := webfetch.New(webfetch.Config{
AllowHosts: hosts,
DenyHosts: cc.DenyHosts,
Timeout: time.Duration(cc.Timeout),
MaxBytes: cc.MaxBytes,
UserAgent: ua,
})
// The user-agent handed to the crawler is the one the fetcher sends: obeying
// robots rules written for a different name would be a lie.
return crawl.New(&crawlFetcher{f: fetcher}, crawl.Config{
UserAgent: ua,
MaxRunes: cc.MaxRunes,
})
}
// onDemandCrawler returns a crawler for the answer path, or nil when on-demand
// reading is off. The scheduled watches can be on while this is off: reading a
// fixed list of pages on a timer and reading whatever URL is in an utterance are
// different permissions, and the config keeps them separate.
func onDemandCrawler(cfg *config.Config) *crawl.Crawler {
if cfg.Crawl == nil || !cfg.Crawl.OnDemand {
return nil
}
return newCrawler(cfg)
}
// crawlWorker — ticker + watcher for the scheduled half.
type crawlWorker struct {
watcher *crawl.Watcher
interval time.Duration
}
// crawlTickInterval — how often the worker asks what is due. Per-watch cadence
// is the watcher's business.
const crawlTickInterval = 15 * time.Minute
// newCrawlWorker wires the scheduled crawls, or nil when nothing is watched.
func newCrawlWorker(c *crawl.Crawler, api ipc.CoreAPI, emb router.Embedder, cfg *config.Config) *crawlWorker {
if c == nil || cfg.Crawl == nil || len(cfg.Crawl.Watches) == 0 {
return nil
}
watches := make([]crawl.WatchConfig, 0, len(cfg.Crawl.Watches))
for _, w := range cfg.Crawl.Watches {
watches = append(watches, crawl.WatchConfig{
Name: w.Name,
URL: w.URL,
Interval: time.Duration(w.Interval),
})
}
watcher := crawl.NewWatcher(c, watches, api, &factHashes{api: api},
crawlEmbedder(emb), time.Duration(cfg.Crawl.Interval))
if watcher == nil {
log.Printf("crawl: configured but nothing watchable — scheduled crawls disabled")
return nil
}
log.Printf("crawl: watching %d page(s), checking what is due every %s", len(watches), crawlTickInterval)
return &crawlWorker{watcher: watcher, interval: crawlTickInterval}
}
// run checks what is due until ctx is canceled. The first round runs
// immediately; it writes notes only, so an early round startles nobody.
func (w *crawlWorker) run(ctx context.Context) {
w.watcher.CheckDue(ctx, time.Now())
t := time.NewTicker(w.interval)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case now := <-t.C:
w.watcher.CheckDue(ctx, now)
}
}
}
// crawlFetcher adapts webfetch to crawl.Fetcher, which is the seam that keeps
// net/http out of the crawler package.
type crawlFetcher struct{ f *webfetch.Fetcher }
func (a *crawlFetcher) Get(ctx context.Context, u string) (*crawl.Response, error) {
resp, err := a.f.Get(ctx, u)
if err != nil {
return nil, err
}
return &crawl.Response{URL: resp.URL, ContentType: resp.ContentType, Body: resp.Body}, nil
}
// factHashes stores each watch's last content hash as a config fact, so a
// restart does not re-note an unchanged page. Same mechanism the feed reader
// uses for its marks, and inspectable on /dash.
type factHashes struct{ api ipc.CoreAPI }
func hashKey(name string) string { return "crawl:hash:" + name }
func (h *factHashes) LastHash(ctx context.Context, name string) (string, error) {
f, err := h.api.LatestFact(ctx, hashKey(name))
if err != nil {
// No hash yet is not an error: the watcher treats "" as "never read".
return "", nil
}
return f.Value, nil
}
func (h *factHashes) SetHash(ctx context.Context, name, hash string) error {
_, err := h.api.WriteFact(ctx, ipc.WriteFactReq{
Ts: time.Now(),
Kind: "config",
Key: hashKey(name),
Value: hash,
Source: "poll:crawl",
Confidence: 1.0,
})
return err
}
// crawlEmbedder adapts router.Embedder for the watcher, embedding with
// EmbedPassage (a page is text being searched FOR, and the e5 embedder is
// asymmetric).
func crawlEmbedder(emb router.Embedder) crawl.Embedder {
if emb == nil {
return nil
}
return passageEmbedder{emb}
}
+186
View File
@@ -0,0 +1,186 @@
package main
import (
"context"
"net/http"
"net/http/httptest"
"strings"
"testing"
"github.com/kami/maven/internal/config"
"github.com/kami/maven/internal/crawl"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/voice"
)
// The default config reads nothing. This is the whole "off unless configured"
// contract for the crawler, asserted at the wiring level rather than trusted.
func TestCrawlOffByDefault(t *testing.T) {
cfg := &config.Config{}
if c := newCrawler(cfg); c != nil {
t.Error("newCrawler with no crawl block returned a crawler")
}
if c := onDemandCrawler(cfg); c != nil {
t.Error("onDemandCrawler with no crawl block returned a crawler")
}
if w := newCrawlWorker(nil, nil, nil, cfg); w != nil {
t.Error("newCrawlWorker with no crawl block returned a worker")
}
// Watches configured but on_demand off ⇒ the answer path still reads
// nothing: a timer over a fixed list is not permission for arbitrary URLs.
withWatch := &config.Config{Crawl: &config.CrawlConfig{
Watches: []config.CrawlWatchConfig{{Name: "p", URL: "https://example.org/p"}},
}}
if c := onDemandCrawler(withWatch); c != nil {
t.Error("onDemandCrawler honoured a watch list as on-demand permission")
}
if c := newCrawler(withWatch); c == nil {
t.Error("newCrawler returned nil for a configured watch")
}
}
// The wired fetcher must refuse a private address, because the crawler on this
// box sits one hop from the whole homelab. Same guard the webfetch tests cover;
// this asserts the daemon actually wires it.
func TestCrawlerRefusesPrivateAddress(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "text/html")
w.Write([]byte("<html><body>secret</body></html>"))
}))
defer srv.Close()
c := newCrawler(&config.Config{Crawl: &config.CrawlConfig{OnDemand: true}})
if c == nil {
t.Fatal("newCrawler returned nil for an on-demand config")
}
if _, err := c.Page(context.Background(), srv.URL); err == nil {
t.Fatalf("reading %s succeeded; a loopback address must be refused", srv.URL)
}
}
func TestFactHashesRoundTrip(t *testing.T) {
ctx := context.Background()
st := newTestStore(t)
h := &factHashes{api: ipc.NewStoreAPI(st)}
got, err := h.LastHash(ctx, "page")
if err != nil {
t.Fatalf("LastHash on a fresh store: %v", err)
}
if got != "" {
t.Errorf("LastHash = %q, want empty for a never-read page", got)
}
if err := h.SetHash(ctx, "page", "deadbeef"); err != nil {
t.Fatalf("SetHash: %v", err)
}
got, err = h.LastHash(ctx, "page")
if err != nil {
t.Fatalf("LastHash: %v", err)
}
if got != "deadbeef" {
t.Errorf("LastHash = %q, want deadbeef", got)
}
if key := hashKey("page"); key != "crawl:hash:page" {
t.Errorf("hashKey = %q", key)
}
}
// stubCrawlFetcher serves one fixed page to every URL, so queryWeb can be
// exercised without a network or an allowlist.
type stubCrawlFetcher struct{ body, ctype string }
func (s *stubCrawlFetcher) Get(_ context.Context, u string) (*crawl.Response, error) {
ct := s.ctype
if ct == "" {
ct = "text/html"
}
if strings.HasSuffix(u, "/robots.txt") {
return &crawl.Response{URL: u, ContentType: "text/plain", Body: []byte("")}, nil
}
return &crawl.Response{URL: u, ContentType: ct, Body: []byte(s.body)}, nil
}
func buildWebHandler(c *crawl.Crawler) *reactiveHandler {
return &reactiveHandler{
replier: voice.NewStubReplier(),
phraser: phraser.NewStub(),
crawler: c,
}
}
func askWeb(h *reactiveHandler, q string) (string, bool) {
return h.queryWeb(context.Background(), &queryTurn{
dec: router.Decision{Intent: router.IntentQuery, Utterance: q},
})
}
func TestQueryWebPassesWithoutAURL(t *testing.T) {
h := buildWebHandler(crawl.New(&stubCrawlFetcher{body: "<html><body>x</body></html>"}, crawl.Config{}))
if reply, ok := askWeb(h, "почему небо синее?"); ok {
t.Errorf("the web source claimed a question with no URL: %q", reply)
}
}
// Not configured is said out loud rather than falling through, so a small model
// never invents a page's contents from its URL.
func TestQueryWebSaysWhenNotConfigured(t *testing.T) {
h := buildWebHandler(nil)
reply, ok := askWeb(h, "посмотри https://example.org/page")
if !ok {
t.Fatal("the web source did not claim a question with a URL")
}
if !strings.Contains(reply, "не настроено") {
t.Errorf("reply = %q, want the not-configured answer", reply)
}
}
func TestQueryWebReadsThePage(t *testing.T) {
h := buildWebHandler(crawl.New(&stubCrawlFetcher{
body: "<html><head><title>Заголовок</title></head><body><p>текст страницы</p></body></html>",
}, crawl.Config{}))
reply, ok := askWeb(h, "посмотри https://example.org/page — что там?")
if !ok {
t.Fatal("the web source did not claim a question with a URL")
}
if !strings.Contains(reply, "текст страницы") {
t.Errorf("reply = %q, want the page text read back", reply)
}
}
func TestQueryWebRefusesNonHTML(t *testing.T) {
h := buildWebHandler(crawl.New(&stubCrawlFetcher{
body: "\x00\x01binary", ctype: "application/octet-stream",
}, crawl.Config{}))
reply, ok := askWeb(h, "почитай https://example.org/blob.bin")
if !ok {
t.Fatal("the web source did not claim a question with a URL")
}
if !strings.Contains(reply, "не получилось") {
t.Errorf("reply = %q, want the read-failed answer", reply)
}
}
// robots.txt is honoured on the answer path too, and she says so instead of
// reporting a generic failure.
func TestQueryWebObeysRobots(t *testing.T) {
h := buildWebHandler(crawl.New(&robotsDenyFetcher{}, crawl.Config{}))
reply, ok := askWeb(h, "посмотри https://example.org/private")
if !ok {
t.Fatal("the web source did not claim a question with a URL")
}
if !strings.Contains(reply, "robots.txt") {
t.Errorf("reply = %q, want the robots answer", reply)
}
}
type robotsDenyFetcher struct{}
func (robotsDenyFetcher) Get(_ context.Context, u string) (*crawl.Response, error) {
if strings.HasSuffix(u, "/robots.txt") {
return &crawl.Response{URL: u, ContentType: "text/plain",
Body: []byte("User-agent: *\nDisallow: /private\n")}, nil
}
return &crawl.Response{URL: u, ContentType: "text/html", Body: []byte("<html>nope</html>")}, nil
}
+17
View File
@@ -171,6 +171,7 @@ func run(args []string) error {
factWorker *factEnrichmentWorker
evalWorker *memoryEvalWorker // nil ⇒ memory evaluation off (the default)
feedWkr *feedWorker // nil ⇒ no feed is read (the default)
crawlWkr *crawlWorker // nil ⇒ no page is watched (the default)
)
if !locked {
@@ -267,6 +268,7 @@ func run(args []string) error {
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
evalWorker = newMemoryEvalWorker(st, phr, cfg)
feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
coreAPI = &daemonAPI{
CoreAPI: ipc.NewStoreAPI(st),
@@ -454,6 +456,7 @@ func run(args []string) error {
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
evalWorker = newMemoryEvalWorker(st, phr, cfg)
feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
// Swap the CoreAPI from the locked placeholder to the real store adapter.
newAPI := &daemonAPI{
@@ -506,6 +509,13 @@ func run(args []string) error {
}()
}
// Start the watched-page crawls (nil unless configured).
if crawlWkr != nil {
go func() {
crawlWkr.run(ctx)
}()
}
dl.unlock()
log.Printf("mavend: unlocked via passkey assertion")
return nil
@@ -558,6 +568,13 @@ func run(args []string) error {
feedWkr.run(ctx)
}()
}
if crawlWkr != nil {
wg.Add(1)
go func() {
defer wg.Done()
crawlWkr.run(ctx)
}()
}
}
<-ctx.Done()
+5
View File
@@ -52,6 +52,7 @@ import (
"time"
"github.com/kami/maven/internal/audio"
"github.com/kami/maven/internal/crawl"
"github.com/kami/maven/internal/dialogue"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/memory"
@@ -82,6 +83,10 @@ type reactiveHandler struct {
replier voice.Replier
now func() time.Time
// crawler reads a web page he names out loud (queryWeb). nil ⇒ on-demand
// page reading is off, which is the default: no `crawl` block, no fetch.
crawler *crawl.Crawler
// feedsOn — whether any RSS feed is configured (config.Feeds). It changes
// only what she SAYS when asked and nothing is there: "ленты не настроены"
// instead of "ничего нового", which are different truths.
+14 -11
View File
@@ -201,17 +201,20 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
// ----- the handler (the reactive path; closes over stt / tts / router / coreAPI / memory) -----
h := &reactiveHandler{
stt: transcriber,
tts: synthesizer,
router: rtr,
embedder: emb,
api: coreAPI,
tools: exec,
matcher: matcher,
replier: replier,
phraser: phr,
now: time.Now,
feedsOn: cfg.Feeds != nil,
stt: transcriber,
tts: synthesizer,
router: rtr,
embedder: emb,
api: coreAPI,
tools: exec,
matcher: matcher,
replier: replier,
phraser: phr,
now: time.Now,
feedsOn: cfg.Feeds != nil,
// nil unless `crawl.on_demand` is on: reading a page he names is a
// capability, and capabilities are off unless configured.
crawler: onDemandCrawler(cfg),
weatherProvider: weatherProvider,
weatherLocation: weatherLocation,
memStore: memStore,