diff --git a/cmd/mavend/actions_query.go b/cmd/mavend/actions_query.go
index 4cfa1b7..56f9ede 100644
--- a/cmd/mavend/actions_query.go
+++ b/cmd/mavend/actions_query.go
@@ -8,6 +8,7 @@ import (
"strings"
"time"
+ "github.com/kami/maven/internal/crawl"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/memory"
"github.com/kami/maven/internal/morning"
@@ -78,6 +79,13 @@ var querySources = []querySource{
{"embed", (*reactiveHandler).queryEmbed},
{"memory", (*reactiveHandler).queryMemory},
{"notes", (*reactiveHandler).queryNotes},
+ // LAST before the model answers from memory, and that position is the whole
+ // design (Vikunja #259): local sources first. The model, his own notes and
+ // facts, and — once internal/kiwix is wired into this chain — the offline
+ // ZIMs all get their turn before anything touches the network. This source
+ // only claims a turn where he named a URL out loud, so it never competes
+ // with a local answer.
+ {"web", (*reactiveHandler).queryWeb},
{"general-knowledge", (*reactiveHandler).queryGeneral},
}
@@ -371,6 +379,57 @@ func (h *reactiveHandler) queryNotes(ctx context.Context, t *queryTurn) (string,
return reply, true
}
+// webPageContextRunes — how much of a fetched page is handed to the phraser.
+// Less than the crawler keeps: the rest of the 4096-token window belongs to the
+// prompt, the persona block and the reply.
+const webPageContextRunes = 1500
+
+// queryWeb — "посмотри https://example.org/x — что там?" (Vikunja #259).
+//
+// It claims a turn ONLY when he named a URL, which is what keeps a fallback from
+// becoming a habit: no URL, no fetch, and the model answers from what is local.
+// What leaves the box is the URL and nothing else — no note, no fact, no history
+// travels with it.
+func (h *reactiveHandler) queryWeb(ctx context.Context, t *queryTurn) (string, bool) {
+ link, ok := router.FirstURL(t.dec.Utterance)
+ if !ok {
+ return "", false
+ }
+ if h.crawler == nil {
+ // Claim rather than fall through: he asked about a specific page, and
+ // letting the model answer from the URL's spelling alone is how a small
+ // model invents a page's contents.
+ return "я не читаю страницы — это не настроено.", true
+ }
+ ctxFetch, cancel := context.WithTimeout(ctx, 30*time.Second)
+ defer cancel()
+ page, err := h.crawler.Page(ctxFetch, link)
+ if err != nil {
+ if errors.Is(err, crawl.ErrRobots) {
+ return "эта страница закрыта для чтения — robots.txt не разрешает.", true
+ }
+ log.Printf("voice: web: %v", err)
+ return "не получилось прочитать страницу.", true
+ }
+ if page.Text == "" {
+ return "страница открылась, но читать там нечего.", true
+ }
+ // The page is handed to the phraser the same way a note is: as context for
+ // the question he actually asked. She answers the question, she does not
+ // recite the page.
+ snippet := page.Title + "\n" + crawl.TrimRunes(page.Text, webPageContextRunes)
+ reply, perr := h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet})
+ if perr != nil {
+ log.Printf("voice: web: phrase: %v", perr)
+ }
+ if reply == "" {
+ // No phraser (or it failed): read back the top of the page rather than
+ // pretend the fetch did not happen.
+ return "вот что на странице: " + crawl.TrimRunes(page.Text, 300), true
+ }
+ return reply, true
+}
+
// queryGeneral — general knowledge from the phraser, the last source before
// giving up. It always claims: either the model answers or Maven says she
// doesn't know.
diff --git a/cmd/mavend/crawls.go b/cmd/mavend/crawls.go
new file mode 100644
index 0000000..e20a01a
--- /dev/null
+++ b/cmd/mavend/crawls.go
@@ -0,0 +1,184 @@
+// mavend/crawls.go — the driver for reading web pages (Vikunja #259,
+// docs/plans/14-web-crawler.md). The crawler is pure and lives in
+// internal/crawl; this is the impure half: the guarded fetcher, a ticker for the
+// scheduled watches, and the fact-backed dedup hashes.
+//
+// Two paths, one config block, both off unless configured:
+//
+// - ON DEMAND — he names a URL out loud and she reads it. That is the
+// `queryWeb` source in actions_query.go, LAST in the chain: after his
+// memory, after the notes, and (once Kiwix is wired into the chain) after
+// the local ZIMs. A local read costs nothing and leaks nothing; a fetch puts
+// a URL in someone's log, so it goes last.
+// - SCHEDULED — a watched page is re-read on its interval, and a page whose
+// text changed is written as a note. It does NOT announce itself. Same rule
+// as the feed poller: notes, never nudges.
+//
+// Only the URL goes out. Nothing here reads a note, a fact, the persona block or
+// the history, and internal/crawl has no access to the store at all.
+package main
+
+import (
+ "context"
+ "log"
+ "net/url"
+ "time"
+
+ "github.com/kami/maven/internal/config"
+ "github.com/kami/maven/internal/crawl"
+ "github.com/kami/maven/internal/ipc"
+ "github.com/kami/maven/internal/router"
+ "github.com/kami/maven/internal/webfetch"
+)
+
+// newCrawler builds the crawler from the `crawl` block, or returns nil when
+// there is none. Every caller checks for nil, and nil means no page is ever
+// fetched.
+func newCrawler(cfg *config.Config) *crawl.Crawler {
+ if cfg.Crawl == nil {
+ return nil
+ }
+ cc := cfg.Crawl
+
+ hosts := append([]string(nil), cc.AllowHosts...)
+ // A watched page's own host is always reachable; otherwise an allowlist and
+ // a watch list would have to be kept in sync by hand.
+ for _, w := range cc.Watches {
+ if u, err := url.Parse(w.URL); err == nil && u.Hostname() != "" {
+ hosts = append(hosts, u.Hostname())
+ }
+ }
+ // An allowlist plus on-demand is a contradiction worth logging rather than
+ // silently resolving: he asked for arbitrary pages AND for a fixed list.
+ // The allowlist wins, because it is the narrower instruction.
+ if len(hosts) > 0 && cc.OnDemand && len(cc.AllowHosts) > 0 {
+ log.Printf("crawl: allow_hosts is set, so on-demand reading is limited to those hosts")
+ }
+ ua := cc.UserAgent
+ if ua == "" {
+ ua = webfetch.DefaultUserAgent
+ }
+ fetcher := webfetch.New(webfetch.Config{
+ AllowHosts: hosts,
+ DenyHosts: cc.DenyHosts,
+ Timeout: time.Duration(cc.Timeout),
+ MaxBytes: cc.MaxBytes,
+ UserAgent: ua,
+ })
+ // The user-agent handed to the crawler is the one the fetcher sends: obeying
+ // robots rules written for a different name would be a lie.
+ return crawl.New(&crawlFetcher{f: fetcher}, crawl.Config{
+ UserAgent: ua,
+ MaxRunes: cc.MaxRunes,
+ })
+}
+
+// onDemandCrawler returns a crawler for the answer path, or nil when on-demand
+// reading is off. The scheduled watches can be on while this is off: reading a
+// fixed list of pages on a timer and reading whatever URL is in an utterance are
+// different permissions, and the config keeps them separate.
+func onDemandCrawler(cfg *config.Config) *crawl.Crawler {
+ if cfg.Crawl == nil || !cfg.Crawl.OnDemand {
+ return nil
+ }
+ return newCrawler(cfg)
+}
+
+// crawlWorker — ticker + watcher for the scheduled half.
+type crawlWorker struct {
+ watcher *crawl.Watcher
+ interval time.Duration
+}
+
+// crawlTickInterval — how often the worker asks what is due. Per-watch cadence
+// is the watcher's business.
+const crawlTickInterval = 15 * time.Minute
+
+// newCrawlWorker wires the scheduled crawls, or nil when nothing is watched.
+func newCrawlWorker(c *crawl.Crawler, api ipc.CoreAPI, emb router.Embedder, cfg *config.Config) *crawlWorker {
+ if c == nil || cfg.Crawl == nil || len(cfg.Crawl.Watches) == 0 {
+ return nil
+ }
+ watches := make([]crawl.WatchConfig, 0, len(cfg.Crawl.Watches))
+ for _, w := range cfg.Crawl.Watches {
+ watches = append(watches, crawl.WatchConfig{
+ Name: w.Name,
+ URL: w.URL,
+ Interval: time.Duration(w.Interval),
+ })
+ }
+ watcher := crawl.NewWatcher(c, watches, api, &factHashes{api: api},
+ crawlEmbedder(emb), time.Duration(cfg.Crawl.Interval))
+ if watcher == nil {
+ log.Printf("crawl: configured but nothing watchable — scheduled crawls disabled")
+ return nil
+ }
+ log.Printf("crawl: watching %d page(s), checking what is due every %s", len(watches), crawlTickInterval)
+ return &crawlWorker{watcher: watcher, interval: crawlTickInterval}
+}
+
+// run checks what is due until ctx is canceled. The first round runs
+// immediately; it writes notes only, so an early round startles nobody.
+func (w *crawlWorker) run(ctx context.Context) {
+ w.watcher.CheckDue(ctx, time.Now())
+ t := time.NewTicker(w.interval)
+ defer t.Stop()
+ for {
+ select {
+ case <-ctx.Done():
+ return
+ case now := <-t.C:
+ w.watcher.CheckDue(ctx, now)
+ }
+ }
+}
+
+// crawlFetcher adapts webfetch to crawl.Fetcher, which is the seam that keeps
+// net/http out of the crawler package.
+type crawlFetcher struct{ f *webfetch.Fetcher }
+
+func (a *crawlFetcher) Get(ctx context.Context, u string) (*crawl.Response, error) {
+ resp, err := a.f.Get(ctx, u)
+ if err != nil {
+ return nil, err
+ }
+ return &crawl.Response{URL: resp.URL, ContentType: resp.ContentType, Body: resp.Body}, nil
+}
+
+// factHashes stores each watch's last content hash as a config fact, so a
+// restart does not re-note an unchanged page. Same mechanism the feed reader
+// uses for its marks, and inspectable on /dash.
+type factHashes struct{ api ipc.CoreAPI }
+
+func hashKey(name string) string { return "crawl:hash:" + name }
+
+func (h *factHashes) LastHash(ctx context.Context, name string) (string, error) {
+ f, err := h.api.LatestFact(ctx, hashKey(name))
+ if err != nil {
+ // No hash yet is not an error: the watcher treats "" as "never read".
+ return "", nil
+ }
+ return f.Value, nil
+}
+
+func (h *factHashes) SetHash(ctx context.Context, name, hash string) error {
+ _, err := h.api.WriteFact(ctx, ipc.WriteFactReq{
+ Ts: time.Now(),
+ Kind: "config",
+ Key: hashKey(name),
+ Value: hash,
+ Source: "poll:crawl",
+ Confidence: 1.0,
+ })
+ return err
+}
+
+// crawlEmbedder adapts router.Embedder for the watcher, embedding with
+// EmbedPassage (a page is text being searched FOR, and the e5 embedder is
+// asymmetric).
+func crawlEmbedder(emb router.Embedder) crawl.Embedder {
+ if emb == nil {
+ return nil
+ }
+ return passageEmbedder{emb}
+}
diff --git a/cmd/mavend/crawls_test.go b/cmd/mavend/crawls_test.go
new file mode 100644
index 0000000..2910d5f
--- /dev/null
+++ b/cmd/mavend/crawls_test.go
@@ -0,0 +1,186 @@
+package main
+
+import (
+ "context"
+ "net/http"
+ "net/http/httptest"
+ "strings"
+ "testing"
+
+ "github.com/kami/maven/internal/config"
+ "github.com/kami/maven/internal/crawl"
+ "github.com/kami/maven/internal/ipc"
+ "github.com/kami/maven/internal/phraser"
+ "github.com/kami/maven/internal/router"
+ "github.com/kami/maven/internal/voice"
+)
+
+// The default config reads nothing. This is the whole "off unless configured"
+// contract for the crawler, asserted at the wiring level rather than trusted.
+func TestCrawlOffByDefault(t *testing.T) {
+ cfg := &config.Config{}
+ if c := newCrawler(cfg); c != nil {
+ t.Error("newCrawler with no crawl block returned a crawler")
+ }
+ if c := onDemandCrawler(cfg); c != nil {
+ t.Error("onDemandCrawler with no crawl block returned a crawler")
+ }
+ if w := newCrawlWorker(nil, nil, nil, cfg); w != nil {
+ t.Error("newCrawlWorker with no crawl block returned a worker")
+ }
+ // Watches configured but on_demand off ⇒ the answer path still reads
+ // nothing: a timer over a fixed list is not permission for arbitrary URLs.
+ withWatch := &config.Config{Crawl: &config.CrawlConfig{
+ Watches: []config.CrawlWatchConfig{{Name: "p", URL: "https://example.org/p"}},
+ }}
+ if c := onDemandCrawler(withWatch); c != nil {
+ t.Error("onDemandCrawler honoured a watch list as on-demand permission")
+ }
+ if c := newCrawler(withWatch); c == nil {
+ t.Error("newCrawler returned nil for a configured watch")
+ }
+}
+
+// The wired fetcher must refuse a private address, because the crawler on this
+// box sits one hop from the whole homelab. Same guard the webfetch tests cover;
+// this asserts the daemon actually wires it.
+func TestCrawlerRefusesPrivateAddress(t *testing.T) {
+ srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ w.Header().Set("Content-Type", "text/html")
+ w.Write([]byte("
secret"))
+ }))
+ defer srv.Close()
+
+ c := newCrawler(&config.Config{Crawl: &config.CrawlConfig{OnDemand: true}})
+ if c == nil {
+ t.Fatal("newCrawler returned nil for an on-demand config")
+ }
+ if _, err := c.Page(context.Background(), srv.URL); err == nil {
+ t.Fatalf("reading %s succeeded; a loopback address must be refused", srv.URL)
+ }
+}
+
+func TestFactHashesRoundTrip(t *testing.T) {
+ ctx := context.Background()
+ st := newTestStore(t)
+ h := &factHashes{api: ipc.NewStoreAPI(st)}
+
+ got, err := h.LastHash(ctx, "page")
+ if err != nil {
+ t.Fatalf("LastHash on a fresh store: %v", err)
+ }
+ if got != "" {
+ t.Errorf("LastHash = %q, want empty for a never-read page", got)
+ }
+ if err := h.SetHash(ctx, "page", "deadbeef"); err != nil {
+ t.Fatalf("SetHash: %v", err)
+ }
+ got, err = h.LastHash(ctx, "page")
+ if err != nil {
+ t.Fatalf("LastHash: %v", err)
+ }
+ if got != "deadbeef" {
+ t.Errorf("LastHash = %q, want deadbeef", got)
+ }
+ if key := hashKey("page"); key != "crawl:hash:page" {
+ t.Errorf("hashKey = %q", key)
+ }
+}
+
+// stubCrawlFetcher serves one fixed page to every URL, so queryWeb can be
+// exercised without a network or an allowlist.
+type stubCrawlFetcher struct{ body, ctype string }
+
+func (s *stubCrawlFetcher) Get(_ context.Context, u string) (*crawl.Response, error) {
+ ct := s.ctype
+ if ct == "" {
+ ct = "text/html"
+ }
+ if strings.HasSuffix(u, "/robots.txt") {
+ return &crawl.Response{URL: u, ContentType: "text/plain", Body: []byte("")}, nil
+ }
+ return &crawl.Response{URL: u, ContentType: ct, Body: []byte(s.body)}, nil
+}
+
+func buildWebHandler(c *crawl.Crawler) *reactiveHandler {
+ return &reactiveHandler{
+ replier: voice.NewStubReplier(),
+ phraser: phraser.NewStub(),
+ crawler: c,
+ }
+}
+
+func askWeb(h *reactiveHandler, q string) (string, bool) {
+ return h.queryWeb(context.Background(), &queryTurn{
+ dec: router.Decision{Intent: router.IntentQuery, Utterance: q},
+ })
+}
+
+func TestQueryWebPassesWithoutAURL(t *testing.T) {
+ h := buildWebHandler(crawl.New(&stubCrawlFetcher{body: "x"}, crawl.Config{}))
+ if reply, ok := askWeb(h, "почему небо синее?"); ok {
+ t.Errorf("the web source claimed a question with no URL: %q", reply)
+ }
+}
+
+// Not configured is said out loud rather than falling through, so a small model
+// never invents a page's contents from its URL.
+func TestQueryWebSaysWhenNotConfigured(t *testing.T) {
+ h := buildWebHandler(nil)
+ reply, ok := askWeb(h, "посмотри https://example.org/page")
+ if !ok {
+ t.Fatal("the web source did not claim a question with a URL")
+ }
+ if !strings.Contains(reply, "не настроено") {
+ t.Errorf("reply = %q, want the not-configured answer", reply)
+ }
+}
+
+func TestQueryWebReadsThePage(t *testing.T) {
+ h := buildWebHandler(crawl.New(&stubCrawlFetcher{
+ body: "Заголовок
текст страницы
",
+ }, crawl.Config{}))
+ reply, ok := askWeb(h, "посмотри https://example.org/page — что там?")
+ if !ok {
+ t.Fatal("the web source did not claim a question with a URL")
+ }
+ if !strings.Contains(reply, "текст страницы") {
+ t.Errorf("reply = %q, want the page text read back", reply)
+ }
+}
+
+func TestQueryWebRefusesNonHTML(t *testing.T) {
+ h := buildWebHandler(crawl.New(&stubCrawlFetcher{
+ body: "\x00\x01binary", ctype: "application/octet-stream",
+ }, crawl.Config{}))
+ reply, ok := askWeb(h, "почитай https://example.org/blob.bin")
+ if !ok {
+ t.Fatal("the web source did not claim a question with a URL")
+ }
+ if !strings.Contains(reply, "не получилось") {
+ t.Errorf("reply = %q, want the read-failed answer", reply)
+ }
+}
+
+// robots.txt is honoured on the answer path too, and she says so instead of
+// reporting a generic failure.
+func TestQueryWebObeysRobots(t *testing.T) {
+ h := buildWebHandler(crawl.New(&robotsDenyFetcher{}, crawl.Config{}))
+ reply, ok := askWeb(h, "посмотри https://example.org/private")
+ if !ok {
+ t.Fatal("the web source did not claim a question with a URL")
+ }
+ if !strings.Contains(reply, "robots.txt") {
+ t.Errorf("reply = %q, want the robots answer", reply)
+ }
+}
+
+type robotsDenyFetcher struct{}
+
+func (robotsDenyFetcher) Get(_ context.Context, u string) (*crawl.Response, error) {
+ if strings.HasSuffix(u, "/robots.txt") {
+ return &crawl.Response{URL: u, ContentType: "text/plain",
+ Body: []byte("User-agent: *\nDisallow: /private\n")}, nil
+ }
+ return &crawl.Response{URL: u, ContentType: "text/html", Body: []byte("nope")}, nil
+}
diff --git a/cmd/mavend/main.go b/cmd/mavend/main.go
index 4053be3..0c10d80 100644
--- a/cmd/mavend/main.go
+++ b/cmd/mavend/main.go
@@ -171,6 +171,7 @@ func run(args []string) error {
factWorker *factEnrichmentWorker
evalWorker *memoryEvalWorker // nil ⇒ memory evaluation off (the default)
feedWkr *feedWorker // nil ⇒ no feed is read (the default)
+ crawlWkr *crawlWorker // nil ⇒ no page is watched (the default)
)
if !locked {
@@ -267,6 +268,7 @@ func run(args []string) error {
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
evalWorker = newMemoryEvalWorker(st, phr, cfg)
feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
+ crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
coreAPI = &daemonAPI{
CoreAPI: ipc.NewStoreAPI(st),
@@ -454,6 +456,7 @@ func run(args []string) error {
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
evalWorker = newMemoryEvalWorker(st, phr, cfg)
feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
+ crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
// Swap the CoreAPI from the locked placeholder to the real store adapter.
newAPI := &daemonAPI{
@@ -506,6 +509,13 @@ func run(args []string) error {
}()
}
+ // Start the watched-page crawls (nil unless configured).
+ if crawlWkr != nil {
+ go func() {
+ crawlWkr.run(ctx)
+ }()
+ }
+
dl.unlock()
log.Printf("mavend: unlocked via passkey assertion")
return nil
@@ -558,6 +568,13 @@ func run(args []string) error {
feedWkr.run(ctx)
}()
}
+ if crawlWkr != nil {
+ wg.Add(1)
+ go func() {
+ defer wg.Done()
+ crawlWkr.run(ctx)
+ }()
+ }
}
<-ctx.Done()
diff --git a/cmd/mavend/voice.go b/cmd/mavend/voice.go
index 4ef572c..c9ecbcd 100644
--- a/cmd/mavend/voice.go
+++ b/cmd/mavend/voice.go
@@ -52,6 +52,7 @@ import (
"time"
"github.com/kami/maven/internal/audio"
+ "github.com/kami/maven/internal/crawl"
"github.com/kami/maven/internal/dialogue"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/memory"
@@ -82,6 +83,10 @@ type reactiveHandler struct {
replier voice.Replier
now func() time.Time
+ // crawler reads a web page he names out loud (queryWeb). nil ⇒ on-demand
+ // page reading is off, which is the default: no `crawl` block, no fetch.
+ crawler *crawl.Crawler
+
// feedsOn — whether any RSS feed is configured (config.Feeds). It changes
// only what she SAYS when asked and nothing is there: "ленты не настроены"
// instead of "ничего нового", which are different truths.
diff --git a/cmd/mavend/voicewire.go b/cmd/mavend/voicewire.go
index 21032df..7167511 100644
--- a/cmd/mavend/voicewire.go
+++ b/cmd/mavend/voicewire.go
@@ -201,17 +201,20 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
// ----- the handler (the reactive path; closes over stt / tts / router / coreAPI / memory) -----
h := &reactiveHandler{
- stt: transcriber,
- tts: synthesizer,
- router: rtr,
- embedder: emb,
- api: coreAPI,
- tools: exec,
- matcher: matcher,
- replier: replier,
- phraser: phr,
- now: time.Now,
- feedsOn: cfg.Feeds != nil,
+ stt: transcriber,
+ tts: synthesizer,
+ router: rtr,
+ embedder: emb,
+ api: coreAPI,
+ tools: exec,
+ matcher: matcher,
+ replier: replier,
+ phraser: phr,
+ now: time.Now,
+ feedsOn: cfg.Feeds != nil,
+ // nil unless `crawl.on_demand` is on: reading a page he names is a
+ // capability, and capabilities are off unless configured.
+ crawler: onDemandCrawler(cfg),
weatherProvider: weatherProvider,
weatherLocation: weatherLocation,
memStore: memStore,
diff --git a/deploy/README.md b/deploy/README.md
index 0cb1a24..2f68637 100644
--- a/deploy/README.md
+++ b/deploy/README.md
@@ -67,6 +67,36 @@ What it does and does not do:
- how far each feed was read is stored as a config fact `rss:latest:`, so
a restart does not re-note yesterday's headlines.
+### Reading a page (`crawl`, also off by default)
+
+There is no `crawl` block either, so no page is fetched. Two halves, separately
+switched:
+
+```json
+"crawl": {
+ "on_demand": true,
+ "interval": "6h",
+ "max_runes": 4000,
+ "watches": [
+ { "name": "changelog", "url": "https://example.org/changelog", "interval": "12h" }
+ ]
+}
+```
+
+- `on_demand` lets her read a page he names in the utterance: "посмотри
+ https://example.org/x — что там?". The page becomes context for his question,
+ and only the URL leaves the box. Without a URL nothing is fetched, so this is
+ a fallback and not a habit;
+- `watches` re-reads a fixed list on its interval and writes a note when the
+ text changed. Like the feeds, it announces nothing;
+- the answer path sits **last** in the query chain, behind his memory, his notes
+ and (once wired) the local Kiwix ZIMs. A local read costs nothing;
+- `robots.txt` is fetched first and obeyed with no override; a `Disallow` is a
+ refusal she says out loud. `Crawl-delay` is honoured;
+- same guarded fetcher as the feeds: allowlist/denylist, no private addresses,
+ size cap, redirect cap, timeout, one request per host per second;
+- dedup state is the config fact `crawl:hash:`.
+
## Not yet verified / host-dependent
This stack is correct-by-construction but has **not been build-tested here**
diff --git a/docs/plans/14-web-crawler.md b/docs/plans/14-web-crawler.md
index b5c78ac..a9b22f3 100644
--- a/docs/plans/14-web-crawler.md
+++ b/docs/plans/14-web-crawler.md
@@ -28,3 +28,41 @@
7. Add IPC methods `MethodTriggerCrawl(name)`, `MethodListCrawls`, `MethodGetCrawlResult(name)`
8. Add `crawls` block to `config.Config` and `deploy/mavend.json`
9. Test with a static HTML page — verify extraction matches expected values, verify scheduling fires correctly
+
+## Shipped 2026-08-01 (#259)
+
+Built as `internal/crawl` (pure: robots, extraction, watcher) plus
+`cmd/mavend/crawls.go` (fetcher, ticker, dedup facts), on top of the guarded
+`internal/webfetch` door added with the feed reader (#258). Off unless
+configured, in two separately-switched halves: `crawl.on_demand` for a URL he
+names, `crawl.watches` for a scheduled re-read.
+
+**Limits are code, not documentation** (`internal/webfetch`, tested one test per
+limit): host allowlist/denylist, no private addresses (loopback, RFC1918 —
+hence the LAN and the `10.42.0.0/24` wg range —, link-local incl. cloud
+metadata, CGNAT, v6 ULA) enforced in the dialer's `Control` hook so DNS
+rebinding and every redirect hop are covered, response size cap, redirect cap,
+timeout, one request per host per second. `robots.txt` is fetched first, cached
+per host, and a `Disallow` is refused with no override.
+
+Deliberate deviations from the plan above:
+
+- **No CSS selectors and no LLM structured extraction** (steps 2). The output is
+ plaintext handed to the phraser as context for the question he asked. A 1.7B
+ extracting a JSON price table from 4000 runes is a worse bet than reading, and
+ `goquery` is not vendored.
+- **No `crawl` act verb and no new IPC methods** (steps 5, 7). Reading a page is
+ a query source (`queryWeb` in `actions_query.go`, last in the chain, behind
+ Kiwix once that is wired), not an action he commands. Nothing needs a new wire
+ method to work.
+- **Notes, not facts.** A page's text is not a fact about him. Only the dedup
+ hash is a fact (`crawl:hash:`, kind `config`, source `poll:crawl`).
+- **Nothing is dispatched.** A changed page writes a note; it does not nudge.
+ Not a nag.
+- **No `/tools` crawl history page.** The notes and the hash facts are already
+ visible on `/dash`.
+
+**No new dependency.** The vendored tree has no `x/net/html`, no `goquery` and
+no `temoto/robotstxt`, so robots parsing and HTML-to-text are stdlib
+(`regexp`, `html`) — RE2 has no backreferences, hence the `pairsRE` builder in
+`extract.go`.
diff --git a/internal/config/config.go b/internal/config/config.go
index 47d476c..c3fe4b6 100644
--- a/internal/config/config.go
+++ b/internal/config/config.go
@@ -161,6 +161,10 @@ type Config struct {
// the weather and telegram. See FeedsConfig.
Feeds *FeedsConfig `json:"feeds,omitempty"`
+ // Crawl — reading a web page (Vikunja #259). nil / absent ⇒ Maven never
+ // fetches a page: not on request, not on a schedule. See CrawlConfig.
+ Crawl *CrawlConfig `json:"crawl,omitempty"`
+
// Praxis — the ecosystem attention-state service. When configured, maven
// calls the Praxis HTTP tools API for attention listing and item lifecycle.
// Maven never touches Praxis's database directly (ecosystem invariant: no
@@ -454,6 +458,62 @@ type FeedSourceConfig struct {
Exclude []string `json:"exclude,omitempty"` // drop items containing any of these
}
+// CrawlConfig — the web crawler (Vikunja #259, docs/plans/14-web-crawler.md).
+//
+// Absent ⇒ off, and off means no page is ever fetched. Present with neither
+// `on_demand` nor a `watches` entry is also off: there would be nothing to do.
+//
+// The crawler is the LAST place an answer is looked for, behind the model, his
+// own memory and the local Kiwix ZIMs. That ordering lives in the query-source
+// chain (cmd/mavend/actions_query.go), not here, but it is the reason this block
+// is small: it is a fallback, not a search engine.
+//
+// Only the URL leaves the box. His notes, facts, persona block and history are
+// never part of a request — the crawler package cannot even read the store.
+type CrawlConfig struct {
+ // OnDemand — may he ask her to read a page he names out loud
+ // ("посмотри https://… — что там пишут?"). false ⇒ the on-demand answer
+ // source stays off and only the watches below run.
+ OnDemand bool `json:"on_demand,omitempty"`
+
+ // Watches — pages re-read on a schedule. A page whose text changed is
+ // written as a note (source "crawl:"); nothing is announced.
+ Watches []CrawlWatchConfig `json:"watches,omitempty"`
+
+ // Interval — default watch cadence. 0 ⇒ crawl.DefaultWatchInterval (6h).
+ Interval Duration `json:"interval,omitempty"`
+
+ // AllowHosts — when set, the ONLY hosts the crawler may reach (subdomains
+ // included). Watched pages' own hosts are added automatically. Setting this
+ // is how "she may read the arch wiki and nothing else" is expressed.
+ AllowHosts []string `json:"allow_hosts,omitempty"`
+
+ // DenyHosts — never reachable, checked first. Private addresses do not need
+ // to be listed: they are refused unconditionally (see internal/webfetch).
+ DenyHosts []string `json:"deny_hosts,omitempty"`
+
+ // UserAgent — sent on every request AND matched against robots.txt groups.
+ // Empty ⇒ webfetch.DefaultUserAgent.
+ UserAgent string `json:"user_agent,omitempty"`
+
+ // Timeout — per-request budget. 0 ⇒ webfetch.DefaultTimeout.
+ Timeout Duration `json:"timeout,omitempty"`
+
+ // MaxBytes — response size cap. 0 ⇒ webfetch.DefaultMaxBytes (2 MiB).
+ MaxBytes int64 `json:"max_bytes,omitempty"`
+
+ // MaxRunes — how much extracted text is kept. 0 ⇒ crawl.DefaultMaxRunes
+ // (4000), which is what fits a 4096-token context alongside a prompt.
+ MaxRunes int `json:"max_runes,omitempty"`
+}
+
+// CrawlWatchConfig — one page kept an eye on.
+type CrawlWatchConfig struct {
+ Name string `json:"name"` // note source is "crawl:"
+ URL string `json:"url"`
+ Interval Duration `json:"interval,omitempty"` // 0 ⇒ CrawlConfig.Interval
+}
+
// MemoryEvalConfig — the background memory-evaluation loop (Vikunja #248).
// Absent ⇒ off, like every other capability that costs something the owner did
// not ask for. Each evaluation is a full LLM round-trip on the one resident
@@ -697,6 +757,12 @@ func (c *Config) applyDefaults() {
c.Feeds = nil
}
+ // Same rule for the crawler: a block that neither answers on demand nor
+ // watches anything has nothing to do, so it is normalised to "off".
+ if c.Crawl != nil && !c.Crawl.OnDemand && len(c.Crawl.Watches) == 0 {
+ c.Crawl = nil
+ }
+
if c.Voice != nil {
if c.Voice.RouterThreshold <= 0 {
c.Voice.RouterThreshold = DefaultRouterThreshold
diff --git a/internal/crawl/crawl.go b/internal/crawl/crawl.go
new file mode 100644
index 0000000..96bec10
--- /dev/null
+++ b/internal/crawl/crawl.go
@@ -0,0 +1,174 @@
+// Package crawl reads a web page: fetch, robots check, HTML to text.
+//
+// It is the LAST place Maven looks for an answer, and that ordering is the whole
+// design. "Never phones home" is deprecated, but what replaced it puts local
+// sources first: the resident model, then his own memory, then the Kiwix ZIMs on
+// the box (internal/kiwix), and only then the network. A local read costs
+// nothing and leaks nothing; a fetch costs a round-trip and puts a URL in
+// someone's access log. So this package exists to be the fallback, not the
+// front door — see the querySources chain in cmd/mavend/actions_query.go for
+// where it actually sits.
+//
+// What never leaves the box: his notes, his facts, the persona block, the
+// conversation history. Only the URL is requested and, for the on-demand path,
+// only because he said it out loud. Nothing here reads the store.
+//
+// The limits are not in this package — they are in internal/webfetch, which is
+// the only way anything here touches a socket: http(s) only, host allow/deny,
+// private-address refusal, size cap, redirect cap, per-host rate limit. What
+// this package adds is politeness (robots.txt) and dedup.
+package crawl
+
+import (
+ "context"
+ "crypto/sha256"
+ "encoding/hex"
+ "errors"
+ "fmt"
+ "net/url"
+ "strings"
+ "time"
+)
+
+// Errors callers distinguish.
+var (
+ ErrRobots = errors.New("crawl: robots.txt disallows this path")
+ ErrNotHTML = errors.New("crawl: response is not html or text")
+)
+
+// Fetcher is the guarded HTTP door (internal/webfetch adapted by the daemon). An
+// interface so this package constructs no http.Client of its own and can be
+// tested without a network.
+type Fetcher interface {
+ Get(ctx context.Context, url string) (*Response, error)
+}
+
+// Response is the minimum a crawl needs from a fetch.
+type Response struct {
+ URL string
+ ContentType string
+ Body []byte
+}
+
+// Config — crawler knobs.
+type Config struct {
+ // UserAgent is the name matched against robots.txt groups. It must be the
+ // same string the fetcher sends, or Maven would be claiming one identity
+ // and obeying the rules for another.
+ UserAgent string
+ // MaxRunes caps extracted text. 0 ⇒ DefaultMaxRunes.
+ MaxRunes int
+ // RobotsTTL — how long a parsed robots.txt is trusted. 0 ⇒ 1h.
+ RobotsTTL time.Duration
+ // Now is injectable for tests. nil ⇒ time.Now.
+ Now func() time.Time
+}
+
+// Crawler fetches and extracts pages. Safe for concurrent use.
+type Crawler struct {
+ fetch Fetcher
+ cfg Config
+ robots *robotsCache
+}
+
+// New builds a crawler. Returns nil when there is no fetcher, which is how the
+// daemon expresses "crawling is off unless configured".
+func New(fetch Fetcher, cfg Config) *Crawler {
+ if fetch == nil {
+ return nil
+ }
+ if cfg.UserAgent == "" {
+ cfg.UserAgent = "Maven"
+ }
+ if cfg.MaxRunes <= 0 {
+ cfg.MaxRunes = DefaultMaxRunes
+ }
+ if cfg.RobotsTTL <= 0 {
+ cfg.RobotsTTL = time.Hour
+ }
+ if cfg.Now == nil {
+ cfg.Now = time.Now
+ }
+ return &Crawler{fetch: fetch, cfg: cfg, robots: newRobotsCache(cfg.RobotsTTL)}
+}
+
+// Page fetches rawURL and returns its text. It checks robots.txt first and
+// refuses a disallowed path with ErrRobots — there is no override.
+func (c *Crawler) Page(ctx context.Context, rawURL string) (Page, error) {
+ u, err := url.Parse(strings.TrimSpace(rawURL))
+ if err != nil {
+ return Page{}, fmt.Errorf("crawl: bad url %q: %w", rawURL, err)
+ }
+ ok, err := c.allowed(ctx, u)
+ if err != nil {
+ return Page{}, err
+ }
+ if !ok {
+ return Page{}, fmt.Errorf("%w: %s", ErrRobots, u.Path)
+ }
+ resp, err := c.fetch.Get(ctx, u.String())
+ if err != nil {
+ return Page{}, err
+ }
+ // A PDF or an image is bytes Maven cannot read; saying so beats storing
+ // binary garbage as a "note".
+ ct := strings.ToLower(resp.ContentType)
+ if ct != "" && !strings.Contains(ct, "html") && !strings.Contains(ct, "text/") &&
+ !strings.Contains(ct, "xml") && !strings.Contains(ct, "json") {
+ return Page{}, fmt.Errorf("%w: %s", ErrNotHTML, resp.ContentType)
+ }
+ return Extract(resp.URL, resp.Body, c.cfg.MaxRunes), nil
+}
+
+// allowed consults robots.txt for u's host, reading it at most once per TTL.
+//
+// A robots.txt that cannot be fetched (404, a timeout, a blocked host) means
+// allow, per the standard. The one thing that is NOT fail-open is an explicit
+// Disallow.
+func (c *Crawler) allowed(ctx context.Context, u *url.URL) (bool, error) {
+ host := u.Host
+ now := c.cfg.Now()
+ rules, ok := c.robots.get(host, now)
+ if !ok {
+ robotsURL := u.Scheme + "://" + host + "/robots.txt"
+ resp, err := c.fetch.Get(ctx, robotsURL)
+ switch {
+ case err != nil:
+ // Note what is NOT swallowed: a refusal from the guarded fetcher.
+ // If webfetch says this host is denied or private, the page fetch
+ // would fail the same way, and reporting the real reason beats
+ // reporting a robots verdict we never got.
+ if isFatalFetchError(err) {
+ return false, err
+ }
+ rules = Rules{}
+ default:
+ rules = ParseRobots(string(resp.Body), c.cfg.UserAgent)
+ }
+ c.robots.put(host, rules, now)
+ }
+ path := u.EscapedPath()
+ if u.RawQuery != "" {
+ path += "?" + u.RawQuery
+ }
+ return rules.Allowed(path), nil
+}
+
+// isFatalFetchError — a fetch failure that means "this host is off limits"
+// rather than "there is no robots.txt here". The sentinel set is webfetch's, but
+// this package must not import it (the interface exists precisely so it does
+// not), so the check is on the message. Ugly and honest: the alternative is a
+// dependency inversion for two strings.
+func isFatalFetchError(err error) bool {
+ s := err.Error()
+ return strings.Contains(s, "not allowed") || strings.Contains(s, "private address") ||
+ strings.Contains(s, "only http and https")
+}
+
+// Hash is the dedup key for a crawl result: the sha256 of the extracted text,
+// hex, first 16 chars. Text and not raw HTML, because a page whose only change
+// is a rotating ad slot or a CSRF token has not changed.
+func Hash(text string) string {
+ sum := sha256.Sum256([]byte(strings.TrimSpace(text)))
+ return hex.EncodeToString(sum[:])[:16]
+}
diff --git a/internal/crawl/crawl_test.go b/internal/crawl/crawl_test.go
new file mode 100644
index 0000000..0dbaf4d
--- /dev/null
+++ b/internal/crawl/crawl_test.go
@@ -0,0 +1,155 @@
+package crawl
+
+import (
+ "context"
+ "errors"
+ "strings"
+ "testing"
+ "time"
+)
+
+// fakeFetcher serves canned pages by URL and counts requests, so a test can
+// assert that robots.txt was read once and that a refusal never reached the page.
+type fakeFetcher struct {
+ pages map[string]Response
+ err error
+ calls []string
+}
+
+func (f *fakeFetcher) Get(_ context.Context, u string) (*Response, error) {
+ f.calls = append(f.calls, u)
+ if f.err != nil {
+ return nil, f.err
+ }
+ r, ok := f.pages[u]
+ if !ok {
+ return nil, errors.New("http 404")
+ }
+ if r.URL == "" {
+ r.URL = u
+ }
+ if r.ContentType == "" {
+ r.ContentType = "text/html; charset=utf-8"
+ }
+ return &r, nil
+}
+
+const htmlPage = `Почему небо синее
+
+
"
+ f := &fakeFetcher{pages: map[string]Response{"https://example.org/l": {Body: []byte(long)}}}
+ c := New(f, Config{MaxRunes: 50})
+ page, err := c.Page(context.Background(), "https://example.org/l")
+ if err != nil {
+ t.Fatal(err)
+ }
+ if n := len([]rune(page.Text)); n > 51 {
+ t.Fatalf("text = %d runes, want the 50-rune cap", n)
+ }
+}
+
+func TestNewWithoutFetcherIsNil(t *testing.T) {
+ if New(nil, Config{}) != nil {
+ t.Fatal("a crawler with no fetcher must be nil — crawling is off unless configured")
+ }
+}
+
+func TestHashIgnoresNothingButText(t *testing.T) {
+ if Hash("a") == Hash("b") {
+ t.Fatal("different text hashed the same")
+ }
+ if Hash(" same \n") != Hash("same") {
+ t.Fatal("surrounding whitespace changed the hash")
+ }
+}
diff --git a/internal/crawl/extract.go b/internal/crawl/extract.go
new file mode 100644
index 0000000..5828722
--- /dev/null
+++ b/internal/crawl/extract.go
@@ -0,0 +1,106 @@
+package crawl
+
+import (
+ "html"
+ "regexp"
+ "strings"
+)
+
+// HTML → text, with a regexp and no tokenizer.
+//
+// golang.org/x/net/html is not vendored and the network is not assumed, so this
+// is stdlib. That is less of a compromise than it sounds: the unit of context
+// here is a few hundred words for a 4096-token model to read, exactly like the
+// Kiwix snippet, so what matters is dropping script/style/nav noise and keeping
+// paragraph boundaries. A DOM would buy correctness on malformed markup that is
+// then thrown away by truncation anyway.
+//
+// What this deliberately does NOT do: run JavaScript, follow links, or extract
+// structured fields with CSS selectors or an LLM prompt. The plan's step 2 asked
+// for the last of those; see docs/plans/14-web-crawler.md for why it was left
+// out for now.
+
+var (
+ // RE2 has no backreferences, so each tag pair is spelled out rather than
+ // captured and matched against itself.
+ dropRE = regexp.MustCompile(pairsRE("script", "style", "noscript", "svg", "head", "nav", "footer", "form"))
+ titleRE = regexp.MustCompile(`(?is)]*>(.*?)`)
+ h1RE = regexp.MustCompile(`(?is)
]*>(.*?)
`)
+ // Block-level tags become newlines so paragraphs survive as paragraphs.
+ blockRE = regexp.MustCompile(`(?is)?(p|div|br|li|tr|h[1-6]|section|article|blockquote|pre)\b[^>]*>`)
+ tagRE = regexp.MustCompile(`(?s)<[^>]*>`)
+ commentRE = regexp.MustCompile(`(?s)`)
+ spaceRE = regexp.MustCompile(`[ \t\f\v]+`)
+ blankRE = regexp.MustCompile(`\n{2,}`)
+)
+
+// pairsRE builds `(?is)…|…` for the given tags.
+func pairsRE(tags ...string) string {
+ parts := make([]string, 0, len(tags))
+ for _, t := range tags {
+ parts = append(parts, `<`+t+`\b[^>]*>.*?`+t+`>`)
+ }
+ return `(?is)` + strings.Join(parts, "|")
+}
+
+// Page is an extracted page.
+type Page struct {
+ URL string
+ Title string
+ Text string // plain text, paragraphs separated by single newlines
+}
+
+// Extract turns a fetched HTML document into a Page. maxRunes caps the text (0 ⇒
+// DefaultMaxRunes); the cap is on runes, not bytes, because a Russian page cut
+// at a byte boundary ends in half a letter.
+func Extract(url string, body []byte, maxRunes int) Page {
+ if maxRunes <= 0 {
+ maxRunes = DefaultMaxRunes
+ }
+ s := string(body)
+ s = commentRE.ReplaceAllString(s, " ")
+
+ title := firstGroup(titleRE, s)
+ if title == "" {
+ title = firstGroup(h1RE, s)
+ }
+
+ s = dropRE.ReplaceAllString(s, "\n")
+ s = blockRE.ReplaceAllString(s, "\n")
+ s = tagRE.ReplaceAllString(s, " ")
+ s = html.UnescapeString(s)
+ s = spaceRE.ReplaceAllString(s, " ")
+
+ var lines []string
+ for _, l := range strings.Split(s, "\n") {
+ if l = strings.TrimSpace(l); l != "" {
+ lines = append(lines, l)
+ }
+ }
+ text := blankRE.ReplaceAllString(strings.Join(lines, "\n"), "\n")
+
+ return Page{URL: url, Title: title, Text: TrimRunes(text, maxRunes)}
+}
+
+// DefaultMaxRunes — how much of a page is kept. ~4000 runes is a long answer's
+// worth of context and still leaves room in a 4096-token window for the prompt
+// and the reply.
+const DefaultMaxRunes = 4000
+
+func firstGroup(re *regexp.Regexp, s string) string {
+ m := re.FindStringSubmatch(s)
+ if len(m) < 2 {
+ return ""
+ }
+ t := tagRE.ReplaceAllString(m[1], " ")
+ return strings.TrimSpace(strings.Join(strings.Fields(html.UnescapeString(t)), " "))
+}
+
+// TrimRunes cuts s to at most max runes, on a rune boundary.
+func TrimRunes(s string, max int) string {
+ r := []rune(s)
+ if len(r) <= max {
+ return s
+ }
+ return strings.TrimSpace(string(r[:max])) + "…"
+}
diff --git a/internal/crawl/robots.go b/internal/crawl/robots.go
new file mode 100644
index 0000000..3774aa6
--- /dev/null
+++ b/internal/crawl/robots.go
@@ -0,0 +1,211 @@
+package crawl
+
+import (
+ "regexp"
+ "strings"
+ "sync"
+ "time"
+)
+
+// robots.txt, parsed the small way: no wildcards beyond the two the standard
+// actually defines (`*` inside a path and `$` at the end), no sitemaps, no
+// crawl-delay-per-agent gymnastics. A personal assistant reading a handful of
+// pages does not need a spec-complete implementation; it needs to not be rude,
+// and to be auditable in one sitting.
+//
+// Two rules worth stating because they are choices, not accidents:
+//
+// - a missing or unreadable robots.txt means ALLOW. That is what the standard
+// says (404 ⇒ unrestricted), and the alternative would make a site that
+// simply has no robots.txt unreadable;
+// - an explicit Disallow means REFUSE, and Maven does not offer an override.
+// There is no "but he asked me to" flag: the page is not read.
+
+// Rules is a parsed robots.txt for one user-agent.
+type Rules struct {
+ allow []string
+ disallow []string
+ // Delay is Crawl-delay in seconds when the group named one, 0 otherwise.
+ // The fetcher's own per-host rate limit is the floor; this can only make
+ // Maven slower, never faster.
+ Delay time.Duration
+}
+
+// ParseRobots reads robots.txt and returns the rules that apply to agent.
+//
+// Group selection follows the standard: the most specific matching group wins,
+// which here means an exact user-agent match beats `*`. Lines that are neither
+// are ignored rather than guessed at.
+func ParseRobots(body string, agent string) Rules {
+ agent = strings.ToLower(agent)
+
+ type group struct {
+ agents []string
+ allow []string
+ disallow []string
+ delay time.Duration
+ }
+ var groups []group
+ var cur *group
+ // startNew tracks whether the next User-agent line opens a new group or
+ // joins the current one: consecutive User-agent lines share their rules.
+ startNew := true
+
+ for _, raw := range strings.Split(body, "\n") {
+ line := raw
+ if i := strings.IndexByte(line, '#'); i >= 0 {
+ line = line[:i]
+ }
+ line = strings.TrimSpace(line)
+ if line == "" {
+ continue
+ }
+ key, val, ok := strings.Cut(line, ":")
+ if !ok {
+ continue
+ }
+ key = strings.ToLower(strings.TrimSpace(key))
+ val = strings.TrimSpace(val)
+
+ switch key {
+ case "user-agent":
+ if startNew || cur == nil {
+ groups = append(groups, group{})
+ cur = &groups[len(groups)-1]
+ startNew = false
+ }
+ cur.agents = append(cur.agents, strings.ToLower(val))
+ case "disallow":
+ if cur == nil {
+ continue
+ }
+ startNew = true
+ // "Disallow:" with an empty value allows everything, and is not a
+ // path rule at all.
+ if val != "" {
+ cur.disallow = append(cur.disallow, val)
+ }
+ case "allow":
+ if cur == nil {
+ continue
+ }
+ startNew = true
+ if val != "" {
+ cur.allow = append(cur.allow, val)
+ }
+ case "crawl-delay":
+ if cur == nil {
+ continue
+ }
+ startNew = true
+ if d, err := time.ParseDuration(val + "s"); err == nil && d > 0 {
+ cur.delay = d
+ }
+ }
+ }
+
+ var star, exact *group
+ for i := range groups {
+ for _, a := range groups[i].agents {
+ if a == "*" && star == nil {
+ star = &groups[i]
+ }
+ // A robots.txt names "maven", we send "Maven/1.0 (…)": match on
+ // prefix, which is how every crawler reads this field.
+ if a != "*" && a != "" && strings.HasPrefix(agent, a) {
+ exact = &groups[i]
+ }
+ }
+ }
+ g := exact
+ if g == nil {
+ g = star
+ }
+ if g == nil {
+ return Rules{}
+ }
+ return Rules{allow: g.allow, disallow: g.disallow, Delay: g.delay}
+}
+
+// Allowed reports whether path may be fetched. Longest matching rule wins, and
+// Allow beats Disallow at equal length — the standard's tie-break, and the one
+// that makes "Disallow: /" plus "Allow: /public" mean what it looks like.
+func (r Rules) Allowed(path string) bool {
+ if path == "" {
+ path = "/"
+ }
+ best, allowed := -1, true
+ for _, p := range r.disallow {
+ if n, ok := matchPath(p, path); ok && n > best {
+ best, allowed = n, false
+ }
+ }
+ for _, p := range r.allow {
+ if n, ok := matchPath(p, path); ok && n >= best {
+ best, allowed = n, true
+ }
+ }
+ return allowed
+}
+
+// matchPath applies a robots path pattern and returns the pattern's length as
+// the specificity score. `*` matches any run of characters, `$` anchors the end.
+// A pattern is a PREFIX match otherwise, which is what "Disallow: /admin" means.
+func matchPath(pattern, path string) (int, bool) {
+ score := len(pattern)
+ re, err := robotsRegexp(pattern)
+ if err != nil {
+ return 0, false
+ }
+ return score, re.MatchString(path)
+}
+
+// robotsRegexp turns a robots path pattern into an anchored-at-the-start
+// regexp. Everything but `*` and a trailing `$` is a literal, so the pattern is
+// quoted first and the two metacharacters are put back afterwards.
+func robotsRegexp(pattern string) (*regexp.Regexp, error) {
+ end := ""
+ if strings.HasSuffix(pattern, "$") {
+ pattern = strings.TrimSuffix(pattern, "$")
+ end = "$"
+ }
+ parts := strings.Split(pattern, "*")
+ for i, p := range parts {
+ parts[i] = regexp.QuoteMeta(p)
+ }
+ return regexp.Compile("^" + strings.Join(parts, ".*") + end)
+}
+
+// robotsCache holds parsed rules per host so a crawl of ten pages on one site
+// reads robots.txt once. TTL because a site may change its mind, and a daemon
+// that runs for weeks would otherwise never notice.
+type robotsCache struct {
+ ttl time.Duration
+ mu sync.Mutex
+ m map[string]robotsEntry
+}
+
+type robotsEntry struct {
+ rules Rules
+ at time.Time
+}
+
+func newRobotsCache(ttl time.Duration) *robotsCache {
+ return &robotsCache{ttl: ttl, m: map[string]robotsEntry{}}
+}
+
+func (c *robotsCache) get(host string, now time.Time) (Rules, bool) {
+ c.mu.Lock()
+ defer c.mu.Unlock()
+ e, ok := c.m[host]
+ if !ok || now.Sub(e.at) > c.ttl {
+ return Rules{}, false
+ }
+ return e.rules, true
+}
+
+func (c *robotsCache) put(host string, r Rules, now time.Time) {
+ c.mu.Lock()
+ defer c.mu.Unlock()
+ c.m[host] = robotsEntry{rules: r, at: now}
+}
diff --git a/internal/crawl/robots_test.go b/internal/crawl/robots_test.go
new file mode 100644
index 0000000..93e6bb9
--- /dev/null
+++ b/internal/crawl/robots_test.go
@@ -0,0 +1,84 @@
+package crawl
+
+import (
+ "testing"
+ "time"
+)
+
+const robotsBody = `# a comment
+User-agent: *
+Disallow: /private
+Disallow: /tmp/
+Crawl-delay: 5
+
+User-agent: Maven
+Disallow: /
+Allow: /public
+`
+
+func TestParseRobotsPicksTheMostSpecificGroup(t *testing.T) {
+ // The Maven group applies to us even though we send a longer UA string.
+ r := ParseRobots(robotsBody, "Maven/1.0 (self-hosted personal assistant)")
+ if r.Allowed("/anything") {
+ t.Error("Disallow: / in our own group was ignored")
+ }
+ if !r.Allowed("/public/page") {
+ t.Error("Allow: /public must beat the shorter Disallow: /")
+ }
+
+ // A different agent falls into the * group.
+ star := ParseRobots(robotsBody, "SomeoneElse/2")
+ if !star.Allowed("/anything") {
+ t.Error("the * group disallows nothing but /private and /tmp/")
+ }
+ if star.Allowed("/private/x") || star.Allowed("/tmp/") {
+ t.Error("the * group's disallows were not applied")
+ }
+ if star.Delay != 5*time.Second {
+ t.Errorf("crawl-delay = %v, want 5s", star.Delay)
+ }
+}
+
+func TestParseRobotsEmptyMeansAllowAll(t *testing.T) {
+ for _, body := range []string{"", "# nothing here\n", "User-agent: *\nDisallow:\n"} {
+ if !ParseRobots(body, "Maven").Allowed("/whatever") {
+ t.Errorf("body %q must allow everything", body)
+ }
+ }
+}
+
+func TestRobotsWildcards(t *testing.T) {
+ r := ParseRobots("User-agent: *\nDisallow: /*.pdf$\nDisallow: /a/*/secret\n", "Maven")
+ if r.Allowed("/docs/manual.pdf") {
+ t.Error("*.pdf$ did not match")
+ }
+ if !r.Allowed("/docs/manual.pdf.html") {
+ t.Error("$ must anchor at the end")
+ }
+ if r.Allowed("/a/b/secret") {
+ t.Error("/a/*/secret did not match")
+ }
+ if !r.Allowed("/a/b/public") {
+ t.Error("unrelated path was refused")
+ }
+}
+
+// Consecutive User-agent lines share one group, which is common in the wild.
+func TestRobotsSharedGroup(t *testing.T) {
+ r := ParseRobots("User-agent: Googlebot\nUser-agent: Maven\nDisallow: /x\n", "Maven/1.0")
+ if r.Allowed("/x/y") {
+ t.Fatal("a shared group's rules were not applied to the second agent")
+ }
+}
+
+func TestRobotsCacheTTL(t *testing.T) {
+ c := newRobotsCache(time.Minute)
+ now := time.Now()
+ c.put("example.com", ParseRobots("User-agent: *\nDisallow: /\n", "Maven"), now)
+ if _, ok := c.get("example.com", now.Add(30*time.Second)); !ok {
+ t.Error("a fresh entry must be served from cache")
+ }
+ if _, ok := c.get("example.com", now.Add(2*time.Minute)); ok {
+ t.Error("an expired entry must be re-read")
+ }
+}
diff --git a/internal/crawl/watch.go b/internal/crawl/watch.go
new file mode 100644
index 0000000..704cef7
--- /dev/null
+++ b/internal/crawl/watch.go
@@ -0,0 +1,177 @@
+package crawl
+
+import (
+ "context"
+ "fmt"
+ "log"
+ "strings"
+ "time"
+)
+
+// Scheduled crawls: a page is re-read on an interval, and when its TEXT changed
+// the new text is written as a note. Nothing is dispatched — same rule as the
+// feed poller (Vikunja #258). A page that announced its own change would be a
+// nag, and "the docs page changed" is not worth interrupting anyone for.
+//
+// Dedup is by content hash, so a page that re-renders identically writes nothing
+// and a rotating ad slot does not count as news.
+
+// WatchConfig — one page to keep an eye on.
+type WatchConfig struct {
+ Name string // note source is "crawl:"
+ URL string // http(s), guarded by the fetcher
+ Interval time.Duration // 0 ⇒ Watcher's default
+}
+
+// Notes is core's note-writing half (same shape as ipc.CoreAPI's method).
+type Notes interface {
+ WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error)
+}
+
+// Hashes remembers the last text hash per watch, durably, so a restart does not
+// re-note an unchanged page. The daemon backs this with config facts
+// ("crawl:hash:").
+type Hashes interface {
+ LastHash(ctx context.Context, name string) (string, error)
+ SetHash(ctx context.Context, name, hash string) error
+}
+
+// Embedder embeds a note on its way into the store. nil ⇒ no vector.
+type Embedder interface {
+ Embed(ctx context.Context, text string) ([]float32, error)
+}
+
+// DefaultWatchInterval — pages change slowly, and every check is a request in
+// someone's log.
+const DefaultWatchInterval = 6 * time.Hour
+
+// Watcher re-reads watched pages on their interval.
+type Watcher struct {
+ c *Crawler
+ watches []WatchConfig
+ notes Notes
+ hashes Hashes
+ embed Embedder
+ interval time.Duration
+ nextDue map[string]time.Time
+}
+
+// NewWatcher wires the scheduled half, or returns nil when there is nothing to
+// watch. Callers check for nil: no watches, no goroutine, no request.
+func NewWatcher(c *Crawler, watches []WatchConfig, notes Notes, hashes Hashes, embed Embedder, defaultInterval time.Duration) *Watcher {
+ if c == nil || notes == nil {
+ return nil
+ }
+ var valid []WatchConfig
+ for _, w := range watches {
+ if strings.TrimSpace(w.Name) == "" || strings.TrimSpace(w.URL) == "" {
+ log.Printf("crawl: skipping a watch with no name or no url")
+ continue
+ }
+ valid = append(valid, w)
+ }
+ if len(valid) == 0 {
+ return nil
+ }
+ if defaultInterval <= 0 {
+ defaultInterval = DefaultWatchInterval
+ }
+ return &Watcher{
+ c: c, watches: valid, notes: notes, hashes: hashes, embed: embed,
+ interval: defaultInterval, nextDue: map[string]time.Time{},
+ }
+}
+
+// Watches returns the configured watches.
+func (w *Watcher) Watches() []WatchConfig { return w.watches }
+
+// CheckDue re-reads every watch whose interval elapsed and returns how many
+// notes were written. Errors are logged per watch, never returned: one dead page
+// must not stop the others.
+func (w *Watcher) CheckDue(ctx context.Context, now time.Time) int {
+ written := 0
+ for _, watch := range w.watches {
+ if due, ok := w.nextDue[watch.Name]; ok && now.Before(due) {
+ continue
+ }
+ interval := watch.Interval
+ if interval <= 0 {
+ interval = w.interval
+ }
+ w.nextDue[watch.Name] = now.Add(interval)
+ changed, err := w.Check(ctx, watch, now)
+ if err != nil {
+ log.Printf("crawl: watch %s: %v", watch.Name, err)
+ continue
+ }
+ if changed {
+ log.Printf("crawl: watch %s: page changed, noted", watch.Name)
+ written++
+ }
+ }
+ return written
+}
+
+// Check re-reads one watch now and reports whether it wrote a note.
+func (w *Watcher) Check(ctx context.Context, watch WatchConfig, now time.Time) (bool, error) {
+ page, err := w.c.Page(ctx, watch.URL)
+ if err != nil {
+ return false, err
+ }
+ // Title included: a page whose headline changed has changed.
+ h := Hash(page.Title + "\n" + page.Text)
+ if w.hashes != nil {
+ prev, err := w.hashes.LastHash(ctx, watch.Name)
+ if err != nil {
+ log.Printf("crawl: watch %s: read hash: %v", watch.Name, err)
+ }
+ if prev == h {
+ return false, nil
+ }
+ }
+ text := NoteText(watch, page)
+ var vec []float32
+ if w.embed != nil {
+ v, err := w.embed.Embed(ctx, text)
+ if err != nil {
+ log.Printf("crawl: watch %s: embed: %v", watch.Name, err)
+ } else {
+ vec = v
+ }
+ }
+ if _, err := w.notes.WriteNote(ctx, now, text, vec, SourceFor(watch.Name)); err != nil {
+ return false, fmt.Errorf("write note: %w", err)
+ }
+ if w.hashes != nil {
+ if err := w.hashes.SetHash(ctx, watch.Name, h); err != nil {
+ log.Printf("crawl: watch %s: save hash: %v", watch.Name, err)
+ }
+ }
+ return true, nil
+}
+
+// SourceFor is the note source for a watch, and SourcePrefix is what the answer
+// path matches to recognise one.
+func SourceFor(name string) string { return SourcePrefix + name }
+
+// SourcePrefix — provenance for anything read off the network on a schedule.
+const SourcePrefix = "crawl:"
+
+// noteRunes — how much of a watched page goes into a note. Shorter than what the
+// on-demand path reads: a note is a record of a change, not an archive.
+const noteRunes = 800
+
+// NoteText renders a watched page as a note body.
+func NoteText(watch WatchConfig, page Page) string {
+ var b strings.Builder
+ if page.Title != "" {
+ b.WriteString(page.Title)
+ } else {
+ b.WriteString(watch.Name)
+ }
+ b.WriteString("\n")
+ b.WriteString(TrimRunes(page.Text, noteRunes))
+ b.WriteString("\n")
+ b.WriteString(watch.URL)
+ return b.String()
+}
diff --git a/internal/crawl/watch_test.go b/internal/crawl/watch_test.go
new file mode 100644
index 0000000..1f266fd
--- /dev/null
+++ b/internal/crawl/watch_test.go
@@ -0,0 +1,117 @@
+package crawl
+
+import (
+ "context"
+ "strings"
+ "testing"
+ "time"
+)
+
+type note struct {
+ text string
+ source string
+}
+
+type fakeNotes struct{ notes []note }
+
+func (n *fakeNotes) WriteNote(_ context.Context, _ time.Time, text string, _ []float32, source string) (int64, error) {
+ n.notes = append(n.notes, note{text, source})
+ return int64(len(n.notes)), nil
+}
+
+type fakeHashes struct{ m map[string]string }
+
+func newHashes() *fakeHashes { return &fakeHashes{m: map[string]string{}} }
+func (f *fakeHashes) LastHash(_ context.Context, name string) (string, error) {
+ return f.m[name], nil
+}
+func (f *fakeHashes) SetHash(_ context.Context, name, h string) error { f.m[name] = h; return nil }
+
+var t0 = time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC)
+
+func TestWatchNotesAChangedPage(t *testing.T) {
+ f := &fakeFetcher{pages: map[string]Response{
+ "https://example.org/docs": {Body: []byte(htmlPage)},
+ }}
+ notes := &fakeNotes{}
+ hashes := newHashes()
+ w := NewWatcher(newTestCrawler(f), []WatchConfig{{Name: "docs", URL: "https://example.org/docs"}},
+ notes, hashes, nil, time.Hour)
+ if w == nil {
+ t.Fatal("NewWatcher returned nil for a configured watch")
+ }
+ if n := w.CheckDue(context.Background(), t0); n != 1 {
+ t.Fatalf("first check wrote %d notes, want 1", n)
+ }
+ if notes.notes[0].source != "crawl:docs" {
+ t.Errorf("source = %q, want crawl:docs", notes.notes[0].source)
+ }
+ if !strings.Contains(notes.notes[0].text, "https://example.org/docs") {
+ t.Errorf("note does not carry the url: %q", notes.notes[0].text)
+ }
+
+ // Unchanged page, interval elapsed: nothing written.
+ if n := w.CheckDue(context.Background(), t0.Add(2*time.Hour)); n != 0 {
+ t.Fatalf("an unchanged page wrote %d notes", n)
+ }
+
+ // Changed page: one note.
+ f.pages["https://example.org/docs"] = Response{Body: []byte(strings.Replace(htmlPage, "синее", "серое", 1))}
+ if n := w.CheckDue(context.Background(), t0.Add(4*time.Hour)); n != 1 {
+ t.Fatalf("a changed page wrote %d notes, want 1", n)
+ }
+}
+
+func TestWatchIntervalIsRespected(t *testing.T) {
+ f := &fakeFetcher{pages: map[string]Response{"https://example.org/d": {Body: []byte(htmlPage)}}}
+ w := NewWatcher(newTestCrawler(f), []WatchConfig{{Name: "d", URL: "https://example.org/d", Interval: time.Hour}},
+ &fakeNotes{}, newHashes(), nil, 0)
+ w.CheckDue(context.Background(), t0)
+ before := len(f.calls)
+ w.CheckDue(context.Background(), t0.Add(time.Minute))
+ if len(f.calls) != before {
+ t.Fatal("the page was re-read inside its interval")
+ }
+}
+
+// The hash is durable so a restart does not re-note an unchanged page.
+func TestWatchHashSurvivesRestart(t *testing.T) {
+ f := &fakeFetcher{pages: map[string]Response{"https://example.org/d": {Body: []byte(htmlPage)}}}
+ hashes := newHashes()
+ watches := []WatchConfig{{Name: "d", URL: "https://example.org/d"}}
+ NewWatcher(newTestCrawler(f), watches, &fakeNotes{}, hashes, nil, time.Hour).CheckDue(context.Background(), t0)
+
+ notes2 := &fakeNotes{}
+ NewWatcher(newTestCrawler(f), watches, notes2, hashes, nil, time.Hour).CheckDue(context.Background(), t0.Add(time.Hour))
+ if len(notes2.notes) != 0 {
+ t.Fatalf("a fresh watcher re-noted an unchanged page: %q", notes2.notes[0].text)
+ }
+}
+
+func TestWatchDeadPageDoesNotStopTheOthers(t *testing.T) {
+ f := &fakeFetcher{pages: map[string]Response{"https://example.org/live": {Body: []byte(htmlPage)}}}
+ notes := &fakeNotes{}
+ w := NewWatcher(newTestCrawler(f), []WatchConfig{
+ {Name: "dead", URL: "https://example.org/gone"},
+ {Name: "live", URL: "https://example.org/live"},
+ }, notes, newHashes(), nil, time.Hour)
+ if n := w.CheckDue(context.Background(), t0); n != 1 {
+ t.Fatalf("wrote %d notes, want 1 (the live page)", n)
+ }
+ if notes.notes[0].source != "crawl:live" {
+ t.Fatalf("source = %q", notes.notes[0].source)
+ }
+}
+
+func TestNoWatchesMeansNoWatcher(t *testing.T) {
+ c := newTestCrawler(&fakeFetcher{})
+ if NewWatcher(c, nil, &fakeNotes{}, nil, nil, 0) != nil {
+ t.Fatal("no watches must mean no watcher")
+ }
+ if NewWatcher(nil, []WatchConfig{{Name: "a", URL: "u"}}, &fakeNotes{}, nil, nil, 0) != nil {
+ t.Fatal("no crawler must mean no watcher")
+ }
+ if NewWatcher(c, []WatchConfig{{Name: "", URL: ""}}, &fakeNotes{}, nil, nil, 0) != nil {
+ t.Fatal("a watch with no name or url is not a configuration")
+ }
+}
diff --git a/internal/router/url.go b/internal/router/url.go
new file mode 100644
index 0000000..8cdea09
--- /dev/null
+++ b/internal/router/url.go
@@ -0,0 +1,39 @@
+package router
+
+import (
+ "regexp"
+ "strings"
+)
+
+// Finding a URL in an utterance (Vikunja #259).
+//
+// This is deliberately strict: a scheme is required. "посмотри на example.org"
+// is not treated as a fetch request, because a bare dotted word is also how
+// people say file names, versions and Russian abbreviations, and the cost of a
+// false positive here is an outbound request nobody asked for.
+//
+// Note where this runs: an utterance from STT. Whisper will mangle a spoken URL,
+// which is fine — the URL that survives is one he pasted into the web chat, and
+// a mangled one simply fails to match.
+var urlRE = regexp.MustCompile(`(?i)\bhttps?://[^\s<>"']+`)
+
+// FirstURL returns the first http(s) URL in text.
+//
+// Trailing punctuation is trimmed: he ends sentences, and "…/page." is not a
+// path component. A closing bracket is only trimmed when it has no opener,
+// because a wikipedia URL legitimately ends in one.
+func FirstURL(text string) (string, bool) {
+ m := urlRE.FindString(text)
+ if m == "" {
+ return "", false
+ }
+ m = strings.TrimRight(m, ".,;:!?…")
+ if strings.HasSuffix(m, ")") && strings.Count(m, "(") == 0 {
+ m = strings.TrimSuffix(m, ")")
+ }
+ // A scheme with nothing after it is not a URL.
+ if rest := strings.SplitN(m, "//", 2); len(rest) < 2 || rest[1] == "" {
+ return "", false
+ }
+ return m, true
+}
diff --git a/internal/router/url_test.go b/internal/router/url_test.go
new file mode 100644
index 0000000..6710773
--- /dev/null
+++ b/internal/router/url_test.go
@@ -0,0 +1,34 @@
+package router
+
+import "testing"
+
+func TestFirstURL(t *testing.T) {
+ cases := []struct {
+ text string
+ want string
+ }{
+ {"посмотри https://example.org/page — что там?", "https://example.org/page"},
+ {"почитай http://example.org/a/b?x=1 и скажи", "http://example.org/a/b?x=1"},
+ {"вот ссылка: https://example.org/page.", "https://example.org/page"},
+ {"https://ru.wikipedia.org/wiki/Небо_(значения)", "https://ru.wikipedia.org/wiki/Небо_(значения)"},
+ // No scheme ⇒ no fetch. A bare dotted word is not an instruction to
+ // reach out to the network.
+ {"посмотри на example.org", ""},
+ {"открой файл config.json", ""},
+ {"что нового?", ""},
+ {"https://", ""},
+ {"", ""},
+ }
+ for _, c := range cases {
+ got, ok := FirstURL(c.text)
+ if c.want == "" {
+ if ok {
+ t.Errorf("FirstURL(%q) = %q, want no match", c.text, got)
+ }
+ continue
+ }
+ if !ok || got != c.want {
+ t.Errorf("FirstURL(%q) = %q, %v; want %q", c.text, got, ok, c.want)
+ }
+ }
+}