diff --git a/cmd/mavend/actions_query.go b/cmd/mavend/actions_query.go index 4cfa1b7..56f9ede 100644 --- a/cmd/mavend/actions_query.go +++ b/cmd/mavend/actions_query.go @@ -8,6 +8,7 @@ import ( "strings" "time" + "github.com/kami/maven/internal/crawl" "github.com/kami/maven/internal/ipc" "github.com/kami/maven/internal/memory" "github.com/kami/maven/internal/morning" @@ -78,6 +79,13 @@ var querySources = []querySource{ {"embed", (*reactiveHandler).queryEmbed}, {"memory", (*reactiveHandler).queryMemory}, {"notes", (*reactiveHandler).queryNotes}, + // LAST before the model answers from memory, and that position is the whole + // design (Vikunja #259): local sources first. The model, his own notes and + // facts, and — once internal/kiwix is wired into this chain — the offline + // ZIMs all get their turn before anything touches the network. This source + // only claims a turn where he named a URL out loud, so it never competes + // with a local answer. + {"web", (*reactiveHandler).queryWeb}, {"general-knowledge", (*reactiveHandler).queryGeneral}, } @@ -371,6 +379,57 @@ func (h *reactiveHandler) queryNotes(ctx context.Context, t *queryTurn) (string, return reply, true } +// webPageContextRunes — how much of a fetched page is handed to the phraser. +// Less than the crawler keeps: the rest of the 4096-token window belongs to the +// prompt, the persona block and the reply. +const webPageContextRunes = 1500 + +// queryWeb — "посмотри https://example.org/x — что там?" (Vikunja #259). +// +// It claims a turn ONLY when he named a URL, which is what keeps a fallback from +// becoming a habit: no URL, no fetch, and the model answers from what is local. +// What leaves the box is the URL and nothing else — no note, no fact, no history +// travels with it. +func (h *reactiveHandler) queryWeb(ctx context.Context, t *queryTurn) (string, bool) { + link, ok := router.FirstURL(t.dec.Utterance) + if !ok { + return "", false + } + if h.crawler == nil { + // Claim rather than fall through: he asked about a specific page, and + // letting the model answer from the URL's spelling alone is how a small + // model invents a page's contents. + return "я не читаю страницы — это не настроено.", true + } + ctxFetch, cancel := context.WithTimeout(ctx, 30*time.Second) + defer cancel() + page, err := h.crawler.Page(ctxFetch, link) + if err != nil { + if errors.Is(err, crawl.ErrRobots) { + return "эта страница закрыта для чтения — robots.txt не разрешает.", true + } + log.Printf("voice: web: %v", err) + return "не получилось прочитать страницу.", true + } + if page.Text == "" { + return "страница открылась, но читать там нечего.", true + } + // The page is handed to the phraser the same way a note is: as context for + // the question he actually asked. She answers the question, she does not + // recite the page. + snippet := page.Title + "\n" + crawl.TrimRunes(page.Text, webPageContextRunes) + reply, perr := h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet}) + if perr != nil { + log.Printf("voice: web: phrase: %v", perr) + } + if reply == "" { + // No phraser (or it failed): read back the top of the page rather than + // pretend the fetch did not happen. + return "вот что на странице: " + crawl.TrimRunes(page.Text, 300), true + } + return reply, true +} + // queryGeneral — general knowledge from the phraser, the last source before // giving up. It always claims: either the model answers or Maven says she // doesn't know. diff --git a/cmd/mavend/crawls.go b/cmd/mavend/crawls.go new file mode 100644 index 0000000..e20a01a --- /dev/null +++ b/cmd/mavend/crawls.go @@ -0,0 +1,184 @@ +// mavend/crawls.go — the driver for reading web pages (Vikunja #259, +// docs/plans/14-web-crawler.md). The crawler is pure and lives in +// internal/crawl; this is the impure half: the guarded fetcher, a ticker for the +// scheduled watches, and the fact-backed dedup hashes. +// +// Two paths, one config block, both off unless configured: +// +// - ON DEMAND — he names a URL out loud and she reads it. That is the +// `queryWeb` source in actions_query.go, LAST in the chain: after his +// memory, after the notes, and (once Kiwix is wired into the chain) after +// the local ZIMs. A local read costs nothing and leaks nothing; a fetch puts +// a URL in someone's log, so it goes last. +// - SCHEDULED — a watched page is re-read on its interval, and a page whose +// text changed is written as a note. It does NOT announce itself. Same rule +// as the feed poller: notes, never nudges. +// +// Only the URL goes out. Nothing here reads a note, a fact, the persona block or +// the history, and internal/crawl has no access to the store at all. +package main + +import ( + "context" + "log" + "net/url" + "time" + + "github.com/kami/maven/internal/config" + "github.com/kami/maven/internal/crawl" + "github.com/kami/maven/internal/ipc" + "github.com/kami/maven/internal/router" + "github.com/kami/maven/internal/webfetch" +) + +// newCrawler builds the crawler from the `crawl` block, or returns nil when +// there is none. Every caller checks for nil, and nil means no page is ever +// fetched. +func newCrawler(cfg *config.Config) *crawl.Crawler { + if cfg.Crawl == nil { + return nil + } + cc := cfg.Crawl + + hosts := append([]string(nil), cc.AllowHosts...) + // A watched page's own host is always reachable; otherwise an allowlist and + // a watch list would have to be kept in sync by hand. + for _, w := range cc.Watches { + if u, err := url.Parse(w.URL); err == nil && u.Hostname() != "" { + hosts = append(hosts, u.Hostname()) + } + } + // An allowlist plus on-demand is a contradiction worth logging rather than + // silently resolving: he asked for arbitrary pages AND for a fixed list. + // The allowlist wins, because it is the narrower instruction. + if len(hosts) > 0 && cc.OnDemand && len(cc.AllowHosts) > 0 { + log.Printf("crawl: allow_hosts is set, so on-demand reading is limited to those hosts") + } + ua := cc.UserAgent + if ua == "" { + ua = webfetch.DefaultUserAgent + } + fetcher := webfetch.New(webfetch.Config{ + AllowHosts: hosts, + DenyHosts: cc.DenyHosts, + Timeout: time.Duration(cc.Timeout), + MaxBytes: cc.MaxBytes, + UserAgent: ua, + }) + // The user-agent handed to the crawler is the one the fetcher sends: obeying + // robots rules written for a different name would be a lie. + return crawl.New(&crawlFetcher{f: fetcher}, crawl.Config{ + UserAgent: ua, + MaxRunes: cc.MaxRunes, + }) +} + +// onDemandCrawler returns a crawler for the answer path, or nil when on-demand +// reading is off. The scheduled watches can be on while this is off: reading a +// fixed list of pages on a timer and reading whatever URL is in an utterance are +// different permissions, and the config keeps them separate. +func onDemandCrawler(cfg *config.Config) *crawl.Crawler { + if cfg.Crawl == nil || !cfg.Crawl.OnDemand { + return nil + } + return newCrawler(cfg) +} + +// crawlWorker — ticker + watcher for the scheduled half. +type crawlWorker struct { + watcher *crawl.Watcher + interval time.Duration +} + +// crawlTickInterval — how often the worker asks what is due. Per-watch cadence +// is the watcher's business. +const crawlTickInterval = 15 * time.Minute + +// newCrawlWorker wires the scheduled crawls, or nil when nothing is watched. +func newCrawlWorker(c *crawl.Crawler, api ipc.CoreAPI, emb router.Embedder, cfg *config.Config) *crawlWorker { + if c == nil || cfg.Crawl == nil || len(cfg.Crawl.Watches) == 0 { + return nil + } + watches := make([]crawl.WatchConfig, 0, len(cfg.Crawl.Watches)) + for _, w := range cfg.Crawl.Watches { + watches = append(watches, crawl.WatchConfig{ + Name: w.Name, + URL: w.URL, + Interval: time.Duration(w.Interval), + }) + } + watcher := crawl.NewWatcher(c, watches, api, &factHashes{api: api}, + crawlEmbedder(emb), time.Duration(cfg.Crawl.Interval)) + if watcher == nil { + log.Printf("crawl: configured but nothing watchable — scheduled crawls disabled") + return nil + } + log.Printf("crawl: watching %d page(s), checking what is due every %s", len(watches), crawlTickInterval) + return &crawlWorker{watcher: watcher, interval: crawlTickInterval} +} + +// run checks what is due until ctx is canceled. The first round runs +// immediately; it writes notes only, so an early round startles nobody. +func (w *crawlWorker) run(ctx context.Context) { + w.watcher.CheckDue(ctx, time.Now()) + t := time.NewTicker(w.interval) + defer t.Stop() + for { + select { + case <-ctx.Done(): + return + case now := <-t.C: + w.watcher.CheckDue(ctx, now) + } + } +} + +// crawlFetcher adapts webfetch to crawl.Fetcher, which is the seam that keeps +// net/http out of the crawler package. +type crawlFetcher struct{ f *webfetch.Fetcher } + +func (a *crawlFetcher) Get(ctx context.Context, u string) (*crawl.Response, error) { + resp, err := a.f.Get(ctx, u) + if err != nil { + return nil, err + } + return &crawl.Response{URL: resp.URL, ContentType: resp.ContentType, Body: resp.Body}, nil +} + +// factHashes stores each watch's last content hash as a config fact, so a +// restart does not re-note an unchanged page. Same mechanism the feed reader +// uses for its marks, and inspectable on /dash. +type factHashes struct{ api ipc.CoreAPI } + +func hashKey(name string) string { return "crawl:hash:" + name } + +func (h *factHashes) LastHash(ctx context.Context, name string) (string, error) { + f, err := h.api.LatestFact(ctx, hashKey(name)) + if err != nil { + // No hash yet is not an error: the watcher treats "" as "never read". + return "", nil + } + return f.Value, nil +} + +func (h *factHashes) SetHash(ctx context.Context, name, hash string) error { + _, err := h.api.WriteFact(ctx, ipc.WriteFactReq{ + Ts: time.Now(), + Kind: "config", + Key: hashKey(name), + Value: hash, + Source: "poll:crawl", + Confidence: 1.0, + }) + return err +} + +// crawlEmbedder adapts router.Embedder for the watcher, embedding with +// EmbedPassage (a page is text being searched FOR, and the e5 embedder is +// asymmetric). +func crawlEmbedder(emb router.Embedder) crawl.Embedder { + if emb == nil { + return nil + } + return passageEmbedder{emb} +} diff --git a/cmd/mavend/crawls_test.go b/cmd/mavend/crawls_test.go new file mode 100644 index 0000000..2910d5f --- /dev/null +++ b/cmd/mavend/crawls_test.go @@ -0,0 +1,186 @@ +package main + +import ( + "context" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/kami/maven/internal/config" + "github.com/kami/maven/internal/crawl" + "github.com/kami/maven/internal/ipc" + "github.com/kami/maven/internal/phraser" + "github.com/kami/maven/internal/router" + "github.com/kami/maven/internal/voice" +) + +// The default config reads nothing. This is the whole "off unless configured" +// contract for the crawler, asserted at the wiring level rather than trusted. +func TestCrawlOffByDefault(t *testing.T) { + cfg := &config.Config{} + if c := newCrawler(cfg); c != nil { + t.Error("newCrawler with no crawl block returned a crawler") + } + if c := onDemandCrawler(cfg); c != nil { + t.Error("onDemandCrawler with no crawl block returned a crawler") + } + if w := newCrawlWorker(nil, nil, nil, cfg); w != nil { + t.Error("newCrawlWorker with no crawl block returned a worker") + } + // Watches configured but on_demand off ⇒ the answer path still reads + // nothing: a timer over a fixed list is not permission for arbitrary URLs. + withWatch := &config.Config{Crawl: &config.CrawlConfig{ + Watches: []config.CrawlWatchConfig{{Name: "p", URL: "https://example.org/p"}}, + }} + if c := onDemandCrawler(withWatch); c != nil { + t.Error("onDemandCrawler honoured a watch list as on-demand permission") + } + if c := newCrawler(withWatch); c == nil { + t.Error("newCrawler returned nil for a configured watch") + } +} + +// The wired fetcher must refuse a private address, because the crawler on this +// box sits one hop from the whole homelab. Same guard the webfetch tests cover; +// this asserts the daemon actually wires it. +func TestCrawlerRefusesPrivateAddress(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/html") + w.Write([]byte("secret")) + })) + defer srv.Close() + + c := newCrawler(&config.Config{Crawl: &config.CrawlConfig{OnDemand: true}}) + if c == nil { + t.Fatal("newCrawler returned nil for an on-demand config") + } + if _, err := c.Page(context.Background(), srv.URL); err == nil { + t.Fatalf("reading %s succeeded; a loopback address must be refused", srv.URL) + } +} + +func TestFactHashesRoundTrip(t *testing.T) { + ctx := context.Background() + st := newTestStore(t) + h := &factHashes{api: ipc.NewStoreAPI(st)} + + got, err := h.LastHash(ctx, "page") + if err != nil { + t.Fatalf("LastHash on a fresh store: %v", err) + } + if got != "" { + t.Errorf("LastHash = %q, want empty for a never-read page", got) + } + if err := h.SetHash(ctx, "page", "deadbeef"); err != nil { + t.Fatalf("SetHash: %v", err) + } + got, err = h.LastHash(ctx, "page") + if err != nil { + t.Fatalf("LastHash: %v", err) + } + if got != "deadbeef" { + t.Errorf("LastHash = %q, want deadbeef", got) + } + if key := hashKey("page"); key != "crawl:hash:page" { + t.Errorf("hashKey = %q", key) + } +} + +// stubCrawlFetcher serves one fixed page to every URL, so queryWeb can be +// exercised without a network or an allowlist. +type stubCrawlFetcher struct{ body, ctype string } + +func (s *stubCrawlFetcher) Get(_ context.Context, u string) (*crawl.Response, error) { + ct := s.ctype + if ct == "" { + ct = "text/html" + } + if strings.HasSuffix(u, "/robots.txt") { + return &crawl.Response{URL: u, ContentType: "text/plain", Body: []byte("")}, nil + } + return &crawl.Response{URL: u, ContentType: ct, Body: []byte(s.body)}, nil +} + +func buildWebHandler(c *crawl.Crawler) *reactiveHandler { + return &reactiveHandler{ + replier: voice.NewStubReplier(), + phraser: phraser.NewStub(), + crawler: c, + } +} + +func askWeb(h *reactiveHandler, q string) (string, bool) { + return h.queryWeb(context.Background(), &queryTurn{ + dec: router.Decision{Intent: router.IntentQuery, Utterance: q}, + }) +} + +func TestQueryWebPassesWithoutAURL(t *testing.T) { + h := buildWebHandler(crawl.New(&stubCrawlFetcher{body: "x"}, crawl.Config{})) + if reply, ok := askWeb(h, "почему небо синее?"); ok { + t.Errorf("the web source claimed a question with no URL: %q", reply) + } +} + +// Not configured is said out loud rather than falling through, so a small model +// never invents a page's contents from its URL. +func TestQueryWebSaysWhenNotConfigured(t *testing.T) { + h := buildWebHandler(nil) + reply, ok := askWeb(h, "посмотри https://example.org/page") + if !ok { + t.Fatal("the web source did not claim a question with a URL") + } + if !strings.Contains(reply, "не настроено") { + t.Errorf("reply = %q, want the not-configured answer", reply) + } +} + +func TestQueryWebReadsThePage(t *testing.T) { + h := buildWebHandler(crawl.New(&stubCrawlFetcher{ + body: "Заголовок

текст страницы

", + }, crawl.Config{})) + reply, ok := askWeb(h, "посмотри https://example.org/page — что там?") + if !ok { + t.Fatal("the web source did not claim a question with a URL") + } + if !strings.Contains(reply, "текст страницы") { + t.Errorf("reply = %q, want the page text read back", reply) + } +} + +func TestQueryWebRefusesNonHTML(t *testing.T) { + h := buildWebHandler(crawl.New(&stubCrawlFetcher{ + body: "\x00\x01binary", ctype: "application/octet-stream", + }, crawl.Config{})) + reply, ok := askWeb(h, "почитай https://example.org/blob.bin") + if !ok { + t.Fatal("the web source did not claim a question with a URL") + } + if !strings.Contains(reply, "не получилось") { + t.Errorf("reply = %q, want the read-failed answer", reply) + } +} + +// robots.txt is honoured on the answer path too, and she says so instead of +// reporting a generic failure. +func TestQueryWebObeysRobots(t *testing.T) { + h := buildWebHandler(crawl.New(&robotsDenyFetcher{}, crawl.Config{})) + reply, ok := askWeb(h, "посмотри https://example.org/private") + if !ok { + t.Fatal("the web source did not claim a question with a URL") + } + if !strings.Contains(reply, "robots.txt") { + t.Errorf("reply = %q, want the robots answer", reply) + } +} + +type robotsDenyFetcher struct{} + +func (robotsDenyFetcher) Get(_ context.Context, u string) (*crawl.Response, error) { + if strings.HasSuffix(u, "/robots.txt") { + return &crawl.Response{URL: u, ContentType: "text/plain", + Body: []byte("User-agent: *\nDisallow: /private\n")}, nil + } + return &crawl.Response{URL: u, ContentType: "text/html", Body: []byte("nope")}, nil +} diff --git a/cmd/mavend/main.go b/cmd/mavend/main.go index 4053be3..0c10d80 100644 --- a/cmd/mavend/main.go +++ b/cmd/mavend/main.go @@ -171,6 +171,7 @@ func run(args []string) error { factWorker *factEnrichmentWorker evalWorker *memoryEvalWorker // nil ⇒ memory evaluation off (the default) feedWkr *feedWorker // nil ⇒ no feed is read (the default) + crawlWkr *crawlWorker // nil ⇒ no page is watched (the default) ) if !locked { @@ -267,6 +268,7 @@ func run(args []string) error { factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval)) evalWorker = newMemoryEvalWorker(st, phr, cfg) feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg) + crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg) coreAPI = &daemonAPI{ CoreAPI: ipc.NewStoreAPI(st), @@ -454,6 +456,7 @@ func run(args []string) error { factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval)) evalWorker = newMemoryEvalWorker(st, phr, cfg) feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg) + crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg) // Swap the CoreAPI from the locked placeholder to the real store adapter. newAPI := &daemonAPI{ @@ -506,6 +509,13 @@ func run(args []string) error { }() } + // Start the watched-page crawls (nil unless configured). + if crawlWkr != nil { + go func() { + crawlWkr.run(ctx) + }() + } + dl.unlock() log.Printf("mavend: unlocked via passkey assertion") return nil @@ -558,6 +568,13 @@ func run(args []string) error { feedWkr.run(ctx) }() } + if crawlWkr != nil { + wg.Add(1) + go func() { + defer wg.Done() + crawlWkr.run(ctx) + }() + } } <-ctx.Done() diff --git a/cmd/mavend/voice.go b/cmd/mavend/voice.go index 4ef572c..c9ecbcd 100644 --- a/cmd/mavend/voice.go +++ b/cmd/mavend/voice.go @@ -52,6 +52,7 @@ import ( "time" "github.com/kami/maven/internal/audio" + "github.com/kami/maven/internal/crawl" "github.com/kami/maven/internal/dialogue" "github.com/kami/maven/internal/ipc" "github.com/kami/maven/internal/memory" @@ -82,6 +83,10 @@ type reactiveHandler struct { replier voice.Replier now func() time.Time + // crawler reads a web page he names out loud (queryWeb). nil ⇒ on-demand + // page reading is off, which is the default: no `crawl` block, no fetch. + crawler *crawl.Crawler + // feedsOn — whether any RSS feed is configured (config.Feeds). It changes // only what she SAYS when asked and nothing is there: "ленты не настроены" // instead of "ничего нового", which are different truths. diff --git a/cmd/mavend/voicewire.go b/cmd/mavend/voicewire.go index 21032df..7167511 100644 --- a/cmd/mavend/voicewire.go +++ b/cmd/mavend/voicewire.go @@ -201,17 +201,20 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem // ----- the handler (the reactive path; closes over stt / tts / router / coreAPI / memory) ----- h := &reactiveHandler{ - stt: transcriber, - tts: synthesizer, - router: rtr, - embedder: emb, - api: coreAPI, - tools: exec, - matcher: matcher, - replier: replier, - phraser: phr, - now: time.Now, - feedsOn: cfg.Feeds != nil, + stt: transcriber, + tts: synthesizer, + router: rtr, + embedder: emb, + api: coreAPI, + tools: exec, + matcher: matcher, + replier: replier, + phraser: phr, + now: time.Now, + feedsOn: cfg.Feeds != nil, + // nil unless `crawl.on_demand` is on: reading a page he names is a + // capability, and capabilities are off unless configured. + crawler: onDemandCrawler(cfg), weatherProvider: weatherProvider, weatherLocation: weatherLocation, memStore: memStore, diff --git a/deploy/README.md b/deploy/README.md index 0cb1a24..2f68637 100644 --- a/deploy/README.md +++ b/deploy/README.md @@ -67,6 +67,36 @@ What it does and does not do: - how far each feed was read is stored as a config fact `rss:latest:`, so a restart does not re-note yesterday's headlines. +### Reading a page (`crawl`, also off by default) + +There is no `crawl` block either, so no page is fetched. Two halves, separately +switched: + +```json +"crawl": { + "on_demand": true, + "interval": "6h", + "max_runes": 4000, + "watches": [ + { "name": "changelog", "url": "https://example.org/changelog", "interval": "12h" } + ] +} +``` + +- `on_demand` lets her read a page he names in the utterance: "посмотри + https://example.org/x — что там?". The page becomes context for his question, + and only the URL leaves the box. Without a URL nothing is fetched, so this is + a fallback and not a habit; +- `watches` re-reads a fixed list on its interval and writes a note when the + text changed. Like the feeds, it announces nothing; +- the answer path sits **last** in the query chain, behind his memory, his notes + and (once wired) the local Kiwix ZIMs. A local read costs nothing; +- `robots.txt` is fetched first and obeyed with no override; a `Disallow` is a + refusal she says out loud. `Crawl-delay` is honoured; +- same guarded fetcher as the feeds: allowlist/denylist, no private addresses, + size cap, redirect cap, timeout, one request per host per second; +- dedup state is the config fact `crawl:hash:`. + ## Not yet verified / host-dependent This stack is correct-by-construction but has **not been build-tested here** diff --git a/docs/plans/14-web-crawler.md b/docs/plans/14-web-crawler.md index b5c78ac..a9b22f3 100644 --- a/docs/plans/14-web-crawler.md +++ b/docs/plans/14-web-crawler.md @@ -28,3 +28,41 @@ 7. Add IPC methods `MethodTriggerCrawl(name)`, `MethodListCrawls`, `MethodGetCrawlResult(name)` 8. Add `crawls` block to `config.Config` and `deploy/mavend.json` 9. Test with a static HTML page — verify extraction matches expected values, verify scheduling fires correctly + +## Shipped 2026-08-01 (#259) + +Built as `internal/crawl` (pure: robots, extraction, watcher) plus +`cmd/mavend/crawls.go` (fetcher, ticker, dedup facts), on top of the guarded +`internal/webfetch` door added with the feed reader (#258). Off unless +configured, in two separately-switched halves: `crawl.on_demand` for a URL he +names, `crawl.watches` for a scheduled re-read. + +**Limits are code, not documentation** (`internal/webfetch`, tested one test per +limit): host allowlist/denylist, no private addresses (loopback, RFC1918 — +hence the LAN and the `10.42.0.0/24` wg range —, link-local incl. cloud +metadata, CGNAT, v6 ULA) enforced in the dialer's `Control` hook so DNS +rebinding and every redirect hop are covered, response size cap, redirect cap, +timeout, one request per host per second. `robots.txt` is fetched first, cached +per host, and a `Disallow` is refused with no override. + +Deliberate deviations from the plan above: + +- **No CSS selectors and no LLM structured extraction** (steps 2). The output is + plaintext handed to the phraser as context for the question he asked. A 1.7B + extracting a JSON price table from 4000 runes is a worse bet than reading, and + `goquery` is not vendored. +- **No `crawl` act verb and no new IPC methods** (steps 5, 7). Reading a page is + a query source (`queryWeb` in `actions_query.go`, last in the chain, behind + Kiwix once that is wired), not an action he commands. Nothing needs a new wire + method to work. +- **Notes, not facts.** A page's text is not a fact about him. Only the dedup + hash is a fact (`crawl:hash:`, kind `config`, source `poll:crawl`). +- **Nothing is dispatched.** A changed page writes a note; it does not nudge. + Not a nag. +- **No `/tools` crawl history page.** The notes and the hash facts are already + visible on `/dash`. + +**No new dependency.** The vendored tree has no `x/net/html`, no `goquery` and +no `temoto/robotstxt`, so robots parsing and HTML-to-text are stdlib +(`regexp`, `html`) — RE2 has no backreferences, hence the `pairsRE` builder in +`extract.go`. diff --git a/internal/config/config.go b/internal/config/config.go index 47d476c..c3fe4b6 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -161,6 +161,10 @@ type Config struct { // the weather and telegram. See FeedsConfig. Feeds *FeedsConfig `json:"feeds,omitempty"` + // Crawl — reading a web page (Vikunja #259). nil / absent ⇒ Maven never + // fetches a page: not on request, not on a schedule. See CrawlConfig. + Crawl *CrawlConfig `json:"crawl,omitempty"` + // Praxis — the ecosystem attention-state service. When configured, maven // calls the Praxis HTTP tools API for attention listing and item lifecycle. // Maven never touches Praxis's database directly (ecosystem invariant: no @@ -454,6 +458,62 @@ type FeedSourceConfig struct { Exclude []string `json:"exclude,omitempty"` // drop items containing any of these } +// CrawlConfig — the web crawler (Vikunja #259, docs/plans/14-web-crawler.md). +// +// Absent ⇒ off, and off means no page is ever fetched. Present with neither +// `on_demand` nor a `watches` entry is also off: there would be nothing to do. +// +// The crawler is the LAST place an answer is looked for, behind the model, his +// own memory and the local Kiwix ZIMs. That ordering lives in the query-source +// chain (cmd/mavend/actions_query.go), not here, but it is the reason this block +// is small: it is a fallback, not a search engine. +// +// Only the URL leaves the box. His notes, facts, persona block and history are +// never part of a request — the crawler package cannot even read the store. +type CrawlConfig struct { + // OnDemand — may he ask her to read a page he names out loud + // ("посмотри https://… — что там пишут?"). false ⇒ the on-demand answer + // source stays off and only the watches below run. + OnDemand bool `json:"on_demand,omitempty"` + + // Watches — pages re-read on a schedule. A page whose text changed is + // written as a note (source "crawl:"); nothing is announced. + Watches []CrawlWatchConfig `json:"watches,omitempty"` + + // Interval — default watch cadence. 0 ⇒ crawl.DefaultWatchInterval (6h). + Interval Duration `json:"interval,omitempty"` + + // AllowHosts — when set, the ONLY hosts the crawler may reach (subdomains + // included). Watched pages' own hosts are added automatically. Setting this + // is how "she may read the arch wiki and nothing else" is expressed. + AllowHosts []string `json:"allow_hosts,omitempty"` + + // DenyHosts — never reachable, checked first. Private addresses do not need + // to be listed: they are refused unconditionally (see internal/webfetch). + DenyHosts []string `json:"deny_hosts,omitempty"` + + // UserAgent — sent on every request AND matched against robots.txt groups. + // Empty ⇒ webfetch.DefaultUserAgent. + UserAgent string `json:"user_agent,omitempty"` + + // Timeout — per-request budget. 0 ⇒ webfetch.DefaultTimeout. + Timeout Duration `json:"timeout,omitempty"` + + // MaxBytes — response size cap. 0 ⇒ webfetch.DefaultMaxBytes (2 MiB). + MaxBytes int64 `json:"max_bytes,omitempty"` + + // MaxRunes — how much extracted text is kept. 0 ⇒ crawl.DefaultMaxRunes + // (4000), which is what fits a 4096-token context alongside a prompt. + MaxRunes int `json:"max_runes,omitempty"` +} + +// CrawlWatchConfig — one page kept an eye on. +type CrawlWatchConfig struct { + Name string `json:"name"` // note source is "crawl:" + URL string `json:"url"` + Interval Duration `json:"interval,omitempty"` // 0 ⇒ CrawlConfig.Interval +} + // MemoryEvalConfig — the background memory-evaluation loop (Vikunja #248). // Absent ⇒ off, like every other capability that costs something the owner did // not ask for. Each evaluation is a full LLM round-trip on the one resident @@ -697,6 +757,12 @@ func (c *Config) applyDefaults() { c.Feeds = nil } + // Same rule for the crawler: a block that neither answers on demand nor + // watches anything has nothing to do, so it is normalised to "off". + if c.Crawl != nil && !c.Crawl.OnDemand && len(c.Crawl.Watches) == 0 { + c.Crawl = nil + } + if c.Voice != nil { if c.Voice.RouterThreshold <= 0 { c.Voice.RouterThreshold = DefaultRouterThreshold diff --git a/internal/crawl/crawl.go b/internal/crawl/crawl.go new file mode 100644 index 0000000..96bec10 --- /dev/null +++ b/internal/crawl/crawl.go @@ -0,0 +1,174 @@ +// Package crawl reads a web page: fetch, robots check, HTML to text. +// +// It is the LAST place Maven looks for an answer, and that ordering is the whole +// design. "Never phones home" is deprecated, but what replaced it puts local +// sources first: the resident model, then his own memory, then the Kiwix ZIMs on +// the box (internal/kiwix), and only then the network. A local read costs +// nothing and leaks nothing; a fetch costs a round-trip and puts a URL in +// someone's access log. So this package exists to be the fallback, not the +// front door — see the querySources chain in cmd/mavend/actions_query.go for +// where it actually sits. +// +// What never leaves the box: his notes, his facts, the persona block, the +// conversation history. Only the URL is requested and, for the on-demand path, +// only because he said it out loud. Nothing here reads the store. +// +// The limits are not in this package — they are in internal/webfetch, which is +// the only way anything here touches a socket: http(s) only, host allow/deny, +// private-address refusal, size cap, redirect cap, per-host rate limit. What +// this package adds is politeness (robots.txt) and dedup. +package crawl + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "errors" + "fmt" + "net/url" + "strings" + "time" +) + +// Errors callers distinguish. +var ( + ErrRobots = errors.New("crawl: robots.txt disallows this path") + ErrNotHTML = errors.New("crawl: response is not html or text") +) + +// Fetcher is the guarded HTTP door (internal/webfetch adapted by the daemon). An +// interface so this package constructs no http.Client of its own and can be +// tested without a network. +type Fetcher interface { + Get(ctx context.Context, url string) (*Response, error) +} + +// Response is the minimum a crawl needs from a fetch. +type Response struct { + URL string + ContentType string + Body []byte +} + +// Config — crawler knobs. +type Config struct { + // UserAgent is the name matched against robots.txt groups. It must be the + // same string the fetcher sends, or Maven would be claiming one identity + // and obeying the rules for another. + UserAgent string + // MaxRunes caps extracted text. 0 ⇒ DefaultMaxRunes. + MaxRunes int + // RobotsTTL — how long a parsed robots.txt is trusted. 0 ⇒ 1h. + RobotsTTL time.Duration + // Now is injectable for tests. nil ⇒ time.Now. + Now func() time.Time +} + +// Crawler fetches and extracts pages. Safe for concurrent use. +type Crawler struct { + fetch Fetcher + cfg Config + robots *robotsCache +} + +// New builds a crawler. Returns nil when there is no fetcher, which is how the +// daemon expresses "crawling is off unless configured". +func New(fetch Fetcher, cfg Config) *Crawler { + if fetch == nil { + return nil + } + if cfg.UserAgent == "" { + cfg.UserAgent = "Maven" + } + if cfg.MaxRunes <= 0 { + cfg.MaxRunes = DefaultMaxRunes + } + if cfg.RobotsTTL <= 0 { + cfg.RobotsTTL = time.Hour + } + if cfg.Now == nil { + cfg.Now = time.Now + } + return &Crawler{fetch: fetch, cfg: cfg, robots: newRobotsCache(cfg.RobotsTTL)} +} + +// Page fetches rawURL and returns its text. It checks robots.txt first and +// refuses a disallowed path with ErrRobots — there is no override. +func (c *Crawler) Page(ctx context.Context, rawURL string) (Page, error) { + u, err := url.Parse(strings.TrimSpace(rawURL)) + if err != nil { + return Page{}, fmt.Errorf("crawl: bad url %q: %w", rawURL, err) + } + ok, err := c.allowed(ctx, u) + if err != nil { + return Page{}, err + } + if !ok { + return Page{}, fmt.Errorf("%w: %s", ErrRobots, u.Path) + } + resp, err := c.fetch.Get(ctx, u.String()) + if err != nil { + return Page{}, err + } + // A PDF or an image is bytes Maven cannot read; saying so beats storing + // binary garbage as a "note". + ct := strings.ToLower(resp.ContentType) + if ct != "" && !strings.Contains(ct, "html") && !strings.Contains(ct, "text/") && + !strings.Contains(ct, "xml") && !strings.Contains(ct, "json") { + return Page{}, fmt.Errorf("%w: %s", ErrNotHTML, resp.ContentType) + } + return Extract(resp.URL, resp.Body, c.cfg.MaxRunes), nil +} + +// allowed consults robots.txt for u's host, reading it at most once per TTL. +// +// A robots.txt that cannot be fetched (404, a timeout, a blocked host) means +// allow, per the standard. The one thing that is NOT fail-open is an explicit +// Disallow. +func (c *Crawler) allowed(ctx context.Context, u *url.URL) (bool, error) { + host := u.Host + now := c.cfg.Now() + rules, ok := c.robots.get(host, now) + if !ok { + robotsURL := u.Scheme + "://" + host + "/robots.txt" + resp, err := c.fetch.Get(ctx, robotsURL) + switch { + case err != nil: + // Note what is NOT swallowed: a refusal from the guarded fetcher. + // If webfetch says this host is denied or private, the page fetch + // would fail the same way, and reporting the real reason beats + // reporting a robots verdict we never got. + if isFatalFetchError(err) { + return false, err + } + rules = Rules{} + default: + rules = ParseRobots(string(resp.Body), c.cfg.UserAgent) + } + c.robots.put(host, rules, now) + } + path := u.EscapedPath() + if u.RawQuery != "" { + path += "?" + u.RawQuery + } + return rules.Allowed(path), nil +} + +// isFatalFetchError — a fetch failure that means "this host is off limits" +// rather than "there is no robots.txt here". The sentinel set is webfetch's, but +// this package must not import it (the interface exists precisely so it does +// not), so the check is on the message. Ugly and honest: the alternative is a +// dependency inversion for two strings. +func isFatalFetchError(err error) bool { + s := err.Error() + return strings.Contains(s, "not allowed") || strings.Contains(s, "private address") || + strings.Contains(s, "only http and https") +} + +// Hash is the dedup key for a crawl result: the sha256 of the extracted text, +// hex, first 16 chars. Text and not raw HTML, because a page whose only change +// is a rotating ad slot or a CSRF token has not changed. +func Hash(text string) string { + sum := sha256.Sum256([]byte(strings.TrimSpace(text))) + return hex.EncodeToString(sum[:])[:16] +} diff --git a/internal/crawl/crawl_test.go b/internal/crawl/crawl_test.go new file mode 100644 index 0000000..0dbaf4d --- /dev/null +++ b/internal/crawl/crawl_test.go @@ -0,0 +1,155 @@ +package crawl + +import ( + "context" + "errors" + "strings" + "testing" + "time" +) + +// fakeFetcher serves canned pages by URL and counts requests, so a test can +// assert that robots.txt was read once and that a refusal never reached the page. +type fakeFetcher struct { + pages map[string]Response + err error + calls []string +} + +func (f *fakeFetcher) Get(_ context.Context, u string) (*Response, error) { + f.calls = append(f.calls, u) + if f.err != nil { + return nil, f.err + } + r, ok := f.pages[u] + if !ok { + return nil, errors.New("http 404") + } + if r.URL == "" { + r.URL = u + } + if r.ContentType == "" { + r.ContentType = "text/html; charset=utf-8" + } + return &r, nil +} + +const htmlPage = `Почему небо синее + +

Небо

+

Свет рассеивается на молекулах воздуха.

+

Короткие волны рассеиваются сильнее.

+
© 2026
` + +func newTestCrawler(f *fakeFetcher) *Crawler { + return New(f, Config{UserAgent: "Maven/1.0", Now: func() time.Time { return time.Unix(0, 0) }}) +} + +func TestPageExtractsText(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{ + "https://example.org/sky": {Body: []byte(htmlPage)}, + }} + page, err := newTestCrawler(f).Page(context.Background(), "https://example.org/sky") + if err != nil { + t.Fatal(err) + } + if page.Title != "Почему небо синее" { + t.Errorf("title = %q", page.Title) + } + if !strings.Contains(page.Text, "Свет рассеивается") { + t.Errorf("body text missing: %q", page.Text) + } + for _, junk := range []string{"track()", "color:red", "меню", "© 2026"} { + if strings.Contains(page.Text, junk) { + t.Errorf("%q survived extraction: %q", junk, page.Text) + } + } +} + +func TestRobotsIsCheckedAndObeyed(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{ + "https://example.org/robots.txt": {Body: []byte("User-agent: *\nDisallow: /secret\n"), ContentType: "text/plain"}, + "https://example.org/secret/x": {Body: []byte(htmlPage)}, + "https://example.org/open": {Body: []byte(htmlPage)}, + }} + c := newTestCrawler(f) + if _, err := c.Page(context.Background(), "https://example.org/secret/x"); !errors.Is(err, ErrRobots) { + t.Fatalf("error = %v, want ErrRobots", err) + } + for _, u := range f.calls { + if strings.Contains(u, "/secret") { + t.Fatal("the disallowed page was fetched anyway") + } + } + if _, err := c.Page(context.Background(), "https://example.org/open"); err != nil { + t.Fatalf("allowed page: %v", err) + } + // robots.txt was read once for the host, not once per page. + robotsReads := 0 + for _, u := range f.calls { + if strings.HasSuffix(u, "/robots.txt") { + robotsReads++ + } + } + if robotsReads != 1 { + t.Fatalf("robots.txt read %d times, want 1", robotsReads) + } +} + +// No robots.txt means allow — that is the standard, and the alternative makes +// most of the web unreadable. +func TestMissingRobotsAllows(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{ + "https://example.org/page": {Body: []byte(htmlPage)}, + }} + if _, err := newTestCrawler(f).Page(context.Background(), "https://example.org/page"); err != nil { + t.Fatalf("err = %v, want the page", err) + } +} + +// A refusal from the guarded fetcher must surface as itself, not be laundered +// into "no robots.txt, go ahead". +func TestFetcherRefusalIsNotSwallowed(t *testing.T) { + f := &fakeFetcher{err: errors.New("webfetch: refusing to connect to a private address: 127.0.0.1")} + _, err := newTestCrawler(f).Page(context.Background(), "http://127.0.0.1:9100/mcp") + if err == nil || !strings.Contains(err.Error(), "private address") { + t.Fatalf("error = %v, want the fetcher's refusal", err) + } +} + +func TestNonTextIsRefused(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{ + "https://example.org/f.pdf": {Body: []byte("%PDF-1.7"), ContentType: "application/pdf"}, + }} + if _, err := newTestCrawler(f).Page(context.Background(), "https://example.org/f.pdf"); !errors.Is(err, ErrNotHTML) { + t.Fatalf("error = %v, want ErrNotHTML", err) + } +} + +func TestMaxRunesCapsText(t *testing.T) { + long := "

" + strings.Repeat("привет ", 2000) + "

" + f := &fakeFetcher{pages: map[string]Response{"https://example.org/l": {Body: []byte(long)}}} + c := New(f, Config{MaxRunes: 50}) + page, err := c.Page(context.Background(), "https://example.org/l") + if err != nil { + t.Fatal(err) + } + if n := len([]rune(page.Text)); n > 51 { + t.Fatalf("text = %d runes, want the 50-rune cap", n) + } +} + +func TestNewWithoutFetcherIsNil(t *testing.T) { + if New(nil, Config{}) != nil { + t.Fatal("a crawler with no fetcher must be nil — crawling is off unless configured") + } +} + +func TestHashIgnoresNothingButText(t *testing.T) { + if Hash("a") == Hash("b") { + t.Fatal("different text hashed the same") + } + if Hash(" same \n") != Hash("same") { + t.Fatal("surrounding whitespace changed the hash") + } +} diff --git a/internal/crawl/extract.go b/internal/crawl/extract.go new file mode 100644 index 0000000..5828722 --- /dev/null +++ b/internal/crawl/extract.go @@ -0,0 +1,106 @@ +package crawl + +import ( + "html" + "regexp" + "strings" +) + +// HTML → text, with a regexp and no tokenizer. +// +// golang.org/x/net/html is not vendored and the network is not assumed, so this +// is stdlib. That is less of a compromise than it sounds: the unit of context +// here is a few hundred words for a 4096-token model to read, exactly like the +// Kiwix snippet, so what matters is dropping script/style/nav noise and keeping +// paragraph boundaries. A DOM would buy correctness on malformed markup that is +// then thrown away by truncation anyway. +// +// What this deliberately does NOT do: run JavaScript, follow links, or extract +// structured fields with CSS selectors or an LLM prompt. The plan's step 2 asked +// for the last of those; see docs/plans/14-web-crawler.md for why it was left +// out for now. + +var ( + // RE2 has no backreferences, so each tag pair is spelled out rather than + // captured and matched against itself. + dropRE = regexp.MustCompile(pairsRE("script", "style", "noscript", "svg", "head", "nav", "footer", "form")) + titleRE = regexp.MustCompile(`(?is)]*>(.*?)`) + h1RE = regexp.MustCompile(`(?is)]*>(.*?)`) + // Block-level tags become newlines so paragraphs survive as paragraphs. + blockRE = regexp.MustCompile(`(?is)]*>`) + tagRE = regexp.MustCompile(`(?s)<[^>]*>`) + commentRE = regexp.MustCompile(`(?s)`) + spaceRE = regexp.MustCompile(`[ \t\f\v]+`) + blankRE = regexp.MustCompile(`\n{2,}`) +) + +// pairsRE builds `(?is)|…` for the given tags. +func pairsRE(tags ...string) string { + parts := make([]string, 0, len(tags)) + for _, t := range tags { + parts = append(parts, `<`+t+`\b[^>]*>.*?`) + } + return `(?is)` + strings.Join(parts, "|") +} + +// Page is an extracted page. +type Page struct { + URL string + Title string + Text string // plain text, paragraphs separated by single newlines +} + +// Extract turns a fetched HTML document into a Page. maxRunes caps the text (0 ⇒ +// DefaultMaxRunes); the cap is on runes, not bytes, because a Russian page cut +// at a byte boundary ends in half a letter. +func Extract(url string, body []byte, maxRunes int) Page { + if maxRunes <= 0 { + maxRunes = DefaultMaxRunes + } + s := string(body) + s = commentRE.ReplaceAllString(s, " ") + + title := firstGroup(titleRE, s) + if title == "" { + title = firstGroup(h1RE, s) + } + + s = dropRE.ReplaceAllString(s, "\n") + s = blockRE.ReplaceAllString(s, "\n") + s = tagRE.ReplaceAllString(s, " ") + s = html.UnescapeString(s) + s = spaceRE.ReplaceAllString(s, " ") + + var lines []string + for _, l := range strings.Split(s, "\n") { + if l = strings.TrimSpace(l); l != "" { + lines = append(lines, l) + } + } + text := blankRE.ReplaceAllString(strings.Join(lines, "\n"), "\n") + + return Page{URL: url, Title: title, Text: TrimRunes(text, maxRunes)} +} + +// DefaultMaxRunes — how much of a page is kept. ~4000 runes is a long answer's +// worth of context and still leaves room in a 4096-token window for the prompt +// and the reply. +const DefaultMaxRunes = 4000 + +func firstGroup(re *regexp.Regexp, s string) string { + m := re.FindStringSubmatch(s) + if len(m) < 2 { + return "" + } + t := tagRE.ReplaceAllString(m[1], " ") + return strings.TrimSpace(strings.Join(strings.Fields(html.UnescapeString(t)), " ")) +} + +// TrimRunes cuts s to at most max runes, on a rune boundary. +func TrimRunes(s string, max int) string { + r := []rune(s) + if len(r) <= max { + return s + } + return strings.TrimSpace(string(r[:max])) + "…" +} diff --git a/internal/crawl/robots.go b/internal/crawl/robots.go new file mode 100644 index 0000000..3774aa6 --- /dev/null +++ b/internal/crawl/robots.go @@ -0,0 +1,211 @@ +package crawl + +import ( + "regexp" + "strings" + "sync" + "time" +) + +// robots.txt, parsed the small way: no wildcards beyond the two the standard +// actually defines (`*` inside a path and `$` at the end), no sitemaps, no +// crawl-delay-per-agent gymnastics. A personal assistant reading a handful of +// pages does not need a spec-complete implementation; it needs to not be rude, +// and to be auditable in one sitting. +// +// Two rules worth stating because they are choices, not accidents: +// +// - a missing or unreadable robots.txt means ALLOW. That is what the standard +// says (404 ⇒ unrestricted), and the alternative would make a site that +// simply has no robots.txt unreadable; +// - an explicit Disallow means REFUSE, and Maven does not offer an override. +// There is no "but he asked me to" flag: the page is not read. + +// Rules is a parsed robots.txt for one user-agent. +type Rules struct { + allow []string + disallow []string + // Delay is Crawl-delay in seconds when the group named one, 0 otherwise. + // The fetcher's own per-host rate limit is the floor; this can only make + // Maven slower, never faster. + Delay time.Duration +} + +// ParseRobots reads robots.txt and returns the rules that apply to agent. +// +// Group selection follows the standard: the most specific matching group wins, +// which here means an exact user-agent match beats `*`. Lines that are neither +// are ignored rather than guessed at. +func ParseRobots(body string, agent string) Rules { + agent = strings.ToLower(agent) + + type group struct { + agents []string + allow []string + disallow []string + delay time.Duration + } + var groups []group + var cur *group + // startNew tracks whether the next User-agent line opens a new group or + // joins the current one: consecutive User-agent lines share their rules. + startNew := true + + for _, raw := range strings.Split(body, "\n") { + line := raw + if i := strings.IndexByte(line, '#'); i >= 0 { + line = line[:i] + } + line = strings.TrimSpace(line) + if line == "" { + continue + } + key, val, ok := strings.Cut(line, ":") + if !ok { + continue + } + key = strings.ToLower(strings.TrimSpace(key)) + val = strings.TrimSpace(val) + + switch key { + case "user-agent": + if startNew || cur == nil { + groups = append(groups, group{}) + cur = &groups[len(groups)-1] + startNew = false + } + cur.agents = append(cur.agents, strings.ToLower(val)) + case "disallow": + if cur == nil { + continue + } + startNew = true + // "Disallow:" with an empty value allows everything, and is not a + // path rule at all. + if val != "" { + cur.disallow = append(cur.disallow, val) + } + case "allow": + if cur == nil { + continue + } + startNew = true + if val != "" { + cur.allow = append(cur.allow, val) + } + case "crawl-delay": + if cur == nil { + continue + } + startNew = true + if d, err := time.ParseDuration(val + "s"); err == nil && d > 0 { + cur.delay = d + } + } + } + + var star, exact *group + for i := range groups { + for _, a := range groups[i].agents { + if a == "*" && star == nil { + star = &groups[i] + } + // A robots.txt names "maven", we send "Maven/1.0 (…)": match on + // prefix, which is how every crawler reads this field. + if a != "*" && a != "" && strings.HasPrefix(agent, a) { + exact = &groups[i] + } + } + } + g := exact + if g == nil { + g = star + } + if g == nil { + return Rules{} + } + return Rules{allow: g.allow, disallow: g.disallow, Delay: g.delay} +} + +// Allowed reports whether path may be fetched. Longest matching rule wins, and +// Allow beats Disallow at equal length — the standard's tie-break, and the one +// that makes "Disallow: /" plus "Allow: /public" mean what it looks like. +func (r Rules) Allowed(path string) bool { + if path == "" { + path = "/" + } + best, allowed := -1, true + for _, p := range r.disallow { + if n, ok := matchPath(p, path); ok && n > best { + best, allowed = n, false + } + } + for _, p := range r.allow { + if n, ok := matchPath(p, path); ok && n >= best { + best, allowed = n, true + } + } + return allowed +} + +// matchPath applies a robots path pattern and returns the pattern's length as +// the specificity score. `*` matches any run of characters, `$` anchors the end. +// A pattern is a PREFIX match otherwise, which is what "Disallow: /admin" means. +func matchPath(pattern, path string) (int, bool) { + score := len(pattern) + re, err := robotsRegexp(pattern) + if err != nil { + return 0, false + } + return score, re.MatchString(path) +} + +// robotsRegexp turns a robots path pattern into an anchored-at-the-start +// regexp. Everything but `*` and a trailing `$` is a literal, so the pattern is +// quoted first and the two metacharacters are put back afterwards. +func robotsRegexp(pattern string) (*regexp.Regexp, error) { + end := "" + if strings.HasSuffix(pattern, "$") { + pattern = strings.TrimSuffix(pattern, "$") + end = "$" + } + parts := strings.Split(pattern, "*") + for i, p := range parts { + parts[i] = regexp.QuoteMeta(p) + } + return regexp.Compile("^" + strings.Join(parts, ".*") + end) +} + +// robotsCache holds parsed rules per host so a crawl of ten pages on one site +// reads robots.txt once. TTL because a site may change its mind, and a daemon +// that runs for weeks would otherwise never notice. +type robotsCache struct { + ttl time.Duration + mu sync.Mutex + m map[string]robotsEntry +} + +type robotsEntry struct { + rules Rules + at time.Time +} + +func newRobotsCache(ttl time.Duration) *robotsCache { + return &robotsCache{ttl: ttl, m: map[string]robotsEntry{}} +} + +func (c *robotsCache) get(host string, now time.Time) (Rules, bool) { + c.mu.Lock() + defer c.mu.Unlock() + e, ok := c.m[host] + if !ok || now.Sub(e.at) > c.ttl { + return Rules{}, false + } + return e.rules, true +} + +func (c *robotsCache) put(host string, r Rules, now time.Time) { + c.mu.Lock() + defer c.mu.Unlock() + c.m[host] = robotsEntry{rules: r, at: now} +} diff --git a/internal/crawl/robots_test.go b/internal/crawl/robots_test.go new file mode 100644 index 0000000..93e6bb9 --- /dev/null +++ b/internal/crawl/robots_test.go @@ -0,0 +1,84 @@ +package crawl + +import ( + "testing" + "time" +) + +const robotsBody = `# a comment +User-agent: * +Disallow: /private +Disallow: /tmp/ +Crawl-delay: 5 + +User-agent: Maven +Disallow: / +Allow: /public +` + +func TestParseRobotsPicksTheMostSpecificGroup(t *testing.T) { + // The Maven group applies to us even though we send a longer UA string. + r := ParseRobots(robotsBody, "Maven/1.0 (self-hosted personal assistant)") + if r.Allowed("/anything") { + t.Error("Disallow: / in our own group was ignored") + } + if !r.Allowed("/public/page") { + t.Error("Allow: /public must beat the shorter Disallow: /") + } + + // A different agent falls into the * group. + star := ParseRobots(robotsBody, "SomeoneElse/2") + if !star.Allowed("/anything") { + t.Error("the * group disallows nothing but /private and /tmp/") + } + if star.Allowed("/private/x") || star.Allowed("/tmp/") { + t.Error("the * group's disallows were not applied") + } + if star.Delay != 5*time.Second { + t.Errorf("crawl-delay = %v, want 5s", star.Delay) + } +} + +func TestParseRobotsEmptyMeansAllowAll(t *testing.T) { + for _, body := range []string{"", "# nothing here\n", "User-agent: *\nDisallow:\n"} { + if !ParseRobots(body, "Maven").Allowed("/whatever") { + t.Errorf("body %q must allow everything", body) + } + } +} + +func TestRobotsWildcards(t *testing.T) { + r := ParseRobots("User-agent: *\nDisallow: /*.pdf$\nDisallow: /a/*/secret\n", "Maven") + if r.Allowed("/docs/manual.pdf") { + t.Error("*.pdf$ did not match") + } + if !r.Allowed("/docs/manual.pdf.html") { + t.Error("$ must anchor at the end") + } + if r.Allowed("/a/b/secret") { + t.Error("/a/*/secret did not match") + } + if !r.Allowed("/a/b/public") { + t.Error("unrelated path was refused") + } +} + +// Consecutive User-agent lines share one group, which is common in the wild. +func TestRobotsSharedGroup(t *testing.T) { + r := ParseRobots("User-agent: Googlebot\nUser-agent: Maven\nDisallow: /x\n", "Maven/1.0") + if r.Allowed("/x/y") { + t.Fatal("a shared group's rules were not applied to the second agent") + } +} + +func TestRobotsCacheTTL(t *testing.T) { + c := newRobotsCache(time.Minute) + now := time.Now() + c.put("example.com", ParseRobots("User-agent: *\nDisallow: /\n", "Maven"), now) + if _, ok := c.get("example.com", now.Add(30*time.Second)); !ok { + t.Error("a fresh entry must be served from cache") + } + if _, ok := c.get("example.com", now.Add(2*time.Minute)); ok { + t.Error("an expired entry must be re-read") + } +} diff --git a/internal/crawl/watch.go b/internal/crawl/watch.go new file mode 100644 index 0000000..704cef7 --- /dev/null +++ b/internal/crawl/watch.go @@ -0,0 +1,177 @@ +package crawl + +import ( + "context" + "fmt" + "log" + "strings" + "time" +) + +// Scheduled crawls: a page is re-read on an interval, and when its TEXT changed +// the new text is written as a note. Nothing is dispatched — same rule as the +// feed poller (Vikunja #258). A page that announced its own change would be a +// nag, and "the docs page changed" is not worth interrupting anyone for. +// +// Dedup is by content hash, so a page that re-renders identically writes nothing +// and a rotating ad slot does not count as news. + +// WatchConfig — one page to keep an eye on. +type WatchConfig struct { + Name string // note source is "crawl:" + URL string // http(s), guarded by the fetcher + Interval time.Duration // 0 ⇒ Watcher's default +} + +// Notes is core's note-writing half (same shape as ipc.CoreAPI's method). +type Notes interface { + WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error) +} + +// Hashes remembers the last text hash per watch, durably, so a restart does not +// re-note an unchanged page. The daemon backs this with config facts +// ("crawl:hash:"). +type Hashes interface { + LastHash(ctx context.Context, name string) (string, error) + SetHash(ctx context.Context, name, hash string) error +} + +// Embedder embeds a note on its way into the store. nil ⇒ no vector. +type Embedder interface { + Embed(ctx context.Context, text string) ([]float32, error) +} + +// DefaultWatchInterval — pages change slowly, and every check is a request in +// someone's log. +const DefaultWatchInterval = 6 * time.Hour + +// Watcher re-reads watched pages on their interval. +type Watcher struct { + c *Crawler + watches []WatchConfig + notes Notes + hashes Hashes + embed Embedder + interval time.Duration + nextDue map[string]time.Time +} + +// NewWatcher wires the scheduled half, or returns nil when there is nothing to +// watch. Callers check for nil: no watches, no goroutine, no request. +func NewWatcher(c *Crawler, watches []WatchConfig, notes Notes, hashes Hashes, embed Embedder, defaultInterval time.Duration) *Watcher { + if c == nil || notes == nil { + return nil + } + var valid []WatchConfig + for _, w := range watches { + if strings.TrimSpace(w.Name) == "" || strings.TrimSpace(w.URL) == "" { + log.Printf("crawl: skipping a watch with no name or no url") + continue + } + valid = append(valid, w) + } + if len(valid) == 0 { + return nil + } + if defaultInterval <= 0 { + defaultInterval = DefaultWatchInterval + } + return &Watcher{ + c: c, watches: valid, notes: notes, hashes: hashes, embed: embed, + interval: defaultInterval, nextDue: map[string]time.Time{}, + } +} + +// Watches returns the configured watches. +func (w *Watcher) Watches() []WatchConfig { return w.watches } + +// CheckDue re-reads every watch whose interval elapsed and returns how many +// notes were written. Errors are logged per watch, never returned: one dead page +// must not stop the others. +func (w *Watcher) CheckDue(ctx context.Context, now time.Time) int { + written := 0 + for _, watch := range w.watches { + if due, ok := w.nextDue[watch.Name]; ok && now.Before(due) { + continue + } + interval := watch.Interval + if interval <= 0 { + interval = w.interval + } + w.nextDue[watch.Name] = now.Add(interval) + changed, err := w.Check(ctx, watch, now) + if err != nil { + log.Printf("crawl: watch %s: %v", watch.Name, err) + continue + } + if changed { + log.Printf("crawl: watch %s: page changed, noted", watch.Name) + written++ + } + } + return written +} + +// Check re-reads one watch now and reports whether it wrote a note. +func (w *Watcher) Check(ctx context.Context, watch WatchConfig, now time.Time) (bool, error) { + page, err := w.c.Page(ctx, watch.URL) + if err != nil { + return false, err + } + // Title included: a page whose headline changed has changed. + h := Hash(page.Title + "\n" + page.Text) + if w.hashes != nil { + prev, err := w.hashes.LastHash(ctx, watch.Name) + if err != nil { + log.Printf("crawl: watch %s: read hash: %v", watch.Name, err) + } + if prev == h { + return false, nil + } + } + text := NoteText(watch, page) + var vec []float32 + if w.embed != nil { + v, err := w.embed.Embed(ctx, text) + if err != nil { + log.Printf("crawl: watch %s: embed: %v", watch.Name, err) + } else { + vec = v + } + } + if _, err := w.notes.WriteNote(ctx, now, text, vec, SourceFor(watch.Name)); err != nil { + return false, fmt.Errorf("write note: %w", err) + } + if w.hashes != nil { + if err := w.hashes.SetHash(ctx, watch.Name, h); err != nil { + log.Printf("crawl: watch %s: save hash: %v", watch.Name, err) + } + } + return true, nil +} + +// SourceFor is the note source for a watch, and SourcePrefix is what the answer +// path matches to recognise one. +func SourceFor(name string) string { return SourcePrefix + name } + +// SourcePrefix — provenance for anything read off the network on a schedule. +const SourcePrefix = "crawl:" + +// noteRunes — how much of a watched page goes into a note. Shorter than what the +// on-demand path reads: a note is a record of a change, not an archive. +const noteRunes = 800 + +// NoteText renders a watched page as a note body. +func NoteText(watch WatchConfig, page Page) string { + var b strings.Builder + if page.Title != "" { + b.WriteString(page.Title) + } else { + b.WriteString(watch.Name) + } + b.WriteString("\n") + b.WriteString(TrimRunes(page.Text, noteRunes)) + b.WriteString("\n") + b.WriteString(watch.URL) + return b.String() +} diff --git a/internal/crawl/watch_test.go b/internal/crawl/watch_test.go new file mode 100644 index 0000000..1f266fd --- /dev/null +++ b/internal/crawl/watch_test.go @@ -0,0 +1,117 @@ +package crawl + +import ( + "context" + "strings" + "testing" + "time" +) + +type note struct { + text string + source string +} + +type fakeNotes struct{ notes []note } + +func (n *fakeNotes) WriteNote(_ context.Context, _ time.Time, text string, _ []float32, source string) (int64, error) { + n.notes = append(n.notes, note{text, source}) + return int64(len(n.notes)), nil +} + +type fakeHashes struct{ m map[string]string } + +func newHashes() *fakeHashes { return &fakeHashes{m: map[string]string{}} } +func (f *fakeHashes) LastHash(_ context.Context, name string) (string, error) { + return f.m[name], nil +} +func (f *fakeHashes) SetHash(_ context.Context, name, h string) error { f.m[name] = h; return nil } + +var t0 = time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC) + +func TestWatchNotesAChangedPage(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{ + "https://example.org/docs": {Body: []byte(htmlPage)}, + }} + notes := &fakeNotes{} + hashes := newHashes() + w := NewWatcher(newTestCrawler(f), []WatchConfig{{Name: "docs", URL: "https://example.org/docs"}}, + notes, hashes, nil, time.Hour) + if w == nil { + t.Fatal("NewWatcher returned nil for a configured watch") + } + if n := w.CheckDue(context.Background(), t0); n != 1 { + t.Fatalf("first check wrote %d notes, want 1", n) + } + if notes.notes[0].source != "crawl:docs" { + t.Errorf("source = %q, want crawl:docs", notes.notes[0].source) + } + if !strings.Contains(notes.notes[0].text, "https://example.org/docs") { + t.Errorf("note does not carry the url: %q", notes.notes[0].text) + } + + // Unchanged page, interval elapsed: nothing written. + if n := w.CheckDue(context.Background(), t0.Add(2*time.Hour)); n != 0 { + t.Fatalf("an unchanged page wrote %d notes", n) + } + + // Changed page: one note. + f.pages["https://example.org/docs"] = Response{Body: []byte(strings.Replace(htmlPage, "синее", "серое", 1))} + if n := w.CheckDue(context.Background(), t0.Add(4*time.Hour)); n != 1 { + t.Fatalf("a changed page wrote %d notes, want 1", n) + } +} + +func TestWatchIntervalIsRespected(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{"https://example.org/d": {Body: []byte(htmlPage)}}} + w := NewWatcher(newTestCrawler(f), []WatchConfig{{Name: "d", URL: "https://example.org/d", Interval: time.Hour}}, + &fakeNotes{}, newHashes(), nil, 0) + w.CheckDue(context.Background(), t0) + before := len(f.calls) + w.CheckDue(context.Background(), t0.Add(time.Minute)) + if len(f.calls) != before { + t.Fatal("the page was re-read inside its interval") + } +} + +// The hash is durable so a restart does not re-note an unchanged page. +func TestWatchHashSurvivesRestart(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{"https://example.org/d": {Body: []byte(htmlPage)}}} + hashes := newHashes() + watches := []WatchConfig{{Name: "d", URL: "https://example.org/d"}} + NewWatcher(newTestCrawler(f), watches, &fakeNotes{}, hashes, nil, time.Hour).CheckDue(context.Background(), t0) + + notes2 := &fakeNotes{} + NewWatcher(newTestCrawler(f), watches, notes2, hashes, nil, time.Hour).CheckDue(context.Background(), t0.Add(time.Hour)) + if len(notes2.notes) != 0 { + t.Fatalf("a fresh watcher re-noted an unchanged page: %q", notes2.notes[0].text) + } +} + +func TestWatchDeadPageDoesNotStopTheOthers(t *testing.T) { + f := &fakeFetcher{pages: map[string]Response{"https://example.org/live": {Body: []byte(htmlPage)}}} + notes := &fakeNotes{} + w := NewWatcher(newTestCrawler(f), []WatchConfig{ + {Name: "dead", URL: "https://example.org/gone"}, + {Name: "live", URL: "https://example.org/live"}, + }, notes, newHashes(), nil, time.Hour) + if n := w.CheckDue(context.Background(), t0); n != 1 { + t.Fatalf("wrote %d notes, want 1 (the live page)", n) + } + if notes.notes[0].source != "crawl:live" { + t.Fatalf("source = %q", notes.notes[0].source) + } +} + +func TestNoWatchesMeansNoWatcher(t *testing.T) { + c := newTestCrawler(&fakeFetcher{}) + if NewWatcher(c, nil, &fakeNotes{}, nil, nil, 0) != nil { + t.Fatal("no watches must mean no watcher") + } + if NewWatcher(nil, []WatchConfig{{Name: "a", URL: "u"}}, &fakeNotes{}, nil, nil, 0) != nil { + t.Fatal("no crawler must mean no watcher") + } + if NewWatcher(c, []WatchConfig{{Name: "", URL: ""}}, &fakeNotes{}, nil, nil, 0) != nil { + t.Fatal("a watch with no name or url is not a configuration") + } +} diff --git a/internal/router/url.go b/internal/router/url.go new file mode 100644 index 0000000..8cdea09 --- /dev/null +++ b/internal/router/url.go @@ -0,0 +1,39 @@ +package router + +import ( + "regexp" + "strings" +) + +// Finding a URL in an utterance (Vikunja #259). +// +// This is deliberately strict: a scheme is required. "посмотри на example.org" +// is not treated as a fetch request, because a bare dotted word is also how +// people say file names, versions and Russian abbreviations, and the cost of a +// false positive here is an outbound request nobody asked for. +// +// Note where this runs: an utterance from STT. Whisper will mangle a spoken URL, +// which is fine — the URL that survives is one he pasted into the web chat, and +// a mangled one simply fails to match. +var urlRE = regexp.MustCompile(`(?i)\bhttps?://[^\s<>"']+`) + +// FirstURL returns the first http(s) URL in text. +// +// Trailing punctuation is trimmed: he ends sentences, and "…/page." is not a +// path component. A closing bracket is only trimmed when it has no opener, +// because a wikipedia URL legitimately ends in one. +func FirstURL(text string) (string, bool) { + m := urlRE.FindString(text) + if m == "" { + return "", false + } + m = strings.TrimRight(m, ".,;:!?…") + if strings.HasSuffix(m, ")") && strings.Count(m, "(") == 0 { + m = strings.TrimSuffix(m, ")") + } + // A scheme with nothing after it is not a URL. + if rest := strings.SplitN(m, "//", 2); len(rest) < 2 || rest[1] == "" { + return "", false + } + return m, true +} diff --git a/internal/router/url_test.go b/internal/router/url_test.go new file mode 100644 index 0000000..6710773 --- /dev/null +++ b/internal/router/url_test.go @@ -0,0 +1,34 @@ +package router + +import "testing" + +func TestFirstURL(t *testing.T) { + cases := []struct { + text string + want string + }{ + {"посмотри https://example.org/page — что там?", "https://example.org/page"}, + {"почитай http://example.org/a/b?x=1 и скажи", "http://example.org/a/b?x=1"}, + {"вот ссылка: https://example.org/page.", "https://example.org/page"}, + {"https://ru.wikipedia.org/wiki/Небо_(значения)", "https://ru.wikipedia.org/wiki/Небо_(значения)"}, + // No scheme ⇒ no fetch. A bare dotted word is not an instruction to + // reach out to the network. + {"посмотри на example.org", ""}, + {"открой файл config.json", ""}, + {"что нового?", ""}, + {"https://", ""}, + {"", ""}, + } + for _, c := range cases { + got, ok := FirstURL(c.text) + if c.want == "" { + if ok { + t.Errorf("FirstURL(%q) = %q, want no match", c.text, got) + } + continue + } + if !ok || got != c.want { + t.Errorf("FirstURL(%q) = %q, %v; want %q", c.text, got, ok, c.want) + } + } +}