Read a web page when he names one, and watch a few on a timer (#259)
The network fallback behind the local sources, off unless configured. internal/crawl is pure: a stdlib robots.txt parser (group specificity, wildcards, Crawl-delay, cached per host), HTML-to-plaintext extraction, and a watcher that notes a watched page only when its text changed. It has no store access and no net/http; cmd/mavend/crawls.go is the impure half. Every limit is code and tested: the guarded fetcher from #258 enforces the host allowlist/denylist, refuses private addresses in the dialer Control hook (so DNS rebinding and each redirect hop are covered), caps size and redirects, times out, and spaces requests per host. A robots.txt Disallow is refused with no override. On demand, reading is a query source placed last in the chain, after his memory, his notes, and the local Kiwix ZIMs once those are wired: no URL in the utterance means no fetch, and only the URL ever leaves the box. Scheduled watches write notes and announce nothing. The vendored tree has no x/net/html, goquery or temoto/robotstxt, so the parsers are stdlib. No new dependency.
This commit is contained in:
@@ -8,6 +8,7 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/crawl"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/memory"
|
||||
"github.com/kami/maven/internal/morning"
|
||||
@@ -78,6 +79,13 @@ var querySources = []querySource{
|
||||
{"embed", (*reactiveHandler).queryEmbed},
|
||||
{"memory", (*reactiveHandler).queryMemory},
|
||||
{"notes", (*reactiveHandler).queryNotes},
|
||||
// LAST before the model answers from memory, and that position is the whole
|
||||
// design (Vikunja #259): local sources first. The model, his own notes and
|
||||
// facts, and — once internal/kiwix is wired into this chain — the offline
|
||||
// ZIMs all get their turn before anything touches the network. This source
|
||||
// only claims a turn where he named a URL out loud, so it never competes
|
||||
// with a local answer.
|
||||
{"web", (*reactiveHandler).queryWeb},
|
||||
{"general-knowledge", (*reactiveHandler).queryGeneral},
|
||||
}
|
||||
|
||||
@@ -371,6 +379,57 @@ func (h *reactiveHandler) queryNotes(ctx context.Context, t *queryTurn) (string,
|
||||
return reply, true
|
||||
}
|
||||
|
||||
// webPageContextRunes — how much of a fetched page is handed to the phraser.
|
||||
// Less than the crawler keeps: the rest of the 4096-token window belongs to the
|
||||
// prompt, the persona block and the reply.
|
||||
const webPageContextRunes = 1500
|
||||
|
||||
// queryWeb — "посмотри https://example.org/x — что там?" (Vikunja #259).
|
||||
//
|
||||
// It claims a turn ONLY when he named a URL, which is what keeps a fallback from
|
||||
// becoming a habit: no URL, no fetch, and the model answers from what is local.
|
||||
// What leaves the box is the URL and nothing else — no note, no fact, no history
|
||||
// travels with it.
|
||||
func (h *reactiveHandler) queryWeb(ctx context.Context, t *queryTurn) (string, bool) {
|
||||
link, ok := router.FirstURL(t.dec.Utterance)
|
||||
if !ok {
|
||||
return "", false
|
||||
}
|
||||
if h.crawler == nil {
|
||||
// Claim rather than fall through: he asked about a specific page, and
|
||||
// letting the model answer from the URL's spelling alone is how a small
|
||||
// model invents a page's contents.
|
||||
return "я не читаю страницы — это не настроено.", true
|
||||
}
|
||||
ctxFetch, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
defer cancel()
|
||||
page, err := h.crawler.Page(ctxFetch, link)
|
||||
if err != nil {
|
||||
if errors.Is(err, crawl.ErrRobots) {
|
||||
return "эта страница закрыта для чтения — robots.txt не разрешает.", true
|
||||
}
|
||||
log.Printf("voice: web: %v", err)
|
||||
return "не получилось прочитать страницу.", true
|
||||
}
|
||||
if page.Text == "" {
|
||||
return "страница открылась, но читать там нечего.", true
|
||||
}
|
||||
// The page is handed to the phraser the same way a note is: as context for
|
||||
// the question he actually asked. She answers the question, she does not
|
||||
// recite the page.
|
||||
snippet := page.Title + "\n" + crawl.TrimRunes(page.Text, webPageContextRunes)
|
||||
reply, perr := h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet})
|
||||
if perr != nil {
|
||||
log.Printf("voice: web: phrase: %v", perr)
|
||||
}
|
||||
if reply == "" {
|
||||
// No phraser (or it failed): read back the top of the page rather than
|
||||
// pretend the fetch did not happen.
|
||||
return "вот что на странице: " + crawl.TrimRunes(page.Text, 300), true
|
||||
}
|
||||
return reply, true
|
||||
}
|
||||
|
||||
// queryGeneral — general knowledge from the phraser, the last source before
|
||||
// giving up. It always claims: either the model answers or Maven says she
|
||||
// doesn't know.
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
// mavend/crawls.go — the driver for reading web pages (Vikunja #259,
|
||||
// docs/plans/14-web-crawler.md). The crawler is pure and lives in
|
||||
// internal/crawl; this is the impure half: the guarded fetcher, a ticker for the
|
||||
// scheduled watches, and the fact-backed dedup hashes.
|
||||
//
|
||||
// Two paths, one config block, both off unless configured:
|
||||
//
|
||||
// - ON DEMAND — he names a URL out loud and she reads it. That is the
|
||||
// `queryWeb` source in actions_query.go, LAST in the chain: after his
|
||||
// memory, after the notes, and (once Kiwix is wired into the chain) after
|
||||
// the local ZIMs. A local read costs nothing and leaks nothing; a fetch puts
|
||||
// a URL in someone's log, so it goes last.
|
||||
// - SCHEDULED — a watched page is re-read on its interval, and a page whose
|
||||
// text changed is written as a note. It does NOT announce itself. Same rule
|
||||
// as the feed poller: notes, never nudges.
|
||||
//
|
||||
// Only the URL goes out. Nothing here reads a note, a fact, the persona block or
|
||||
// the history, and internal/crawl has no access to the store at all.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"net/url"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/crawl"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/webfetch"
|
||||
)
|
||||
|
||||
// newCrawler builds the crawler from the `crawl` block, or returns nil when
|
||||
// there is none. Every caller checks for nil, and nil means no page is ever
|
||||
// fetched.
|
||||
func newCrawler(cfg *config.Config) *crawl.Crawler {
|
||||
if cfg.Crawl == nil {
|
||||
return nil
|
||||
}
|
||||
cc := cfg.Crawl
|
||||
|
||||
hosts := append([]string(nil), cc.AllowHosts...)
|
||||
// A watched page's own host is always reachable; otherwise an allowlist and
|
||||
// a watch list would have to be kept in sync by hand.
|
||||
for _, w := range cc.Watches {
|
||||
if u, err := url.Parse(w.URL); err == nil && u.Hostname() != "" {
|
||||
hosts = append(hosts, u.Hostname())
|
||||
}
|
||||
}
|
||||
// An allowlist plus on-demand is a contradiction worth logging rather than
|
||||
// silently resolving: he asked for arbitrary pages AND for a fixed list.
|
||||
// The allowlist wins, because it is the narrower instruction.
|
||||
if len(hosts) > 0 && cc.OnDemand && len(cc.AllowHosts) > 0 {
|
||||
log.Printf("crawl: allow_hosts is set, so on-demand reading is limited to those hosts")
|
||||
}
|
||||
ua := cc.UserAgent
|
||||
if ua == "" {
|
||||
ua = webfetch.DefaultUserAgent
|
||||
}
|
||||
fetcher := webfetch.New(webfetch.Config{
|
||||
AllowHosts: hosts,
|
||||
DenyHosts: cc.DenyHosts,
|
||||
Timeout: time.Duration(cc.Timeout),
|
||||
MaxBytes: cc.MaxBytes,
|
||||
UserAgent: ua,
|
||||
})
|
||||
// The user-agent handed to the crawler is the one the fetcher sends: obeying
|
||||
// robots rules written for a different name would be a lie.
|
||||
return crawl.New(&crawlFetcher{f: fetcher}, crawl.Config{
|
||||
UserAgent: ua,
|
||||
MaxRunes: cc.MaxRunes,
|
||||
})
|
||||
}
|
||||
|
||||
// onDemandCrawler returns a crawler for the answer path, or nil when on-demand
|
||||
// reading is off. The scheduled watches can be on while this is off: reading a
|
||||
// fixed list of pages on a timer and reading whatever URL is in an utterance are
|
||||
// different permissions, and the config keeps them separate.
|
||||
func onDemandCrawler(cfg *config.Config) *crawl.Crawler {
|
||||
if cfg.Crawl == nil || !cfg.Crawl.OnDemand {
|
||||
return nil
|
||||
}
|
||||
return newCrawler(cfg)
|
||||
}
|
||||
|
||||
// crawlWorker — ticker + watcher for the scheduled half.
|
||||
type crawlWorker struct {
|
||||
watcher *crawl.Watcher
|
||||
interval time.Duration
|
||||
}
|
||||
|
||||
// crawlTickInterval — how often the worker asks what is due. Per-watch cadence
|
||||
// is the watcher's business.
|
||||
const crawlTickInterval = 15 * time.Minute
|
||||
|
||||
// newCrawlWorker wires the scheduled crawls, or nil when nothing is watched.
|
||||
func newCrawlWorker(c *crawl.Crawler, api ipc.CoreAPI, emb router.Embedder, cfg *config.Config) *crawlWorker {
|
||||
if c == nil || cfg.Crawl == nil || len(cfg.Crawl.Watches) == 0 {
|
||||
return nil
|
||||
}
|
||||
watches := make([]crawl.WatchConfig, 0, len(cfg.Crawl.Watches))
|
||||
for _, w := range cfg.Crawl.Watches {
|
||||
watches = append(watches, crawl.WatchConfig{
|
||||
Name: w.Name,
|
||||
URL: w.URL,
|
||||
Interval: time.Duration(w.Interval),
|
||||
})
|
||||
}
|
||||
watcher := crawl.NewWatcher(c, watches, api, &factHashes{api: api},
|
||||
crawlEmbedder(emb), time.Duration(cfg.Crawl.Interval))
|
||||
if watcher == nil {
|
||||
log.Printf("crawl: configured but nothing watchable — scheduled crawls disabled")
|
||||
return nil
|
||||
}
|
||||
log.Printf("crawl: watching %d page(s), checking what is due every %s", len(watches), crawlTickInterval)
|
||||
return &crawlWorker{watcher: watcher, interval: crawlTickInterval}
|
||||
}
|
||||
|
||||
// run checks what is due until ctx is canceled. The first round runs
|
||||
// immediately; it writes notes only, so an early round startles nobody.
|
||||
func (w *crawlWorker) run(ctx context.Context) {
|
||||
w.watcher.CheckDue(ctx, time.Now())
|
||||
t := time.NewTicker(w.interval)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case now := <-t.C:
|
||||
w.watcher.CheckDue(ctx, now)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// crawlFetcher adapts webfetch to crawl.Fetcher, which is the seam that keeps
|
||||
// net/http out of the crawler package.
|
||||
type crawlFetcher struct{ f *webfetch.Fetcher }
|
||||
|
||||
func (a *crawlFetcher) Get(ctx context.Context, u string) (*crawl.Response, error) {
|
||||
resp, err := a.f.Get(ctx, u)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &crawl.Response{URL: resp.URL, ContentType: resp.ContentType, Body: resp.Body}, nil
|
||||
}
|
||||
|
||||
// factHashes stores each watch's last content hash as a config fact, so a
|
||||
// restart does not re-note an unchanged page. Same mechanism the feed reader
|
||||
// uses for its marks, and inspectable on /dash.
|
||||
type factHashes struct{ api ipc.CoreAPI }
|
||||
|
||||
func hashKey(name string) string { return "crawl:hash:" + name }
|
||||
|
||||
func (h *factHashes) LastHash(ctx context.Context, name string) (string, error) {
|
||||
f, err := h.api.LatestFact(ctx, hashKey(name))
|
||||
if err != nil {
|
||||
// No hash yet is not an error: the watcher treats "" as "never read".
|
||||
return "", nil
|
||||
}
|
||||
return f.Value, nil
|
||||
}
|
||||
|
||||
func (h *factHashes) SetHash(ctx context.Context, name, hash string) error {
|
||||
_, err := h.api.WriteFact(ctx, ipc.WriteFactReq{
|
||||
Ts: time.Now(),
|
||||
Kind: "config",
|
||||
Key: hashKey(name),
|
||||
Value: hash,
|
||||
Source: "poll:crawl",
|
||||
Confidence: 1.0,
|
||||
})
|
||||
return err
|
||||
}
|
||||
|
||||
// crawlEmbedder adapts router.Embedder for the watcher, embedding with
|
||||
// EmbedPassage (a page is text being searched FOR, and the e5 embedder is
|
||||
// asymmetric).
|
||||
func crawlEmbedder(emb router.Embedder) crawl.Embedder {
|
||||
if emb == nil {
|
||||
return nil
|
||||
}
|
||||
return passageEmbedder{emb}
|
||||
}
|
||||
@@ -0,0 +1,186 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/crawl"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
)
|
||||
|
||||
// The default config reads nothing. This is the whole "off unless configured"
|
||||
// contract for the crawler, asserted at the wiring level rather than trusted.
|
||||
func TestCrawlOffByDefault(t *testing.T) {
|
||||
cfg := &config.Config{}
|
||||
if c := newCrawler(cfg); c != nil {
|
||||
t.Error("newCrawler with no crawl block returned a crawler")
|
||||
}
|
||||
if c := onDemandCrawler(cfg); c != nil {
|
||||
t.Error("onDemandCrawler with no crawl block returned a crawler")
|
||||
}
|
||||
if w := newCrawlWorker(nil, nil, nil, cfg); w != nil {
|
||||
t.Error("newCrawlWorker with no crawl block returned a worker")
|
||||
}
|
||||
// Watches configured but on_demand off ⇒ the answer path still reads
|
||||
// nothing: a timer over a fixed list is not permission for arbitrary URLs.
|
||||
withWatch := &config.Config{Crawl: &config.CrawlConfig{
|
||||
Watches: []config.CrawlWatchConfig{{Name: "p", URL: "https://example.org/p"}},
|
||||
}}
|
||||
if c := onDemandCrawler(withWatch); c != nil {
|
||||
t.Error("onDemandCrawler honoured a watch list as on-demand permission")
|
||||
}
|
||||
if c := newCrawler(withWatch); c == nil {
|
||||
t.Error("newCrawler returned nil for a configured watch")
|
||||
}
|
||||
}
|
||||
|
||||
// The wired fetcher must refuse a private address, because the crawler on this
|
||||
// box sits one hop from the whole homelab. Same guard the webfetch tests cover;
|
||||
// this asserts the daemon actually wires it.
|
||||
func TestCrawlerRefusesPrivateAddress(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "text/html")
|
||||
w.Write([]byte("<html><body>secret</body></html>"))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
c := newCrawler(&config.Config{Crawl: &config.CrawlConfig{OnDemand: true}})
|
||||
if c == nil {
|
||||
t.Fatal("newCrawler returned nil for an on-demand config")
|
||||
}
|
||||
if _, err := c.Page(context.Background(), srv.URL); err == nil {
|
||||
t.Fatalf("reading %s succeeded; a loopback address must be refused", srv.URL)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFactHashesRoundTrip(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
st := newTestStore(t)
|
||||
h := &factHashes{api: ipc.NewStoreAPI(st)}
|
||||
|
||||
got, err := h.LastHash(ctx, "page")
|
||||
if err != nil {
|
||||
t.Fatalf("LastHash on a fresh store: %v", err)
|
||||
}
|
||||
if got != "" {
|
||||
t.Errorf("LastHash = %q, want empty for a never-read page", got)
|
||||
}
|
||||
if err := h.SetHash(ctx, "page", "deadbeef"); err != nil {
|
||||
t.Fatalf("SetHash: %v", err)
|
||||
}
|
||||
got, err = h.LastHash(ctx, "page")
|
||||
if err != nil {
|
||||
t.Fatalf("LastHash: %v", err)
|
||||
}
|
||||
if got != "deadbeef" {
|
||||
t.Errorf("LastHash = %q, want deadbeef", got)
|
||||
}
|
||||
if key := hashKey("page"); key != "crawl:hash:page" {
|
||||
t.Errorf("hashKey = %q", key)
|
||||
}
|
||||
}
|
||||
|
||||
// stubCrawlFetcher serves one fixed page to every URL, so queryWeb can be
|
||||
// exercised without a network or an allowlist.
|
||||
type stubCrawlFetcher struct{ body, ctype string }
|
||||
|
||||
func (s *stubCrawlFetcher) Get(_ context.Context, u string) (*crawl.Response, error) {
|
||||
ct := s.ctype
|
||||
if ct == "" {
|
||||
ct = "text/html"
|
||||
}
|
||||
if strings.HasSuffix(u, "/robots.txt") {
|
||||
return &crawl.Response{URL: u, ContentType: "text/plain", Body: []byte("")}, nil
|
||||
}
|
||||
return &crawl.Response{URL: u, ContentType: ct, Body: []byte(s.body)}, nil
|
||||
}
|
||||
|
||||
func buildWebHandler(c *crawl.Crawler) *reactiveHandler {
|
||||
return &reactiveHandler{
|
||||
replier: voice.NewStubReplier(),
|
||||
phraser: phraser.NewStub(),
|
||||
crawler: c,
|
||||
}
|
||||
}
|
||||
|
||||
func askWeb(h *reactiveHandler, q string) (string, bool) {
|
||||
return h.queryWeb(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Intent: router.IntentQuery, Utterance: q},
|
||||
})
|
||||
}
|
||||
|
||||
func TestQueryWebPassesWithoutAURL(t *testing.T) {
|
||||
h := buildWebHandler(crawl.New(&stubCrawlFetcher{body: "<html><body>x</body></html>"}, crawl.Config{}))
|
||||
if reply, ok := askWeb(h, "почему небо синее?"); ok {
|
||||
t.Errorf("the web source claimed a question with no URL: %q", reply)
|
||||
}
|
||||
}
|
||||
|
||||
// Not configured is said out loud rather than falling through, so a small model
|
||||
// never invents a page's contents from its URL.
|
||||
func TestQueryWebSaysWhenNotConfigured(t *testing.T) {
|
||||
h := buildWebHandler(nil)
|
||||
reply, ok := askWeb(h, "посмотри https://example.org/page")
|
||||
if !ok {
|
||||
t.Fatal("the web source did not claim a question with a URL")
|
||||
}
|
||||
if !strings.Contains(reply, "не настроено") {
|
||||
t.Errorf("reply = %q, want the not-configured answer", reply)
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueryWebReadsThePage(t *testing.T) {
|
||||
h := buildWebHandler(crawl.New(&stubCrawlFetcher{
|
||||
body: "<html><head><title>Заголовок</title></head><body><p>текст страницы</p></body></html>",
|
||||
}, crawl.Config{}))
|
||||
reply, ok := askWeb(h, "посмотри https://example.org/page — что там?")
|
||||
if !ok {
|
||||
t.Fatal("the web source did not claim a question with a URL")
|
||||
}
|
||||
if !strings.Contains(reply, "текст страницы") {
|
||||
t.Errorf("reply = %q, want the page text read back", reply)
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueryWebRefusesNonHTML(t *testing.T) {
|
||||
h := buildWebHandler(crawl.New(&stubCrawlFetcher{
|
||||
body: "\x00\x01binary", ctype: "application/octet-stream",
|
||||
}, crawl.Config{}))
|
||||
reply, ok := askWeb(h, "почитай https://example.org/blob.bin")
|
||||
if !ok {
|
||||
t.Fatal("the web source did not claim a question with a URL")
|
||||
}
|
||||
if !strings.Contains(reply, "не получилось") {
|
||||
t.Errorf("reply = %q, want the read-failed answer", reply)
|
||||
}
|
||||
}
|
||||
|
||||
// robots.txt is honoured on the answer path too, and she says so instead of
|
||||
// reporting a generic failure.
|
||||
func TestQueryWebObeysRobots(t *testing.T) {
|
||||
h := buildWebHandler(crawl.New(&robotsDenyFetcher{}, crawl.Config{}))
|
||||
reply, ok := askWeb(h, "посмотри https://example.org/private")
|
||||
if !ok {
|
||||
t.Fatal("the web source did not claim a question with a URL")
|
||||
}
|
||||
if !strings.Contains(reply, "robots.txt") {
|
||||
t.Errorf("reply = %q, want the robots answer", reply)
|
||||
}
|
||||
}
|
||||
|
||||
type robotsDenyFetcher struct{}
|
||||
|
||||
func (robotsDenyFetcher) Get(_ context.Context, u string) (*crawl.Response, error) {
|
||||
if strings.HasSuffix(u, "/robots.txt") {
|
||||
return &crawl.Response{URL: u, ContentType: "text/plain",
|
||||
Body: []byte("User-agent: *\nDisallow: /private\n")}, nil
|
||||
}
|
||||
return &crawl.Response{URL: u, ContentType: "text/html", Body: []byte("<html>nope</html>")}, nil
|
||||
}
|
||||
@@ -171,6 +171,7 @@ func run(args []string) error {
|
||||
factWorker *factEnrichmentWorker
|
||||
evalWorker *memoryEvalWorker // nil ⇒ memory evaluation off (the default)
|
||||
feedWkr *feedWorker // nil ⇒ no feed is read (the default)
|
||||
crawlWkr *crawlWorker // nil ⇒ no page is watched (the default)
|
||||
)
|
||||
|
||||
if !locked {
|
||||
@@ -267,6 +268,7 @@ func run(args []string) error {
|
||||
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
||||
evalWorker = newMemoryEvalWorker(st, phr, cfg)
|
||||
feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
|
||||
crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
|
||||
|
||||
coreAPI = &daemonAPI{
|
||||
CoreAPI: ipc.NewStoreAPI(st),
|
||||
@@ -454,6 +456,7 @@ func run(args []string) error {
|
||||
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
||||
evalWorker = newMemoryEvalWorker(st, phr, cfg)
|
||||
feedWkr = newFeedWorker(ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
|
||||
crawlWkr = newCrawlWorker(newCrawler(cfg), ipc.NewStoreAPI(st), embedderOf(voiceW), cfg)
|
||||
|
||||
// Swap the CoreAPI from the locked placeholder to the real store adapter.
|
||||
newAPI := &daemonAPI{
|
||||
@@ -506,6 +509,13 @@ func run(args []string) error {
|
||||
}()
|
||||
}
|
||||
|
||||
// Start the watched-page crawls (nil unless configured).
|
||||
if crawlWkr != nil {
|
||||
go func() {
|
||||
crawlWkr.run(ctx)
|
||||
}()
|
||||
}
|
||||
|
||||
dl.unlock()
|
||||
log.Printf("mavend: unlocked via passkey assertion")
|
||||
return nil
|
||||
@@ -558,6 +568,13 @@ func run(args []string) error {
|
||||
feedWkr.run(ctx)
|
||||
}()
|
||||
}
|
||||
if crawlWkr != nil {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
crawlWkr.run(ctx)
|
||||
}()
|
||||
}
|
||||
}
|
||||
|
||||
<-ctx.Done()
|
||||
|
||||
@@ -52,6 +52,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
"github.com/kami/maven/internal/crawl"
|
||||
"github.com/kami/maven/internal/dialogue"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/memory"
|
||||
@@ -82,6 +83,10 @@ type reactiveHandler struct {
|
||||
replier voice.Replier
|
||||
now func() time.Time
|
||||
|
||||
// crawler reads a web page he names out loud (queryWeb). nil ⇒ on-demand
|
||||
// page reading is off, which is the default: no `crawl` block, no fetch.
|
||||
crawler *crawl.Crawler
|
||||
|
||||
// feedsOn — whether any RSS feed is configured (config.Feeds). It changes
|
||||
// only what she SAYS when asked and nothing is there: "ленты не настроены"
|
||||
// instead of "ничего нового", which are different truths.
|
||||
|
||||
+14
-11
@@ -201,17 +201,20 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
|
||||
// ----- the handler (the reactive path; closes over stt / tts / router / coreAPI / memory) -----
|
||||
h := &reactiveHandler{
|
||||
stt: transcriber,
|
||||
tts: synthesizer,
|
||||
router: rtr,
|
||||
embedder: emb,
|
||||
api: coreAPI,
|
||||
tools: exec,
|
||||
matcher: matcher,
|
||||
replier: replier,
|
||||
phraser: phr,
|
||||
now: time.Now,
|
||||
feedsOn: cfg.Feeds != nil,
|
||||
stt: transcriber,
|
||||
tts: synthesizer,
|
||||
router: rtr,
|
||||
embedder: emb,
|
||||
api: coreAPI,
|
||||
tools: exec,
|
||||
matcher: matcher,
|
||||
replier: replier,
|
||||
phraser: phr,
|
||||
now: time.Now,
|
||||
feedsOn: cfg.Feeds != nil,
|
||||
// nil unless `crawl.on_demand` is on: reading a page he names is a
|
||||
// capability, and capabilities are off unless configured.
|
||||
crawler: onDemandCrawler(cfg),
|
||||
weatherProvider: weatherProvider,
|
||||
weatherLocation: weatherLocation,
|
||||
memStore: memStore,
|
||||
|
||||
Reference in New Issue
Block a user