// Package crawl reads a web page: fetch, robots check, HTML to text. // // It is the last LOCAL-FIRST step, and the ordering is the whole design. // "Never phones home" is deprecated, but what replaced it puts local sources // first. Where this actually sits in querySources (cmd/mavend/actions_query.go): // after his memory and after his notes, and BEFORE the model answers from what // it remembers. Not after the model, which is what this comment used to claim. // // That position is deliberate. A fetch only happens where he named a URL out // loud, and a named URL is an instruction, not a guess; letting a 1.7B answer // about a page it cannot read is how a small model invents contents. Kiwix // (internal/kiwix) is not in the chain yet, so nothing here describes it. // // A page's text is written by whoever owns the page. It reaches PhraseQuery as // context beside his question, and for a watch it becomes a note. It cannot // reach a tool or an act — the query path executes nothing — but it can steer // what she says, which is the same trust level as a mail body and lower than // anything he said himself. // // What never leaves the box: his notes, his facts, the persona block, the // conversation history. Only the URL is requested and, for the on-demand path, // only because he said it out loud. Nothing here reads the store. // // The limits are not in this package — they are in internal/webfetch, which is // the only way anything here touches a socket: http(s) only, host allow/deny, // private-address refusal, size cap, redirect cap, per-host rate limit. What // this package adds is politeness (robots.txt) and dedup. package crawl import ( "context" "crypto/sha256" "encoding/hex" "errors" "fmt" "net/url" "strings" "sync" "time" ) // Errors callers distinguish. var ( ErrRobots = errors.New("crawl: robots.txt disallows this path") ErrNotHTML = errors.New("crawl: response is not html or text") // ErrFetchRefused — the fetcher would not go: a denied host, a private // address, a scheme that is not http(s). The adapter that owns both // packages (cmd/mavend/crawls.go) maps webfetch's sentinels onto this one, // so this package tells "off limits" from "no robots.txt here" without // importing webfetch and without matching on message text. ErrFetchRefused = errors.New("crawl: the fetcher refused this url") // ErrFetchStatus — the server answered, badly (5xx, and anything else // non-2xx). Separate from ErrFetchRefused because robots treats them // differently: a broken server is not permission to crawl. ErrFetchStatus = errors.New("crawl: the server answered with an error status") ) // Fetcher is the guarded HTTP door (internal/webfetch adapted by the daemon). An // interface so this package constructs no http.Client of its own and can be // tested without a network. type Fetcher interface { Get(ctx context.Context, url string) (*Response, error) } // Response is the minimum a crawl needs from a fetch. type Response struct { URL string ContentType string Body []byte } // Config — crawler knobs. type Config struct { // UserAgent is the name matched against robots.txt groups. It must be the // same string the fetcher sends, or Maven would be claiming one identity // and obeying the rules for another. UserAgent string // MaxRunes caps extracted text. 0 ⇒ DefaultMaxRunes. MaxRunes int // RobotsTTL — how long a parsed robots.txt is trusted. 0 ⇒ 1h. RobotsTTL time.Duration // Now is injectable for tests. nil ⇒ time.Now. Now func() time.Time } // Crawler fetches and extracts pages. Safe for concurrent use. type Crawler struct { fetch Fetcher cfg Config robots *robotsCache // mu guards last, the per-host time of the previous fetch. It is what makes // Crawl-delay real: the fetcher's own limiter is a flat one request per // host per second and knows nothing about what a site asked for. mu sync.Mutex last map[string]time.Time } // robotsTimeout — the robots fetch gets its own, shorter deadline. It shares the // caller's budget with the page fetch (30s for the on-demand path, against a 20s // default fetch timeout each), so a slow robots.txt used to eat the page's half // and he heard "не получилось прочитать страницу" about a site that was fine. const robotsTimeout = 8 * time.Second // New builds a crawler. Returns nil when there is no fetcher, which is how the // daemon expresses "crawling is off unless configured". func New(fetch Fetcher, cfg Config) *Crawler { if fetch == nil { return nil } if cfg.UserAgent == "" { cfg.UserAgent = "Maven" } if cfg.MaxRunes <= 0 { cfg.MaxRunes = DefaultMaxRunes } if cfg.RobotsTTL <= 0 { cfg.RobotsTTL = time.Hour } if cfg.Now == nil { cfg.Now = time.Now } return &Crawler{fetch: fetch, cfg: cfg, robots: newRobotsCache(cfg.RobotsTTL), last: map[string]time.Time{}} } // Page fetches rawURL and returns its text. It checks robots.txt first and // refuses a disallowed path with ErrRobots — there is no override. func (c *Crawler) Page(ctx context.Context, rawURL string) (Page, error) { u, err := url.Parse(strings.TrimSpace(rawURL)) if err != nil { return Page{}, fmt.Errorf("crawl: bad url %q: %w", rawURL, err) } rules, err := c.rulesFor(ctx, u) if err != nil { return Page{}, err } path := u.EscapedPath() if u.RawQuery != "" { path += "?" + u.RawQuery } if !rules.Allowed(path) { return Page{}, fmt.Errorf("%w: %s", ErrRobots, u.Path) } if err := c.waitCrawlDelay(ctx, u.Host, rules.Delay); err != nil { return Page{}, err } resp, err := c.fetch.Get(ctx, u.String()) c.markFetched(u.Host) if err != nil { return Page{}, err } // A PDF or an image is bytes Maven cannot read; saying so beats storing // binary garbage as a "note". ct := strings.ToLower(resp.ContentType) if ct != "" && !strings.Contains(ct, "html") && !strings.Contains(ct, "text/") && !strings.Contains(ct, "xml") && !strings.Contains(ct, "json") { return Page{}, fmt.Errorf("%w: %s", ErrNotHTML, resp.ContentType) } return Extract(resp.URL, resp.Body, c.cfg.MaxRunes), nil } // rulesFor consults robots.txt for u's host, reading it at most once per TTL. // // A robots.txt that is not there (404) means allow, per the standard. What does // NOT mean allow: a server that answered with an error. The standard asks for // the opposite there, and "the site is broken, so crawl it" is the wrong way to // resolve an unknown. func (c *Crawler) rulesFor(ctx context.Context, u *url.URL) (Rules, error) { host := u.Host now := c.cfg.Now() rules, ok := c.robots.get(host, now) if ok { return rules, nil } robotsURL := u.Scheme + "://" + host + "/robots.txt" rctx, cancel := context.WithTimeout(ctx, robotsTimeout) defer cancel() resp, err := c.fetch.Get(rctx, robotsURL) c.markFetched(host) switch { case err == nil: rules = ParseRobots(string(resp.Body), c.cfg.UserAgent) case errors.Is(err, ErrFetchRefused): // Not swallowed: if the fetcher says this host is denied or private, // the page fetch would fail the same way, and the real reason beats a // robots verdict we never got. return Rules{}, err case errors.Is(err, ErrFetchStatus) && isServerError(err): return Rules{}, fmt.Errorf("%w: robots.txt at %s could not be read", ErrFetchStatus, host) default: rules = Rules{} } c.robots.put(host, rules, now) return rules, nil } // waitCrawlDelay honours a Crawl-delay the site asked for. The fetcher's flat // one-per-host-per-second is the floor and cannot express "30 seconds"; without // this, deploy/README's claim that Crawl-delay is honoured was false. // // It waits on ctx, so a delay longer than the caller's budget fails the read // rather than blocking a turn. That is the honest outcome: a site that wants a // minute between requests is not a site to answer a voice question from. func (c *Crawler) waitCrawlDelay(ctx context.Context, host string, delay time.Duration) error { if delay <= 0 { return nil } c.mu.Lock() last, ok := c.last[host] c.mu.Unlock() if !ok { return nil } wait := delay - c.cfg.Now().Sub(last) if wait <= 0 { return nil } t := time.NewTimer(wait) defer t.Stop() select { case <-ctx.Done(): return fmt.Errorf("crawl: %s asks for %s between requests, longer than this read has: %w", host, delay, ctx.Err()) case <-t.C: return nil } } func (c *Crawler) markFetched(host string) { c.mu.Lock() c.last[host] = c.cfg.Now() c.mu.Unlock() } // isServerError — a 5xx rather than any other non-2xx. The adapter formats the // status into the message, which is the only place it survives. func isServerError(err error) bool { s := err.Error() for _, code := range []string{" 50", " 51", " 52", " 53"} { if strings.Contains(s, code) { return true } } return false } // Hash is the dedup key for a crawl result: the sha256 of the extracted text, // hex, first 16 chars. Text and not raw HTML, because a page whose only change // is a rotating ad slot or a CSRF token has not changed. func Hash(text string) string { sum := sha256.Sum256([]byte(strings.TrimSpace(text))) return hex.EncodeToString(sum[:])[:16] }