Files
Maven/internal/crawl/robots.go
T
kami 2c1b0eede0 Read a web page when he names one, and watch a few on a timer (#259)
The network fallback behind the local sources, off unless configured.

internal/crawl is pure: a stdlib robots.txt parser (group specificity,
wildcards, Crawl-delay, cached per host), HTML-to-plaintext extraction, and a
watcher that notes a watched page only when its text changed. It has no store
access and no net/http; cmd/mavend/crawls.go is the impure half.

Every limit is code and tested: the guarded fetcher from #258 enforces the host
allowlist/denylist, refuses private addresses in the dialer Control hook (so DNS
rebinding and each redirect hop are covered), caps size and redirects, times out,
and spaces requests per host. A robots.txt Disallow is refused with no override.

On demand, reading is a query source placed last in the chain, after his memory,
his notes, and the local Kiwix ZIMs once those are wired: no URL in the
utterance means no fetch, and only the URL ever leaves the box. Scheduled
watches write notes and announce nothing.

The vendored tree has no x/net/html, goquery or temoto/robotstxt, so the parsers
are stdlib. No new dependency.
2026-08-01 03:40:21 +04:00

212 lines
5.7 KiB
Go

package crawl
import (
"regexp"
"strings"
"sync"
"time"
)
// robots.txt, parsed the small way: no wildcards beyond the two the standard
// actually defines (`*` inside a path and `$` at the end), no sitemaps, no
// crawl-delay-per-agent gymnastics. A personal assistant reading a handful of
// pages does not need a spec-complete implementation; it needs to not be rude,
// and to be auditable in one sitting.
//
// Two rules worth stating because they are choices, not accidents:
//
// - a missing or unreadable robots.txt means ALLOW. That is what the standard
// says (404 ⇒ unrestricted), and the alternative would make a site that
// simply has no robots.txt unreadable;
// - an explicit Disallow means REFUSE, and Maven does not offer an override.
// There is no "but he asked me to" flag: the page is not read.
// Rules is a parsed robots.txt for one user-agent.
type Rules struct {
allow []string
disallow []string
// Delay is Crawl-delay in seconds when the group named one, 0 otherwise.
// The fetcher's own per-host rate limit is the floor; this can only make
// Maven slower, never faster.
Delay time.Duration
}
// ParseRobots reads robots.txt and returns the rules that apply to agent.
//
// Group selection follows the standard: the most specific matching group wins,
// which here means an exact user-agent match beats `*`. Lines that are neither
// are ignored rather than guessed at.
func ParseRobots(body string, agent string) Rules {
agent = strings.ToLower(agent)
type group struct {
agents []string
allow []string
disallow []string
delay time.Duration
}
var groups []group
var cur *group
// startNew tracks whether the next User-agent line opens a new group or
// joins the current one: consecutive User-agent lines share their rules.
startNew := true
for _, raw := range strings.Split(body, "\n") {
line := raw
if i := strings.IndexByte(line, '#'); i >= 0 {
line = line[:i]
}
line = strings.TrimSpace(line)
if line == "" {
continue
}
key, val, ok := strings.Cut(line, ":")
if !ok {
continue
}
key = strings.ToLower(strings.TrimSpace(key))
val = strings.TrimSpace(val)
switch key {
case "user-agent":
if startNew || cur == nil {
groups = append(groups, group{})
cur = &groups[len(groups)-1]
startNew = false
}
cur.agents = append(cur.agents, strings.ToLower(val))
case "disallow":
if cur == nil {
continue
}
startNew = true
// "Disallow:" with an empty value allows everything, and is not a
// path rule at all.
if val != "" {
cur.disallow = append(cur.disallow, val)
}
case "allow":
if cur == nil {
continue
}
startNew = true
if val != "" {
cur.allow = append(cur.allow, val)
}
case "crawl-delay":
if cur == nil {
continue
}
startNew = true
if d, err := time.ParseDuration(val + "s"); err == nil && d > 0 {
cur.delay = d
}
}
}
var star, exact *group
for i := range groups {
for _, a := range groups[i].agents {
if a == "*" && star == nil {
star = &groups[i]
}
// A robots.txt names "maven", we send "Maven/1.0 (…)": match on
// prefix, which is how every crawler reads this field.
if a != "*" && a != "" && strings.HasPrefix(agent, a) {
exact = &groups[i]
}
}
}
g := exact
if g == nil {
g = star
}
if g == nil {
return Rules{}
}
return Rules{allow: g.allow, disallow: g.disallow, Delay: g.delay}
}
// Allowed reports whether path may be fetched. Longest matching rule wins, and
// Allow beats Disallow at equal length — the standard's tie-break, and the one
// that makes "Disallow: /" plus "Allow: /public" mean what it looks like.
func (r Rules) Allowed(path string) bool {
if path == "" {
path = "/"
}
best, allowed := -1, true
for _, p := range r.disallow {
if n, ok := matchPath(p, path); ok && n > best {
best, allowed = n, false
}
}
for _, p := range r.allow {
if n, ok := matchPath(p, path); ok && n >= best {
best, allowed = n, true
}
}
return allowed
}
// matchPath applies a robots path pattern and returns the pattern's length as
// the specificity score. `*` matches any run of characters, `$` anchors the end.
// A pattern is a PREFIX match otherwise, which is what "Disallow: /admin" means.
func matchPath(pattern, path string) (int, bool) {
score := len(pattern)
re, err := robotsRegexp(pattern)
if err != nil {
return 0, false
}
return score, re.MatchString(path)
}
// robotsRegexp turns a robots path pattern into an anchored-at-the-start
// regexp. Everything but `*` and a trailing `$` is a literal, so the pattern is
// quoted first and the two metacharacters are put back afterwards.
func robotsRegexp(pattern string) (*regexp.Regexp, error) {
end := ""
if strings.HasSuffix(pattern, "$") {
pattern = strings.TrimSuffix(pattern, "$")
end = "$"
}
parts := strings.Split(pattern, "*")
for i, p := range parts {
parts[i] = regexp.QuoteMeta(p)
}
return regexp.Compile("^" + strings.Join(parts, ".*") + end)
}
// robotsCache holds parsed rules per host so a crawl of ten pages on one site
// reads robots.txt once. TTL because a site may change its mind, and a daemon
// that runs for weeks would otherwise never notice.
type robotsCache struct {
ttl time.Duration
mu sync.Mutex
m map[string]robotsEntry
}
type robotsEntry struct {
rules Rules
at time.Time
}
func newRobotsCache(ttl time.Duration) *robotsCache {
return &robotsCache{ttl: ttl, m: map[string]robotsEntry{}}
}
func (c *robotsCache) get(host string, now time.Time) (Rules, bool) {
c.mu.Lock()
defer c.mu.Unlock()
e, ok := c.m[host]
if !ok || now.Sub(e.at) > c.ttl {
return Rules{}, false
}
return e.rules, true
}
func (c *robotsCache) put(host string, r Rules, now time.Time) {
c.mu.Lock()
defer c.mu.Unlock()
c.m[host] = robotsEntry{rules: r, at: now}
}