Read a web page when he names one, and watch a few on a timer (#259)
The network fallback behind the local sources, off unless configured. internal/crawl is pure: a stdlib robots.txt parser (group specificity, wildcards, Crawl-delay, cached per host), HTML-to-plaintext extraction, and a watcher that notes a watched page only when its text changed. It has no store access and no net/http; cmd/mavend/crawls.go is the impure half. Every limit is code and tested: the guarded fetcher from #258 enforces the host allowlist/denylist, refuses private addresses in the dialer Control hook (so DNS rebinding and each redirect hop are covered), caps size and redirects, times out, and spaces requests per host. A robots.txt Disallow is refused with no override. On demand, reading is a query source placed last in the chain, after his memory, his notes, and the local Kiwix ZIMs once those are wired: no URL in the utterance means no fetch, and only the URL ever leaves the box. Scheduled watches write notes and announce nothing. The vendored tree has no x/net/html, goquery or temoto/robotstxt, so the parsers are stdlib. No new dependency.
This commit is contained in:
@@ -0,0 +1,211 @@
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// robots.txt, parsed the small way: no wildcards beyond the two the standard
|
||||
// actually defines (`*` inside a path and `$` at the end), no sitemaps, no
|
||||
// crawl-delay-per-agent gymnastics. A personal assistant reading a handful of
|
||||
// pages does not need a spec-complete implementation; it needs to not be rude,
|
||||
// and to be auditable in one sitting.
|
||||
//
|
||||
// Two rules worth stating because they are choices, not accidents:
|
||||
//
|
||||
// - a missing or unreadable robots.txt means ALLOW. That is what the standard
|
||||
// says (404 ⇒ unrestricted), and the alternative would make a site that
|
||||
// simply has no robots.txt unreadable;
|
||||
// - an explicit Disallow means REFUSE, and Maven does not offer an override.
|
||||
// There is no "but he asked me to" flag: the page is not read.
|
||||
|
||||
// Rules is a parsed robots.txt for one user-agent.
|
||||
type Rules struct {
|
||||
allow []string
|
||||
disallow []string
|
||||
// Delay is Crawl-delay in seconds when the group named one, 0 otherwise.
|
||||
// The fetcher's own per-host rate limit is the floor; this can only make
|
||||
// Maven slower, never faster.
|
||||
Delay time.Duration
|
||||
}
|
||||
|
||||
// ParseRobots reads robots.txt and returns the rules that apply to agent.
|
||||
//
|
||||
// Group selection follows the standard: the most specific matching group wins,
|
||||
// which here means an exact user-agent match beats `*`. Lines that are neither
|
||||
// are ignored rather than guessed at.
|
||||
func ParseRobots(body string, agent string) Rules {
|
||||
agent = strings.ToLower(agent)
|
||||
|
||||
type group struct {
|
||||
agents []string
|
||||
allow []string
|
||||
disallow []string
|
||||
delay time.Duration
|
||||
}
|
||||
var groups []group
|
||||
var cur *group
|
||||
// startNew tracks whether the next User-agent line opens a new group or
|
||||
// joins the current one: consecutive User-agent lines share their rules.
|
||||
startNew := true
|
||||
|
||||
for _, raw := range strings.Split(body, "\n") {
|
||||
line := raw
|
||||
if i := strings.IndexByte(line, '#'); i >= 0 {
|
||||
line = line[:i]
|
||||
}
|
||||
line = strings.TrimSpace(line)
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
key, val, ok := strings.Cut(line, ":")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
key = strings.ToLower(strings.TrimSpace(key))
|
||||
val = strings.TrimSpace(val)
|
||||
|
||||
switch key {
|
||||
case "user-agent":
|
||||
if startNew || cur == nil {
|
||||
groups = append(groups, group{})
|
||||
cur = &groups[len(groups)-1]
|
||||
startNew = false
|
||||
}
|
||||
cur.agents = append(cur.agents, strings.ToLower(val))
|
||||
case "disallow":
|
||||
if cur == nil {
|
||||
continue
|
||||
}
|
||||
startNew = true
|
||||
// "Disallow:" with an empty value allows everything, and is not a
|
||||
// path rule at all.
|
||||
if val != "" {
|
||||
cur.disallow = append(cur.disallow, val)
|
||||
}
|
||||
case "allow":
|
||||
if cur == nil {
|
||||
continue
|
||||
}
|
||||
startNew = true
|
||||
if val != "" {
|
||||
cur.allow = append(cur.allow, val)
|
||||
}
|
||||
case "crawl-delay":
|
||||
if cur == nil {
|
||||
continue
|
||||
}
|
||||
startNew = true
|
||||
if d, err := time.ParseDuration(val + "s"); err == nil && d > 0 {
|
||||
cur.delay = d
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var star, exact *group
|
||||
for i := range groups {
|
||||
for _, a := range groups[i].agents {
|
||||
if a == "*" && star == nil {
|
||||
star = &groups[i]
|
||||
}
|
||||
// A robots.txt names "maven", we send "Maven/1.0 (…)": match on
|
||||
// prefix, which is how every crawler reads this field.
|
||||
if a != "*" && a != "" && strings.HasPrefix(agent, a) {
|
||||
exact = &groups[i]
|
||||
}
|
||||
}
|
||||
}
|
||||
g := exact
|
||||
if g == nil {
|
||||
g = star
|
||||
}
|
||||
if g == nil {
|
||||
return Rules{}
|
||||
}
|
||||
return Rules{allow: g.allow, disallow: g.disallow, Delay: g.delay}
|
||||
}
|
||||
|
||||
// Allowed reports whether path may be fetched. Longest matching rule wins, and
|
||||
// Allow beats Disallow at equal length — the standard's tie-break, and the one
|
||||
// that makes "Disallow: /" plus "Allow: /public" mean what it looks like.
|
||||
func (r Rules) Allowed(path string) bool {
|
||||
if path == "" {
|
||||
path = "/"
|
||||
}
|
||||
best, allowed := -1, true
|
||||
for _, p := range r.disallow {
|
||||
if n, ok := matchPath(p, path); ok && n > best {
|
||||
best, allowed = n, false
|
||||
}
|
||||
}
|
||||
for _, p := range r.allow {
|
||||
if n, ok := matchPath(p, path); ok && n >= best {
|
||||
best, allowed = n, true
|
||||
}
|
||||
}
|
||||
return allowed
|
||||
}
|
||||
|
||||
// matchPath applies a robots path pattern and returns the pattern's length as
|
||||
// the specificity score. `*` matches any run of characters, `$` anchors the end.
|
||||
// A pattern is a PREFIX match otherwise, which is what "Disallow: /admin" means.
|
||||
func matchPath(pattern, path string) (int, bool) {
|
||||
score := len(pattern)
|
||||
re, err := robotsRegexp(pattern)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
return score, re.MatchString(path)
|
||||
}
|
||||
|
||||
// robotsRegexp turns a robots path pattern into an anchored-at-the-start
|
||||
// regexp. Everything but `*` and a trailing `$` is a literal, so the pattern is
|
||||
// quoted first and the two metacharacters are put back afterwards.
|
||||
func robotsRegexp(pattern string) (*regexp.Regexp, error) {
|
||||
end := ""
|
||||
if strings.HasSuffix(pattern, "$") {
|
||||
pattern = strings.TrimSuffix(pattern, "$")
|
||||
end = "$"
|
||||
}
|
||||
parts := strings.Split(pattern, "*")
|
||||
for i, p := range parts {
|
||||
parts[i] = regexp.QuoteMeta(p)
|
||||
}
|
||||
return regexp.Compile("^" + strings.Join(parts, ".*") + end)
|
||||
}
|
||||
|
||||
// robotsCache holds parsed rules per host so a crawl of ten pages on one site
|
||||
// reads robots.txt once. TTL because a site may change its mind, and a daemon
|
||||
// that runs for weeks would otherwise never notice.
|
||||
type robotsCache struct {
|
||||
ttl time.Duration
|
||||
mu sync.Mutex
|
||||
m map[string]robotsEntry
|
||||
}
|
||||
|
||||
type robotsEntry struct {
|
||||
rules Rules
|
||||
at time.Time
|
||||
}
|
||||
|
||||
func newRobotsCache(ttl time.Duration) *robotsCache {
|
||||
return &robotsCache{ttl: ttl, m: map[string]robotsEntry{}}
|
||||
}
|
||||
|
||||
func (c *robotsCache) get(host string, now time.Time) (Rules, bool) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
e, ok := c.m[host]
|
||||
if !ok || now.Sub(e.at) > c.ttl {
|
||||
return Rules{}, false
|
||||
}
|
||||
return e.rules, true
|
||||
}
|
||||
|
||||
func (c *robotsCache) put(host string, r Rules, now time.Time) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.m[host] = robotsEntry{rules: r, at: now}
|
||||
}
|
||||
Reference in New Issue
Block a user