Read a web page when he names one, and watch a few on a timer (#259)
The network fallback behind the local sources, off unless configured. internal/crawl is pure: a stdlib robots.txt parser (group specificity, wildcards, Crawl-delay, cached per host), HTML-to-plaintext extraction, and a watcher that notes a watched page only when its text changed. It has no store access and no net/http; cmd/mavend/crawls.go is the impure half. Every limit is code and tested: the guarded fetcher from #258 enforces the host allowlist/denylist, refuses private addresses in the dialer Control hook (so DNS rebinding and each redirect hop are covered), caps size and redirects, times out, and spaces requests per host. A robots.txt Disallow is refused with no override. On demand, reading is a query source placed last in the chain, after his memory, his notes, and the local Kiwix ZIMs once those are wired: no URL in the utterance means no fetch, and only the URL ever leaves the box. Scheduled watches write notes and announce nothing. The vendored tree has no x/net/html, goquery or temoto/robotstxt, so the parsers are stdlib. No new dependency.
This commit is contained in:
@@ -161,6 +161,10 @@ type Config struct {
|
||||
// the weather and telegram. See FeedsConfig.
|
||||
Feeds *FeedsConfig `json:"feeds,omitempty"`
|
||||
|
||||
// Crawl — reading a web page (Vikunja #259). nil / absent ⇒ Maven never
|
||||
// fetches a page: not on request, not on a schedule. See CrawlConfig.
|
||||
Crawl *CrawlConfig `json:"crawl,omitempty"`
|
||||
|
||||
// Praxis — the ecosystem attention-state service. When configured, maven
|
||||
// calls the Praxis HTTP tools API for attention listing and item lifecycle.
|
||||
// Maven never touches Praxis's database directly (ecosystem invariant: no
|
||||
@@ -454,6 +458,62 @@ type FeedSourceConfig struct {
|
||||
Exclude []string `json:"exclude,omitempty"` // drop items containing any of these
|
||||
}
|
||||
|
||||
// CrawlConfig — the web crawler (Vikunja #259, docs/plans/14-web-crawler.md).
|
||||
//
|
||||
// Absent ⇒ off, and off means no page is ever fetched. Present with neither
|
||||
// `on_demand` nor a `watches` entry is also off: there would be nothing to do.
|
||||
//
|
||||
// The crawler is the LAST place an answer is looked for, behind the model, his
|
||||
// own memory and the local Kiwix ZIMs. That ordering lives in the query-source
|
||||
// chain (cmd/mavend/actions_query.go), not here, but it is the reason this block
|
||||
// is small: it is a fallback, not a search engine.
|
||||
//
|
||||
// Only the URL leaves the box. His notes, facts, persona block and history are
|
||||
// never part of a request — the crawler package cannot even read the store.
|
||||
type CrawlConfig struct {
|
||||
// OnDemand — may he ask her to read a page he names out loud
|
||||
// ("посмотри https://… — что там пишут?"). false ⇒ the on-demand answer
|
||||
// source stays off and only the watches below run.
|
||||
OnDemand bool `json:"on_demand,omitempty"`
|
||||
|
||||
// Watches — pages re-read on a schedule. A page whose text changed is
|
||||
// written as a note (source "crawl:<name>"); nothing is announced.
|
||||
Watches []CrawlWatchConfig `json:"watches,omitempty"`
|
||||
|
||||
// Interval — default watch cadence. 0 ⇒ crawl.DefaultWatchInterval (6h).
|
||||
Interval Duration `json:"interval,omitempty"`
|
||||
|
||||
// AllowHosts — when set, the ONLY hosts the crawler may reach (subdomains
|
||||
// included). Watched pages' own hosts are added automatically. Setting this
|
||||
// is how "she may read the arch wiki and nothing else" is expressed.
|
||||
AllowHosts []string `json:"allow_hosts,omitempty"`
|
||||
|
||||
// DenyHosts — never reachable, checked first. Private addresses do not need
|
||||
// to be listed: they are refused unconditionally (see internal/webfetch).
|
||||
DenyHosts []string `json:"deny_hosts,omitempty"`
|
||||
|
||||
// UserAgent — sent on every request AND matched against robots.txt groups.
|
||||
// Empty ⇒ webfetch.DefaultUserAgent.
|
||||
UserAgent string `json:"user_agent,omitempty"`
|
||||
|
||||
// Timeout — per-request budget. 0 ⇒ webfetch.DefaultTimeout.
|
||||
Timeout Duration `json:"timeout,omitempty"`
|
||||
|
||||
// MaxBytes — response size cap. 0 ⇒ webfetch.DefaultMaxBytes (2 MiB).
|
||||
MaxBytes int64 `json:"max_bytes,omitempty"`
|
||||
|
||||
// MaxRunes — how much extracted text is kept. 0 ⇒ crawl.DefaultMaxRunes
|
||||
// (4000), which is what fits a 4096-token context alongside a prompt.
|
||||
MaxRunes int `json:"max_runes,omitempty"`
|
||||
}
|
||||
|
||||
// CrawlWatchConfig — one page kept an eye on.
|
||||
type CrawlWatchConfig struct {
|
||||
Name string `json:"name"` // note source is "crawl:<name>"
|
||||
URL string `json:"url"`
|
||||
Interval Duration `json:"interval,omitempty"` // 0 ⇒ CrawlConfig.Interval
|
||||
}
|
||||
|
||||
// MemoryEvalConfig — the background memory-evaluation loop (Vikunja #248).
|
||||
// Absent ⇒ off, like every other capability that costs something the owner did
|
||||
// not ask for. Each evaluation is a full LLM round-trip on the one resident
|
||||
@@ -697,6 +757,12 @@ func (c *Config) applyDefaults() {
|
||||
c.Feeds = nil
|
||||
}
|
||||
|
||||
// Same rule for the crawler: a block that neither answers on demand nor
|
||||
// watches anything has nothing to do, so it is normalised to "off".
|
||||
if c.Crawl != nil && !c.Crawl.OnDemand && len(c.Crawl.Watches) == 0 {
|
||||
c.Crawl = nil
|
||||
}
|
||||
|
||||
if c.Voice != nil {
|
||||
if c.Voice.RouterThreshold <= 0 {
|
||||
c.Voice.RouterThreshold = DefaultRouterThreshold
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
// Package crawl reads a web page: fetch, robots check, HTML to text.
|
||||
//
|
||||
// It is the LAST place Maven looks for an answer, and that ordering is the whole
|
||||
// design. "Never phones home" is deprecated, but what replaced it puts local
|
||||
// sources first: the resident model, then his own memory, then the Kiwix ZIMs on
|
||||
// the box (internal/kiwix), and only then the network. A local read costs
|
||||
// nothing and leaks nothing; a fetch costs a round-trip and puts a URL in
|
||||
// someone's access log. So this package exists to be the fallback, not the
|
||||
// front door — see the querySources chain in cmd/mavend/actions_query.go for
|
||||
// where it actually sits.
|
||||
//
|
||||
// What never leaves the box: his notes, his facts, the persona block, the
|
||||
// conversation history. Only the URL is requested and, for the on-demand path,
|
||||
// only because he said it out loud. Nothing here reads the store.
|
||||
//
|
||||
// The limits are not in this package — they are in internal/webfetch, which is
|
||||
// the only way anything here touches a socket: http(s) only, host allow/deny,
|
||||
// private-address refusal, size cap, redirect cap, per-host rate limit. What
|
||||
// this package adds is politeness (robots.txt) and dedup.
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Errors callers distinguish.
|
||||
var (
|
||||
ErrRobots = errors.New("crawl: robots.txt disallows this path")
|
||||
ErrNotHTML = errors.New("crawl: response is not html or text")
|
||||
)
|
||||
|
||||
// Fetcher is the guarded HTTP door (internal/webfetch adapted by the daemon). An
|
||||
// interface so this package constructs no http.Client of its own and can be
|
||||
// tested without a network.
|
||||
type Fetcher interface {
|
||||
Get(ctx context.Context, url string) (*Response, error)
|
||||
}
|
||||
|
||||
// Response is the minimum a crawl needs from a fetch.
|
||||
type Response struct {
|
||||
URL string
|
||||
ContentType string
|
||||
Body []byte
|
||||
}
|
||||
|
||||
// Config — crawler knobs.
|
||||
type Config struct {
|
||||
// UserAgent is the name matched against robots.txt groups. It must be the
|
||||
// same string the fetcher sends, or Maven would be claiming one identity
|
||||
// and obeying the rules for another.
|
||||
UserAgent string
|
||||
// MaxRunes caps extracted text. 0 ⇒ DefaultMaxRunes.
|
||||
MaxRunes int
|
||||
// RobotsTTL — how long a parsed robots.txt is trusted. 0 ⇒ 1h.
|
||||
RobotsTTL time.Duration
|
||||
// Now is injectable for tests. nil ⇒ time.Now.
|
||||
Now func() time.Time
|
||||
}
|
||||
|
||||
// Crawler fetches and extracts pages. Safe for concurrent use.
|
||||
type Crawler struct {
|
||||
fetch Fetcher
|
||||
cfg Config
|
||||
robots *robotsCache
|
||||
}
|
||||
|
||||
// New builds a crawler. Returns nil when there is no fetcher, which is how the
|
||||
// daemon expresses "crawling is off unless configured".
|
||||
func New(fetch Fetcher, cfg Config) *Crawler {
|
||||
if fetch == nil {
|
||||
return nil
|
||||
}
|
||||
if cfg.UserAgent == "" {
|
||||
cfg.UserAgent = "Maven"
|
||||
}
|
||||
if cfg.MaxRunes <= 0 {
|
||||
cfg.MaxRunes = DefaultMaxRunes
|
||||
}
|
||||
if cfg.RobotsTTL <= 0 {
|
||||
cfg.RobotsTTL = time.Hour
|
||||
}
|
||||
if cfg.Now == nil {
|
||||
cfg.Now = time.Now
|
||||
}
|
||||
return &Crawler{fetch: fetch, cfg: cfg, robots: newRobotsCache(cfg.RobotsTTL)}
|
||||
}
|
||||
|
||||
// Page fetches rawURL and returns its text. It checks robots.txt first and
|
||||
// refuses a disallowed path with ErrRobots — there is no override.
|
||||
func (c *Crawler) Page(ctx context.Context, rawURL string) (Page, error) {
|
||||
u, err := url.Parse(strings.TrimSpace(rawURL))
|
||||
if err != nil {
|
||||
return Page{}, fmt.Errorf("crawl: bad url %q: %w", rawURL, err)
|
||||
}
|
||||
ok, err := c.allowed(ctx, u)
|
||||
if err != nil {
|
||||
return Page{}, err
|
||||
}
|
||||
if !ok {
|
||||
return Page{}, fmt.Errorf("%w: %s", ErrRobots, u.Path)
|
||||
}
|
||||
resp, err := c.fetch.Get(ctx, u.String())
|
||||
if err != nil {
|
||||
return Page{}, err
|
||||
}
|
||||
// A PDF or an image is bytes Maven cannot read; saying so beats storing
|
||||
// binary garbage as a "note".
|
||||
ct := strings.ToLower(resp.ContentType)
|
||||
if ct != "" && !strings.Contains(ct, "html") && !strings.Contains(ct, "text/") &&
|
||||
!strings.Contains(ct, "xml") && !strings.Contains(ct, "json") {
|
||||
return Page{}, fmt.Errorf("%w: %s", ErrNotHTML, resp.ContentType)
|
||||
}
|
||||
return Extract(resp.URL, resp.Body, c.cfg.MaxRunes), nil
|
||||
}
|
||||
|
||||
// allowed consults robots.txt for u's host, reading it at most once per TTL.
|
||||
//
|
||||
// A robots.txt that cannot be fetched (404, a timeout, a blocked host) means
|
||||
// allow, per the standard. The one thing that is NOT fail-open is an explicit
|
||||
// Disallow.
|
||||
func (c *Crawler) allowed(ctx context.Context, u *url.URL) (bool, error) {
|
||||
host := u.Host
|
||||
now := c.cfg.Now()
|
||||
rules, ok := c.robots.get(host, now)
|
||||
if !ok {
|
||||
robotsURL := u.Scheme + "://" + host + "/robots.txt"
|
||||
resp, err := c.fetch.Get(ctx, robotsURL)
|
||||
switch {
|
||||
case err != nil:
|
||||
// Note what is NOT swallowed: a refusal from the guarded fetcher.
|
||||
// If webfetch says this host is denied or private, the page fetch
|
||||
// would fail the same way, and reporting the real reason beats
|
||||
// reporting a robots verdict we never got.
|
||||
if isFatalFetchError(err) {
|
||||
return false, err
|
||||
}
|
||||
rules = Rules{}
|
||||
default:
|
||||
rules = ParseRobots(string(resp.Body), c.cfg.UserAgent)
|
||||
}
|
||||
c.robots.put(host, rules, now)
|
||||
}
|
||||
path := u.EscapedPath()
|
||||
if u.RawQuery != "" {
|
||||
path += "?" + u.RawQuery
|
||||
}
|
||||
return rules.Allowed(path), nil
|
||||
}
|
||||
|
||||
// isFatalFetchError — a fetch failure that means "this host is off limits"
|
||||
// rather than "there is no robots.txt here". The sentinel set is webfetch's, but
|
||||
// this package must not import it (the interface exists precisely so it does
|
||||
// not), so the check is on the message. Ugly and honest: the alternative is a
|
||||
// dependency inversion for two strings.
|
||||
func isFatalFetchError(err error) bool {
|
||||
s := err.Error()
|
||||
return strings.Contains(s, "not allowed") || strings.Contains(s, "private address") ||
|
||||
strings.Contains(s, "only http and https")
|
||||
}
|
||||
|
||||
// Hash is the dedup key for a crawl result: the sha256 of the extracted text,
|
||||
// hex, first 16 chars. Text and not raw HTML, because a page whose only change
|
||||
// is a rotating ad slot or a CSRF token has not changed.
|
||||
func Hash(text string) string {
|
||||
sum := sha256.Sum256([]byte(strings.TrimSpace(text)))
|
||||
return hex.EncodeToString(sum[:])[:16]
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// fakeFetcher serves canned pages by URL and counts requests, so a test can
|
||||
// assert that robots.txt was read once and that a refusal never reached the page.
|
||||
type fakeFetcher struct {
|
||||
pages map[string]Response
|
||||
err error
|
||||
calls []string
|
||||
}
|
||||
|
||||
func (f *fakeFetcher) Get(_ context.Context, u string) (*Response, error) {
|
||||
f.calls = append(f.calls, u)
|
||||
if f.err != nil {
|
||||
return nil, f.err
|
||||
}
|
||||
r, ok := f.pages[u]
|
||||
if !ok {
|
||||
return nil, errors.New("http 404")
|
||||
}
|
||||
if r.URL == "" {
|
||||
r.URL = u
|
||||
}
|
||||
if r.ContentType == "" {
|
||||
r.ContentType = "text/html; charset=utf-8"
|
||||
}
|
||||
return &r, nil
|
||||
}
|
||||
|
||||
const htmlPage = `<html><head><title>Почему небо синее</title>
|
||||
<style>body{color:red}</style><script>track()</script></head>
|
||||
<body><nav>меню</nav><h1>Небо</h1>
|
||||
<p>Свет рассеивается на молекулах воздуха.</p>
|
||||
<p>Короткие волны рассеиваются сильнее.</p>
|
||||
<footer>© 2026</footer></body></html>`
|
||||
|
||||
func newTestCrawler(f *fakeFetcher) *Crawler {
|
||||
return New(f, Config{UserAgent: "Maven/1.0", Now: func() time.Time { return time.Unix(0, 0) }})
|
||||
}
|
||||
|
||||
func TestPageExtractsText(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{
|
||||
"https://example.org/sky": {Body: []byte(htmlPage)},
|
||||
}}
|
||||
page, err := newTestCrawler(f).Page(context.Background(), "https://example.org/sky")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if page.Title != "Почему небо синее" {
|
||||
t.Errorf("title = %q", page.Title)
|
||||
}
|
||||
if !strings.Contains(page.Text, "Свет рассеивается") {
|
||||
t.Errorf("body text missing: %q", page.Text)
|
||||
}
|
||||
for _, junk := range []string{"track()", "color:red", "меню", "© 2026"} {
|
||||
if strings.Contains(page.Text, junk) {
|
||||
t.Errorf("%q survived extraction: %q", junk, page.Text)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRobotsIsCheckedAndObeyed(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{
|
||||
"https://example.org/robots.txt": {Body: []byte("User-agent: *\nDisallow: /secret\n"), ContentType: "text/plain"},
|
||||
"https://example.org/secret/x": {Body: []byte(htmlPage)},
|
||||
"https://example.org/open": {Body: []byte(htmlPage)},
|
||||
}}
|
||||
c := newTestCrawler(f)
|
||||
if _, err := c.Page(context.Background(), "https://example.org/secret/x"); !errors.Is(err, ErrRobots) {
|
||||
t.Fatalf("error = %v, want ErrRobots", err)
|
||||
}
|
||||
for _, u := range f.calls {
|
||||
if strings.Contains(u, "/secret") {
|
||||
t.Fatal("the disallowed page was fetched anyway")
|
||||
}
|
||||
}
|
||||
if _, err := c.Page(context.Background(), "https://example.org/open"); err != nil {
|
||||
t.Fatalf("allowed page: %v", err)
|
||||
}
|
||||
// robots.txt was read once for the host, not once per page.
|
||||
robotsReads := 0
|
||||
for _, u := range f.calls {
|
||||
if strings.HasSuffix(u, "/robots.txt") {
|
||||
robotsReads++
|
||||
}
|
||||
}
|
||||
if robotsReads != 1 {
|
||||
t.Fatalf("robots.txt read %d times, want 1", robotsReads)
|
||||
}
|
||||
}
|
||||
|
||||
// No robots.txt means allow — that is the standard, and the alternative makes
|
||||
// most of the web unreadable.
|
||||
func TestMissingRobotsAllows(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{
|
||||
"https://example.org/page": {Body: []byte(htmlPage)},
|
||||
}}
|
||||
if _, err := newTestCrawler(f).Page(context.Background(), "https://example.org/page"); err != nil {
|
||||
t.Fatalf("err = %v, want the page", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A refusal from the guarded fetcher must surface as itself, not be laundered
|
||||
// into "no robots.txt, go ahead".
|
||||
func TestFetcherRefusalIsNotSwallowed(t *testing.T) {
|
||||
f := &fakeFetcher{err: errors.New("webfetch: refusing to connect to a private address: 127.0.0.1")}
|
||||
_, err := newTestCrawler(f).Page(context.Background(), "http://127.0.0.1:9100/mcp")
|
||||
if err == nil || !strings.Contains(err.Error(), "private address") {
|
||||
t.Fatalf("error = %v, want the fetcher's refusal", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonTextIsRefused(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{
|
||||
"https://example.org/f.pdf": {Body: []byte("%PDF-1.7"), ContentType: "application/pdf"},
|
||||
}}
|
||||
if _, err := newTestCrawler(f).Page(context.Background(), "https://example.org/f.pdf"); !errors.Is(err, ErrNotHTML) {
|
||||
t.Fatalf("error = %v, want ErrNotHTML", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMaxRunesCapsText(t *testing.T) {
|
||||
long := "<html><body><p>" + strings.Repeat("привет ", 2000) + "</p></body></html>"
|
||||
f := &fakeFetcher{pages: map[string]Response{"https://example.org/l": {Body: []byte(long)}}}
|
||||
c := New(f, Config{MaxRunes: 50})
|
||||
page, err := c.Page(context.Background(), "https://example.org/l")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if n := len([]rune(page.Text)); n > 51 {
|
||||
t.Fatalf("text = %d runes, want the 50-rune cap", n)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewWithoutFetcherIsNil(t *testing.T) {
|
||||
if New(nil, Config{}) != nil {
|
||||
t.Fatal("a crawler with no fetcher must be nil — crawling is off unless configured")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHashIgnoresNothingButText(t *testing.T) {
|
||||
if Hash("a") == Hash("b") {
|
||||
t.Fatal("different text hashed the same")
|
||||
}
|
||||
if Hash(" same \n") != Hash("same") {
|
||||
t.Fatal("surrounding whitespace changed the hash")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,106 @@
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"html"
|
||||
"regexp"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// HTML → text, with a regexp and no tokenizer.
|
||||
//
|
||||
// golang.org/x/net/html is not vendored and the network is not assumed, so this
|
||||
// is stdlib. That is less of a compromise than it sounds: the unit of context
|
||||
// here is a few hundred words for a 4096-token model to read, exactly like the
|
||||
// Kiwix snippet, so what matters is dropping script/style/nav noise and keeping
|
||||
// paragraph boundaries. A DOM would buy correctness on malformed markup that is
|
||||
// then thrown away by truncation anyway.
|
||||
//
|
||||
// What this deliberately does NOT do: run JavaScript, follow links, or extract
|
||||
// structured fields with CSS selectors or an LLM prompt. The plan's step 2 asked
|
||||
// for the last of those; see docs/plans/14-web-crawler.md for why it was left
|
||||
// out for now.
|
||||
|
||||
var (
|
||||
// RE2 has no backreferences, so each tag pair is spelled out rather than
|
||||
// captured and matched against itself.
|
||||
dropRE = regexp.MustCompile(pairsRE("script", "style", "noscript", "svg", "head", "nav", "footer", "form"))
|
||||
titleRE = regexp.MustCompile(`(?is)<title\b[^>]*>(.*?)</title>`)
|
||||
h1RE = regexp.MustCompile(`(?is)<h1\b[^>]*>(.*?)</h1>`)
|
||||
// Block-level tags become newlines so paragraphs survive as paragraphs.
|
||||
blockRE = regexp.MustCompile(`(?is)</?(p|div|br|li|tr|h[1-6]|section|article|blockquote|pre)\b[^>]*>`)
|
||||
tagRE = regexp.MustCompile(`(?s)<[^>]*>`)
|
||||
commentRE = regexp.MustCompile(`(?s)<!--.*?-->`)
|
||||
spaceRE = regexp.MustCompile(`[ \t\f\v]+`)
|
||||
blankRE = regexp.MustCompile(`\n{2,}`)
|
||||
)
|
||||
|
||||
// pairsRE builds `(?is)<tag …>…</tag>|…` for the given tags.
|
||||
func pairsRE(tags ...string) string {
|
||||
parts := make([]string, 0, len(tags))
|
||||
for _, t := range tags {
|
||||
parts = append(parts, `<`+t+`\b[^>]*>.*?</`+t+`>`)
|
||||
}
|
||||
return `(?is)` + strings.Join(parts, "|")
|
||||
}
|
||||
|
||||
// Page is an extracted page.
|
||||
type Page struct {
|
||||
URL string
|
||||
Title string
|
||||
Text string // plain text, paragraphs separated by single newlines
|
||||
}
|
||||
|
||||
// Extract turns a fetched HTML document into a Page. maxRunes caps the text (0 ⇒
|
||||
// DefaultMaxRunes); the cap is on runes, not bytes, because a Russian page cut
|
||||
// at a byte boundary ends in half a letter.
|
||||
func Extract(url string, body []byte, maxRunes int) Page {
|
||||
if maxRunes <= 0 {
|
||||
maxRunes = DefaultMaxRunes
|
||||
}
|
||||
s := string(body)
|
||||
s = commentRE.ReplaceAllString(s, " ")
|
||||
|
||||
title := firstGroup(titleRE, s)
|
||||
if title == "" {
|
||||
title = firstGroup(h1RE, s)
|
||||
}
|
||||
|
||||
s = dropRE.ReplaceAllString(s, "\n")
|
||||
s = blockRE.ReplaceAllString(s, "\n")
|
||||
s = tagRE.ReplaceAllString(s, " ")
|
||||
s = html.UnescapeString(s)
|
||||
s = spaceRE.ReplaceAllString(s, " ")
|
||||
|
||||
var lines []string
|
||||
for _, l := range strings.Split(s, "\n") {
|
||||
if l = strings.TrimSpace(l); l != "" {
|
||||
lines = append(lines, l)
|
||||
}
|
||||
}
|
||||
text := blankRE.ReplaceAllString(strings.Join(lines, "\n"), "\n")
|
||||
|
||||
return Page{URL: url, Title: title, Text: TrimRunes(text, maxRunes)}
|
||||
}
|
||||
|
||||
// DefaultMaxRunes — how much of a page is kept. ~4000 runes is a long answer's
|
||||
// worth of context and still leaves room in a 4096-token window for the prompt
|
||||
// and the reply.
|
||||
const DefaultMaxRunes = 4000
|
||||
|
||||
func firstGroup(re *regexp.Regexp, s string) string {
|
||||
m := re.FindStringSubmatch(s)
|
||||
if len(m) < 2 {
|
||||
return ""
|
||||
}
|
||||
t := tagRE.ReplaceAllString(m[1], " ")
|
||||
return strings.TrimSpace(strings.Join(strings.Fields(html.UnescapeString(t)), " "))
|
||||
}
|
||||
|
||||
// TrimRunes cuts s to at most max runes, on a rune boundary.
|
||||
func TrimRunes(s string, max int) string {
|
||||
r := []rune(s)
|
||||
if len(r) <= max {
|
||||
return s
|
||||
}
|
||||
return strings.TrimSpace(string(r[:max])) + "…"
|
||||
}
|
||||
@@ -0,0 +1,211 @@
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// robots.txt, parsed the small way: no wildcards beyond the two the standard
|
||||
// actually defines (`*` inside a path and `$` at the end), no sitemaps, no
|
||||
// crawl-delay-per-agent gymnastics. A personal assistant reading a handful of
|
||||
// pages does not need a spec-complete implementation; it needs to not be rude,
|
||||
// and to be auditable in one sitting.
|
||||
//
|
||||
// Two rules worth stating because they are choices, not accidents:
|
||||
//
|
||||
// - a missing or unreadable robots.txt means ALLOW. That is what the standard
|
||||
// says (404 ⇒ unrestricted), and the alternative would make a site that
|
||||
// simply has no robots.txt unreadable;
|
||||
// - an explicit Disallow means REFUSE, and Maven does not offer an override.
|
||||
// There is no "but he asked me to" flag: the page is not read.
|
||||
|
||||
// Rules is a parsed robots.txt for one user-agent.
|
||||
type Rules struct {
|
||||
allow []string
|
||||
disallow []string
|
||||
// Delay is Crawl-delay in seconds when the group named one, 0 otherwise.
|
||||
// The fetcher's own per-host rate limit is the floor; this can only make
|
||||
// Maven slower, never faster.
|
||||
Delay time.Duration
|
||||
}
|
||||
|
||||
// ParseRobots reads robots.txt and returns the rules that apply to agent.
|
||||
//
|
||||
// Group selection follows the standard: the most specific matching group wins,
|
||||
// which here means an exact user-agent match beats `*`. Lines that are neither
|
||||
// are ignored rather than guessed at.
|
||||
func ParseRobots(body string, agent string) Rules {
|
||||
agent = strings.ToLower(agent)
|
||||
|
||||
type group struct {
|
||||
agents []string
|
||||
allow []string
|
||||
disallow []string
|
||||
delay time.Duration
|
||||
}
|
||||
var groups []group
|
||||
var cur *group
|
||||
// startNew tracks whether the next User-agent line opens a new group or
|
||||
// joins the current one: consecutive User-agent lines share their rules.
|
||||
startNew := true
|
||||
|
||||
for _, raw := range strings.Split(body, "\n") {
|
||||
line := raw
|
||||
if i := strings.IndexByte(line, '#'); i >= 0 {
|
||||
line = line[:i]
|
||||
}
|
||||
line = strings.TrimSpace(line)
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
key, val, ok := strings.Cut(line, ":")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
key = strings.ToLower(strings.TrimSpace(key))
|
||||
val = strings.TrimSpace(val)
|
||||
|
||||
switch key {
|
||||
case "user-agent":
|
||||
if startNew || cur == nil {
|
||||
groups = append(groups, group{})
|
||||
cur = &groups[len(groups)-1]
|
||||
startNew = false
|
||||
}
|
||||
cur.agents = append(cur.agents, strings.ToLower(val))
|
||||
case "disallow":
|
||||
if cur == nil {
|
||||
continue
|
||||
}
|
||||
startNew = true
|
||||
// "Disallow:" with an empty value allows everything, and is not a
|
||||
// path rule at all.
|
||||
if val != "" {
|
||||
cur.disallow = append(cur.disallow, val)
|
||||
}
|
||||
case "allow":
|
||||
if cur == nil {
|
||||
continue
|
||||
}
|
||||
startNew = true
|
||||
if val != "" {
|
||||
cur.allow = append(cur.allow, val)
|
||||
}
|
||||
case "crawl-delay":
|
||||
if cur == nil {
|
||||
continue
|
||||
}
|
||||
startNew = true
|
||||
if d, err := time.ParseDuration(val + "s"); err == nil && d > 0 {
|
||||
cur.delay = d
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var star, exact *group
|
||||
for i := range groups {
|
||||
for _, a := range groups[i].agents {
|
||||
if a == "*" && star == nil {
|
||||
star = &groups[i]
|
||||
}
|
||||
// A robots.txt names "maven", we send "Maven/1.0 (…)": match on
|
||||
// prefix, which is how every crawler reads this field.
|
||||
if a != "*" && a != "" && strings.HasPrefix(agent, a) {
|
||||
exact = &groups[i]
|
||||
}
|
||||
}
|
||||
}
|
||||
g := exact
|
||||
if g == nil {
|
||||
g = star
|
||||
}
|
||||
if g == nil {
|
||||
return Rules{}
|
||||
}
|
||||
return Rules{allow: g.allow, disallow: g.disallow, Delay: g.delay}
|
||||
}
|
||||
|
||||
// Allowed reports whether path may be fetched. Longest matching rule wins, and
|
||||
// Allow beats Disallow at equal length — the standard's tie-break, and the one
|
||||
// that makes "Disallow: /" plus "Allow: /public" mean what it looks like.
|
||||
func (r Rules) Allowed(path string) bool {
|
||||
if path == "" {
|
||||
path = "/"
|
||||
}
|
||||
best, allowed := -1, true
|
||||
for _, p := range r.disallow {
|
||||
if n, ok := matchPath(p, path); ok && n > best {
|
||||
best, allowed = n, false
|
||||
}
|
||||
}
|
||||
for _, p := range r.allow {
|
||||
if n, ok := matchPath(p, path); ok && n >= best {
|
||||
best, allowed = n, true
|
||||
}
|
||||
}
|
||||
return allowed
|
||||
}
|
||||
|
||||
// matchPath applies a robots path pattern and returns the pattern's length as
|
||||
// the specificity score. `*` matches any run of characters, `$` anchors the end.
|
||||
// A pattern is a PREFIX match otherwise, which is what "Disallow: /admin" means.
|
||||
func matchPath(pattern, path string) (int, bool) {
|
||||
score := len(pattern)
|
||||
re, err := robotsRegexp(pattern)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
return score, re.MatchString(path)
|
||||
}
|
||||
|
||||
// robotsRegexp turns a robots path pattern into an anchored-at-the-start
|
||||
// regexp. Everything but `*` and a trailing `$` is a literal, so the pattern is
|
||||
// quoted first and the two metacharacters are put back afterwards.
|
||||
func robotsRegexp(pattern string) (*regexp.Regexp, error) {
|
||||
end := ""
|
||||
if strings.HasSuffix(pattern, "$") {
|
||||
pattern = strings.TrimSuffix(pattern, "$")
|
||||
end = "$"
|
||||
}
|
||||
parts := strings.Split(pattern, "*")
|
||||
for i, p := range parts {
|
||||
parts[i] = regexp.QuoteMeta(p)
|
||||
}
|
||||
return regexp.Compile("^" + strings.Join(parts, ".*") + end)
|
||||
}
|
||||
|
||||
// robotsCache holds parsed rules per host so a crawl of ten pages on one site
|
||||
// reads robots.txt once. TTL because a site may change its mind, and a daemon
|
||||
// that runs for weeks would otherwise never notice.
|
||||
type robotsCache struct {
|
||||
ttl time.Duration
|
||||
mu sync.Mutex
|
||||
m map[string]robotsEntry
|
||||
}
|
||||
|
||||
type robotsEntry struct {
|
||||
rules Rules
|
||||
at time.Time
|
||||
}
|
||||
|
||||
func newRobotsCache(ttl time.Duration) *robotsCache {
|
||||
return &robotsCache{ttl: ttl, m: map[string]robotsEntry{}}
|
||||
}
|
||||
|
||||
func (c *robotsCache) get(host string, now time.Time) (Rules, bool) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
e, ok := c.m[host]
|
||||
if !ok || now.Sub(e.at) > c.ttl {
|
||||
return Rules{}, false
|
||||
}
|
||||
return e.rules, true
|
||||
}
|
||||
|
||||
func (c *robotsCache) put(host string, r Rules, now time.Time) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.m[host] = robotsEntry{rules: r, at: now}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
const robotsBody = `# a comment
|
||||
User-agent: *
|
||||
Disallow: /private
|
||||
Disallow: /tmp/
|
||||
Crawl-delay: 5
|
||||
|
||||
User-agent: Maven
|
||||
Disallow: /
|
||||
Allow: /public
|
||||
`
|
||||
|
||||
func TestParseRobotsPicksTheMostSpecificGroup(t *testing.T) {
|
||||
// The Maven group applies to us even though we send a longer UA string.
|
||||
r := ParseRobots(robotsBody, "Maven/1.0 (self-hosted personal assistant)")
|
||||
if r.Allowed("/anything") {
|
||||
t.Error("Disallow: / in our own group was ignored")
|
||||
}
|
||||
if !r.Allowed("/public/page") {
|
||||
t.Error("Allow: /public must beat the shorter Disallow: /")
|
||||
}
|
||||
|
||||
// A different agent falls into the * group.
|
||||
star := ParseRobots(robotsBody, "SomeoneElse/2")
|
||||
if !star.Allowed("/anything") {
|
||||
t.Error("the * group disallows nothing but /private and /tmp/")
|
||||
}
|
||||
if star.Allowed("/private/x") || star.Allowed("/tmp/") {
|
||||
t.Error("the * group's disallows were not applied")
|
||||
}
|
||||
if star.Delay != 5*time.Second {
|
||||
t.Errorf("crawl-delay = %v, want 5s", star.Delay)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseRobotsEmptyMeansAllowAll(t *testing.T) {
|
||||
for _, body := range []string{"", "# nothing here\n", "User-agent: *\nDisallow:\n"} {
|
||||
if !ParseRobots(body, "Maven").Allowed("/whatever") {
|
||||
t.Errorf("body %q must allow everything", body)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRobotsWildcards(t *testing.T) {
|
||||
r := ParseRobots("User-agent: *\nDisallow: /*.pdf$\nDisallow: /a/*/secret\n", "Maven")
|
||||
if r.Allowed("/docs/manual.pdf") {
|
||||
t.Error("*.pdf$ did not match")
|
||||
}
|
||||
if !r.Allowed("/docs/manual.pdf.html") {
|
||||
t.Error("$ must anchor at the end")
|
||||
}
|
||||
if r.Allowed("/a/b/secret") {
|
||||
t.Error("/a/*/secret did not match")
|
||||
}
|
||||
if !r.Allowed("/a/b/public") {
|
||||
t.Error("unrelated path was refused")
|
||||
}
|
||||
}
|
||||
|
||||
// Consecutive User-agent lines share one group, which is common in the wild.
|
||||
func TestRobotsSharedGroup(t *testing.T) {
|
||||
r := ParseRobots("User-agent: Googlebot\nUser-agent: Maven\nDisallow: /x\n", "Maven/1.0")
|
||||
if r.Allowed("/x/y") {
|
||||
t.Fatal("a shared group's rules were not applied to the second agent")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRobotsCacheTTL(t *testing.T) {
|
||||
c := newRobotsCache(time.Minute)
|
||||
now := time.Now()
|
||||
c.put("example.com", ParseRobots("User-agent: *\nDisallow: /\n", "Maven"), now)
|
||||
if _, ok := c.get("example.com", now.Add(30*time.Second)); !ok {
|
||||
t.Error("a fresh entry must be served from cache")
|
||||
}
|
||||
if _, ok := c.get("example.com", now.Add(2*time.Minute)); ok {
|
||||
t.Error("an expired entry must be re-read")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,177 @@
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Scheduled crawls: a page is re-read on an interval, and when its TEXT changed
|
||||
// the new text is written as a note. Nothing is dispatched — same rule as the
|
||||
// feed poller (Vikunja #258). A page that announced its own change would be a
|
||||
// nag, and "the docs page changed" is not worth interrupting anyone for.
|
||||
//
|
||||
// Dedup is by content hash, so a page that re-renders identically writes nothing
|
||||
// and a rotating ad slot does not count as news.
|
||||
|
||||
// WatchConfig — one page to keep an eye on.
|
||||
type WatchConfig struct {
|
||||
Name string // note source is "crawl:<Name>"
|
||||
URL string // http(s), guarded by the fetcher
|
||||
Interval time.Duration // 0 ⇒ Watcher's default
|
||||
}
|
||||
|
||||
// Notes is core's note-writing half (same shape as ipc.CoreAPI's method).
|
||||
type Notes interface {
|
||||
WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error)
|
||||
}
|
||||
|
||||
// Hashes remembers the last text hash per watch, durably, so a restart does not
|
||||
// re-note an unchanged page. The daemon backs this with config facts
|
||||
// ("crawl:hash:<name>").
|
||||
type Hashes interface {
|
||||
LastHash(ctx context.Context, name string) (string, error)
|
||||
SetHash(ctx context.Context, name, hash string) error
|
||||
}
|
||||
|
||||
// Embedder embeds a note on its way into the store. nil ⇒ no vector.
|
||||
type Embedder interface {
|
||||
Embed(ctx context.Context, text string) ([]float32, error)
|
||||
}
|
||||
|
||||
// DefaultWatchInterval — pages change slowly, and every check is a request in
|
||||
// someone's log.
|
||||
const DefaultWatchInterval = 6 * time.Hour
|
||||
|
||||
// Watcher re-reads watched pages on their interval.
|
||||
type Watcher struct {
|
||||
c *Crawler
|
||||
watches []WatchConfig
|
||||
notes Notes
|
||||
hashes Hashes
|
||||
embed Embedder
|
||||
interval time.Duration
|
||||
nextDue map[string]time.Time
|
||||
}
|
||||
|
||||
// NewWatcher wires the scheduled half, or returns nil when there is nothing to
|
||||
// watch. Callers check for nil: no watches, no goroutine, no request.
|
||||
func NewWatcher(c *Crawler, watches []WatchConfig, notes Notes, hashes Hashes, embed Embedder, defaultInterval time.Duration) *Watcher {
|
||||
if c == nil || notes == nil {
|
||||
return nil
|
||||
}
|
||||
var valid []WatchConfig
|
||||
for _, w := range watches {
|
||||
if strings.TrimSpace(w.Name) == "" || strings.TrimSpace(w.URL) == "" {
|
||||
log.Printf("crawl: skipping a watch with no name or no url")
|
||||
continue
|
||||
}
|
||||
valid = append(valid, w)
|
||||
}
|
||||
if len(valid) == 0 {
|
||||
return nil
|
||||
}
|
||||
if defaultInterval <= 0 {
|
||||
defaultInterval = DefaultWatchInterval
|
||||
}
|
||||
return &Watcher{
|
||||
c: c, watches: valid, notes: notes, hashes: hashes, embed: embed,
|
||||
interval: defaultInterval, nextDue: map[string]time.Time{},
|
||||
}
|
||||
}
|
||||
|
||||
// Watches returns the configured watches.
|
||||
func (w *Watcher) Watches() []WatchConfig { return w.watches }
|
||||
|
||||
// CheckDue re-reads every watch whose interval elapsed and returns how many
|
||||
// notes were written. Errors are logged per watch, never returned: one dead page
|
||||
// must not stop the others.
|
||||
func (w *Watcher) CheckDue(ctx context.Context, now time.Time) int {
|
||||
written := 0
|
||||
for _, watch := range w.watches {
|
||||
if due, ok := w.nextDue[watch.Name]; ok && now.Before(due) {
|
||||
continue
|
||||
}
|
||||
interval := watch.Interval
|
||||
if interval <= 0 {
|
||||
interval = w.interval
|
||||
}
|
||||
w.nextDue[watch.Name] = now.Add(interval)
|
||||
changed, err := w.Check(ctx, watch, now)
|
||||
if err != nil {
|
||||
log.Printf("crawl: watch %s: %v", watch.Name, err)
|
||||
continue
|
||||
}
|
||||
if changed {
|
||||
log.Printf("crawl: watch %s: page changed, noted", watch.Name)
|
||||
written++
|
||||
}
|
||||
}
|
||||
return written
|
||||
}
|
||||
|
||||
// Check re-reads one watch now and reports whether it wrote a note.
|
||||
func (w *Watcher) Check(ctx context.Context, watch WatchConfig, now time.Time) (bool, error) {
|
||||
page, err := w.c.Page(ctx, watch.URL)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
// Title included: a page whose headline changed has changed.
|
||||
h := Hash(page.Title + "\n" + page.Text)
|
||||
if w.hashes != nil {
|
||||
prev, err := w.hashes.LastHash(ctx, watch.Name)
|
||||
if err != nil {
|
||||
log.Printf("crawl: watch %s: read hash: %v", watch.Name, err)
|
||||
}
|
||||
if prev == h {
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
text := NoteText(watch, page)
|
||||
var vec []float32
|
||||
if w.embed != nil {
|
||||
v, err := w.embed.Embed(ctx, text)
|
||||
if err != nil {
|
||||
log.Printf("crawl: watch %s: embed: %v", watch.Name, err)
|
||||
} else {
|
||||
vec = v
|
||||
}
|
||||
}
|
||||
if _, err := w.notes.WriteNote(ctx, now, text, vec, SourceFor(watch.Name)); err != nil {
|
||||
return false, fmt.Errorf("write note: %w", err)
|
||||
}
|
||||
if w.hashes != nil {
|
||||
if err := w.hashes.SetHash(ctx, watch.Name, h); err != nil {
|
||||
log.Printf("crawl: watch %s: save hash: %v", watch.Name, err)
|
||||
}
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// SourceFor is the note source for a watch, and SourcePrefix is what the answer
|
||||
// path matches to recognise one.
|
||||
func SourceFor(name string) string { return SourcePrefix + name }
|
||||
|
||||
// SourcePrefix — provenance for anything read off the network on a schedule.
|
||||
const SourcePrefix = "crawl:"
|
||||
|
||||
// noteRunes — how much of a watched page goes into a note. Shorter than what the
|
||||
// on-demand path reads: a note is a record of a change, not an archive.
|
||||
const noteRunes = 800
|
||||
|
||||
// NoteText renders a watched page as a note body.
|
||||
func NoteText(watch WatchConfig, page Page) string {
|
||||
var b strings.Builder
|
||||
if page.Title != "" {
|
||||
b.WriteString(page.Title)
|
||||
} else {
|
||||
b.WriteString(watch.Name)
|
||||
}
|
||||
b.WriteString("\n")
|
||||
b.WriteString(TrimRunes(page.Text, noteRunes))
|
||||
b.WriteString("\n")
|
||||
b.WriteString(watch.URL)
|
||||
return b.String()
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
package crawl
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
type note struct {
|
||||
text string
|
||||
source string
|
||||
}
|
||||
|
||||
type fakeNotes struct{ notes []note }
|
||||
|
||||
func (n *fakeNotes) WriteNote(_ context.Context, _ time.Time, text string, _ []float32, source string) (int64, error) {
|
||||
n.notes = append(n.notes, note{text, source})
|
||||
return int64(len(n.notes)), nil
|
||||
}
|
||||
|
||||
type fakeHashes struct{ m map[string]string }
|
||||
|
||||
func newHashes() *fakeHashes { return &fakeHashes{m: map[string]string{}} }
|
||||
func (f *fakeHashes) LastHash(_ context.Context, name string) (string, error) {
|
||||
return f.m[name], nil
|
||||
}
|
||||
func (f *fakeHashes) SetHash(_ context.Context, name, h string) error { f.m[name] = h; return nil }
|
||||
|
||||
var t0 = time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC)
|
||||
|
||||
func TestWatchNotesAChangedPage(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{
|
||||
"https://example.org/docs": {Body: []byte(htmlPage)},
|
||||
}}
|
||||
notes := &fakeNotes{}
|
||||
hashes := newHashes()
|
||||
w := NewWatcher(newTestCrawler(f), []WatchConfig{{Name: "docs", URL: "https://example.org/docs"}},
|
||||
notes, hashes, nil, time.Hour)
|
||||
if w == nil {
|
||||
t.Fatal("NewWatcher returned nil for a configured watch")
|
||||
}
|
||||
if n := w.CheckDue(context.Background(), t0); n != 1 {
|
||||
t.Fatalf("first check wrote %d notes, want 1", n)
|
||||
}
|
||||
if notes.notes[0].source != "crawl:docs" {
|
||||
t.Errorf("source = %q, want crawl:docs", notes.notes[0].source)
|
||||
}
|
||||
if !strings.Contains(notes.notes[0].text, "https://example.org/docs") {
|
||||
t.Errorf("note does not carry the url: %q", notes.notes[0].text)
|
||||
}
|
||||
|
||||
// Unchanged page, interval elapsed: nothing written.
|
||||
if n := w.CheckDue(context.Background(), t0.Add(2*time.Hour)); n != 0 {
|
||||
t.Fatalf("an unchanged page wrote %d notes", n)
|
||||
}
|
||||
|
||||
// Changed page: one note.
|
||||
f.pages["https://example.org/docs"] = Response{Body: []byte(strings.Replace(htmlPage, "синее", "серое", 1))}
|
||||
if n := w.CheckDue(context.Background(), t0.Add(4*time.Hour)); n != 1 {
|
||||
t.Fatalf("a changed page wrote %d notes, want 1", n)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWatchIntervalIsRespected(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{"https://example.org/d": {Body: []byte(htmlPage)}}}
|
||||
w := NewWatcher(newTestCrawler(f), []WatchConfig{{Name: "d", URL: "https://example.org/d", Interval: time.Hour}},
|
||||
&fakeNotes{}, newHashes(), nil, 0)
|
||||
w.CheckDue(context.Background(), t0)
|
||||
before := len(f.calls)
|
||||
w.CheckDue(context.Background(), t0.Add(time.Minute))
|
||||
if len(f.calls) != before {
|
||||
t.Fatal("the page was re-read inside its interval")
|
||||
}
|
||||
}
|
||||
|
||||
// The hash is durable so a restart does not re-note an unchanged page.
|
||||
func TestWatchHashSurvivesRestart(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{"https://example.org/d": {Body: []byte(htmlPage)}}}
|
||||
hashes := newHashes()
|
||||
watches := []WatchConfig{{Name: "d", URL: "https://example.org/d"}}
|
||||
NewWatcher(newTestCrawler(f), watches, &fakeNotes{}, hashes, nil, time.Hour).CheckDue(context.Background(), t0)
|
||||
|
||||
notes2 := &fakeNotes{}
|
||||
NewWatcher(newTestCrawler(f), watches, notes2, hashes, nil, time.Hour).CheckDue(context.Background(), t0.Add(time.Hour))
|
||||
if len(notes2.notes) != 0 {
|
||||
t.Fatalf("a fresh watcher re-noted an unchanged page: %q", notes2.notes[0].text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWatchDeadPageDoesNotStopTheOthers(t *testing.T) {
|
||||
f := &fakeFetcher{pages: map[string]Response{"https://example.org/live": {Body: []byte(htmlPage)}}}
|
||||
notes := &fakeNotes{}
|
||||
w := NewWatcher(newTestCrawler(f), []WatchConfig{
|
||||
{Name: "dead", URL: "https://example.org/gone"},
|
||||
{Name: "live", URL: "https://example.org/live"},
|
||||
}, notes, newHashes(), nil, time.Hour)
|
||||
if n := w.CheckDue(context.Background(), t0); n != 1 {
|
||||
t.Fatalf("wrote %d notes, want 1 (the live page)", n)
|
||||
}
|
||||
if notes.notes[0].source != "crawl:live" {
|
||||
t.Fatalf("source = %q", notes.notes[0].source)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNoWatchesMeansNoWatcher(t *testing.T) {
|
||||
c := newTestCrawler(&fakeFetcher{})
|
||||
if NewWatcher(c, nil, &fakeNotes{}, nil, nil, 0) != nil {
|
||||
t.Fatal("no watches must mean no watcher")
|
||||
}
|
||||
if NewWatcher(nil, []WatchConfig{{Name: "a", URL: "u"}}, &fakeNotes{}, nil, nil, 0) != nil {
|
||||
t.Fatal("no crawler must mean no watcher")
|
||||
}
|
||||
if NewWatcher(c, []WatchConfig{{Name: "", URL: ""}}, &fakeNotes{}, nil, nil, 0) != nil {
|
||||
t.Fatal("a watch with no name or url is not a configuration")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package router
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Finding a URL in an utterance (Vikunja #259).
|
||||
//
|
||||
// This is deliberately strict: a scheme is required. "посмотри на example.org"
|
||||
// is not treated as a fetch request, because a bare dotted word is also how
|
||||
// people say file names, versions and Russian abbreviations, and the cost of a
|
||||
// false positive here is an outbound request nobody asked for.
|
||||
//
|
||||
// Note where this runs: an utterance from STT. Whisper will mangle a spoken URL,
|
||||
// which is fine — the URL that survives is one he pasted into the web chat, and
|
||||
// a mangled one simply fails to match.
|
||||
var urlRE = regexp.MustCompile(`(?i)\bhttps?://[^\s<>"']+`)
|
||||
|
||||
// FirstURL returns the first http(s) URL in text.
|
||||
//
|
||||
// Trailing punctuation is trimmed: he ends sentences, and "…/page." is not a
|
||||
// path component. A closing bracket is only trimmed when it has no opener,
|
||||
// because a wikipedia URL legitimately ends in one.
|
||||
func FirstURL(text string) (string, bool) {
|
||||
m := urlRE.FindString(text)
|
||||
if m == "" {
|
||||
return "", false
|
||||
}
|
||||
m = strings.TrimRight(m, ".,;:!?…")
|
||||
if strings.HasSuffix(m, ")") && strings.Count(m, "(") == 0 {
|
||||
m = strings.TrimSuffix(m, ")")
|
||||
}
|
||||
// A scheme with nothing after it is not a URL.
|
||||
if rest := strings.SplitN(m, "//", 2); len(rest) < 2 || rest[1] == "" {
|
||||
return "", false
|
||||
}
|
||||
return m, true
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
package router
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestFirstURL(t *testing.T) {
|
||||
cases := []struct {
|
||||
text string
|
||||
want string
|
||||
}{
|
||||
{"посмотри https://example.org/page — что там?", "https://example.org/page"},
|
||||
{"почитай http://example.org/a/b?x=1 и скажи", "http://example.org/a/b?x=1"},
|
||||
{"вот ссылка: https://example.org/page.", "https://example.org/page"},
|
||||
{"https://ru.wikipedia.org/wiki/Небо_(значения)", "https://ru.wikipedia.org/wiki/Небо_(значения)"},
|
||||
// No scheme ⇒ no fetch. A bare dotted word is not an instruction to
|
||||
// reach out to the network.
|
||||
{"посмотри на example.org", ""},
|
||||
{"открой файл config.json", ""},
|
||||
{"что нового?", ""},
|
||||
{"https://", ""},
|
||||
{"", ""},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, ok := FirstURL(c.text)
|
||||
if c.want == "" {
|
||||
if ok {
|
||||
t.Errorf("FirstURL(%q) = %q, want no match", c.text, got)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if !ok || got != c.want {
|
||||
t.Errorf("FirstURL(%q) = %q, %v; want %q", c.text, got, ok, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user