indexer/scraper/robots.go
2026-09-28 15:36:42 +02:00

157 lines
4.3 KiB
Go

package scraper
import (
"fmt"
"io"
"net/http"
"net/url"
"strings"
"time"
)
// UserAgent identifies this crawler to servers. Keep it stable so that
// instances can set robots rules for us specifically.
const UserAgent = "ffd-indexer/0.1"
// robotsHTTPClient is used for fetching robots.txt. It does not follow the
// SDK client's settings; it is a plain, short-timeout client.
var robotsHTTPClient = &http.Client{Timeout: 10 * time.Second}
// Allowed reports whether an instance at baseURL permits crawling its API
// according to its robots.txt. It only consults the rules for the
// User-Agent "*"; a block of /api/ (the API root) disallows scraping.
//
// A failure to fetch or parse robots.txt is treated as an error; callers
// decide whether that is fatal.
func Allowed(baseURL string) (bool, error) {
u, err := url.Parse(baseURL)
if err != nil {
return false, fmt.Errorf("parse base URL: %w", err)
}
u.Path = "/robots.txt"
req, err := http.NewRequest(http.MethodGet, u.String(), nil)
if err != nil {
return false, fmt.Errorf("build robots request: %w", err)
}
req.Header.Set("User-Agent", UserAgent)
resp, err := robotsHTTPClient.Do(req)
if err != nil {
return false, fmt.Errorf("fetch robots.txt: %w", err)
}
defer func() {
_ = resp.Body.Close()
}()
// A 404 or similar means "no robots.txt": assume allowed.
if resp.StatusCode == http.StatusNotFound {
return true, nil
}
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
return false, fmt.Errorf("robots.txt returned %s", resp.Status)
}
body, err := io.ReadAll(resp.Body)
if err != nil {
return false, fmt.Errorf("read robots.txt: %w", err)
}
return allowedFor(string(body), UserAgent), nil
}
// allowedFor parses a robots.txt body and reports whether a crawler with the
// given user agent may crawl the path "/api/". It implements a reasonable
// subset of the robots.txt standard:
//
// - Only the "*" group and an explicit group matching userAgent are
// considered; later groups override earlier ones for the same agent.
// - Disallow entries are matched as path prefixes (a trailing "*" is
// stripped). No other wildcard/glob support.
// - An empty group (no rules) is "allow everything".
func allowedFor(body, userAgent string) bool {
groups := parseRobotGroups(body)
// A group that explicitly names our user agent takes precedence over a
// "*" group, regardless of order. Otherwise fall back to "*".
var star *robotGroup
for _, g := range groups {
for _, a := range g.agents {
if strings.EqualFold(a, "*") {
star = g
} else if strings.EqualFold(a, userAgent) {
return !matchDisallow(g, "/api/")
}
}
}
if star != nil {
return !matchDisallow(star, "/api/")
}
return true
}
type robotGroup struct {
agents []string
disallow []string
}
// parseRobotGroups splits a robots.txt body into agent groups.
func parseRobotGroups(body string) []*robotGroup {
var groups []*robotGroup
var cur *robotGroup
for _, raw := range strings.Split(body, "\n") {
line := strings.TrimSpace(raw)
// Strip an inline comment (not part of a rule's result).
if i := strings.IndexByte(line, '#'); i >= 0 {
line = strings.TrimSpace(line[:i])
}
if line == "" {
continue
}
key, val, ok := strings.Cut(line, ":")
if !ok {
continue
}
key = strings.ToLower(strings.TrimSpace(key))
val = strings.TrimSpace(val)
switch key {
case "user-agent":
// A User-agent line starts a new group, unless the current
// group is still collecting agents (has no Disallow yet).
if cur == nil || len(cur.disallow) > 0 {
cur = &robotGroup{}
groups = append(groups, cur)
}
cur.agents = append(cur.agents, val)
case "disallow":
if cur != nil {
cur.disallow = append(cur.disallow, val)
}
}
}
return groups
}
// matchDisallow reports whether a path is disallowed by a group's rules.
// A trailing "*" is stripped. Rules are applied in order, later rules
// overriding earlier ones: an empty Disallow rule revokes all preceding
// disallows (allow everything).
func matchDisallow(g *robotGroup, path string) bool {
disallowed := false
for _, rule := range g.disallow {
rule = strings.TrimSpace(rule)
rule = strings.TrimSuffix(rule, "*")
if rule == "" {
disallowed = false // empty Disallow: allow everything
continue
}
if strings.HasPrefix(path, rule) {
disallowed = true
}
}
return disallowed
}