157 lines
4.3 KiB
Go
157 lines
4.3 KiB
Go
package scraper
|
|
|
|
import (
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"net/url"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// UserAgent identifies this crawler to servers. Keep it stable so that
|
|
// instances can set robots rules for us specifically.
|
|
const UserAgent = "ffd-indexer/0.1"
|
|
|
|
// robotsHTTPClient is used for fetching robots.txt. It does not follow the
|
|
// SDK client's settings; it is a plain, short-timeout client.
|
|
var robotsHTTPClient = &http.Client{Timeout: 10 * time.Second}
|
|
|
|
// Allowed reports whether an instance at baseURL permits crawling its API
|
|
// according to its robots.txt. It only consults the rules for the
|
|
// User-Agent "*"; a block of /api/ (the API root) disallows scraping.
|
|
//
|
|
// A failure to fetch or parse robots.txt is treated as an error; callers
|
|
// decide whether that is fatal.
|
|
func Allowed(baseURL string) (bool, error) {
|
|
u, err := url.Parse(baseURL)
|
|
if err != nil {
|
|
return false, fmt.Errorf("parse base URL: %w", err)
|
|
}
|
|
u.Path = "/robots.txt"
|
|
|
|
req, err := http.NewRequest(http.MethodGet, u.String(), nil)
|
|
if err != nil {
|
|
return false, fmt.Errorf("build robots request: %w", err)
|
|
}
|
|
req.Header.Set("User-Agent", UserAgent)
|
|
|
|
resp, err := robotsHTTPClient.Do(req)
|
|
if err != nil {
|
|
return false, fmt.Errorf("fetch robots.txt: %w", err)
|
|
}
|
|
defer func() {
|
|
_ = resp.Body.Close()
|
|
}()
|
|
|
|
// A 404 or similar means "no robots.txt": assume allowed.
|
|
if resp.StatusCode == http.StatusNotFound {
|
|
return true, nil
|
|
}
|
|
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
|
return false, fmt.Errorf("robots.txt returned %s", resp.Status)
|
|
}
|
|
|
|
body, err := io.ReadAll(resp.Body)
|
|
if err != nil {
|
|
return false, fmt.Errorf("read robots.txt: %w", err)
|
|
}
|
|
|
|
return allowedFor(string(body), UserAgent), nil
|
|
}
|
|
|
|
// allowedFor parses a robots.txt body and reports whether a crawler with the
|
|
// given user agent may crawl the path "/api/". It implements a reasonable
|
|
// subset of the robots.txt standard:
|
|
//
|
|
// - Only the "*" group and an explicit group matching userAgent are
|
|
// considered; later groups override earlier ones for the same agent.
|
|
// - Disallow entries are matched as path prefixes (a trailing "*" is
|
|
// stripped). No other wildcard/glob support.
|
|
// - An empty group (no rules) is "allow everything".
|
|
func allowedFor(body, userAgent string) bool {
|
|
groups := parseRobotGroups(body)
|
|
|
|
// A group that explicitly names our user agent takes precedence over a
|
|
// "*" group, regardless of order. Otherwise fall back to "*".
|
|
var star *robotGroup
|
|
for _, g := range groups {
|
|
for _, a := range g.agents {
|
|
if strings.EqualFold(a, "*") {
|
|
star = g
|
|
} else if strings.EqualFold(a, userAgent) {
|
|
return !matchDisallow(g, "/api/")
|
|
}
|
|
}
|
|
}
|
|
if star != nil {
|
|
return !matchDisallow(star, "/api/")
|
|
}
|
|
return true
|
|
}
|
|
|
|
type robotGroup struct {
|
|
agents []string
|
|
disallow []string
|
|
}
|
|
|
|
// parseRobotGroups splits a robots.txt body into agent groups.
|
|
func parseRobotGroups(body string) []*robotGroup {
|
|
var groups []*robotGroup
|
|
var cur *robotGroup
|
|
|
|
for _, raw := range strings.Split(body, "\n") {
|
|
line := strings.TrimSpace(raw)
|
|
// Strip an inline comment (not part of a rule's result).
|
|
if i := strings.IndexByte(line, '#'); i >= 0 {
|
|
line = strings.TrimSpace(line[:i])
|
|
}
|
|
if line == "" {
|
|
continue
|
|
}
|
|
|
|
key, val, ok := strings.Cut(line, ":")
|
|
if !ok {
|
|
continue
|
|
}
|
|
key = strings.ToLower(strings.TrimSpace(key))
|
|
val = strings.TrimSpace(val)
|
|
|
|
switch key {
|
|
case "user-agent":
|
|
// A User-agent line starts a new group, unless the current
|
|
// group is still collecting agents (has no Disallow yet).
|
|
if cur == nil || len(cur.disallow) > 0 {
|
|
cur = &robotGroup{}
|
|
groups = append(groups, cur)
|
|
}
|
|
cur.agents = append(cur.agents, val)
|
|
case "disallow":
|
|
if cur != nil {
|
|
cur.disallow = append(cur.disallow, val)
|
|
}
|
|
}
|
|
}
|
|
|
|
return groups
|
|
}
|
|
|
|
// matchDisallow reports whether a path is disallowed by a group's rules.
|
|
// A trailing "*" is stripped. Rules are applied in order, later rules
|
|
// overriding earlier ones: an empty Disallow rule revokes all preceding
|
|
// disallows (allow everything).
|
|
func matchDisallow(g *robotGroup, path string) bool {
|
|
disallowed := false
|
|
for _, rule := range g.disallow {
|
|
rule = strings.TrimSpace(rule)
|
|
rule = strings.TrimSuffix(rule, "*")
|
|
if rule == "" {
|
|
disallowed = false // empty Disallow: allow everything
|
|
continue
|
|
}
|
|
if strings.HasPrefix(path, rule) {
|
|
disallowed = true
|
|
}
|
|
}
|
|
return disallowed
|
|
}
|