package scraper import ( "fmt" "io" "net/http" "net/url" "strings" "time" ) // UserAgent identifies this crawler to servers. Keep it stable so that // instances can set robots rules for us specifically. const UserAgent = "ffd-indexer/0.1" // robotsHTTPClient is used for fetching robots.txt. It does not follow the // SDK client's settings; it is a plain, short-timeout client. var robotsHTTPClient = &http.Client{Timeout: 10 * time.Second} // Allowed reports whether an instance at baseURL permits crawling its API // according to its robots.txt. It only consults the rules for the // User-Agent "*"; a block of /api/ (the API root) disallows scraping. // // A failure to fetch or parse robots.txt is treated as an error; callers // decide whether that is fatal. func Allowed(baseURL string) (bool, error) { u, err := url.Parse(baseURL) if err != nil { return false, fmt.Errorf("parse base URL: %w", err) } u.Path = "/robots.txt" req, err := http.NewRequest(http.MethodGet, u.String(), nil) if err != nil { return false, fmt.Errorf("build robots request: %w", err) } req.Header.Set("User-Agent", UserAgent) resp, err := robotsHTTPClient.Do(req) if err != nil { return false, fmt.Errorf("fetch robots.txt: %w", err) } defer func() { _ = resp.Body.Close() }() // A 404 or similar means "no robots.txt": assume allowed. if resp.StatusCode == http.StatusNotFound { return true, nil } if resp.StatusCode < 200 || resp.StatusCode >= 300 { return false, fmt.Errorf("robots.txt returned %s", resp.Status) } body, err := io.ReadAll(resp.Body) if err != nil { return false, fmt.Errorf("read robots.txt: %w", err) } return allowedFor(string(body), UserAgent), nil } // allowedFor parses a robots.txt body and reports whether a crawler with the // given user agent may crawl the path "/api/". It implements a reasonable // subset of the robots.txt standard: // // - Only the "*" group and an explicit group matching userAgent are // considered; later groups override earlier ones for the same agent. // - Disallow entries are matched as path prefixes (a trailing "*" is // stripped). No other wildcard/glob support. // - An empty group (no rules) is "allow everything". func allowedFor(body, userAgent string) bool { groups := parseRobotGroups(body) // A group that explicitly names our user agent takes precedence over a // "*" group, regardless of order. Otherwise fall back to "*". var star *robotGroup for _, g := range groups { for _, a := range g.agents { if strings.EqualFold(a, "*") { star = g } else if strings.EqualFold(a, userAgent) { return !matchDisallow(g, "/api/") } } } if star != nil { return !matchDisallow(star, "/api/") } return true } type robotGroup struct { agents []string disallow []string } // parseRobotGroups splits a robots.txt body into agent groups. func parseRobotGroups(body string) []*robotGroup { var groups []*robotGroup var cur *robotGroup for _, raw := range strings.Split(body, "\n") { line := strings.TrimSpace(raw) // Strip an inline comment (not part of a rule's result). if i := strings.IndexByte(line, '#'); i >= 0 { line = strings.TrimSpace(line[:i]) } if line == "" { continue } key, val, ok := strings.Cut(line, ":") if !ok { continue } key = strings.ToLower(strings.TrimSpace(key)) val = strings.TrimSpace(val) switch key { case "user-agent": // A User-agent line starts a new group, unless the current // group is still collecting agents (has no Disallow yet). if cur == nil || len(cur.disallow) > 0 { cur = &robotGroup{} groups = append(groups, cur) } cur.agents = append(cur.agents, val) case "disallow": if cur != nil { cur.disallow = append(cur.disallow, val) } } } return groups } // matchDisallow reports whether a path is disallowed by a group's rules. // A trailing "*" is stripped. Rules are applied in order, later rules // overriding earlier ones: an empty Disallow rule revokes all preceding // disallows (allow everything). func matchDisallow(g *robotGroup, path string) bool { disallowed := false for _, rule := range g.disallow { rule = strings.TrimSpace(rule) rule = strings.TrimSuffix(rule, "*") if rule == "" { disallowed = false // empty Disallow: allow everything continue } if strings.HasPrefix(path, rule) { disallowed = true } } return disallowed }