scraper done; scrapes to local sqlite db
This commit is contained in:
parent
448552016b
commit
bde13e9b68
18 changed files with 1585 additions and 164 deletions
230
scraper/forgejo.go
Normal file
230
scraper/forgejo.go
Normal file
|
|
@ -0,0 +1,230 @@
|
|||
// Package scraper forgejo.go - Forgejo scraper.
|
||||
package scraper
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
forgejo "codeberg.org/mvdkleijn/forgejo-sdk/forgejo/v3"
|
||||
|
||||
"github.com/go-enry/go-license-detector/v4/licensedb/filer"
|
||||
)
|
||||
|
||||
// ForgejoScraper retrieves public repositories from a Forgejo instance.
|
||||
type ForgejoScraper struct {
|
||||
Client *forgejo.Client
|
||||
BaseURL string
|
||||
PageSize int
|
||||
}
|
||||
|
||||
// NewForgejoScraper creates a scraper for a Forgejo instance.
|
||||
// BaseURL should be the instance URL, e.g. https://codeberg.org.
|
||||
func NewForgejoScraper(baseURL string) (*ForgejoScraper, error) {
|
||||
client, err := forgejo.NewClient(baseURL)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create Forgejo client: %w", err)
|
||||
}
|
||||
client.SetUserAgent(UserAgent)
|
||||
|
||||
return &ForgejoScraper{
|
||||
Client: client,
|
||||
BaseURL: strings.TrimSuffix(baseURL, "/"),
|
||||
PageSize: 50,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// GetProjects retrieves every public repository visible through the
|
||||
// Forgejo repository search API and maps it to the generic Project type.
|
||||
func (s *ForgejoScraper) GetProjects() ([]Project, error) {
|
||||
// Honor the instance's robots.txt before making any API call:
|
||||
// if /api/ is disallowed, do not scrape this instance.
|
||||
ok, err := Allowed(s.BaseURL)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("check robots.txt: %w", err)
|
||||
}
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("instance robots.txt disallows /api/")
|
||||
}
|
||||
|
||||
pageSize := s.PageSize
|
||||
if pageSize <= 0 {
|
||||
pageSize = 50
|
||||
}
|
||||
|
||||
isPrivate := false
|
||||
var projects []Project
|
||||
|
||||
for page := 1; ; page++ {
|
||||
repos, _, err := s.Client.SearchRepos(forgejo.SearchRepoOptions{
|
||||
Page: page,
|
||||
PageSize: pageSize,
|
||||
IsPrivate: &isPrivate,
|
||||
Type: forgejo.RepoTypeSource,
|
||||
Sort: "alpha",
|
||||
Order: "asc",
|
||||
})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf(
|
||||
"list public Forgejo repositories (page %d): %w",
|
||||
page,
|
||||
err,
|
||||
)
|
||||
}
|
||||
|
||||
if len(repos) == 0 {
|
||||
break
|
||||
}
|
||||
|
||||
for _, repo := range repos {
|
||||
if repo == nil {
|
||||
continue
|
||||
}
|
||||
|
||||
owner, name, ok := strings.Cut(repo.FullName, "/")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
|
||||
// List the repository root once and reuse it for readme selection
|
||||
// and license detection (they previously each listed the root).
|
||||
entries, _, err := s.Client.ListContents(owner, name, repo.DefaultBranch, "")
|
||||
if err != nil {
|
||||
// best-effort: fall back to empty root so the other fields still work
|
||||
entries = nil
|
||||
}
|
||||
root := make([]filer.File, 0, len(entries))
|
||||
for _, e := range entries {
|
||||
if e == nil {
|
||||
continue
|
||||
}
|
||||
root = append(root, filer.File{Name: e.Name, IsDir: e.Type == "dir"})
|
||||
}
|
||||
|
||||
// topics currently requires one extra HTTP call per repository.
|
||||
// Future optimization: the /repos/search response already includes
|
||||
// the topics field, so they could be gathered by parsing the
|
||||
// SearchRepos payload itself (e.g. via ListRepos) instead.
|
||||
// On failure the repo is kept in the results with empty topics.
|
||||
topics, err := s.topics(owner, name)
|
||||
if err != nil {
|
||||
topics = nil
|
||||
}
|
||||
|
||||
// languages requires one extra HTTP call per repository
|
||||
// (GET /repos/{owner}/{repo}/languages, via GetRepoLanguages).
|
||||
// It returns a map[language]bytes; we order it most-dominant-first
|
||||
// below. On failure the repo is kept with no languages.
|
||||
langs, _, err := s.Client.GetRepoLanguages(owner, name)
|
||||
if err != nil {
|
||||
langs = nil
|
||||
}
|
||||
|
||||
// Latest commit: one HTTP call per repository
|
||||
// (GET /repos/{owner}/{repo}/commits?limit=1, via ListRepoCommits), returns the
|
||||
// newest commit first. SHA + date are mapped into LatestCommit.
|
||||
latestCommit, _ := s.latestCommit(owner, name)
|
||||
|
||||
// Latest release: one HTTP call per repository
|
||||
// (GET /repos/{owner}/{repo}/releases/latest, via GetLatestRelease).
|
||||
// Returns a single Release (server already excludes drafts and prereleases).
|
||||
latestRelease, _ := s.latestRelease(owner, name)
|
||||
|
||||
// Licenses: scan repo files on its default branch with
|
||||
// go-license-detector, reusing the root listing. May require reads
|
||||
// of candidate license files. On failure the repo is kept with no
|
||||
// licenses.
|
||||
licenses := s.licensesFor(owner, name, repo.DefaultBranch, root)
|
||||
|
||||
// Readme: pick the default README (md/rst/txt/no extension) from the
|
||||
// shared root listing and fetch it. Non-fatal when absent.
|
||||
readme := s.readmeFor(owner, name, repo.DefaultBranch, root)
|
||||
|
||||
projects = append(projects, Project{
|
||||
Name: repo.FullName,
|
||||
Url: repo.HTMLURL,
|
||||
Description: repo.Description,
|
||||
OpenIssues: repo.OpenIssues,
|
||||
OpenPRs: repo.OpenPulls,
|
||||
Topics: topics,
|
||||
Languages: orderedLanguages(langs),
|
||||
LatestCommit: latestCommit,
|
||||
LatestRelease: latestRelease,
|
||||
Licenses: licenses,
|
||||
Readme: readme,
|
||||
})
|
||||
}
|
||||
|
||||
// A short page indicates that there are no further results.
|
||||
if len(repos) < pageSize {
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
return projects, nil
|
||||
}
|
||||
|
||||
// topics returns the repository's topics.
|
||||
func (s *ForgejoScraper) topics(owner, name string) ([]string, error) {
|
||||
topics, _, err := s.Client.ListRepoTopics(owner, name, forgejo.ListRepoTopicsOptions{})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list topics for %s/%s: %w", owner, name, err)
|
||||
}
|
||||
return topics, nil
|
||||
}
|
||||
|
||||
// latestCommit returns the most recent commit's time. On failure it returns
|
||||
// the zero time and the error, leaving the Project's LatestCommit unset.
|
||||
func (s *ForgejoScraper) latestCommit(owner, name string) (time.Time, error) {
|
||||
commits, _, err := s.Client.ListRepoCommits(owner, name, forgejo.ListCommitOptions{
|
||||
PageSize: 1,
|
||||
})
|
||||
if err != nil {
|
||||
return time.Time{}, err
|
||||
}
|
||||
if len(commits) == 0 || commits[0] == nil {
|
||||
return time.Time{}, nil
|
||||
}
|
||||
return commits[0].Created, nil
|
||||
}
|
||||
|
||||
// latestRelease returns the most recent non-draft, non-prerelease release.
|
||||
// On failure (e.g. no releases) it returns a zero Release and the error,
|
||||
// leaving the Project's LatestRelease unset.
|
||||
func (s *ForgejoScraper) latestRelease(owner, name string) (Release, error) {
|
||||
r, _, err := s.Client.GetLatestRelease(owner, name)
|
||||
if err != nil {
|
||||
return Release{}, err
|
||||
}
|
||||
if r == nil {
|
||||
return Release{}, nil
|
||||
}
|
||||
return Release{
|
||||
Name: r.Title,
|
||||
Date: r.PublishedAt,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// orderedLanguages returns the languages of a repo ordered by byte share,
|
||||
// most dominant first. Ties are broken alphabetically for stable output.
|
||||
func orderedLanguages(langs map[string]int64) []string {
|
||||
if len(langs) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
names := make([]string, 0, len(langs))
|
||||
for name := range langs {
|
||||
names = append(names, name)
|
||||
}
|
||||
|
||||
// Descending byte count; alphabetically as a stable tie-breaker.
|
||||
sort.Slice(names, func(i, j int) bool {
|
||||
if langs[names[i]] != langs[names[j]] {
|
||||
return langs[names[i]] > langs[names[j]]
|
||||
}
|
||||
return names[i] < names[j]
|
||||
})
|
||||
|
||||
return names
|
||||
}
|
||||
96
scraper/forgejo_licenses.go
Normal file
96
scraper/forgejo_licenses.go
Normal file
|
|
@ -0,0 +1,96 @@
|
|||
package scraper
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
forgejo "codeberg.org/mvdkleijn/forgejo-sdk/forgejo/v3"
|
||||
|
||||
"github.com/go-enry/go-license-detector/v4/licensedb"
|
||||
"github.com/go-enry/go-license-detector/v4/licensedb/filer"
|
||||
)
|
||||
|
||||
// forgejoFiler adapts the Forgejo content API to the filer.Filer interface so
|
||||
// that go-license-detector can inspect a repository's files without a local
|
||||
// checkout. ReadDir serves a cached root listing when available (to avoid an
|
||||
// extra ListContents call) and falls back to the API otherwise. ReadFile
|
||||
// downloads a file's bytes at the given ref (default branch).
|
||||
type forgejoFiler struct {
|
||||
client *forgejo.Client
|
||||
owner string
|
||||
repo string
|
||||
ref string
|
||||
// cachedRoot, when non-nil, is served for ReadDir("") to avoid a second
|
||||
// root listing that was already fetched by the caller.
|
||||
cachedRoot []filer.File
|
||||
}
|
||||
|
||||
var _ filer.Filer = (*forgejoFiler)(nil)
|
||||
|
||||
func (f *forgejoFiler) ReadDir(dirPath string) ([]filer.File, error) {
|
||||
if dirPath == "" && f.cachedRoot != nil {
|
||||
return f.cachedRoot, nil
|
||||
}
|
||||
entries, _, err := f.client.ListContents(f.owner, f.repo, f.ref, dirPath)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("list contents %q: %w", dirPath, err)
|
||||
}
|
||||
files := make([]filer.File, 0, len(entries))
|
||||
for _, e := range entries {
|
||||
if e == nil {
|
||||
continue
|
||||
}
|
||||
files = append(files, filer.File{
|
||||
Name: e.Name,
|
||||
IsDir: e.Type == "dir",
|
||||
})
|
||||
}
|
||||
return files, nil
|
||||
}
|
||||
|
||||
func (f *forgejoFiler) ReadFile(path string) ([]byte, error) {
|
||||
// GetFile returns raw file bytes (binary-safe, base64-decoded).
|
||||
data, _, err := f.client.GetFile(f.owner, f.repo, f.ref, path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read file %q: %w", path, err)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
func (f *forgejoFiler) Close() {}
|
||||
|
||||
func (f *forgejoFiler) PathsAreAlwaysSlash() bool { return true }
|
||||
|
||||
// licensesFor returns the SPDX identifiers detected for the repository. It
|
||||
// uses go-license-detector over the repository's files on its default branch.
|
||||
// The given root listing is served to the detector to avoid a duplicate,
|
||||
// already-available API call. An empty slice is returned when no license can
|
||||
// be determined (the repo is still kept in results -- license failures are
|
||||
// non-fatal).
|
||||
func (s *ForgejoScraper) licensesFor(owner, repo, ref string, root []filer.File) []string {
|
||||
detector := &forgejoFiler{
|
||||
client: s.Client,
|
||||
owner: owner,
|
||||
repo: repo,
|
||||
ref: ref,
|
||||
cachedRoot: root,
|
||||
}
|
||||
|
||||
matches, err := licensedb.Detect(detector)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
// Keep only the highest-confidence match, sorted deterministically.
|
||||
ids := make([]string, 0, len(matches))
|
||||
bestID, bestConf := "", float32(-1)
|
||||
for id, m := range matches {
|
||||
if m.Confidence > bestConf {
|
||||
bestConf = m.Confidence
|
||||
bestID = id
|
||||
}
|
||||
}
|
||||
if bestID != "" {
|
||||
ids = append(ids, bestID)
|
||||
}
|
||||
return ids
|
||||
}
|
||||
70
scraper/forgejo_readme.go
Normal file
70
scraper/forgejo_readme.go
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
package scraper
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/go-enry/go-license-detector/v4/licensedb/filer"
|
||||
)
|
||||
|
||||
// readmeCandidates are the filenames looked for, in priority order, when
|
||||
// retrieving a repository's README.
|
||||
var readmeCandidates = []string{"README.md", "README.rst", "README.txt", "README"}
|
||||
|
||||
// readmeFor returns the content of the repository's README on the given ref
|
||||
// (default branch), picking from root (the pre-fetched root file listing) the
|
||||
// first candidate found (case-insensitively). An empty string is returned when
|
||||
// no README can be found; the repo is still kept in results either way.
|
||||
func (s *ForgejoScraper) readmeFor(owner, name, ref string, root []filer.File) string {
|
||||
// Collect root file names, lowercased for case-insensitive matching, while
|
||||
// keeping the real name for the GetFile call.
|
||||
type rootFile struct {
|
||||
real string
|
||||
low string
|
||||
}
|
||||
files := make([]rootFile, 0, len(root))
|
||||
for _, e := range root {
|
||||
if e.IsDir {
|
||||
continue
|
||||
}
|
||||
files = append(files, rootFile{real: e.Name, low: strings.ToLower(e.Name)})
|
||||
}
|
||||
|
||||
// Candidate priority wins; ties broken by alphabetical real name for
|
||||
// deterministic picks when a repo has e.g. both README and README.md.
|
||||
sort.SliceStable(files, func(i, j int) bool {
|
||||
pi, pj := readmePriority(files[i].low), readmePriority(files[j].low)
|
||||
if pi != pj {
|
||||
return pi < pj // lower = higher priority
|
||||
}
|
||||
return files[i].real < files[j].real
|
||||
})
|
||||
|
||||
var chosen string
|
||||
for i := range files {
|
||||
if readmePriority(files[i].low) >= 0 {
|
||||
chosen = files[i].real
|
||||
break
|
||||
}
|
||||
}
|
||||
if chosen == "" {
|
||||
return ""
|
||||
}
|
||||
|
||||
data, _, err := s.Client.GetFile(owner, name, ref, chosen)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
return string(data)
|
||||
}
|
||||
|
||||
// readmePriority returns the index of a lowercased filename in
|
||||
// readmeCandidates, or -1 if it is not a readme candidate.
|
||||
func readmePriority(low string) int {
|
||||
for i, c := range readmeCandidates {
|
||||
if low == strings.ToLower(c) {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
157
scraper/robots.go
Normal file
157
scraper/robots.go
Normal file
|
|
@ -0,0 +1,157 @@
|
|||
package scraper
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// UserAgent identifies this crawler to servers. Keep it stable so that
|
||||
// instances can set robots rules for us specifically.
|
||||
const UserAgent = "ffd-indexer/0.1"
|
||||
|
||||
// robotsHTTPClient is used for fetching robots.txt. It does not follow the
|
||||
// SDK client's settings; it is a plain, short-timeout client.
|
||||
var robotsHTTPClient = &http.Client{Timeout: 10 * time.Second}
|
||||
|
||||
// Allowed reports whether an instance at baseURL permits crawling its API
|
||||
// according to its robots.txt. It only consults the rules for the
|
||||
// User-Agent "*"; a block of /api/ (the API root) disallows scraping.
|
||||
//
|
||||
// A failure to fetch or parse robots.txt is treated as an error; callers
|
||||
// decide whether that is fatal.
|
||||
func Allowed(baseURL string) (bool, error) {
|
||||
u, err := url.Parse(baseURL)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("parse base URL: %w", err)
|
||||
}
|
||||
u.Path = "/robots.txt"
|
||||
|
||||
req, err := http.NewRequest(http.MethodGet, u.String(), nil)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("build robots request: %w", err)
|
||||
}
|
||||
req.Header.Set("User-Agent", UserAgent)
|
||||
|
||||
resp, err := robotsHTTPClient.Do(req)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("fetch robots.txt: %w", err)
|
||||
}
|
||||
defer func() {
|
||||
_ = resp.Body.Close()
|
||||
}()
|
||||
|
||||
// A 404 or similar means "no robots.txt": assume allowed.
|
||||
if resp.StatusCode == http.StatusNotFound {
|
||||
return true, nil
|
||||
}
|
||||
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
||||
return false, fmt.Errorf("robots.txt returned %s", resp.Status)
|
||||
}
|
||||
|
||||
body, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("read robots.txt: %w", err)
|
||||
}
|
||||
|
||||
return allowedFor(string(body), UserAgent), nil
|
||||
}
|
||||
|
||||
// allowedFor parses a robots.txt body and reports whether a crawler with the
|
||||
// given user agent may crawl the path "/api/". It implements a reasonable
|
||||
// subset of the robots.txt standard:
|
||||
//
|
||||
// - Only the "*" group and an explicit group matching userAgent are
|
||||
// considered; later groups override earlier ones for the same agent.
|
||||
// - Disallow entries are matched as path prefixes (a trailing "*" is
|
||||
// stripped). No other wildcard/glob support.
|
||||
// - An empty group (no rules) is "allow everything".
|
||||
func allowedFor(body, userAgent string) bool {
|
||||
groups := parseRobotGroups(body)
|
||||
|
||||
// A group that explicitly names our user agent takes precedence over a
|
||||
// "*" group, regardless of order. Otherwise fall back to "*".
|
||||
var star *robotGroup
|
||||
for _, g := range groups {
|
||||
for _, a := range g.agents {
|
||||
if strings.EqualFold(a, "*") {
|
||||
star = g
|
||||
} else if strings.EqualFold(a, userAgent) {
|
||||
return !matchDisallow(g, "/api/")
|
||||
}
|
||||
}
|
||||
}
|
||||
if star != nil {
|
||||
return !matchDisallow(star, "/api/")
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
type robotGroup struct {
|
||||
agents []string
|
||||
disallow []string
|
||||
}
|
||||
|
||||
// parseRobotGroups splits a robots.txt body into agent groups.
|
||||
func parseRobotGroups(body string) []*robotGroup {
|
||||
var groups []*robotGroup
|
||||
var cur *robotGroup
|
||||
|
||||
for _, raw := range strings.Split(body, "\n") {
|
||||
line := strings.TrimSpace(raw)
|
||||
// Strip an inline comment (not part of a rule's result).
|
||||
if i := strings.IndexByte(line, '#'); i >= 0 {
|
||||
line = strings.TrimSpace(line[:i])
|
||||
}
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
key, val, ok := strings.Cut(line, ":")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
key = strings.ToLower(strings.TrimSpace(key))
|
||||
val = strings.TrimSpace(val)
|
||||
|
||||
switch key {
|
||||
case "user-agent":
|
||||
// A User-agent line starts a new group, unless the current
|
||||
// group is still collecting agents (has no Disallow yet).
|
||||
if cur == nil || len(cur.disallow) > 0 {
|
||||
cur = &robotGroup{}
|
||||
groups = append(groups, cur)
|
||||
}
|
||||
cur.agents = append(cur.agents, val)
|
||||
case "disallow":
|
||||
if cur != nil {
|
||||
cur.disallow = append(cur.disallow, val)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return groups
|
||||
}
|
||||
|
||||
// matchDisallow reports whether a path is disallowed by a group's rules.
|
||||
// A trailing "*" is stripped. Rules are applied in order, later rules
|
||||
// overriding earlier ones: an empty Disallow rule revokes all preceding
|
||||
// disallows (allow everything).
|
||||
func matchDisallow(g *robotGroup, path string) bool {
|
||||
disallowed := false
|
||||
for _, rule := range g.disallow {
|
||||
rule = strings.TrimSpace(rule)
|
||||
rule = strings.TrimSuffix(rule, "*")
|
||||
if rule == "" {
|
||||
disallowed = false // empty Disallow: allow everything
|
||||
continue
|
||||
}
|
||||
if strings.HasPrefix(path, rule) {
|
||||
disallowed = true
|
||||
}
|
||||
}
|
||||
return disallowed
|
||||
}
|
||||
60
scraper/robots_test.go
Normal file
60
scraper/robots_test.go
Normal file
|
|
@ -0,0 +1,60 @@
|
|||
package scraper
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestAllowedFor(t *testing.T) {
|
||||
const ua = "ffd-indexer/0.1"
|
||||
|
||||
// xengi / cccb: only /archive/ blocked -> /api/ allowed.
|
||||
if !allowedFor("User-agent: *\nDisallow: /*/*/archive/\n", ua) {
|
||||
t.Error("expected /api/ allowed when only archive is disallowed")
|
||||
}
|
||||
|
||||
// codeberg: /api/ disallowed under "*".
|
||||
codeberg := `User-agent: *
|
||||
Disallow: /api/
|
||||
Disallow: /avatars/
|
||||
`
|
||||
if allowedFor(codeberg, ua) {
|
||||
t.Error("expected /api/ blocked when /api/ is disallowed")
|
||||
}
|
||||
|
||||
// Explicit agent group takes precedence over "*" (agent name matches
|
||||
// UserAgent exactly, including the version).
|
||||
// Here the explicit group is for "ffd-indexer" (no version), so it does
|
||||
// not match "ffd-indexer/0.1"; the "*" group (blocks /api/) applies.
|
||||
if allowedFor("User-agent: *\nDisallow: /api/\n\nUser-agent: ffd-indexer\nDisallow: /whatever/\n", ua) {
|
||||
t.Error("expected unmatched explicit agent to fall back to * which blocks /api/")
|
||||
}
|
||||
|
||||
// A matching explicit agent group takes precedence over "*".
|
||||
if !allowedFor("User-agent: *\nDisallow: /api/\n\nUser-agent: ffd-indexer/0.1\nDisallow: /whatever/\n", ua) {
|
||||
t.Error("expected explicit agent rule to permit /api/")
|
||||
}
|
||||
|
||||
// No rules at all -> allow.
|
||||
if !allowedFor("", ua) {
|
||||
t.Error("expected empty robots to allow")
|
||||
}
|
||||
|
||||
// Only an unrelated agent listed -> no star, no match -> allow.
|
||||
if !allowedFor("User-agent: SomeBot\nDisallow: /api/\n", ua) {
|
||||
t.Error("expected unrelated agent not to block us")
|
||||
}
|
||||
|
||||
// Disallow "/" blocks everything including /api/.
|
||||
if allowedFor("User-agent: *\nDisallow: /\n", ua) {
|
||||
t.Error("expected Disallow / to block /api/")
|
||||
}
|
||||
|
||||
// Empty Disallow (allowed) overrides a previous block in the same group.
|
||||
if !allowedFor("User-agent: *\nDisallow: /api/\nDisallow:\n", ua) {
|
||||
t.Error("expected empty Disallow to allow /api/")
|
||||
}
|
||||
|
||||
// Wildcard on a non-prefix rule should not over-block (substring-only).
|
||||
// "/*.pdf" should not match /api/.
|
||||
if !allowedFor("User-agent: *\nDisallow: /*.pdf$\n", ua) {
|
||||
t.Error("expected unrelated wildcard to not block /api/")
|
||||
}
|
||||
}
|
||||
47
scraper/scraper.go
Normal file
47
scraper/scraper.go
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
// Package scraper defines the interfaces and shared types used to discover
|
||||
// projects from forge instances (Forgejo, Gogs, ...).
|
||||
package scraper
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Scraper is implemented by concrete, per-forge scrapers.
|
||||
type Scraper interface {
|
||||
// GetProjects retrieves every project discoverable through this forge's
|
||||
// public API and maps it to the generic Project type.
|
||||
GetProjects() ([]Project, error)
|
||||
}
|
||||
|
||||
// NewScraper returns a Scraper for the given forge type, e.g. "forgejo".
|
||||
func NewScraper(typ, baseURL string) (Scraper, error) {
|
||||
switch typ {
|
||||
case "forgejo":
|
||||
return NewForgejoScraper(baseURL)
|
||||
default:
|
||||
return nil, fmt.Errorf("unknown forge type %q", typ)
|
||||
}
|
||||
}
|
||||
|
||||
// Project is a generic, forge-agnostic representation of an open source project.
|
||||
type Project struct {
|
||||
Name string `json:"name"`
|
||||
Url string `json:"url"`
|
||||
Description string `json:"description"`
|
||||
Topics []string `json:"topics"`
|
||||
OpenIssues int `json:"open_issues"`
|
||||
OpenPRs int `json:"open_prs"`
|
||||
LatestCommit time.Time `json:"latest_commit"`
|
||||
LatestRelease Release `json:"latest_release"`
|
||||
Languages []string `json:"languages"`
|
||||
Readme string `json:"readme"`
|
||||
Licenses []string `json:"license"`
|
||||
UsesAI bool `json:"uses_ai"`
|
||||
}
|
||||
|
||||
// Release is a named release with the date it was published.
|
||||
type Release struct {
|
||||
Name string `json:"name"`
|
||||
Date time.Time `json:"date"`
|
||||
}
|
||||
Loading…
Reference in a new issue