scraper done; scrapes to local sqlite db

This commit is contained in:
Ricardo (XenGi) Band 2026-09-28 15:36:42 +02:00
commit bde13e9b68
No known key found for this signature in database
18 changed files with 1585 additions and 164 deletions

230
scraper/forgejo.go Normal file
View file

@ -0,0 +1,230 @@
// Package scraper forgejo.go - Forgejo scraper.
package scraper
import (
"fmt"
"sort"
"strings"
"time"
forgejo "codeberg.org/mvdkleijn/forgejo-sdk/forgejo/v3"
"github.com/go-enry/go-license-detector/v4/licensedb/filer"
)
// ForgejoScraper retrieves public repositories from a Forgejo instance.
type ForgejoScraper struct {
Client *forgejo.Client
BaseURL string
PageSize int
}
// NewForgejoScraper creates a scraper for a Forgejo instance.
// BaseURL should be the instance URL, e.g. https://codeberg.org.
func NewForgejoScraper(baseURL string) (*ForgejoScraper, error) {
client, err := forgejo.NewClient(baseURL)
if err != nil {
return nil, fmt.Errorf("create Forgejo client: %w", err)
}
client.SetUserAgent(UserAgent)
return &ForgejoScraper{
Client: client,
BaseURL: strings.TrimSuffix(baseURL, "/"),
PageSize: 50,
}, nil
}
// GetProjects retrieves every public repository visible through the
// Forgejo repository search API and maps it to the generic Project type.
func (s *ForgejoScraper) GetProjects() ([]Project, error) {
// Honor the instance's robots.txt before making any API call:
// if /api/ is disallowed, do not scrape this instance.
ok, err := Allowed(s.BaseURL)
if err != nil {
return nil, fmt.Errorf("check robots.txt: %w", err)
}
if !ok {
return nil, fmt.Errorf("instance robots.txt disallows /api/")
}
pageSize := s.PageSize
if pageSize <= 0 {
pageSize = 50
}
isPrivate := false
var projects []Project
for page := 1; ; page++ {
repos, _, err := s.Client.SearchRepos(forgejo.SearchRepoOptions{
Page: page,
PageSize: pageSize,
IsPrivate: &isPrivate,
Type: forgejo.RepoTypeSource,
Sort: "alpha",
Order: "asc",
})
if err != nil {
return nil, fmt.Errorf(
"list public Forgejo repositories (page %d): %w",
page,
err,
)
}
if len(repos) == 0 {
break
}
for _, repo := range repos {
if repo == nil {
continue
}
owner, name, ok := strings.Cut(repo.FullName, "/")
if !ok {
continue
}
// List the repository root once and reuse it for readme selection
// and license detection (they previously each listed the root).
entries, _, err := s.Client.ListContents(owner, name, repo.DefaultBranch, "")
if err != nil {
// best-effort: fall back to empty root so the other fields still work
entries = nil
}
root := make([]filer.File, 0, len(entries))
for _, e := range entries {
if e == nil {
continue
}
root = append(root, filer.File{Name: e.Name, IsDir: e.Type == "dir"})
}
// topics currently requires one extra HTTP call per repository.
// Future optimization: the /repos/search response already includes
// the topics field, so they could be gathered by parsing the
// SearchRepos payload itself (e.g. via ListRepos) instead.
// On failure the repo is kept in the results with empty topics.
topics, err := s.topics(owner, name)
if err != nil {
topics = nil
}
// languages requires one extra HTTP call per repository
// (GET /repos/{owner}/{repo}/languages, via GetRepoLanguages).
// It returns a map[language]bytes; we order it most-dominant-first
// below. On failure the repo is kept with no languages.
langs, _, err := s.Client.GetRepoLanguages(owner, name)
if err != nil {
langs = nil
}
// Latest commit: one HTTP call per repository
// (GET /repos/{owner}/{repo}/commits?limit=1, via ListRepoCommits), returns the
// newest commit first. SHA + date are mapped into LatestCommit.
latestCommit, _ := s.latestCommit(owner, name)
// Latest release: one HTTP call per repository
// (GET /repos/{owner}/{repo}/releases/latest, via GetLatestRelease).
// Returns a single Release (server already excludes drafts and prereleases).
latestRelease, _ := s.latestRelease(owner, name)
// Licenses: scan repo files on its default branch with
// go-license-detector, reusing the root listing. May require reads
// of candidate license files. On failure the repo is kept with no
// licenses.
licenses := s.licensesFor(owner, name, repo.DefaultBranch, root)
// Readme: pick the default README (md/rst/txt/no extension) from the
// shared root listing and fetch it. Non-fatal when absent.
readme := s.readmeFor(owner, name, repo.DefaultBranch, root)
projects = append(projects, Project{
Name: repo.FullName,
Url: repo.HTMLURL,
Description: repo.Description,
OpenIssues: repo.OpenIssues,
OpenPRs: repo.OpenPulls,
Topics: topics,
Languages: orderedLanguages(langs),
LatestCommit: latestCommit,
LatestRelease: latestRelease,
Licenses: licenses,
Readme: readme,
})
}
// A short page indicates that there are no further results.
if len(repos) < pageSize {
break
}
}
return projects, nil
}
// topics returns the repository's topics.
func (s *ForgejoScraper) topics(owner, name string) ([]string, error) {
topics, _, err := s.Client.ListRepoTopics(owner, name, forgejo.ListRepoTopicsOptions{})
if err != nil {
return nil, fmt.Errorf("list topics for %s/%s: %w", owner, name, err)
}
return topics, nil
}
// latestCommit returns the most recent commit's time. On failure it returns
// the zero time and the error, leaving the Project's LatestCommit unset.
func (s *ForgejoScraper) latestCommit(owner, name string) (time.Time, error) {
commits, _, err := s.Client.ListRepoCommits(owner, name, forgejo.ListCommitOptions{
PageSize: 1,
})
if err != nil {
return time.Time{}, err
}
if len(commits) == 0 || commits[0] == nil {
return time.Time{}, nil
}
return commits[0].Created, nil
}
// latestRelease returns the most recent non-draft, non-prerelease release.
// On failure (e.g. no releases) it returns a zero Release and the error,
// leaving the Project's LatestRelease unset.
func (s *ForgejoScraper) latestRelease(owner, name string) (Release, error) {
r, _, err := s.Client.GetLatestRelease(owner, name)
if err != nil {
return Release{}, err
}
if r == nil {
return Release{}, nil
}
return Release{
Name: r.Title,
Date: r.PublishedAt,
}, nil
}
// orderedLanguages returns the languages of a repo ordered by byte share,
// most dominant first. Ties are broken alphabetically for stable output.
func orderedLanguages(langs map[string]int64) []string {
if len(langs) == 0 {
return nil
}
names := make([]string, 0, len(langs))
for name := range langs {
names = append(names, name)
}
// Descending byte count; alphabetically as a stable tie-breaker.
sort.Slice(names, func(i, j int) bool {
if langs[names[i]] != langs[names[j]] {
return langs[names[i]] > langs[names[j]]
}
return names[i] < names[j]
})
return names
}

View file

@ -0,0 +1,96 @@
package scraper
import (
"fmt"
forgejo "codeberg.org/mvdkleijn/forgejo-sdk/forgejo/v3"
"github.com/go-enry/go-license-detector/v4/licensedb"
"github.com/go-enry/go-license-detector/v4/licensedb/filer"
)
// forgejoFiler adapts the Forgejo content API to the filer.Filer interface so
// that go-license-detector can inspect a repository's files without a local
// checkout. ReadDir serves a cached root listing when available (to avoid an
// extra ListContents call) and falls back to the API otherwise. ReadFile
// downloads a file's bytes at the given ref (default branch).
type forgejoFiler struct {
client *forgejo.Client
owner string
repo string
ref string
// cachedRoot, when non-nil, is served for ReadDir("") to avoid a second
// root listing that was already fetched by the caller.
cachedRoot []filer.File
}
var _ filer.Filer = (*forgejoFiler)(nil)
func (f *forgejoFiler) ReadDir(dirPath string) ([]filer.File, error) {
if dirPath == "" && f.cachedRoot != nil {
return f.cachedRoot, nil
}
entries, _, err := f.client.ListContents(f.owner, f.repo, f.ref, dirPath)
if err != nil {
return nil, fmt.Errorf("list contents %q: %w", dirPath, err)
}
files := make([]filer.File, 0, len(entries))
for _, e := range entries {
if e == nil {
continue
}
files = append(files, filer.File{
Name: e.Name,
IsDir: e.Type == "dir",
})
}
return files, nil
}
func (f *forgejoFiler) ReadFile(path string) ([]byte, error) {
// GetFile returns raw file bytes (binary-safe, base64-decoded).
data, _, err := f.client.GetFile(f.owner, f.repo, f.ref, path)
if err != nil {
return nil, fmt.Errorf("read file %q: %w", path, err)
}
return data, nil
}
func (f *forgejoFiler) Close() {}
func (f *forgejoFiler) PathsAreAlwaysSlash() bool { return true }
// licensesFor returns the SPDX identifiers detected for the repository. It
// uses go-license-detector over the repository's files on its default branch.
// The given root listing is served to the detector to avoid a duplicate,
// already-available API call. An empty slice is returned when no license can
// be determined (the repo is still kept in results -- license failures are
// non-fatal).
func (s *ForgejoScraper) licensesFor(owner, repo, ref string, root []filer.File) []string {
detector := &forgejoFiler{
client: s.Client,
owner: owner,
repo: repo,
ref: ref,
cachedRoot: root,
}
matches, err := licensedb.Detect(detector)
if err != nil {
return nil
}
// Keep only the highest-confidence match, sorted deterministically.
ids := make([]string, 0, len(matches))
bestID, bestConf := "", float32(-1)
for id, m := range matches {
if m.Confidence > bestConf {
bestConf = m.Confidence
bestID = id
}
}
if bestID != "" {
ids = append(ids, bestID)
}
return ids
}

70
scraper/forgejo_readme.go Normal file
View file

@ -0,0 +1,70 @@
package scraper
import (
"sort"
"strings"
"github.com/go-enry/go-license-detector/v4/licensedb/filer"
)
// readmeCandidates are the filenames looked for, in priority order, when
// retrieving a repository's README.
var readmeCandidates = []string{"README.md", "README.rst", "README.txt", "README"}
// readmeFor returns the content of the repository's README on the given ref
// (default branch), picking from root (the pre-fetched root file listing) the
// first candidate found (case-insensitively). An empty string is returned when
// no README can be found; the repo is still kept in results either way.
func (s *ForgejoScraper) readmeFor(owner, name, ref string, root []filer.File) string {
// Collect root file names, lowercased for case-insensitive matching, while
// keeping the real name for the GetFile call.
type rootFile struct {
real string
low string
}
files := make([]rootFile, 0, len(root))
for _, e := range root {
if e.IsDir {
continue
}
files = append(files, rootFile{real: e.Name, low: strings.ToLower(e.Name)})
}
// Candidate priority wins; ties broken by alphabetical real name for
// deterministic picks when a repo has e.g. both README and README.md.
sort.SliceStable(files, func(i, j int) bool {
pi, pj := readmePriority(files[i].low), readmePriority(files[j].low)
if pi != pj {
return pi < pj // lower = higher priority
}
return files[i].real < files[j].real
})
var chosen string
for i := range files {
if readmePriority(files[i].low) >= 0 {
chosen = files[i].real
break
}
}
if chosen == "" {
return ""
}
data, _, err := s.Client.GetFile(owner, name, ref, chosen)
if err != nil {
return ""
}
return string(data)
}
// readmePriority returns the index of a lowercased filename in
// readmeCandidates, or -1 if it is not a readme candidate.
func readmePriority(low string) int {
for i, c := range readmeCandidates {
if low == strings.ToLower(c) {
return i
}
}
return -1
}

157
scraper/robots.go Normal file
View file

@ -0,0 +1,157 @@
package scraper
import (
"fmt"
"io"
"net/http"
"net/url"
"strings"
"time"
)
// UserAgent identifies this crawler to servers. Keep it stable so that
// instances can set robots rules for us specifically.
const UserAgent = "ffd-indexer/0.1"
// robotsHTTPClient is used for fetching robots.txt. It does not follow the
// SDK client's settings; it is a plain, short-timeout client.
var robotsHTTPClient = &http.Client{Timeout: 10 * time.Second}
// Allowed reports whether an instance at baseURL permits crawling its API
// according to its robots.txt. It only consults the rules for the
// User-Agent "*"; a block of /api/ (the API root) disallows scraping.
//
// A failure to fetch or parse robots.txt is treated as an error; callers
// decide whether that is fatal.
func Allowed(baseURL string) (bool, error) {
u, err := url.Parse(baseURL)
if err != nil {
return false, fmt.Errorf("parse base URL: %w", err)
}
u.Path = "/robots.txt"
req, err := http.NewRequest(http.MethodGet, u.String(), nil)
if err != nil {
return false, fmt.Errorf("build robots request: %w", err)
}
req.Header.Set("User-Agent", UserAgent)
resp, err := robotsHTTPClient.Do(req)
if err != nil {
return false, fmt.Errorf("fetch robots.txt: %w", err)
}
defer func() {
_ = resp.Body.Close()
}()
// A 404 or similar means "no robots.txt": assume allowed.
if resp.StatusCode == http.StatusNotFound {
return true, nil
}
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
return false, fmt.Errorf("robots.txt returned %s", resp.Status)
}
body, err := io.ReadAll(resp.Body)
if err != nil {
return false, fmt.Errorf("read robots.txt: %w", err)
}
return allowedFor(string(body), UserAgent), nil
}
// allowedFor parses a robots.txt body and reports whether a crawler with the
// given user agent may crawl the path "/api/". It implements a reasonable
// subset of the robots.txt standard:
//
// - Only the "*" group and an explicit group matching userAgent are
// considered; later groups override earlier ones for the same agent.
// - Disallow entries are matched as path prefixes (a trailing "*" is
// stripped). No other wildcard/glob support.
// - An empty group (no rules) is "allow everything".
func allowedFor(body, userAgent string) bool {
groups := parseRobotGroups(body)
// A group that explicitly names our user agent takes precedence over a
// "*" group, regardless of order. Otherwise fall back to "*".
var star *robotGroup
for _, g := range groups {
for _, a := range g.agents {
if strings.EqualFold(a, "*") {
star = g
} else if strings.EqualFold(a, userAgent) {
return !matchDisallow(g, "/api/")
}
}
}
if star != nil {
return !matchDisallow(star, "/api/")
}
return true
}
type robotGroup struct {
agents []string
disallow []string
}
// parseRobotGroups splits a robots.txt body into agent groups.
func parseRobotGroups(body string) []*robotGroup {
var groups []*robotGroup
var cur *robotGroup
for _, raw := range strings.Split(body, "\n") {
line := strings.TrimSpace(raw)
// Strip an inline comment (not part of a rule's result).
if i := strings.IndexByte(line, '#'); i >= 0 {
line = strings.TrimSpace(line[:i])
}
if line == "" {
continue
}
key, val, ok := strings.Cut(line, ":")
if !ok {
continue
}
key = strings.ToLower(strings.TrimSpace(key))
val = strings.TrimSpace(val)
switch key {
case "user-agent":
// A User-agent line starts a new group, unless the current
// group is still collecting agents (has no Disallow yet).
if cur == nil || len(cur.disallow) > 0 {
cur = &robotGroup{}
groups = append(groups, cur)
}
cur.agents = append(cur.agents, val)
case "disallow":
if cur != nil {
cur.disallow = append(cur.disallow, val)
}
}
}
return groups
}
// matchDisallow reports whether a path is disallowed by a group's rules.
// A trailing "*" is stripped. Rules are applied in order, later rules
// overriding earlier ones: an empty Disallow rule revokes all preceding
// disallows (allow everything).
func matchDisallow(g *robotGroup, path string) bool {
disallowed := false
for _, rule := range g.disallow {
rule = strings.TrimSpace(rule)
rule = strings.TrimSuffix(rule, "*")
if rule == "" {
disallowed = false // empty Disallow: allow everything
continue
}
if strings.HasPrefix(path, rule) {
disallowed = true
}
}
return disallowed
}

60
scraper/robots_test.go Normal file
View file

@ -0,0 +1,60 @@
package scraper
import "testing"
func TestAllowedFor(t *testing.T) {
const ua = "ffd-indexer/0.1"
// xengi / cccb: only /archive/ blocked -> /api/ allowed.
if !allowedFor("User-agent: *\nDisallow: /*/*/archive/\n", ua) {
t.Error("expected /api/ allowed when only archive is disallowed")
}
// codeberg: /api/ disallowed under "*".
codeberg := `User-agent: *
Disallow: /api/
Disallow: /avatars/
`
if allowedFor(codeberg, ua) {
t.Error("expected /api/ blocked when /api/ is disallowed")
}
// Explicit agent group takes precedence over "*" (agent name matches
// UserAgent exactly, including the version).
// Here the explicit group is for "ffd-indexer" (no version), so it does
// not match "ffd-indexer/0.1"; the "*" group (blocks /api/) applies.
if allowedFor("User-agent: *\nDisallow: /api/\n\nUser-agent: ffd-indexer\nDisallow: /whatever/\n", ua) {
t.Error("expected unmatched explicit agent to fall back to * which blocks /api/")
}
// A matching explicit agent group takes precedence over "*".
if !allowedFor("User-agent: *\nDisallow: /api/\n\nUser-agent: ffd-indexer/0.1\nDisallow: /whatever/\n", ua) {
t.Error("expected explicit agent rule to permit /api/")
}
// No rules at all -> allow.
if !allowedFor("", ua) {
t.Error("expected empty robots to allow")
}
// Only an unrelated agent listed -> no star, no match -> allow.
if !allowedFor("User-agent: SomeBot\nDisallow: /api/\n", ua) {
t.Error("expected unrelated agent not to block us")
}
// Disallow "/" blocks everything including /api/.
if allowedFor("User-agent: *\nDisallow: /\n", ua) {
t.Error("expected Disallow / to block /api/")
}
// Empty Disallow (allowed) overrides a previous block in the same group.
if !allowedFor("User-agent: *\nDisallow: /api/\nDisallow:\n", ua) {
t.Error("expected empty Disallow to allow /api/")
}
// Wildcard on a non-prefix rule should not over-block (substring-only).
// "/*.pdf" should not match /api/.
if !allowedFor("User-agent: *\nDisallow: /*.pdf$\n", ua) {
t.Error("expected unrelated wildcard to not block /api/")
}
}

47
scraper/scraper.go Normal file
View file

@ -0,0 +1,47 @@
// Package scraper defines the interfaces and shared types used to discover
// projects from forge instances (Forgejo, Gogs, ...).
package scraper
import (
"fmt"
"time"
)
// Scraper is implemented by concrete, per-forge scrapers.
type Scraper interface {
// GetProjects retrieves every project discoverable through this forge's
// public API and maps it to the generic Project type.
GetProjects() ([]Project, error)
}
// NewScraper returns a Scraper for the given forge type, e.g. "forgejo".
func NewScraper(typ, baseURL string) (Scraper, error) {
switch typ {
case "forgejo":
return NewForgejoScraper(baseURL)
default:
return nil, fmt.Errorf("unknown forge type %q", typ)
}
}
// Project is a generic, forge-agnostic representation of an open source project.
type Project struct {
Name string `json:"name"`
Url string `json:"url"`
Description string `json:"description"`
Topics []string `json:"topics"`
OpenIssues int `json:"open_issues"`
OpenPRs int `json:"open_prs"`
LatestCommit time.Time `json:"latest_commit"`
LatestRelease Release `json:"latest_release"`
Languages []string `json:"languages"`
Readme string `json:"readme"`
Licenses []string `json:"license"`
UsesAI bool `json:"uses_ai"`
}
// Release is a named release with the date it was published.
type Release struct {
Name string `json:"name"`
Date time.Time `json:"date"`
}