// Package scraper forgejo.go - Forgejo scraper. package scraper import ( "fmt" "sort" "strings" "time" forgejo "codeberg.org/mvdkleijn/forgejo-sdk/forgejo/v3" "github.com/go-enry/go-license-detector/v4/licensedb/filer" ) // ForgejoScraper retrieves public repositories from a Forgejo instance. type ForgejoScraper struct { Client *forgejo.Client BaseURL string PageSize int } // NewForgejoScraper creates a scraper for a Forgejo instance. // BaseURL should be the instance URL, e.g. https://codeberg.org. func NewForgejoScraper(baseURL string) (*ForgejoScraper, error) { client, err := forgejo.NewClient(baseURL) if err != nil { return nil, fmt.Errorf("create Forgejo client: %w", err) } client.SetUserAgent(UserAgent) return &ForgejoScraper{ Client: client, BaseURL: strings.TrimSuffix(baseURL, "/"), PageSize: 50, }, nil } // GetProjects retrieves every public repository visible through the // Forgejo repository search API and maps it to the generic Project type. func (s *ForgejoScraper) GetProjects() ([]Project, error) { // Honor the instance's robots.txt before making any API call: // if /api/ is disallowed, do not scrape this instance. ok, err := Allowed(s.BaseURL) if err != nil { return nil, fmt.Errorf("check robots.txt: %w", err) } if !ok { return nil, fmt.Errorf("instance robots.txt disallows /api/") } pageSize := s.PageSize if pageSize <= 0 { pageSize = 50 } isPrivate := false var projects []Project for page := 1; ; page++ { repos, _, err := s.Client.SearchRepos(forgejo.SearchRepoOptions{ Page: page, PageSize: pageSize, IsPrivate: &isPrivate, Type: forgejo.RepoTypeSource, Sort: "alpha", Order: "asc", }) if err != nil { return nil, fmt.Errorf( "list public Forgejo repositories (page %d): %w", page, err, ) } if len(repos) == 0 { break } for _, repo := range repos { if repo == nil { continue } owner, name, ok := strings.Cut(repo.FullName, "/") if !ok { continue } // List the repository root once and reuse it for readme selection // and license detection (they previously each listed the root). entries, _, err := s.Client.ListContents(owner, name, repo.DefaultBranch, "") if err != nil { // best-effort: fall back to empty root so the other fields still work entries = nil } root := make([]filer.File, 0, len(entries)) for _, e := range entries { if e == nil { continue } root = append(root, filer.File{Name: e.Name, IsDir: e.Type == "dir"}) } // topics currently requires one extra HTTP call per repository. // Future optimization: the /repos/search response already includes // the topics field, so they could be gathered by parsing the // SearchRepos payload itself (e.g. via ListRepos) instead. // On failure the repo is kept in the results with empty topics. topics, err := s.topics(owner, name) if err != nil { topics = nil } // languages requires one extra HTTP call per repository // (GET /repos/{owner}/{repo}/languages, via GetRepoLanguages). // It returns a map[language]bytes; we order it most-dominant-first // below. On failure the repo is kept with no languages. langs, _, err := s.Client.GetRepoLanguages(owner, name) if err != nil { langs = nil } // Latest commit: one HTTP call per repository // (GET /repos/{owner}/{repo}/commits?limit=1, via ListRepoCommits), returns the // newest commit first. SHA + date are mapped into LatestCommit. latestCommit, _ := s.latestCommit(owner, name) // Latest release: one HTTP call per repository // (GET /repos/{owner}/{repo}/releases/latest, via GetLatestRelease). // Returns a single Release (server already excludes drafts and prereleases). latestRelease, _ := s.latestRelease(owner, name) // Licenses: scan repo files on its default branch with // go-license-detector, reusing the root listing. May require reads // of candidate license files. On failure the repo is kept with no // licenses. licenses := s.licensesFor(owner, name, repo.DefaultBranch, root) // Readme: pick the default README (md/rst/txt/no extension) from the // shared root listing and fetch it. Non-fatal when absent. readme := s.readmeFor(owner, name, repo.DefaultBranch, root) projects = append(projects, Project{ Name: repo.FullName, Url: repo.HTMLURL, Description: repo.Description, OpenIssues: repo.OpenIssues, OpenPRs: repo.OpenPulls, Topics: topics, Languages: orderedLanguages(langs), LatestCommit: latestCommit, LatestRelease: latestRelease, Licenses: licenses, Readme: readme, }) } // A short page indicates that there are no further results. if len(repos) < pageSize { break } } return projects, nil } // topics returns the repository's topics. func (s *ForgejoScraper) topics(owner, name string) ([]string, error) { topics, _, err := s.Client.ListRepoTopics(owner, name, forgejo.ListRepoTopicsOptions{}) if err != nil { return nil, fmt.Errorf("list topics for %s/%s: %w", owner, name, err) } return topics, nil } // latestCommit returns the most recent commit's time. On failure it returns // the zero time and the error, leaving the Project's LatestCommit unset. func (s *ForgejoScraper) latestCommit(owner, name string) (time.Time, error) { commits, _, err := s.Client.ListRepoCommits(owner, name, forgejo.ListCommitOptions{ PageSize: 1, }) if err != nil { return time.Time{}, err } if len(commits) == 0 || commits[0] == nil { return time.Time{}, nil } return commits[0].Created, nil } // latestRelease returns the most recent non-draft, non-prerelease release. // On failure (e.g. no releases) it returns a zero Release and the error, // leaving the Project's LatestRelease unset. func (s *ForgejoScraper) latestRelease(owner, name string) (Release, error) { r, _, err := s.Client.GetLatestRelease(owner, name) if err != nil { return Release{}, err } if r == nil { return Release{}, nil } return Release{ Name: r.Title, Date: r.PublishedAt, }, nil } // orderedLanguages returns the languages of a repo ordered by byte share, // most dominant first. Ties are broken alphabetically for stable output. func orderedLanguages(langs map[string]int64) []string { if len(langs) == 0 { return nil } names := make([]string, 0, len(langs)) for name := range langs { names = append(names, name) } // Descending byte count; alphabetically as a stable tie-breaker. sort.Slice(names, func(i, j int) bool { if langs[names[i]] != langs[names[j]] { return langs[names[i]] > langs[names[j]] } return names[i] < names[j] }) return names }