aboutsummaryrefslogtreecommitdiff
path: root/src/scraper.ts
diff options
context:
space:
mode:
authorJoe Mou <dev@mou.fo>2026-02-11 01:56:37 -0500
committerJoe Mou <dev@mou.fo>2026-02-11 01:56:49 -0500
commit934b2c39537d6718375170a029fc1d1df01024c5 (patch)
tree0e105a684b16554a19d1ba40b7e977e0c18a94b8 /src/scraper.ts
parentb75d4cfb610b8792e6f1236b613f554d1ac5b78d (diff)
Add GitHub releases support
Scrapes releases page HTML to display version tags, publish dates, and changelog content. Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
Diffstat (limited to 'src/scraper.ts')
-rw-r--r--src/scraper.ts63
1 files changed, 63 insertions, 0 deletions
diff --git a/src/scraper.ts b/src/scraper.ts
index 71d0e3e..e52ddac 100644
--- a/src/scraper.ts
+++ b/src/scraper.ts
@@ -184,6 +184,17 @@ export interface GitHubCommits extends GitHubNav {
commitGroups: CommitGroup[];
}
+interface Release {
+ tagName: string;
+ title: string;
+ publishedAt: string;
+ bodyHtml: string;
+}
+
+export interface GitHubReleases extends GitHubCommon {
+ releases: Release[];
+}
+
async function fetchGitHubPage(path: string): Promise<string> {
// GitHub throttles/blocks requests without realistic browser headers.
// These headers make the request appear as a standard browser visit.
@@ -403,3 +414,55 @@ export async function getGitHubCommits(
commitGroups: payload.commitGroups,
};
}
+
+export async function getGitHubReleases(owner: string, repo: string): Promise<GitHubReleases> {
+ const html = await fetchGitHubPage(`${owner}/${repo}/releases`);
+
+ // Releases page doesn't have embedded JSON, so we parse HTML directly
+ const releases: Release[] = [];
+
+ // Find all h2 tags with version numbers
+ const h2Regex = /<h2[^>]*id="([^"]*)"[^>]*>(v[^<]*)<\/h2>/g;
+
+ for (const h2Match of html.matchAll(h2Regex)) {
+ const id = h2Match[1];
+ const title = h2Match[2].trim();
+
+ // Find the section containing this h2
+ const sectionRegex = new RegExp(
+ `<section[^>]*>\\s*<h2[^>]*id="${id}"[^>]*>[^<]*<\\/h2>([\\s\\S]*?)<\\/section>`,
+ "s",
+ );
+ const sectionMatch = html.match(sectionRegex);
+ if (!sectionMatch) continue;
+
+ const sectionHtml = sectionMatch[1];
+
+ // Extract tag name from link
+ const tagMatch = sectionHtml.match(/<a[^>]*href="[^"]*\/tree\/([^"]+)"[^>]*>/);
+ const tagName = tagMatch?.[1] || title;
+
+ // Extract published date
+ const dateMatch = sectionHtml.match(/<relative-time[^>]*datetime="([^"]*)"[^>]*>/);
+ const publishedAt = dateMatch?.[1] || "";
+
+ // Extract body HTML (markdown content)
+ const bodyMatch = sectionHtml.match(/<div[^>]*class="[^"]*markdown-body[^"]*"[^>]*>([\s\S]*?)<\/div>/);
+ const bodyHtml = bodyMatch?.[1]?.trim() || "";
+
+ releases.push({
+ tagName,
+ title,
+ publishedAt,
+ bodyHtml,
+ });
+ }
+
+ return {
+ repo: {
+ owner,
+ name: repo,
+ },
+ releases,
+ };
+}