From 1435dea09882f110448b034d7cffa32d03e19a8f Mon Sep 17 00:00:00 2001 From: Joe Mou Date: Sat, 21 Feb 2026 16:21:35 +1100 Subject: Parse release page HTML Fixes reading complex release descriptions that had unbalanced tags. Written with Gemini. --- cgithub/src/scraper.test.ts | 1 + cgithub/src/scraper.ts | 46 ++++++++++++++++++++++++--------------------- 2 files changed, 26 insertions(+), 21 deletions(-) (limited to 'cgithub/src') diff --git a/cgithub/src/scraper.test.ts b/cgithub/src/scraper.test.ts index 8c11317..33f00f1 100644 --- a/cgithub/src/scraper.test.ts +++ b/cgithub/src/scraper.test.ts @@ -172,6 +172,7 @@ describe("GitHub scraper", () => { assert.ok(firstRelease.tagName.length > 0); assert.ok(firstRelease.title.match(/^v\d/)); assert.ok(firstRelease.publishedAt.length > 0); + assert.ok(firstRelease.bodyHtml.length > 0); }); }); diff --git a/cgithub/src/scraper.ts b/cgithub/src/scraper.ts index e52ddac..af39660 100644 --- a/cgithub/src/scraper.ts +++ b/cgithub/src/scraper.ts @@ -1,3 +1,8 @@ +import { parseDocument } from "htmlparser2"; +import { type Element } from "domhandler"; +import { getAttributeValue, getInnerHTML, textContent } from "domutils"; +import * as cssSelect from "css-select"; + export class GitHubHTTPError extends Error { status: number; @@ -418,37 +423,36 @@ export async function getGitHubCommits( export async function getGitHubReleases(owner: string, repo: string): Promise { const html = await fetchGitHubPage(`${owner}/${repo}/releases`); - // Releases page doesn't have embedded JSON, so we parse HTML directly + const document = parseDocument(html); const releases: Release[] = []; - // Find all h2 tags with version numbers - const h2Regex = /]*id="([^"]*)"[^>]*>(v[^<]*)<\/h2>/g; - - for (const h2Match of html.matchAll(h2Regex)) { - const id = h2Match[1]; - const title = h2Match[2].trim(); + // Not really sure the right way to express this type. + const sections = cssSelect.selectAll("section", document) as unknown as Element[]; - // Find the section containing this h2 - const sectionRegex = new RegExp( - `]*>\\s*]*id="${id}"[^>]*>[^<]*<\\/h2>([\\s\\S]*?)<\\/section>`, - "s", - ); - const sectionMatch = html.match(sectionRegex); - if (!sectionMatch) continue; + for (const section of sections) { + const h2 = cssSelect.selectOne("h2[id]", section); + if (!h2) continue; - const sectionHtml = sectionMatch[1]; + const title = textContent(h2).trim(); + if (!title.startsWith("v")) continue; // Extract tag name from link - const tagMatch = sectionHtml.match(/]*href="[^"]*\/tree\/([^"]+)"[^>]*>/); - const tagName = tagMatch?.[1] || title; + const tagLink = cssSelect.selectOne('a[href*="/tree/"]', section); + let tagName = title; + if (tagLink) { + const href = getAttributeValue(tagLink, "href"); + if (href) { + tagName = href.split("/").pop() || title; + } + } // Extract published date - const dateMatch = sectionHtml.match(/]*datetime="([^"]*)"[^>]*>/); - const publishedAt = dateMatch?.[1] || ""; + const relativeTime = cssSelect.selectOne("relative-time", section); + const publishedAt = relativeTime ? getAttributeValue(relativeTime, "datetime") || "" : ""; // Extract body HTML (markdown content) - const bodyMatch = sectionHtml.match(/]*class="[^"]*markdown-body[^"]*"[^>]*>([\s\S]*?)<\/div>/); - const bodyHtml = bodyMatch?.[1]?.trim() || ""; + const body = cssSelect.selectOne(".markdown-body", section); + const bodyHtml = body ? getInnerHTML(body).trim() : ""; releases.push({ tagName, -- cgit v1.3.1