From 5e45f900974589c8eead15fc352d50dc45b1a25a Mon Sep 17 00:00:00 2001 From: Joe Mou Date: Thu, 22 Jan 2026 12:10:36 -0500 Subject: GitHub directory scraper Extracts directory listings from GitHub's server-rendered HTML. Avoids using the rate-limited API. Parses JSON embedded within script tags. Co-Authored-By: Claude Sonnet 4.5 --- cgithub/github-scraper.js | 74 +++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 74 insertions(+) create mode 100755 cgithub/github-scraper.js (limited to 'cgithub/github-scraper.js') diff --git a/cgithub/github-scraper.js b/cgithub/github-scraper.js new file mode 100755 index 0000000..7e07adc --- /dev/null +++ b/cgithub/github-scraper.js @@ -0,0 +1,74 @@ +#!/usr/bin/env node +/** + * GitHub Directory Listing Scraper + * + * Gets directory listings from GitHub without using the API (no rate limits!) + * Works by parsing the server-rendered HTML that GitHub sends to browsers. + */ + +async function getGitHubDirectory(owner, repo, path = '') { + // Always use the tree URL pattern, even for root directory + const url = `https://github.com/${owner}/${repo}/tree/main/${path}`; + + console.error(`Fetching: ${url}`); + + const response = await fetch(url); + const html = await response.text(); + + // GitHub embeds the directory data in a JSON script tag + // Using the tree (subdirectory) pattern for all cases + const subdirMatch = html.match(/