summaryrefslogtreecommitdiff
path: root/cgithub/github-scraper.js
diff options
context:
space:
mode:
authorJoe Mou <dev@mou.fo>2026-01-22 12:10:36 -0500
committerJoe Mou <dev@mou.fo>2026-01-22 12:13:26 -0500
commit5e45f900974589c8eead15fc352d50dc45b1a25a (patch)
tree5970784ee61a108c71d297eb28dca3357d45f18d /cgithub/github-scraper.js
GitHub directory scraper
Extracts directory listings from GitHub's server-rendered HTML. Avoids using the rate-limited API. Parses JSON embedded within script tags. Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
Diffstat (limited to 'cgithub/github-scraper.js')
-rwxr-xr-xcgithub/github-scraper.js74
1 files changed, 74 insertions, 0 deletions
diff --git a/cgithub/github-scraper.js b/cgithub/github-scraper.js
new file mode 100755
index 0000000..7e07adc
--- /dev/null
+++ b/cgithub/github-scraper.js
@@ -0,0 +1,74 @@
+#!/usr/bin/env node
+/**
+ * GitHub Directory Listing Scraper
+ *
+ * Gets directory listings from GitHub without using the API (no rate limits!)
+ * Works by parsing the server-rendered HTML that GitHub sends to browsers.
+ */
+
+async function getGitHubDirectory(owner, repo, path = '') {
+ // Always use the tree URL pattern, even for root directory
+ const url = `https://github.com/${owner}/${repo}/tree/main/${path}`;
+
+ console.error(`Fetching: ${url}`);
+
+ const response = await fetch(url);
+ const html = await response.text();
+
+ // GitHub embeds the directory data in a JSON script tag
+ // Using the tree (subdirectory) pattern for all cases
+ const subdirMatch = html.match(/<script type="application\/json" data-target="react-app\.embeddedData">([^<]+)<\/script>/);
+
+ if (!subdirMatch) {
+ throw new Error('Could not find embedded data in HTML');
+ }
+
+ const data = JSON.parse(subdirMatch[1]);
+ const payload = data.payload;
+
+ return {
+ path: payload.path,
+ branch: payload.refInfo.name,
+ items: payload.tree.items,
+ repo: {
+ name: payload.repo.name,
+ owner: payload.repo.ownerLogin,
+ isPublic: payload.repo.public
+ }
+ };
+}
+
+// Example usage
+if (require.main === module) {
+ const [owner, repo, path] = process.argv.slice(2);
+
+ if (!owner || !repo) {
+ console.log('Usage: node github-scraper.js <owner> <repo> [path]');
+ console.log('Example: node github-scraper.js torvalds linux');
+ console.log('Example: node github-scraper.js torvalds linux Documentation');
+ process.exit(1);
+ }
+
+ getGitHubDirectory(owner, repo, path)
+ .then(result => {
+ console.log('\nšŸ“‚ Directory listing');
+ console.log(`Repository: ${result.repo.owner}/${result.repo.name}`);
+ console.log(`Branch: ${result.branch}`);
+ console.log(`Path: ${result.path}`);
+ console.log(`Items: ${result.items.length}`);
+ console.log('\nContents:');
+
+ result.items.forEach(item => {
+ const icon = item.contentType === 'directory' ? 'šŸ“' : 'šŸ“„';
+ console.log(` ${icon} ${item.name}`);
+ });
+
+ console.log('\nāœ“ No API rate limits - this scrapes the HTML!');
+ })
+ .catch(err => {
+ console.error('Error:', err.message);
+ process.exit(1);
+ });
+}
+
+module.exports = { getGitHubDirectory };