summaryrefslogtreecommitdiff
path: root/cgithub/src/scraper.test.ts
diff options
context:
space:
mode:
authorJoe Mou <dev@mou.fo>2026-08-27 14:08:06 -0400
committerJoe Mou <dev@mou.fo>2026-08-27 14:53:38 -0400
commit0d60b8c8235c3c07f675aba7a56dd37b62519f9e (patch)
tree8c94cef7b3530e72fc8542072b037081e5bb18ae /cgithub/src/scraper.test.ts
parent16dfeb61f79e6cd44c667422b8dc50f226269d80 (diff)
Display PDF blobs
GitHub renders a PDF with a viewer of its own and so ships no content for one, leaving the blob view with nothing to show. Its raw bytes come as application/octet-stream, which browsers download rather than display, so serve them from a new embed route that retypes them and point an <object> at that. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01BErPR92xppzGFXkPqXPYj2
Diffstat (limited to 'cgithub/src/scraper.test.ts')
-rw-r--r--cgithub/src/scraper.test.ts32
1 files changed, 32 insertions, 0 deletions
diff --git a/cgithub/src/scraper.test.ts b/cgithub/src/scraper.test.ts
index 836235e..88e9271 100644
--- a/cgithub/src/scraper.test.ts
+++ b/cgithub/src/scraper.test.ts
@@ -7,6 +7,7 @@ import {
getGitHubSidebar,
getGitHubLatestCommit,
getGitHubPulls,
+ getGitHubRaw,
getGitHubRefs,
getGitHubRelease,
getGitHubReleases,
@@ -328,11 +329,41 @@ describe("GitHub scraper", () => {
assert.strictEqual(data.size, "1.26 KB");
assert.strictEqual(data.language, null);
assert.strictEqual(data.image, true);
+ assert.strictEqual(data.renderFileType, null);
assert.strictEqual(data.textLines, null);
assert.strictEqual(data.htmlLines, null);
assert.strictEqual(data.htmlContent, null);
});
+ // GitHub renders a PDF with a viewer of its own, so the page carries no
+ // content for it beyond the file type.
+ it("should support PDF", async () => {
+ const data = await getGitHubBlob("mozilla", "pdf.js", "master", "test/pdfs/basicapi.pdf");
+
+ assert.strictEqual(data.repo.owner, "mozilla");
+ assert.strictEqual(data.repo.name, "pdf.js");
+ assert.strictEqual(data.branch, "master");
+ assert.strictEqual(data.path, "test/pdfs/basicapi.pdf");
+
+ assert.strictEqual(data.size, "103 KB");
+ assert.strictEqual(data.language, null);
+ assert.strictEqual(data.image, false);
+ assert.strictEqual(data.renderFileType, "pdf");
+ assert.strictEqual(data.textLines, null);
+ assert.strictEqual(data.htmlLines, null);
+ assert.strictEqual(data.htmlContent, null);
+ });
+
+ // Why the embed route exists rather than the blob view pointing a viewer
+ // at GitHub: as application/octet-stream a browser downloads the file
+ // instead of displaying it.
+ it("should fetch a raw PDF that GitHub types as a download", async () => {
+ const response = await getGitHubRaw("mozilla", "pdf.js", "master", "test/pdfs/basicapi.pdf");
+
+ assert.strictEqual(response.headers.get("Content-Type"), "application/octet-stream");
+ assert.ok((await response.text()).startsWith("%PDF-"));
+ });
+
// A binary file GitHub has no preview for: no content of any kind, and the
// path exercises escaping of a space.
it("should support an unpreviewable binary file", async () => {
@@ -346,6 +377,7 @@ describe("GitHub scraper", () => {
assert.strictEqual(data.size, "815 KB");
assert.strictEqual(data.language, null);
assert.strictEqual(data.image, false);
+ assert.strictEqual(data.renderFileType, null);
assert.strictEqual(data.textLines, null);
assert.strictEqual(data.htmlLines, null);
assert.strictEqual(data.htmlContent, null);