diff options
| author | Joe Mou <dev@mou.fo> | 2026-08-27 14:08:06 -0400 |
|---|---|---|
| committer | Joe Mou <dev@mou.fo> | 2026-08-27 14:53:38 -0400 |
| commit | 4797a627f461204b289f95f61b64e3dab8bc00ff (patch) | |
| tree | a044ef3d0419c95aad9fa98a093e0c39bdce5459 /src/scraper.test.ts | |
| parent | 7c928d30b16b37624841c4615c0d4bda426e4159 (diff) | |
Display PDF blobs
GitHub renders a PDF with a viewer of its own and so ships no content for
one, leaving the blob view with nothing to show. Its raw bytes come as
application/octet-stream, which browsers download rather than display, so
serve them from a new embed route that retypes them and point an <object>
at that.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BErPR92xppzGFXkPqXPYj2
Diffstat (limited to 'src/scraper.test.ts')
| -rw-r--r-- | src/scraper.test.ts | 32 |
1 files changed, 32 insertions, 0 deletions
diff --git a/src/scraper.test.ts b/src/scraper.test.ts index 836235e..88e9271 100644 --- a/src/scraper.test.ts +++ b/src/scraper.test.ts @@ -7,6 +7,7 @@ import { getGitHubSidebar, getGitHubLatestCommit, getGitHubPulls, + getGitHubRaw, getGitHubRefs, getGitHubRelease, getGitHubReleases, @@ -328,11 +329,41 @@ describe("GitHub scraper", () => { assert.strictEqual(data.size, "1.26 KB"); assert.strictEqual(data.language, null); assert.strictEqual(data.image, true); + assert.strictEqual(data.renderFileType, null); assert.strictEqual(data.textLines, null); assert.strictEqual(data.htmlLines, null); assert.strictEqual(data.htmlContent, null); }); + // GitHub renders a PDF with a viewer of its own, so the page carries no + // content for it beyond the file type. + it("should support PDF", async () => { + const data = await getGitHubBlob("mozilla", "pdf.js", "master", "test/pdfs/basicapi.pdf"); + + assert.strictEqual(data.repo.owner, "mozilla"); + assert.strictEqual(data.repo.name, "pdf.js"); + assert.strictEqual(data.branch, "master"); + assert.strictEqual(data.path, "test/pdfs/basicapi.pdf"); + + assert.strictEqual(data.size, "103 KB"); + assert.strictEqual(data.language, null); + assert.strictEqual(data.image, false); + assert.strictEqual(data.renderFileType, "pdf"); + assert.strictEqual(data.textLines, null); + assert.strictEqual(data.htmlLines, null); + assert.strictEqual(data.htmlContent, null); + }); + + // Why the embed route exists rather than the blob view pointing a viewer + // at GitHub: as application/octet-stream a browser downloads the file + // instead of displaying it. + it("should fetch a raw PDF that GitHub types as a download", async () => { + const response = await getGitHubRaw("mozilla", "pdf.js", "master", "test/pdfs/basicapi.pdf"); + + assert.strictEqual(response.headers.get("Content-Type"), "application/octet-stream"); + assert.ok((await response.text()).startsWith("%PDF-")); + }); + // A binary file GitHub has no preview for: no content of any kind, and the // path exercises escaping of a space. it("should support an unpreviewable binary file", async () => { @@ -346,6 +377,7 @@ describe("GitHub scraper", () => { assert.strictEqual(data.size, "815 KB"); assert.strictEqual(data.language, null); assert.strictEqual(data.image, false); + assert.strictEqual(data.renderFileType, null); assert.strictEqual(data.textLines, null); assert.strictEqual(data.htmlLines, null); assert.strictEqual(data.htmlContent, null); |
