diff --git a/SECURITY.md b/SECURITY.md index b40cb849..4fbc9f8b 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -117,6 +117,29 @@ metadata, or transport fragments. Authorization-context names separate declared access realms, but they cannot detect that the account behind a reused name has changed. Use a new context name when the intended account changes. +`ghostget pdf ` with a browser sign-in option (`--cookie-source`, +`--browser-profile`, `--cookie-profile`, `--auth`, or `--cookies-file`) +downloads the PDF itself instead of passing the link on anonymously. It uses +only HTTPS and DNS-pinned public addresses, follows at most five redirects, +and refuses credential-bearing links, plain HTTP, private IP literals, and +local names such as `.local` or single-label hosts before any cookie is read +for that host. A public name that resolves to a private address is refused +before any request is sent. It reads cookies separately for each host it visits +through the same cookie filter as `read`, so a site receives only the cookies +that belong to it, never another host's. When a redirect leaves the link's own +site (any host other than the link's host or a parent or subdomain of it), the +new site does not receive its `SameSite=Strict` cookies, as a browser would +hold them back after a redirect started elsewhere. It does not launch the +browser profile, so no profile egress consent applies, and +`--trust-profile-egress` or `--mode` copied from `read` are ignored. The whole +download, including every cookie read, is bounded by `--timeout-ms`; the +response is bounded by `--max-pdf-bytes`, must start with a PDF signature, and +is streamed into an owner-only file in an owner-only temporary folder that is +removed after the import. If the site refuses and none of the selected +browser's cookies went to it, the error says no sign-in was found rather than +that the sign-in lacks access. Cookie values are not printed, logged, or +stored. + The unauthenticated direct-media adapter intentionally permits loopback and private-network HTTP(S) targets because its URL is supplied by the local user. It is not an SSRF boundary for remotely supplied URLs. It reads a bounded diff --git a/package.json b/package.json index 1420244c..f3ef7837 100644 --- a/package.json +++ b/package.json @@ -387,6 +387,7 @@ "src/oauth-google.ts", "src/oauth-x.ts", "src/path-helper.ts", + "src/pdf-auth.ts", "src/persistent-helper-bridge.ts", "src/pinned-https.ts", "src/plan-assets.ts", diff --git a/scripts/npm-release-workflow.test.ts b/scripts/npm-release-workflow.test.ts index ab560a4a..2bf9756c 100644 --- a/scripts/npm-release-workflow.test.ts +++ b/scripts/npm-release-workflow.test.ts @@ -1220,7 +1220,7 @@ describe("npm publication contract", () => { (MAX_UNPACKED_BYTES + MAX_PACKED_ENTRIES * 1_023 + 1_024) / 512, ) * 512, ); - expect(MAX_PACKAGE_TAR_BYTES).toBe(24_615_424); + expect(MAX_PACKAGE_TAR_BYTES).toBe(24_658_944); expect(MAX_PACKAGE_TAR_BYTES % 512).toBe(0); expect(artifact).toContain("maxOutputLength: MAX_PACKAGE_TAR_BYTES"); expect(artifact).not.toContain("const maximumTarBytes"); @@ -1430,8 +1430,8 @@ describe("npm publication contract", () => { expect(budget).toContain("785b8fa60c329d7ac46bc8fcf4d959b5fa9d96455bba9e7cdea63f6a3827c4f6"); expect(Object.isFrozen(repairPackageMeasurement)).toBeTrue(); expect(repairPackageMeasurement).toMatchObject({ - archiveSha256: "9641f93ab7dd2c52174cf2ddf3e41f407ebca562c46e0dd0af2e6cd01447c2de", - packedBytes: 12_132_235, unpackedBytes: 23_980_585, entryCount: 619, + archiveSha256: "e34c1d14165fdc4985280f7d56626bcd17055de9e2906bc9db1569807ed568ab", + packedBytes: 12_133_122, unpackedBytes: 24_023_022, entryCount: 620, packedPlatformProjection: 12_387, packedPortabilityAllowance: 4_096, payloadPlatformProjection: 353, payloadAllowance: 65, }); @@ -1441,8 +1441,10 @@ describe("npm publication contract", () => { expect(budget).toContain("12,141,169 packed; 23,937,025 + 353 + 65 = 23,937,443 unpacked"); expect(budget).toContain("23,930,250 + 353 + 65 = 23,930,668 unpacked"); expect(budget).toContain("12,141,373 packed; 23,937,545 + 353 + 65 = 23,937,963 unpacked"); - expect(MAX_PACKED_BYTES).toBe(12_148_718); - expect(MAX_PACKED_BYTES).toBe(12_132_235 + 12_387 + 4_096); + expect(budget).toContain("12,133,122 + 12,387 + 4,096 = 12,149,605 packed"); + expect(budget).toContain("24,023,022 + 353 + 65 = 24,023,440 unpacked"); + expect(MAX_PACKED_BYTES).toBe(12_149_605); + expect(MAX_PACKED_BYTES).toBe(12_133_122 + 12_387 + 4_096); expect(budget).toContain("aa127b3193c9bb3b0cb5deece5927be60ccb7111a50169320d322ffdeaa13f39"); expect(budget).toContain("0c331bab3ab3df69a108e18f5f29845b0db90c281cbd6455c0d90fa0b24081e2"); expect(budget).toContain("873cad8139fda303e2d19c6afd61cf549cf9b4d1d76b2a1d6d632a6afe6bd0d1"); @@ -1537,8 +1539,8 @@ describe("npm publication contract", () => { expect(budget).toContain("11,696,091 + 4,096 = 11,700,187"); expect(budget).toContain("35449445752 attempt 1, package job 105913938839"); expect(budget).toContain("exactly 596 files"); - expect(MAX_PACKED_ENTRIES).toBe(619); - expect(MAX_PACKED_FILES).toBe(619); + expect(MAX_PACKED_ENTRIES).toBe(620); + expect(MAX_PACKED_FILES).toBe(620); expect(budget).toContain("Ghostget 0.18.6 same-boot setup-cleanup candidate over main edbe567"); expect(budget).toContain("11,656,173"); expect(budget).toContain("22,513,450 payload bytes across exactly 557 files"); @@ -1563,7 +1565,7 @@ describe("npm publication contract", () => { expect(budget).toContain("47684b3e2eb5cf3ed07fbb520aade8c7251d993f75262fbf1af627d9081a1a5f"); expect(budget).toContain("23,688,277 + 353 + 65 = 23,688,695"); expect(budget).toContain("23,759,283 + 353 + 65 = 23,759,701"); - expect(MAX_UNPACKED_BYTES).toBe(23_981_003); + expect(MAX_UNPACKED_BYTES).toBe(24_023_440); expect(budget).toContain("23,037,873 + 65 = 23,037,938"); expect(budget).toContain("f9f3ab38a682690ceaa2699a7309997512030f0fa500a9dc29dcd108123dc41f"); expect(budget).toContain("23,038,557 + 65 = 23,038,622"); @@ -1596,7 +1598,7 @@ describe("npm publication contract", () => { expect(budget).toContain("01875f12ab73a49d6c7d6bf520dc3d318db816addee2fa7981889f35c958cf7c"); expect(budget).toContain("b12909f08f7c19460ced56e30619f4860a1183f4b0106170c07837dae577a937"); expect(budget).toContain("0b212ac291218528dcf979370110a36f10850e046ca90a536057d9a44e807d1d"); - expect(MAX_UNPACKED_BYTES).toBe(23_980_585 + 353 + 65); + expect(MAX_UNPACKED_BYTES).toBe(24_023_022 + 353 + 65); expect(budget).toContain("22,794,052 + 65 = 22,794,117"); expect(budget).toContain("c482efe748f880e3717727d6d39fd92a68953e6eea766642b329ba47ae772d80"); expect(budget).toContain("22,759,423 + 65 = 22,759,488"); @@ -1630,10 +1632,10 @@ describe("npm publication contract", () => { expect(Object.isFrozen(range)).toBe(true); } expect(packageArtifactBudget).toEqual({ - entryCount: { min: 619, max: 619 }, - fileCount: { min: 619, max: 619 }, - packedBytes: { min: 1_600_000, max: 12_148_718 }, - unpackedBytes: { min: 9_000_000, max: 23_981_003 }, + entryCount: { min: 620, max: 620 }, + fileCount: { min: 620, max: 620 }, + packedBytes: { min: 1_600_000, max: 12_149_605 }, + unpackedBytes: { min: 9_000_000, max: 24_023_440 }, }); }); diff --git a/scripts/package-budget.ts b/scripts/package-budget.ts index 899dba2a..91ce2bc1 100644 --- a/scripts/package-budget.ts +++ b/scripts/package-budget.ts @@ -2130,15 +2130,27 @@ // Retain the same platform projections and portability allowances: // 12,132,235 + 12,387 + 4,096 = 12,148,718 packed; // 23,980,585 + 353 + 65 = 23,981,003 unpacked. +// +// Signed-in `ghostget pdf ` downloads add src/pdf-auth.ts to the packed +// source and a lazy import from cli.ts; no dist chunk changes. Over merged +// main 65944e7 (which carries the X contacts.list snapshot above), the new +// source file grows the inventory to 620 entries. After `bun run build`, a +// clean npm 11.19.0 pack --ignore-scripts with Node 24.20.0 on darwin arm64 +// measured 620 entries, 12,133,122 packed bytes, and 24,023,022 unpacked +// bytes; archive SHA-256 +// e34c1d14165fdc4985280f7d56626bcd17055de9e2906bc9db1569807ed568ab. +// Retain the same platform projections and portability allowances: +// 12,133,122 + 12,387 + 4,096 = 12,149,605 packed; +// 24,023,022 + 353 + 65 = 24,023,440 unpacked. export const repairPackageMeasurement = Object.freeze({ - scope: "X contacts.list follow-collection qualification over merged main 46e31838 (Ghostget 0.18.46)", + scope: "Ghostget signed-in PDF downloads over merged main 65944e7", command: "npm pack --ignore-scripts", - npmVersion: "11.16.0", + npmVersion: "11.19.0", platform: "darwin-arm64", - archiveSha256: "9641f93ab7dd2c52174cf2ddf3e41f407ebca562c46e0dd0af2e6cd01447c2de", - packedBytes: 12_132_235, - unpackedBytes: 23_980_585, - entryCount: 619, + archiveSha256: "e34c1d14165fdc4985280f7d56626bcd17055de9e2906bc9db1569807ed568ab", + packedBytes: 12_133_122, + unpackedBytes: 24_023_022, + entryCount: 620, packedPlatformProjection: 12_387, packedPortabilityAllowance: 4_096, payloadPlatformProjection: 353, diff --git a/skills/ghostget/SKILL.md b/skills/ghostget/SKILL.md index 3c9a492e..d5b030d4 100644 --- a/skills/ghostget/SKILL.md +++ b/skills/ghostget/SKILL.md @@ -36,6 +36,7 @@ automation. - Capture a URL: `ghostget ` or `ghostget clip `. - Read without persistence: `ghostget read `. +- Save a PDF as a note: `ghostget pdf `. For a paywalled paper the user can open through a library or university sign-in, add that browser: `ghostget pdf --cookie-source chrome` (or `--browser-profile `, `--auth `, `--cookies-file `). Ghostget then downloads over HTTPS itself, sends each site only its own cookies, and fails with "does not seem to have access" when the site returns a sign-in page, or with "no ... sign-in was found" when that browser or profile has none for the site (`ghostget browsers` lists profiles); ask the user to open the link in that browser first rather than retrying. Without these options the download stays anonymous. - Archive media: `ghostget archive ` or `ghostget audio|video|transcript `. - Discover supported article embeds through the provider's bounded semantic media read, then archive each exact returned finite item separately. Do not treat a collection page as one media item or scrape its DOM to manufacture asset routes. - Inspect support: `ghostget plugin list`, `ghostget plugin show `, and `ghostget capabilities [adapter]`. For a typed, schema-backed projection use `ghostget contracts catalog --json`; check a read-only collection plan with `ghostget contracts check --plan --json`. diff --git a/src/cli.test.ts b/src/cli.test.ts index 407c553c..153a367c 100644 --- a/src/cli.test.ts +++ b/src/cli.test.ts @@ -230,6 +230,44 @@ describe("lazy ghostget CLI entrypoint", () => { } }); + test("pdf without sign-in options still delegates unchanged; with them Ghostget handles it first", async () => { + const previousExitCode = process.exitCode; + try { + const received: (readonly string[])[] = []; + const stdout: string[] = []; + const loadKnowledge = () => Promise.resolve({ + main: (raw?: readonly string[]) => { + received.push(raw ?? []); + return Promise.resolve(0); + }, + }); + const noRuntime = () => { + throw new Error("pdf must not load the provider runtime"); + }; + await runGhostgetCliProcess( + ["pdf", "https://example.org/a.pdf", "--json"], + { stdout: () => undefined, stderr: () => undefined }, + noRuntime, + noRuntime, + loadKnowledge, + ); + expect(received).toEqual([["pdf", "https://example.org/a.pdf", "--json"]]); + + await runGhostgetCliProcess( + ["pdf", "https://example.org/a.pdf", "--cookie-source", "netscape", "--json"], + { stdout: (text) => stdout.push(text), stderr: () => undefined }, + noRuntime, + noRuntime, + loadKnowledge, + ); + expect(received).toHaveLength(1); + expect(process.exitCode).toBe(2); + expect(JSON.parse(stdout.join(""))).toMatchObject({ ok: false, error: { code: "usage" } }); + } finally { + process.exitCode = previousExitCode ?? 0; + } + }); + test("routes only complete capabilities and plugin inspection shapes", async () => { expect(routedGhostgetCatalogCommand(["capabilities", "--json"])).toEqual({ command: "capabilities", @@ -490,7 +528,7 @@ throw new Error("private fallback did not load a forbidden module"); test("has only static help and release identity as eager dependencies and bounds startup CPU work", async () => { const source = readFileSync(cliPath, "utf8"); - expect(source).toContain('import { ghostgetBareUsage, ghostgetHelpRequest } from "./usage"'); + expect(source).toContain('import { ghostgetBareUsage, ghostgetHelpRequest, hasPdfSignInOptions } from "./usage"'); expect(source).toContain('import { cliStyle, renderCliError } from "./cli-style"'); expect(source).toContain('import { GHOSTGET_VERSION } from "./version"'); expect(source).toContain('import("./ghostget")'); diff --git a/src/cli.ts b/src/cli.ts index 0334a227..38693944 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -1,6 +1,6 @@ #!/usr/bin/env bun -import { ghostgetBareUsage, ghostgetHelpRequest } from "./usage"; +import { ghostgetBareUsage, ghostgetHelpRequest, hasPdfSignInOptions } from "./usage"; import { cliStyle, renderCliError } from "./cli-style"; import { terminalIntro } from "./cli-intro"; import { GHOSTGET_VERSION } from "./version"; @@ -290,6 +290,16 @@ export async function runGhostgetCliProcess( process.exitCode = await support.runGhostgetSupportCommand(rawArguments.slice(1), resolvedOutput); return; } + if (rawArguments[0] === "pdf" && help === null && hasPdfSignInOptions(rawArguments.slice(1))) { + const { runSignedInPdfCommand, runWordcellPdfWithDownload } = await import("./pdf-auth"); + process.exitCode = await runSignedInPdfCommand(rawArguments.slice(1), resolvedOutput, { + environment: process.env, + runWordcellPdf: async (pdfArguments, download) => download === undefined + ? (await loadKnowledgeCli()).main(["pdf", ...pdfArguments], resolvedOutput) + : runWordcellPdfWithDownload(pdfArguments, download, process.env, resolvedOutput), + }); + return; + } if (isPublicGhostgetCommand(rawArguments)) { const knowledge = await loadKnowledgeCli(); process.exitCode = await knowledge.main(rawArguments, resolvedOutput); diff --git a/src/fixtures/cli-help/topic-notes.txt b/src/fixtures/cli-help/topic-notes.txt index 12302396..721b3a30 100644 --- a/src/fixtures/cli-help/topic-notes.txt +++ b/src/fixtures/cli-help/topic-notes.txt @@ -16,3 +16,12 @@ Commands adapters [--json] List page-capture adapters Most commands take --root to pick the notes folder. + +Paywalled PDFs + Add the browser you use to open the paper, and Ghostget downloads it + with that sign-in: + ghostget pdf --cookie-source chrome + Also: --browser-profile , --cookie-profile , + --auth or --cookies-file . Each site gets only its own cookies. + If the site sends a sign-in page instead, open the link in your browser + first, then run it again. diff --git a/src/pdf-auth.test.ts b/src/pdf-auth.test.ts new file mode 100644 index 00000000..2fd5dc5a --- /dev/null +++ b/src/pdf-auth.test.ts @@ -0,0 +1,775 @@ +import { describe, expect, test } from "bun:test"; +import { chmodSync, existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import fc from "fast-check"; +import { assertProperty } from "./test-support"; + +import type { StrictCookie } from "@hraness/wordcell/clip/cookies"; + +import { + PdfAuthError, + downloadSignedInPdf, + hasPdfSignInOptions, + isSameSiteHop, + pdfFilename, + resolveProfilePath, + runSignedInPdf, + runSignedInPdfCommand, + runWordcellPdfWithDownload, + splitPdfSignInArguments, + type PdfDownloadedSource, + type PdfFetch, +} from "./pdf-auth"; +import { createClassifiedCookieRecordReader } from "./cookie-access"; + +const PDF = new TextEncoder().encode("%PDF-1.7\n1 0 obj\n<<>>\nendobj\n%%EOF\n"); +const LOGIN = new TextEncoder().encode("Sign in through your institution"); + +function cookie(name: string, value: string, domain: string, sameSite: StrictCookie["sameSite"] = null): StrictCookie { + return { name, value, domain, path: "/", secure: true, httpOnly: true, sameSite } as unknown as StrictCookie; +} + +/** A jar where the only host with a sign-in is `host`. */ +function signedIn(host: string) { + return jar({ [host]: [cookie("sid", "1", host)] }); +} + +/** Cookie jar that behaves like the shared filter: only exact host matches. */ +function jar(byHost: Record) { + const reads: string[] = []; + return { + reads, + read: async (url: URL) => { + reads.push(url.hostname); + const cookies = byHost[url.hostname]; + if (cookies === undefined) throw new Error(`no matching cookies were found for ${url.hostname}`); + return { cookies, warnings: [] }; + }, + }; +} + +type Recorded = { readonly url: string; readonly cookie: string | null; readonly init: RequestInit }; + +function server(routes: Record Response>) { + const requests: Recorded[] = []; + const fetch: PdfFetch = async (url, init) => { + requests.push({ url: url.href, cookie: new Headers(init.headers).get("cookie"), init }); + const route = routes[url.href]; + if (route === undefined) return new Response("missing", { status: 404 }); + return route(); + }; + return { requests, fetch }; +} + +function pdfResponse(bytes: Uint8Array = PDF, headers: Record = { "content-type": "application/pdf" }): Response { + return new Response(bytes, { status: 200, headers }); +} + +describe("splitPdfSignInArguments", () => { + test("strips sign-in options and keeps Wordcell options in order", () => { + const split = splitPdfSignInArguments([ + "https://example.org/paper.pdf", + "--cookie-source", + "chrome", + "--root", + "/notes", + "--cookie-profile=Profile 1", + "--json", + ]); + expect(split.signIn).toEqual({ kind: "browser", source: "chrome", profile: "Profile 1" }); + expect(split.remaining).toEqual(["https://example.org/paper.pdf", "--root", "/notes", "--json"]); + }); + + test("--browser-profile means a Chrome profile unless another browser is named", () => { + expect(splitPdfSignInArguments(["u", "--browser-profile", "Work"]).signIn) + .toEqual({ kind: "browser", source: "chrome", profile: "Work" }); + expect(splitPdfSignInArguments(["u", "--browser-profile", "Work", "--cookie-source", "brave"]).signIn) + .toEqual({ kind: "browser", source: "brave", profile: "Work" }); + }); + + test("--auth and --cookies-file are recognised; --mode is left for Wordcell to reject", () => { + expect(splitPdfSignInArguments(["u", "--auth", "uni"])) + .toEqual({ signIn: { kind: "auth", id: "uni" }, remaining: ["u"] }); + expect(hasPdfSignInOptions(["u", "--mode", "browser"])).toBe(false); + expect(splitPdfSignInArguments(["u", "--mode", "browser"])) + .toEqual({ signIn: null, remaining: ["u", "--mode", "browser"] }); + expect(splitPdfSignInArguments(["u", "--cookies-file=/c.json"]).signIn) + .toEqual({ kind: "cookies-file", path: "/c.json" }); + }); + + test("read's --trust-profile-egress and --mode are accepted and dropped next to a sign-in option", () => { + expect(splitPdfSignInArguments(["u", "--browser-profile", "Default", "--trust-profile-egress", "--json"])) + .toEqual({ signIn: { kind: "browser", source: "chrome", profile: "Default" }, remaining: ["u", "--json"] }); + expect(splitPdfSignInArguments(["u", "--cookie-source", "chrome", "--mode", "browser"])) + .toEqual({ signIn: { kind: "browser", source: "chrome", profile: undefined }, remaining: ["u"] }); + expect(splitPdfSignInArguments(["u", "--cookie-source", "chrome", "--mode=http"]).remaining).toEqual(["u"]); + expect(() => splitPdfSignInArguments(["u", "--cookie-source", "chrome", "--mode", "fast"])) + .toThrow("a signed-in PDF download needs neither"); + const without = ["u", "--trust-profile-egress", "--mode", "fast"]; + expect(splitPdfSignInArguments(without)).toEqual({ signIn: null, remaining: without }); + }); + + test("without sign-in options nothing changes", () => { + const argv = ["https://example.org/a.pdf", "--root", "/n", "--json"]; + expect(splitPdfSignInArguments(argv)).toEqual({ signIn: null, remaining: argv }); + expect(hasPdfSignInOptions(argv)).toBe(false); + }); + + test("options after -- are not sign-in options", () => { + expect(hasPdfSignInOptions(["--", "--cookie-source"])).toBe(false); + expect(splitPdfSignInArguments(["--", "--cookie-source"]).signIn).toBeNull(); + }); + + test("rejects unclear combinations before any browser is read", () => { + const cases: readonly (readonly string[])[] = [ + ["u", "--cookie-source"], + ["u", "--cookie-source", "netscape"], + ["u", "--cookie-source", "chrome", "--cookie-source", "arc"], + ["u", "--auth", "a", "--cookie-source", "chrome"], + ["u", "--cookies-file", "/c", "--cookie-source", "chrome"], + ["u", "--browser-profile", "A", "--cookie-profile", "B"], + ["u", "--cookie-profile", "Default"], + ]; + for (const argv of cases) { + expect(() => splitPdfSignInArguments(argv)).toThrow(PdfAuthError); + } + }); + + test("property: stripping never leaves a sign-in option and keeps every other argument", () => { + const plain = fc.constantFrom("--json", "--quiet", "--root", "/notes", "https://e.org/a.pdf", "--slug", "x"); + const auth = fc.constantFrom(["--cookie-source", "chrome"], ["--cookie-profile", "Default"]); + assertProperty(fc.property(fc.array(plain, { maxLength: 8 }), auth, fc.nat(8), (others, pair, at) => { + const position = Math.min(at, others.length); + const argv = [...others.slice(0, position), ...pair, ...others.slice(position)]; + const withSource = pair[0] === "--cookie-profile" ? [...argv, "--cookie-source", "chrome"] : argv; + const split = splitPdfSignInArguments(withSource); + expect(split.remaining).toEqual(others); + expect(split.signIn?.kind).toBe("browser"); + })); + }); +}); + +describe("downloadSignedInPdf", () => { + test("sends only the cookies that belong to each host across a redirect", async () => { + const cookies = jar({ + "doi.example.org": [cookie("doi", "d1", "doi.example.org")], + "publisher.example.com": [cookie("session", "p1", "publisher.example.com")], + }); + const web = server({ + "https://doi.example.org/10.1/x": () => new Response(null, { status: 302, headers: { location: "https://publisher.example.com/pdf/x.pdf" } }), + "https://publisher.example.com/pdf/x.pdf": () => pdfResponse(), + }); + const result = await downloadSignedInPdf(new URL("https://doi.example.org/10.1/x"), { + readCookies: cookies.read, + fetch: web.fetch, + browser: "Chrome", + }); + expect(result.finalUrl.href).toBe("https://publisher.example.com/pdf/x.pdf"); + expect(readFileSync(result.path).equals(Buffer.from(PDF))).toBe(true); + expect(statSync(result.path).mode & 0o777).toBe(0o600); + expect(statSync(dirname(result.path)).mode & 0o777).toBe(0o700); + expect(result.byteLength).toBe(PDF.byteLength); + rmSync(dirname(result.path), { recursive: true, force: true }); + expect(web.requests.map((request) => [request.url, request.cookie])).toEqual([ + ["https://doi.example.org/10.1/x", "doi=d1"], + ["https://publisher.example.com/pdf/x.pdf", "session=p1"], + ]); + expect(cookies.reads).toEqual(["doi.example.org", "publisher.example.com"]); + for (const request of web.requests) expect(request.init.redirect).toBe("error"); + }); + + test("a host with no cookies gets no cookie header, not another host's", async () => { + const cookies = jar({ "a.example.org": [cookie("sid", "secret", "a.example.org")] }); + const web = server({ + "https://a.example.org/x": () => new Response(null, { status: 301, headers: { location: "https://cdn.example.net/x.pdf" } }), + "https://cdn.example.net/x.pdf": () => pdfResponse(), + }); + const result = await downloadSignedInPdf(new URL("https://a.example.org/x"), { + readCookies: cookies.read, fetch: web.fetch, browser: "Chrome", + }); + expect(web.requests[1]?.cookie).toBeNull(); + expect(result.cookieHosts).toEqual(["a.example.org"]); + rmSync(dirname(result.path), { recursive: true, force: true }); + }); + + test("refuses to follow a redirect to plain http", async () => { + const web = server({ + "https://a.example.org/x": () => new Response(null, { status: 302, headers: { location: "http://a.example.org/x.pdf" } }), + }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x"), { + readCookies: jar({}).read, fetch: web.fetch, browser: "Chrome", + })).rejects.toThrow("a.example.org redirected to an unencrypted http link"); + expect(web.requests).toHaveLength(1); + }); + + test("a plain http link says how to fix it", async () => { + await expect(downloadSignedInPdf(new URL("http://dx.doi.org/10.1/x"), { + readCookies: jar({}).read, fetch: server({}).fetch, browser: "Chrome", + })).rejects.toThrow("Change the start of the link from http:// to https://"); + }); + + test("stops after the redirect limit", async () => { + const web = server({ + "https://a.example.org/loop": () => new Response(null, { status: 302, headers: { location: "/loop" } }), + }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/loop"), { + readCookies: jar({}).read, fetch: web.fetch, browser: "Chrome", maxRedirects: 3, + })).rejects.toThrow("more than 3 times"); + expect(web.requests).toHaveLength(4); + }); + + test("an HTML login page becomes a plain-language access error", async () => { + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(LOGIN, { "content-type": "text/html; charset=utf-8" }) }); + const error = await downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: signedIn("a.example.org").read, fetch: web.fetch, browser: "Chrome", + }).catch((caught: unknown) => caught); + expect(error).toBeInstanceOf(PdfAuthError); + expect((error as PdfAuthError).code).toBe("access"); + expect((error as PdfAuthError).message).toContain("Open the link in Chrome"); + }); + + test("an HTML page mislabelled as a PDF is still caught", async () => { + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(LOGIN) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: signedIn("a.example.org").read, fetch: web.fetch, browser: "Chrome", + })).rejects.toThrow("does not seem to have access"); + }); + + test("403 is an access error", async () => { + const web = server({ "https://a.example.org/x.pdf": () => new Response("no", { status: 403 }) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: signedIn("a.example.org").read, fetch: web.fetch, browser: "Safari", + })).rejects.toThrow("your Safari sign-in does not seem to have access"); + }); + + test("a body over the size limit is refused while streaming", async () => { + const big = new Uint8Array(4096); + big.set(PDF); + const web = server({ "https://a.example.org/x.pdf": () => new Response(new ReadableStream({ + start(controller) { + controller.enqueue(big); + controller.enqueue(big); + controller.close(); + }, + }), { status: 200 }) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: jar({}).read, fetch: web.fetch, browser: "Chrome", maxPdfBytes: 5000, + })).rejects.toThrow("the PDF is larger than the 5 KB limit"); + }); + + test("a declared length over the limit is refused before reading", async () => { + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(PDF, { "content-length": "999999" }) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: jar({}).read, fetch: web.fetch, browser: "Chrome", maxPdfBytes: 1000, + })).rejects.toThrow("larger than the 1 KB limit"); + }); + + test("a non-PDF, non-HTML body is refused", async () => { + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(new Uint8Array([0x50, 0x4b, 3, 4, 0])) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: jar({}).read, fetch: web.fetch, browser: "Chrome", + })).rejects.toThrow("did not return a PDF"); + }); + + test("times out", async () => { + const fetch: PdfFetch = (_url, init) => new Promise((_resolve, reject) => { + init.signal?.addEventListener("abort", () => reject(init.signal?.reason)); + }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: jar({}).read, fetch, browser: "Chrome", timeoutMs: 20, + })).rejects.toThrow("did not finish within"); + }); + + test("no sign-in anywhere and a login page is reported as a missing sign-in, not missing access", async () => { + const cookies = jar({}); + const web = server({ + "https://doi.example.org/10.1/x": () => new Response(null, { status: 302, headers: { location: "https://pub.example.com/x" } }), + "https://pub.example.com/x": () => pdfResponse(LOGIN, { "content-type": "text/html" }), + }); + const error = await downloadSignedInPdf(new URL("https://doi.example.org/10.1/x"), { + readCookies: cookies.read, fetch: web.fetch, browser: "Chrome", + }).catch((caught: unknown) => caught); + expect(cookies.reads).toEqual(["doi.example.org", "pub.example.com"]); + expect(web.requests.map((request) => request.cookie)).toEqual([null, null]); + expect((error as PdfAuthError).code).toBe("no-sign-in"); + expect((error as PdfAuthError).message).toContain("no Chrome sign-in was found for pub.example.com"); + expect((error as PdfAuthError).message).toContain("ghostget browsers"); + expect((error as PdfAuthError).message).not.toContain("does not seem to have access"); + }); + + test("a sign-in only for the DOI host still counts as missing for the publisher that refused", async () => { + const web = server({ + "https://doi.example.org/10.1/x": () => new Response(null, { status: 302, headers: { location: "https://pub.example.com/x" } }), + "https://pub.example.com/x": () => new Response("no", { status: 401 }), + }); + await expect(downloadSignedInPdf(new URL("https://doi.example.org/10.1/x"), { + readCookies: signedIn("doi.example.org").read, fetch: web.fetch, browser: "Chrome", + })).rejects.toThrow("no Chrome sign-in was found for pub.example.com"); + }); + + test("a redirect to another site holds back SameSite=Strict cookies; the link's own site keeps them", async () => { + const cookies = jar({ + "www.uni.example.edu": [cookie("strict", "s0", "www.uni.example.edu", "Strict")], + "uni.example.edu": [cookie("strict", "s1", "uni.example.edu", "Strict"), cookie("lax", "l1", "uni.example.edu", "Lax")], + "bank.example.com": [ + cookie("strict", "s2", "bank.example.com", "Strict"), + cookie("lax", "l2", "bank.example.com", "Lax"), + cookie("none", "n2", "bank.example.com", "None"), + cookie("unset", "u2", "bank.example.com"), + ], + }); + const web = server({ + "https://www.uni.example.edu/a": () => new Response(null, { status: 302, headers: { location: "https://uni.example.edu/b" } }), + "https://uni.example.edu/b": () => new Response(null, { status: 302, headers: { location: "https://bank.example.com/statement.pdf" } }), + "https://bank.example.com/statement.pdf": () => pdfResponse(), + }); + const result = await downloadSignedInPdf(new URL("https://www.uni.example.edu/a"), { + readCookies: cookies.read, fetch: web.fetch, browser: "Chrome", + }); + expect(web.requests.map((request) => request.cookie)).toEqual([ + "strict=s0", + "strict=s1; lax=l1", + "lax=l2; none=n2; unset=u2", + ]); + rmSync(dirname(result.path), { recursive: true, force: true }); + }); + + test("property: same-site hops are symmetric and never join unrelated hosts", () => { + expect(isSameSiteHop("doi.org", "doi.org")).toBe(true); + expect(isSameSiteHop("www.pub.example", "pub.example")).toBe(true); + expect(isSameSiteHop("a.pub.example", "b.pub.example")).toBe(false); + expect(isSameSiteHop("evilpub.example", "pub.example")).toBe(false); + expect(isSameSiteHop("1.2.3.4", "2.3.4")).toBe(false); + const label = fc.stringMatching(/^[a-z][a-z0-9]{0,8}$/u); + assertProperty(fc.property(label, label, label, (a, b, tld) => { + const x = `${a}.${tld}`; + const y = `${b}.${tld}`; + expect(isSameSiteHop(x, y)).toBe(isSameSiteHop(y, x)); + expect(isSameSiteHop(x, y)).toBe(a === b); + expect(isSameSiteHop(`www.${x}`, x)).toBe(true); + })); + }); + + test("a slow cookie read counts against the download deadline", async () => { + const timeouts: number[] = []; + const started = Date.now(); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: (_url, timeoutMs) => { + timeouts.push(timeoutMs); + return new Promise(() => undefined); + }, + fetch: server({}).fetch, + browser: "Chrome", + timeoutMs: 30, + })).rejects.toThrow("did not finish within"); + expect(Date.now() - started).toBeLessThan(5_000); + expect(timeouts).toHaveLength(1); + expect(timeouts[0]).toBeLessThanOrEqual(30); + }); + + test("a body that is not a PDF never creates a file", async () => { + const directory = mkdtempSync(join(tmpdir(), "ghostget-pdf-test-")); + try { + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(LOGIN, { "content-type": "text/html" }) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: signedIn("a.example.org").read, fetch: web.fetch, browser: "Chrome", directory, + })).rejects.toThrow("does not seem to have access"); + expect(readdirSync(directory)).toEqual([]); + } finally { + rmSync(directory, { recursive: true, force: true }); + } + }); + + test("an oversize body that started as a PDF leaves no partial file", async () => { + const directory = mkdtempSync(join(tmpdir(), "ghostget-pdf-test-")); + try { + const big = new Uint8Array(4096); + big.set(PDF); + const web = server({ "https://a.example.org/x.pdf": () => new Response(new ReadableStream({ + start(controller) { + controller.enqueue(big); + controller.enqueue(big); + controller.close(); + }, + }), { status: 200 }) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: jar({}).read, fetch: web.fetch, browser: "Chrome", maxPdfBytes: 6000, directory, + })).rejects.toThrow("larger than"); + expect(readdirSync(directory)).toEqual([]); + } finally { + rmSync(directory, { recursive: true, force: true }); + } + }); + + test("a failure other than 'no cookies' from the cookie reader stops the download", async () => { + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse() }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: async () => { throw new Error("keychain denied"); }, + fetch: web.fetch, + browser: "Chrome", + })).rejects.toThrow("keychain denied"); + expect(web.requests).toHaveLength(0); + }); +}); + +describe("downloadSignedInPdf: public hosts only", () => { + test("a redirect to a private address is refused before that host's cookies are read", async () => { + for (const target of ["https://10.0.0.1/x.pdf", "https://[::1]/x.pdf", "https://intranet.local/x.pdf", "https://printer/x.pdf"]) { + const cookies = jar({ "a.example.org": [cookie("sid", "1", "a.example.org")] }); + const web = server({ + "https://a.example.org/x": () => new Response(null, { status: 302, headers: { location: target } }), + }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x"), { + readCookies: cookies.read, fetch: web.fetch, browser: "Chrome", + })).rejects.toThrow("only go to public websites"); + expect(cookies.reads).toEqual(["a.example.org"]); + expect(web.requests).toHaveLength(1); + } + }); + + test("a private first link reads no cookies and sends nothing", async () => { + const cookies = jar({}); + const web = server({}); + await expect(downloadSignedInPdf(new URL("https://192.168.1.4/x.pdf"), { + readCookies: cookies.read, fetch: web.fetch, browser: "Chrome", + })).rejects.toThrow("192.168.1.4 was not contacted and no sign-in was read for it"); + expect(cookies.reads).toEqual([]); + expect(web.requests).toEqual([]); + }); + + test("sizes in messages are readable", async () => { + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(PDF, { "content-length": String(64 * 1024 * 1024) }) }); + await expect(downloadSignedInPdf(new URL("https://a.example.org/x.pdf"), { + readCookies: jar({}).read, fetch: web.fetch, browser: "Chrome", maxPdfBytes: 5 * 1024 * 1024, + })).rejects.toThrow("the PDF is larger than the 5 MB limit. Pass a larger --max-pdf-bytes to allow it"); + }); +}); + +describe("runSignedInPdf with a real cookie file", () => { + test("a DOI link with no cookies redirects to the publisher, which alone gets its cookie", async () => { + const parent = mkdtempSync(join(tmpdir(), "ghostget-pdf-test-")); + try { + const file = join(parent, "cookies.txt"); + writeFileSync(file, "# Netscape HTTP Cookie File\npublisher.example.com\tFALSE\t/\tTRUE\t4102444800\tsession\tp1\n"); + chmodSync(file, 0o600); + const web = server({ + "https://doi.example.org/10.1/x": () => new Response(null, { status: 302, headers: { location: "https://publisher.example.com/pdf/x.pdf" } }), + "https://publisher.example.com/pdf/x.pdf": () => pdfResponse(), + }); + let imported = false; + const code = await runSignedInPdf(["https://doi.example.org/10.1/x", "--cookies-file", file], { + environment: {}, + fetch: web.fetch, + makeTemporaryDirectory: () => mkdtempSync(join(parent, "run-")), + runWordcellPdf: async (_argv, download) => { + imported = download !== undefined; + return 0; + }, + }); + expect(code).toBe(0); + expect(imported).toBe(true); + expect(web.requests.map((request) => [request.url, request.cookie])).toEqual([ + ["https://doi.example.org/10.1/x", null], + ["https://publisher.example.com/pdf/x.pdf", "session=p1"], + ]); + } finally { + rmSync(parent, { recursive: true, force: true }); + } + }); + + test("a cookie file anyone can read is refused before any request", async () => { + const parent = mkdtempSync(join(tmpdir(), "ghostget-pdf-test-")); + try { + const file = join(parent, "cookies.txt"); + writeFileSync(file, "publisher.example.com\tFALSE\t/\tTRUE\t4102444800\tsession\tp1\n"); + chmodSync(file, 0o644); + await expect(runSignedInPdf(["https://publisher.example.com/x.pdf", "--cookies-file", file], { + environment: {}, + fetch: async () => { throw new Error("must not fetch"); }, + runWordcellPdf: async () => 0, + })).rejects.toThrow("must be readable only by you"); + } finally { + rmSync(parent, { recursive: true, force: true }); + } + }); +}); + +describe("runWordcellPdfWithDownload", () => { + test("records the web link as the note's source, as an anonymous download does", async () => { + const seen: { inputPath: string; remoteSource: unknown }[] = []; + const code = await runWordcellPdfWithDownload( + ["https://a.example.org/x.pdf", "--output", "/notes", "--quiet"], + { inputPath: "/private/copy/x.pdf", requestedUrl: "https://doi.example.org/10.1/x", finalUrl: "https://a.example.org/x.pdf" }, + {}, + { stdout: () => undefined, stderr: () => undefined }, + { + runPdfCapture: async (options) => { + seen.push({ inputPath: options.inputPath, remoteSource: options.remoteSource }); + throw new Error("stop after the source step"); + }, + }, + ); + expect(code).not.toBe(0); + expect(seen).toEqual([{ + inputPath: "/private/copy/x.pdf", + remoteSource: { requestedUrl: "https://doi.example.org/10.1/x", finalUrl: "https://a.example.org/x.pdf" }, + }]); + }); +}); + +describe("runSignedInPdf", () => { + test("hands Wordcell a private local copy with the sign-in options removed, then deletes it", async () => { + const parent = mkdtempSync(join(tmpdir(), "ghostget-pdf-test-")); + let seen: readonly string[] = []; + let copy = ""; + let mode = 0; + let source: PdfDownloadedSource | undefined; + const web = server({ "https://a.example.org/papers/Deep%20Sea.pdf": () => pdfResponse() }); + const code = await runSignedInPdf( + ["https://a.example.org/papers/Deep%20Sea.pdf", "--cookie-source", "chrome", "--output", "/notes", "--json"], + { + environment: {}, + readCookies: async () => ({ cookies: [cookie("s", "1", "a.example.org")], warnings: [] }), + fetch: web.fetch, + makeTemporaryDirectory: () => mkdtempSync(join(parent, "run-")), + runWordcellPdf: async (argv, download) => { + seen = argv; + source = download; + copy = download?.inputPath ?? ""; + mode = statSync(copy).mode & 0o777; + expect(readFileSync(copy).subarray(0, 5).toString()).toBe("%PDF-"); + return 0; + }, + }, + ); + expect(code).toBe(0); + expect(seen).toEqual(["https://a.example.org/papers/Deep%20Sea.pdf", "--output", "/notes", "--json"]); + expect(source?.requestedUrl).toBe("https://a.example.org/papers/Deep%20Sea.pdf"); + expect(source?.finalUrl).toBe("https://a.example.org/papers/Deep%20Sea.pdf"); + expect(copy.endsWith("Deep-Sea.pdf")).toBe(true); + expect(mode).toBe(0o600); + expect(existsSync(copy)).toBe(false); + rmSync(parent, { recursive: true, force: true }); + }); + + test("deletes the copy when Wordcell fails", async () => { + const parent = mkdtempSync(join(tmpdir(), "ghostget-pdf-test-")); + let copy = ""; + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse() }); + await expect(runSignedInPdf(["https://a.example.org/x.pdf", "--cookie-source", "chrome"], { + environment: {}, + readCookies: async () => ({ cookies: [], warnings: [] }), + fetch: web.fetch, + makeTemporaryDirectory: () => mkdtempSync(join(parent, "run-")), + runWordcellPdf: async (_argv, download) => { + copy = download?.inputPath ?? ""; + throw new Error("import failed"); + }, + })).rejects.toThrow("import failed"); + expect(copy).not.toBe(""); + expect(existsSync(copy)).toBe(false); + rmSync(parent, { recursive: true, force: true }); + }); + + test("without sign-in options the arguments go to Wordcell unchanged and nothing is downloaded", async () => { + const argv = ["https://a.example.org/x.pdf", "--root", "/n"]; + let seen: readonly string[] = []; + const code = await runSignedInPdf(argv, { + environment: {}, + fetch: async () => { throw new Error("must not fetch"); }, + readCookies: async () => { throw new Error("must not read cookies"); }, + runWordcellPdf: async (forwarded) => { + seen = forwarded; + return 7; + }, + }); + expect(code).toBe(7); + expect(seen).toEqual(argv); + }); + + test("sign-in options with a local file or http link are usage errors", async () => { + const base = { + environment: {}, + fetch: async () => { throw new Error("must not fetch"); }, + readCookies: async () => { throw new Error("must not read cookies"); }, + runWordcellPdf: async () => 0, + }; + await expect(runSignedInPdf(["/tmp/a.pdf", "--cookie-source", "chrome"], base)).rejects.toThrow("only apply to web links"); + await expect(runSignedInPdf(["http://a.example.org/a.pdf", "--cookie-source", "chrome"], base)).rejects.toThrow("only use https"); + }); + + test("a browser data folder resolves to its Default profile; a folder without a sign-in is refused", () => { + const root = mkdtempSync(join(tmpdir(), "ghostget-pdf-profile-")); + try { + const data = join(root, "data"); + mkdirSync(join(data, "Default", "Network"), { recursive: true }); + writeFileSync(join(data, "Default", "Network", "Cookies"), ""); + const profile = join(root, "profile"); + mkdirSync(profile); + writeFileSync(join(profile, "Cookies"), ""); + const browser = (value: string) => ({ kind: "browser", source: "chrome", profile: value }) as const; + expect(resolveProfilePath(browser(data))).toEqual(browser(join(data, "Default"))); + expect(resolveProfilePath(browser(profile))).toEqual(browser(profile)); + expect(resolveProfilePath(browser("~/data"), root)).toEqual(browser(join(data, "Default"))); + expect(resolveProfilePath(browser("Work"))).toEqual(browser("Work")); + expect(() => resolveProfilePath(browser(join(root, "empty")))).toThrow("has no saved sign-in"); + const firefox = { kind: "browser", source: "firefox", profile: join(root, "empty") } as const; + expect(resolveProfilePath(firefox)).toEqual(firefox); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); + + test("a stored account without a browser cookie store is refused", async () => { + await expect(runSignedInPdf(["https://a.example.org/x.pdf", "--auth", "p"], { + environment: {}, + loadAuth: () => ({ schemaVersion: 1, id: "p", kind: "browser-profile", profile: "/p", trustUnfilteredEgress: true }), + fetch: async () => { throw new Error("must not fetch"); }, + runWordcellPdf: async () => 0, + })).rejects.toThrow("does not name a browser"); + }); +}); + +describe("runSignedInPdfCommand", () => { + test("agents get the keychain notice line before the browser store is read, even without a terminal", async () => { + const state = mkdtempSync(join(tmpdir(), "ghostget-pdf-state-")); + try { + const errors: string[] = []; + let beforeRead = ""; + const reader = createClassifiedCookieRecordReader(async () => { + beforeRead = errors.join(""); + return { cookies: [], warnings: [] }; + }, { platform: "darwin" }); + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse() }); + await runSignedInPdfCommand(["https://a.example.org/x.pdf", "--cookie-source", "chrome", "--quiet"], { + stdout: () => undefined, + stderr: (text) => errors.push(text), + }, { + environment: { CLAUDECODE: "1", GHOSTGET_STATE_HOME: state }, + stdinIsTTY: false, + stderrIsTTY: false, + runWordcellPdf: async () => 0, + overrides: { + fetch: web.fetch, + readCookies: (_signIn, _auth, url, timeoutMs) => reader({ + cookieSources: ["chrome"], cookiesFile: undefined, cookieProfile: undefined, timeoutMs, requireExplicitCookieScope: true, + }, url), + }, + }); + const line = JSON.parse(beforeRead.trim().split("\n")[0] ?? "") as Record; + expect(line).toMatchObject({ type: "permission-notice", product: "Ghostget", kind: "keychain" }); + } finally { + rmSync(state, { recursive: true, force: true }); + } + }); + + test("an unknown account name gets a plain sentence and the command that lists accounts", async () => { + const errors: string[] = []; + const code = await runSignedInPdfCommand(["https://a.example.org/x.pdf", "--auth", "nosuch"], { + stdout: () => undefined, + stderr: (text) => errors.push(text), + }, { + environment: { NO_COLOR: "1" }, + stderrIsTTY: false, + runWordcellPdf: async () => 0, + overrides: { loadAuth: () => { throw new Error("auth locator nosuch was not found."); } }, + }); + expect(code).toBe(2); + const text = errors.join(""); + expect(text).toContain("There is no connected account named nosuch; ghostget auth list shows the ones you have."); + expect(text).toContain("ghostget auth list"); + expect(text).not.toContain("auth locator"); + }); + + test("an oversize PDF names its limit in MB", async () => { + const lines: string[] = []; + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(PDF, { "content-length": String(900 * 1024 * 1024) }) }); + const code = await runSignedInPdfCommand(["https://a.example.org/x.pdf", "--cookie-source", "chrome", "--json"], { + stdout: (text) => lines.push(text), + stderr: () => undefined, + }, { + environment: {}, + stderrIsTTY: false, + runWordcellPdf: async () => 0, + overrides: { fetch: web.fetch, readCookies: async () => ({ cookies: [], warnings: [] }) }, + }); + expect(code).toBe(1); + expect(JSON.parse(lines.join(""))).toEqual({ + ok: false, + error: { + code: "download-failed", + message: "The PDF is larger than the 512 MB limit. Pass a larger --max-pdf-bytes to allow it.", + next: "ghostget pdf --help", + }, + }); + }); + test("prints a plain-language error with a next step for a login page", async () => { + const errors: string[] = []; + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(LOGIN, { "content-type": "text/html" }) }); + const code = await runSignedInPdfCommand(["https://a.example.org/x.pdf", "--cookie-source", "chrome"], { + stdout: () => undefined, + stderr: (text) => errors.push(text), + }, { + environment: { NO_COLOR: "1" }, + stderrIsTTY: false, + runWordcellPdf: async () => 0, + overrides: { fetch: web.fetch, readCookies: async () => ({ cookies: [cookie("s", "1", "a.example.org")], warnings: [] }) }, + }); + expect(code).toBe(1); + const text = errors.join(""); + expect(text).toContain("your Chrome sign-in does not seem to have access"); + expect(text).toContain("Open the link in your browser first"); + expect(text).not.toContain(" { + const lines: string[] = []; + const web = server({ "https://a.example.org/x.pdf": () => pdfResponse(LOGIN, { "content-type": "text/html" }) }); + const code = await runSignedInPdfCommand(["https://a.example.org/x.pdf", "--browser-profile", "Work", "--json"], { + stdout: (text) => lines.push(text), + stderr: () => undefined, + }, { + environment: {}, + stderrIsTTY: false, + runWordcellPdf: async () => 0, + overrides: { + fetch: web.fetch, + readCookies: async () => { throw new Error("no matching cookies were found in the selected browser store"); }, + }, + }); + expect(code).toBe(1); + const parsed = JSON.parse(lines.join("")) as { error: { code: string; message: string; next: string } }; + expect(parsed.error.code).toBe("no-sign-in"); + expect(parsed.error.message).toContain("No Chrome sign-in was found for a.example.org"); + expect(parsed.error.next).toBe("ghostget browsers"); + }); + + test("--json reports errors as one JSON line", async () => { + const lines: string[] = []; + const code = await runSignedInPdfCommand(["https://a.example.org/x.pdf", "--cookie-source", "netscape", "--json"], { + stdout: (text) => lines.push(text), + stderr: () => undefined, + }, { environment: {}, stderrIsTTY: false, runWordcellPdf: async () => 0 }); + expect(code).toBe(2); + const parsed = JSON.parse(lines.join("")) as { ok: boolean; error: { code: string } }; + expect(parsed).toMatchObject({ ok: false, error: { code: "usage" } }); + }); +}); + +describe("pdfFilename", () => { + test("keeps a safe .pdf name", () => { + expect(pdfFilename(new URL("https://e.org/a/Deep%20Sea.PDF"))).toBe("Deep-Sea.pdf"); + expect(pdfFilename(new URL("https://e.org/"))).toBe("source.pdf"); + expect(pdfFilename(new URL("https://e.org/..%2F..%2Fetc"))).toBe("etc.pdf"); + }); + + test("property: never contains a path separator and always ends in .pdf", () => { + assertProperty(fc.property(fc.string({ maxLength: 60 }), (segment) => { + const name = pdfFilename(new URL(`https://e.org/${encodeURIComponent(segment)}`)); + expect(name.endsWith(".pdf")).toBe(true); + expect(name.includes("/")).toBe(false); + expect(name.startsWith(".")).toBe(false); + })); + }); +}); diff --git a/src/pdf-auth.ts b/src/pdf-auth.ts new file mode 100644 index 00000000..9b4bb54c --- /dev/null +++ b/src/pdf-auth.ts @@ -0,0 +1,957 @@ +/** + * `ghostget pdf ` with a browser sign-in. + * + * Without sign-in options `pdf` stays a thin delegation to the Wordcell notes + * CLI, which downloads the file anonymously. With them, Ghostget downloads the + * bytes itself through the same cookie readers `read` and `clip` use, checks + * that the result really is a PDF, and hands Wordcell a private local copy. + * + * Cookies are read from the selected browser separately for every origin the + * download visits and filtered to that exact URL by the shared cookie filter, + * so a cookie is only ever sent to a host it belongs to. Every hop is a + * DNS-pinned public HTTPS request. Plain HTTP, credential-bearing URLs, + * non-public IP literals, and local host names are refused before any cookie + * is read; a name that resolves to a private address is refused by the pinned + * fetch before any request is sent. + */ + +import { chmodSync, closeSync, existsSync, mkdtempSync, openSync, rmSync, statSync, writeSync } from "node:fs"; +import { homedir, tmpdir } from "node:os"; +import { basename, isAbsolute, join, resolve } from "node:path"; + +import { cookieSources, type CookieSource } from "@hraness/wordcell/clip/args"; +import type { StrictCookie } from "@hraness/wordcell/clip/cookies"; +import type { CookieSelection } from "@hraness/wordcell/clip/acquire"; + +import type { GhostgetAuth } from "./auth"; +import { isPublicUnicastAddress, parseIpv4, parseIpv6 } from "./public-address"; +import { PDF_SIGN_IN_OPTIONS, hasPdfSignInOptions } from "./usage"; + +export { hasPdfSignInOptions }; + +/** Same limits Wordcell applies to an anonymous PDF download. */ +export const PDF_AUTH_DEFAULTS = Object.freeze({ + timeoutMs: 120_000, + maxPdfBytes: 512 * 1024 * 1024, + maxRedirects: 5, +}); + +/** The browser-like agent `read --mode http` sends by default. */ +const USER_AGENT = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " + + "(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"; + +export type PdfSignIn = + | { readonly kind: "auth"; readonly id: string } + | { + readonly kind: "browser"; + readonly source: CookieSource; + readonly profile: string | undefined; + } + | { readonly kind: "cookies-file"; readonly path: string }; + +export type PdfAuthSplit = { + readonly signIn: PdfSignIn | null; + /** Arguments for Wordcell `pdf`, with every sign-in option removed. */ + readonly remaining: readonly string[]; +}; + +/** + * `access`: the sign-in reached the site but it refused the PDF. + * `no-sign-in`: no cookie at all went to the site that refused, so the chosen + * browser or profile most likely has no sign-in for it. + */ +export type PdfAuthErrorCode = "usage" | "access" | "no-sign-in" | "download"; + +/** A problem the person can fix; `code` picks the exit status. */ +export class PdfAuthError extends Error { + readonly code: PdfAuthErrorCode; + constructor(code: PdfAuthErrorCode, message: string, options?: { readonly cause?: unknown }) { + super(message, options); + this.name = "PdfAuthError"; + this.code = code; + } +} + +function usage(message: string): PdfAuthError { + return new PdfAuthError("usage", message); +} + +function isCookieSource(value: string): value is CookieSource { + return (cookieSources as readonly string[]).includes(value); +} + +/** + * Remove sign-in options from `pdf` arguments (everything after `pdf`). + * Only the combinations `read` accepts are allowed; anything else is a usage + * error raised before a browser store is touched. + */ +export function splitPdfSignInArguments(pdfArguments: readonly string[]): PdfAuthSplit { + const remaining: string[] = []; + const values = new Map(); + // `read` options that mean nothing for a PDF download. They are dropped only + // when a sign-in option is present; otherwise Wordcell rejects them as before. + const readOnly: string[] = []; + for (let index = 0; index < pdfArguments.length; index += 1) { + const argument = pdfArguments[index] ?? ""; + if (argument === "--") { + remaining.push(...pdfArguments.slice(index)); + break; + } + const equals = argument.indexOf("="); + const name = argument.startsWith("--") && equals !== -1 ? argument.slice(0, equals) : argument; + if (argument === "--trust-profile-egress") { + readOnly.push(argument); + continue; + } + if (name === "--mode") { + const mode = name !== argument ? argument.slice(equals + 1) : pdfArguments[index + 1]; + if (name === argument) index += 1; + readOnly.push(name, mode ?? ""); + continue; + } + if (!PDF_SIGN_IN_OPTIONS.has(name)) { + remaining.push(argument); + continue; + } + let value: string | undefined; + if (name !== argument) { + value = argument.slice(equals + 1); + } else { + value = pdfArguments[index + 1]; + index += 1; + } + if (value === undefined || value === "" || value.startsWith("--")) { + throw usage(`${name} needs a value`); + } + if (values.has(name)) throw usage(`${name} can be given only once`); + values.set(name, value); + } + if (values.size === 0) { + if (readOnly.length === 0) return { signIn: null, remaining }; + // Put them back in place so Wordcell reports them exactly as before. + return { signIn: null, remaining: [...pdfArguments] }; + } + for (let index = 0; index < readOnly.length; index += 1) { + if (readOnly[index] !== "--mode") continue; + const mode = readOnly[index + 1]; + if (mode !== "browser" && mode !== "http") { + throw usage("--mode must be browser or http; a signed-in PDF download needs neither, so you can leave it out"); + } + } + + const authId = values.get("--auth"); + const browserProfile = values.get("--browser-profile"); + const cookieSource = values.get("--cookie-source"); + const cookieProfile = values.get("--cookie-profile"); + const cookiesFile = values.get("--cookies-file"); + + if (authId !== undefined) { + if (browserProfile !== undefined || cookieSource !== undefined || cookieProfile !== undefined || cookiesFile !== undefined) { + throw usage("--auth cannot be combined with --browser-profile, --cookie-source, --cookie-profile or --cookies-file"); + } + return { signIn: { kind: "auth", id: authId }, remaining }; + } + if (cookiesFile !== undefined) { + if (browserProfile !== undefined || cookieSource !== undefined || cookieProfile !== undefined) { + throw usage("--cookies-file cannot be combined with a browser or browser profile"); + } + return { signIn: { kind: "cookies-file", path: cookiesFile }, remaining }; + } + if (cookieSource !== undefined && !isCookieSource(cookieSource)) { + throw usage(`--cookie-source must be one of ${cookieSources.join(", ")}`); + } + if (browserProfile !== undefined && cookieProfile !== undefined) { + throw usage("use either --browser-profile or --cookie-profile, not both"); + } + if (cookieSource === undefined && browserProfile === undefined) { + throw usage(cookieProfile !== undefined + ? "--cookie-profile needs --cookie-source, for example --cookie-source chrome" + : "say which browser is signed in, for example --cookie-source chrome"); + } + // `--browser-profile` names a Chrome profile unless another browser is given, + // matching `read --browser-profile`'s "signed-in Chrome profile". + return { + signIn: { + kind: "browser", + source: (cookieSource ?? "chrome") as CookieSource, + profile: browserProfile ?? cookieProfile, + }, + remaining, + }; +} + +/** Reads the cookies for one URL, within `timeoutMs`. */ +export type PdfCookieReader = (url: URL, timeoutMs: number) => Promise<{ + readonly cookies: readonly StrictCookie[]; + readonly warnings: readonly string[]; +}>; + +export type PdfFetch = (url: URL, init: RequestInit, timeoutMs: number) => Promise; + +export type PdfDownload = { + /** Owner-only file holding the PDF, inside the download directory. */ + readonly path: string; + readonly byteLength: number; + readonly finalUrl: URL; + /** Distinct hosts that received at least one cookie, for tests and notices. */ + readonly cookieHosts: readonly string[]; +}; + +/** What the shared reader says when a cookie file has nothing for one URL. */ +const COOKIE_FILE_NO_MATCH = "the explicitly selected cookie file contained no usable cookies for this request"; + +function isNoMatchingCookies(error: unknown): boolean { + const message = error instanceof Error ? error.message : ""; + return message.startsWith("no matching cookies were found") + || message.startsWith("no usable origin-scoped cookies were found") + || message.startsWith("the managed Chromium profile contained no usable origin-scoped cookies") + || message === COOKIE_FILE_NO_MATCH; +} + +function browserLabel(signIn: PdfSignIn | null, auth: GhostgetAuth | undefined): string { + const source = signIn?.kind === "browser" + ? signIn.source + : auth?.kind === "cookie-source" + ? auth.source + : auth?.kind === "browser-profile" + ? auth.cookieSource ?? "chrome" + : undefined; + if (source === undefined) return "browser"; + const names: Record = { + chrome: "Chrome", arc: "Arc", brave: "Brave", chromium: "Chromium", + edge: "Edge", firefox: "Firefox", safari: "Safari", + }; + return names[source] ?? "browser"; +} + +function noAccessMessage(browser: string): string { + return `the site sent a web page instead of the PDF, so your ${browser} sign-in does not seem to have access to it. ` + + `Open the link in ${browser}, sign in through your library or university if it asks, check that the PDF opens there, then run this again`; +} + +function noSignInMessage(browser: string, host: string): string { + return `no ${browser} sign-in was found for ${host}, so the site sent a sign-in page instead of the PDF. ` + + `Check that you chose the browser and profile you use for this site (ghostget browsers lists them), ` + + `or open the link in ${browser}, sign in, then run this again`; +} + +function refused(browser: string, host: string, signedIn: boolean): PdfAuthError { + return signedIn + ? new PdfAuthError("access", noAccessMessage(browser)) + : new PdfAuthError("no-sign-in", noSignInMessage(browser, host)); +} + +function readableSize(bytes: number): string { + if (bytes < 1024 * 1024) return `${Math.max(1, Math.round(bytes / 1024))} KB`; + const value = bytes / (1024 * 1024); + return `${value >= 10 ? Math.round(value) : Math.round(value * 10) / 10} MB`; +} + +function tooLarge(maxBytes: number): PdfAuthError { + return new PdfAuthError( + "download", + `the PDF is larger than the ${readableSize(maxBytes)} limit. Pass a larger --max-pdf-bytes to allow it`, + ); +} + +function contentTypeEssence(response: Response): string { + return (response.headers.get("content-type") ?? "").split(";", 1)[0]?.trim().toLowerCase() ?? ""; +} + +function looksLikeHtml(bytes: Uint8Array, contentType: string): boolean { + if (contentType === "text/html" || contentType === "application/xhtml+xml") return true; + const head = new TextDecoder("utf-8", { fatal: false }) + .decode(bytes.subarray(0, 512)) + .replace(/^/u, "") + .trimStart() + .toLowerCase(); + return head.startsWith("= 5 + && bytes[0] === 0x25 && bytes[1] === 0x50 && bytes[2] === 0x44 && bytes[3] === 0x46 && bytes[4] === 0x2d; +} + +/** Enough of the body to tell a PDF from an HTML page. */ +const SNIFF_BYTES = 512; + +type StreamResult = + | { readonly pdf: true; readonly byteLength: number } + | { readonly pdf: false; readonly head: Uint8Array }; + +function concat(chunks: readonly Uint8Array[], total: number): Uint8Array { + const bytes = new Uint8Array(total); + let offset = 0; + for (const chunk of chunks) { + bytes.set(chunk, offset); + offset += chunk.byteLength; + } + return bytes; +} + +function writeAll(descriptor: number, bytes: Uint8Array): void { + let offset = 0; + while (offset < bytes.byteLength) { + offset += writeSync(descriptor, bytes, offset, bytes.byteLength - offset); + } +} + +/** + * Stream a response into an owner-only file created only once the first bytes + * show a PDF signature. At most `maxBytes` are read, and only the sniffed head + * is ever held in memory as a whole. + */ +async function streamPdf( + response: Response, + maxBytes: number, + signal: AbortSignal, + openPath: () => string, +): Promise { + const declared = Number(response.headers.get("content-length") ?? ""); + if (Number.isFinite(declared) && declared > maxBytes) { + await response.body?.cancel().catch(() => undefined); + throw tooLarge(maxBytes); + } + if (response.body === null) return { pdf: false, head: new Uint8Array() }; + const reader = response.body.getReader(); + const pending: Uint8Array[] = []; + let total = 0; + let path: string | undefined; + let descriptor: number | undefined; + try { + for (;;) { + signal.throwIfAborted(); + const next = await reader.read(); + if (!next.done) { + total += next.value.byteLength; + if (total > maxBytes) throw tooLarge(maxBytes); + } + if (descriptor === undefined) { + if (!next.done) pending.push(next.value); + const buffered = pending.reduce((sum, chunk) => sum + chunk.byteLength, 0); + if (!next.done && buffered < SNIFF_BYTES) continue; + const head = concat(pending, buffered); + pending.length = 0; + if (!isPdfSignature(head)) { + if (!next.done) await reader.cancel().catch(() => undefined); + return { pdf: false, head: head.subarray(0, SNIFF_BYTES) }; + } + path = openPath(); + descriptor = openSync(path, "wx", 0o600); + writeAll(descriptor, head); + } else if (!next.done) { + writeAll(descriptor, next.value); + } + if (next.done) break; + } + closeSync(descriptor); + descriptor = undefined; + return { pdf: true, byteLength: total }; + } catch (error) { + await reader.cancel().catch(() => undefined); + if (descriptor !== undefined) closeSync(descriptor); + if (path !== undefined) rmSync(path, { force: true }); + throw error; + } +} + +/** + * Resolve with `work`, or reject with the deadline's reason as soon as + * `signal` aborts, so a slow cookie read cannot outlast `--timeout-ms`. + */ +function withinDeadline(signal: AbortSignal, work: Promise): Promise { + signal.throwIfAborted(); + return new Promise((resolvePromise, reject) => { + const onAbort = (): void => reject(signal.reason); + signal.addEventListener("abort", onAbort, { once: true }); + work.then( + (value) => { + signal.removeEventListener("abort", onAbort); + resolvePromise(value); + }, + (error: unknown) => { + signal.removeEventListener("abort", onAbort); + reject(error); + }, + ); + }); +} + +function bareHost(hostname: string): string { + return hostname.toLowerCase().replace(/^\[|\]$/gu, "").replace(/\.$/u, ""); +} + +/** + * True when a redirect hop stays on the link's own site: the same host, or a + * parent or subdomain of it. Anything else, including a sibling subdomain, is + * treated as another site, which only ever holds back more cookies than a + * browser would. + */ +export function isSameSiteHop(linkHost: string, hopHost: string): boolean { + const link = bareHost(linkHost); + const hop = bareHost(hopHost); + if (link === hop) return true; + if (parseIpv4(link) !== null || parseIpv6(link) !== null || parseIpv4(hop) !== null || parseIpv6(hop) !== null) return false; + return link.endsWith(`.${hop}`) || hop.endsWith(`.${link}`); +} + +/** + * True for hosts that are plainly not public websites: non-public IP literals + * and local-only names. Names that merely resolve to a private address are + * caught by the pinned fetch, which checks every resolved address. + */ +function isLocalOnlyHost(hostname: string): boolean { + const host = hostname.toLowerCase().replace(/^\[|\]$/gu, "").replace(/\.$/u, ""); + if (parseIpv4(host) !== null || parseIpv6(host) !== null) return !isPublicUnicastAddress(host); + return !host.includes(".") + || host === "localhost" + || [".localhost", ".local", ".internal", ".home.arpa", ".lan", ".intranet"].some((suffix) => host.endsWith(suffix)); +} + +function checkedHttpsUrl(url: URL, first: boolean): URL { + if (url.protocol !== "https:") { + throw new PdfAuthError("download", first + ? "signed-in PDF downloads only use https links, so your sign-in is never sent unencrypted. " + + "Change the start of the link from http:// to https:// and run it again" + : `signed-in PDF downloads only use https links, but ${url.hostname} redirected to an unencrypted http link, ` + + "so the download stopped to keep your sign-in private"); + } + if (url.username !== "" || url.password !== "") { + throw new PdfAuthError("download", "the PDF link must not contain a user name or password"); + } + if (isLocalOnlyHost(url.hostname)) { + throw new PdfAuthError( + "download", + `signed-in PDF downloads only go to public websites, so ${url.hostname} was not contacted and no sign-in was read for it`, + ); + } + const clean = new URL(url); + clean.hash = ""; + return clean; +} + +/** + * Download one PDF with cookies from the selected browser. + * + * Redirects are followed manually, at most `maxRedirects` times, each to an + * https URL. Cookies for every hop come from `readCookies(hopUrl)`, which + * returns only cookies whose domain and path match that URL; results are cached + * per origin and path, so one host's cookies can never be replayed to another. + * A hop on a different site from the link holds back `SameSite=Strict` + * cookies, as a browser does after a redirect. Cookie reads count against the + * same deadline as the download. The PDF is streamed into an owner-only file + * in `directory` (a new private temporary folder when omitted). + */ +export async function downloadSignedInPdf( + input: URL, + options: { + readonly readCookies: PdfCookieReader; + readonly fetch: PdfFetch; + readonly browser: string; + readonly directory?: string; + readonly timeoutMs?: number; + readonly maxPdfBytes?: number; + readonly maxRedirects?: number; + }, +): Promise { + const timeoutMs = options.timeoutMs ?? PDF_AUTH_DEFAULTS.timeoutMs; + const maxBytes = options.maxPdfBytes ?? PDF_AUTH_DEFAULTS.maxPdfBytes; + const maxRedirects = options.maxRedirects ?? PDF_AUTH_DEFAULTS.maxRedirects; + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(new PdfAuthError( + "download", + `the download did not finish within ${Math.round(timeoutMs / 1000)} seconds. Try again, or pass a larger --timeout-ms`, + )), timeoutMs); + const deadline = Date.now() + timeoutMs; + const cookiesByOrigin = new Map(); + const cookieHosts = new Set(); + const rethrowDeadline = (error: unknown): never => { + if (controller.signal.aborted && controller.signal.reason instanceof PdfAuthError) throw controller.signal.reason; + throw error; + }; + try { + const link = checkedHttpsUrl(input, true); + let url = link; + for (let hop = 0; ; hop += 1) { + const sameSite = isSameSiteHop(link.hostname, url.hostname); + const cookieKey = `${sameSite ? "same" : "cross"} ${url.origin}${url.pathname}`; + let header = cookiesByOrigin.get(cookieKey); + if (header === undefined) { + let cookies: readonly StrictCookie[] = []; + try { + cookies = (await withinDeadline( + controller.signal, + options.readCookies(url, Math.max(1, deadline - Date.now())), + )).cookies; + } catch (error) { + if (controller.signal.aborted) rethrowDeadline(error); + if (!isNoMatchingCookies(error)) throw error; + } + header = cookies + .filter((cookie) => sameSite || cookie.sameSite !== "Strict") + .map(({ name, value }) => `${name}=${value}`) + .join("; "); + cookiesByOrigin.set(cookieKey, header); + } + if (header !== "") cookieHosts.add(url.hostname); + controller.signal.throwIfAborted(); + const headers = new Headers({ + accept: "application/pdf,application/octet-stream;q=0.9,*/*;q=0.1", + "accept-encoding": "identity", + "user-agent": USER_AGENT, + }); + if (header !== "") headers.set("cookie", header); + const remaining = Math.max(1_000, Math.min(10 * 60_000, deadline - Date.now())); + let response: Response; + try { + response = await options.fetch(url, { + method: "GET", + redirect: "error", + headers, + signal: controller.signal, + }, remaining); + } catch (error) { + if (controller.signal.aborted && controller.signal.reason instanceof PdfAuthError) throw controller.signal.reason; + throw new PdfAuthError("download", `could not reach ${url.hostname}. Check the link and your connection, then try again`, { cause: error }); + } + if ([301, 302, 303, 307, 308].includes(response.status)) { + await response.body?.cancel().catch(() => undefined); + const location = response.headers.get("location"); + if (location === null || location === "") { + throw new PdfAuthError("download", `${url.hostname} sent a redirect without a destination`); + } + if (hop >= maxRedirects) { + throw new PdfAuthError("download", `the link redirected more than ${maxRedirects} times without reaching a PDF`); + } + let next: URL; + try { + next = new URL(location, url); + } catch { + throw new PdfAuthError("download", `${url.hostname} sent an invalid redirect`); + } + url = checkedHttpsUrl(next, false); + continue; + } + const signedIn = cookieHosts.has(url.hostname); + if (response.status === 401 || response.status === 403) { + await response.body?.cancel().catch(() => undefined); + throw refused(options.browser, url.hostname, signedIn); + } + if (response.status !== 200) { + await response.body?.cancel().catch(() => undefined); + throw new PdfAuthError("download", `${url.hostname} answered with HTTP ${response.status} instead of the PDF`); + } + const contentType = contentTypeEssence(response); + const finalUrl = url; + let directory = options.directory; + let result: StreamResult; + try { + result = await streamPdf(response, maxBytes, controller.signal, () => { + directory ??= privateDirectory(); + return join(directory, pdfFilename(finalUrl)); + }); + } catch (error) { + return rethrowDeadline(error); + } + if (result.pdf) { + return { + path: join(directory ?? "", pdfFilename(finalUrl)), + byteLength: result.byteLength, + finalUrl, + cookieHosts: [...cookieHosts], + }; + } + if (looksLikeHtml(result.head, contentType)) throw refused(options.browser, url.hostname, signedIn); + throw new PdfAuthError("download", "the link did not return a PDF file"); + } + } finally { + clearTimeout(timer); + } +} + +/** A new owner-only temporary folder. */ +function privateDirectory(): string { + const directory = mkdtempSync(join(tmpdir(), "ghostget-pdf-")); + chmodSync(directory, 0o700); + return directory; +} + +/** A filesystem-safe `.pdf` name from the final URL, like Wordcell's own download. */ +export function pdfFilename(url: URL): string { + let decoded: string; + try { + decoded = decodeURIComponent(basename(url.pathname)); + } catch { + decoded = "source.pdf"; + } + const normalized = decoded + .normalize("NFKC") + .replace(/[^\p{Letter}\p{Number}._-]+/gu, "-") + .replace(/^[.-]+|[.-]+$/gu, "") + .slice(0, 180); + const stem = normalized === "" ? "source" : normalized.replace(/\.pdf$/iu, ""); + return `${stem}.pdf`; +} + +export type PdfAuthOutput = { + readonly stdout: (value: string) => void; + readonly stderr: (value: string) => void; +}; + +/** A PDF Ghostget already downloaded, for Wordcell to import as that link. */ +export type PdfDownloadedSource = { + /** Owner-only local copy; Ghostget removes it after the import. */ + readonly inputPath: string; + /** The link the person gave, with secrets in the query redacted. */ + readonly requestedUrl: string; + /** The link the PDF finally came from, redacted the same way. */ + readonly finalUrl: string; +}; + +/** + * Runs Wordcell `pdf`. Without `download` it is the ordinary anonymous path; + * with it, Wordcell imports the local copy but records the web link as the + * note's source, exactly as when it downloads the link itself. + */ +export type RunWordcellPdf = ( + pdfArguments: readonly string[], + download?: PdfDownloadedSource, +) => Promise; + +export type PdfAuthDependencies = { + readonly environment: Readonly>; + readonly runWordcellPdf: RunWordcellPdf; + readonly readCookies?: (signIn: PdfSignIn, auth: GhostgetAuth | undefined, url: URL, timeoutMs: number) => ReturnType; + readonly loadAuth?: (id: string) => GhostgetAuth; + readonly fetch?: PdfFetch; + readonly makeTemporaryDirectory?: () => string; + readonly removeDirectory?: (path: string) => void; + /** Called before any browser store is read, for the keychain notice. */ + readonly beforeCookieRead?: () => void; +}; + +type ValidatedPdf = { + readonly input: string; + readonly json: boolean; + readonly quiet: boolean; + readonly timeoutMs?: number; + readonly maxPdfBytes?: number; +}; + +async function validateWordcellArguments( + remaining: readonly string[], + environment: Readonly>, +): Promise { + const { parsePdfArguments } = await import("@hraness/wordcell/pdf"); + const parsed = parsePdfArguments(remaining, environment); + if (!parsed.ok) throw usage(parsed.message); + if (parsed.value.command !== "capture") throw usage("a PDF link is missing"); + return { + input: parsed.value.input, + json: parsed.value.json, + quiet: parsed.value.quiet, + ...(parsed.value.timeoutMs === undefined ? {} : { timeoutMs: parsed.value.timeoutMs }), + ...(parsed.value.maxPdfBytes === undefined ? {} : { maxPdfBytes: parsed.value.maxPdfBytes }), + }; +} + +function authForSignIn(signIn: PdfSignIn, load: (id: string) => GhostgetAuth): GhostgetAuth | undefined { + if (signIn.kind !== "auth") return undefined; + let auth: GhostgetAuth; + try { + auth = load(signIn.id); + } catch (error) { + const message = error instanceof Error ? error.message : ""; + if (/^auth locator .* was not found/u.test(message)) { + throw new PdfAuthError("usage", `there is no connected account named ${signIn.id}; ghostget auth list shows the ones you have`, { cause: error }); + } + throw error; + } + if (auth.kind === "cookie-source" || auth.kind === "cookies-file") return auth; + if (auth.kind === "browser-profile" && auth.cookieSource !== undefined) return auth; + if (auth.kind === "browser-profile") { + throw usage(`connected account ${signIn.id} does not name a browser to read its sign-in from; use --cookie-source chrome instead`); + } + throw usage(`connected account ${signIn.id} is not a browser sign-in, so it cannot download a PDF`); +} + +function selectionFor(signIn: PdfSignIn, timeoutMs: number): CookieSelection { + if (signIn.kind === "cookies-file") { + return { cookieSources: [], cookiesFile: signIn.path, cookieProfile: undefined, timeoutMs, requireExplicitCookieScope: true }; + } + if (signIn.kind === "browser") { + return { cookieSources: [signIn.source], cookiesFile: undefined, cookieProfile: signIn.profile, timeoutMs, requireExplicitCookieScope: true }; + } + throw new Error("stored sign-ins are read through their auth record"); +} + +function cookieFilePath(signIn: PdfSignIn, auth: GhostgetAuth | undefined): string | undefined { + if (signIn.kind === "cookies-file") return signIn.path; + return auth?.kind === "cookies-file" ? auth.path : undefined; +} + +/** + * Check a cookie file once, before the download, with the same reader `read` + * uses. A file with no cookies for the first link is fine (a DOI link, say, + * redirects to the publisher the cookies belong to); a file that cannot be + * used at all is a plain error instead of an anonymous download. + */ +async function checkCookieFile(path: string, url: URL, requireExplicitScope: boolean): Promise { + const { readCookieFile } = await import("@hraness/wordcell/clip/cookies"); + const checked = readCookieFile(path, url, { requirePrivate: true }); + if (checked.ok) { + if (requireExplicitScope && checked.scopeProvenance !== "explicit") { + throw usage(`the cookie file ${path} must say which website each cookie belongs to; export it as a Netscape cookies.txt file`); + } + return; + } + if (checked.reason === "empty") return; + if (checked.reason === "unavailable") throw usage(`could not open the cookie file ${path}`); + if (checked.reason === "unsafe-permissions") { + throw usage(`the cookie file ${path} must be readable only by you; run chmod 600 on it and try again`); + } + if (checked.reason === "too-large") throw usage(`the cookie file ${path} is too large to be a browser cookie export`); + throw usage(`the cookie file ${path} is not in a format Ghostget can read; export it as a Netscape cookies.txt file`); +} + +async function defaultReadCookies( + signIn: PdfSignIn, + auth: GhostgetAuth | undefined, + url: URL, + timeoutMs: number, +): ReturnType { + const cookieTimeout = Math.min(timeoutMs, 30_000); + if (auth !== undefined) { + const { acquireWebSessionCookieRecords } = await import("./web-session-cookies"); + return acquireWebSessionCookieRecords(auth, url, cookieTimeout); + } + const { acquireCookieRecords } = await import("./cookie-access"); + return acquireCookieRecords(selectionFor(signIn, cookieTimeout), url); +} + +async function defaultFetch(url: URL, init: RequestInit, timeoutMs: number): Promise { + const { pinnedHttpsFetch } = await import("./pinned-https"); + return pinnedHttpsFetch(url, init, timeoutMs); +} + +async function redactedUrl(url: URL): Promise { + const { sanitizeArtifactUrl } = await import("@hraness/wordcell/clip/persist"); + return sanitizeArtifactUrl(url.href); +} + +/** + * Wordcell `pdf` for a PDF Ghostget already downloaded: the arguments keep the + * web link, and Wordcell's source step is replaced by the local copy, so the + * note's `source_url` and the manifest's requested and final URLs are the same + * as for an anonymous download. + */ +export async function runWordcellPdfWithDownload( + pdfArguments: readonly string[], + download: PdfDownloadedSource, + environment: Readonly>, + output: PdfAuthOutput, + /** Test seam for the capture step; the source step is always the local copy. */ + capture: { readonly runPdfCapture?: typeof import("@hraness/wordcell/pdf").runPdfCapture } = {}, +): Promise { + const { runPdfCommand } = await import("@hraness/wordcell/pdf"); + return runPdfCommand(pdfArguments, environment, output, { + ...capture, + preparePdfSource: async () => ({ + inputPath: download.inputPath, + remoteSource: { requestedUrl: download.requestedUrl, finalUrl: download.finalUrl }, + dispose: () => undefined, + }), + }); +} + +/** + * Run `pdf` with sign-in options: download with the browser session, then + * import the private local copy through Wordcell with the other options. + */ +export async function runSignedInPdf( + pdfArguments: readonly string[], + dependencies: PdfAuthDependencies, +): Promise { + const { signIn, remaining } = splitPdfSignInArguments(pdfArguments); + if (signIn === null) return dependencies.runWordcellPdf(pdfArguments); + const validated = await validateWordcellArguments(remaining, dependencies.environment); + if (!/^https?:\/\//iu.test(validated.input)) { + throw usage("sign-in options only apply to web links; open a local PDF without them"); + } + let url: URL; + try { + url = new URL(validated.input); + } catch { + throw usage("the PDF link is not a valid web address"); + } + checkedHttpsUrl(url, true); + const signInResolved = resolveProfilePath(signIn); + const loadAuthRecord = dependencies.loadAuth ?? ((id: string) => { + throw new Error(`auth locator ${id} cannot be loaded`); + }); + const auth = authForSignIn(signInResolved, loadAuthRecord); + const browser = browserLabel(signInResolved, auth); + const timeoutMs = validated.timeoutMs ?? PDF_AUTH_DEFAULTS.timeoutMs; + const readCookies = dependencies.readCookies ?? defaultReadCookies; + const cookiesFile = cookieFilePath(signInResolved, auth); + if (cookiesFile !== undefined && dependencies.readCookies === undefined) { + await checkCookieFile(cookiesFile, url, signInResolved.kind === "cookies-file"); + } + dependencies.beforeCookieRead?.(); + + const directory = (dependencies.makeTemporaryDirectory ?? privateDirectory)(); + const remove = dependencies.removeDirectory + ?? ((path: string) => rmSync(path, { recursive: true, force: true })); + try { + chmodSync(directory, 0o700); + const download = await downloadSignedInPdf(url, { + readCookies: (hop, hopTimeoutMs) => readCookies(signInResolved, auth, hop, hopTimeoutMs), + fetch: dependencies.fetch ?? defaultFetch, + browser, + directory, + timeoutMs, + ...(validated.maxPdfBytes === undefined ? {} : { maxPdfBytes: validated.maxPdfBytes }), + }); + return await dependencies.runWordcellPdf(remaining, { + inputPath: download.path, + requestedUrl: await redactedUrl(url), + finalUrl: await redactedUrl(download.finalUrl), + }); + } finally { + remove(directory); + } +} + +const CHROMIUM_SOURCES: ReadonlySet = new Set(["chrome", "arc", "brave", "chromium", "edge"]); + +function hasCookieStore(directory: string): boolean { + return existsSync(join(directory, "Cookies")) || existsSync(join(directory, "Network", "Cookies")); +} + +/** + * `read --browser-profile ` and `auth add --browser-profile ` take + * a browser data folder whose sign-in lives in its `Default` profile. The + * cookie reader wants the profile folder itself, so a data folder is resolved + * to its `Default` profile, and a folder with no saved sign-in at all is a + * plain error instead of a silent anonymous download. + */ +export function resolveProfilePath(signIn: PdfSignIn, home: string = homedir()): PdfSignIn { + if (signIn.kind !== "browser" || signIn.profile === undefined || !CHROMIUM_SOURCES.has(signIn.source)) return signIn; + const profile = signIn.profile; + if (!profile.includes("/") && !profile.includes("\\")) return signIn; + const expanded = profile.startsWith("~/") + ? join(home, profile.slice(2)) + : isAbsolute(profile) ? profile : resolve(profile); + let isFile = false; + try { + isFile = statSync(expanded).isFile(); + } catch { + isFile = false; + } + if (isFile || hasCookieStore(expanded)) return { ...signIn, profile: expanded }; + const inner = join(expanded, "Default"); + if (hasCookieStore(inner)) return { ...signIn, profile: inner }; + throw usage(`the browser profile folder ${profile} has no saved sign-in; give a profile name instead (ghostget browsers lists them)`); +} + +function terminalSafe(value: string): string { + return value.replace(/[\u0000-\u0008\u000b-\u001f\u007f-\u009f]/gu, "?"); +} + +/** Plain sentences for failures the shared cookie readers report in their own words. */ +function plainCookieFailure(message: string): string | null { + if (/^authenticated API cookie files require an explicit domain or URL/u.test(message)) { + return "the cookie file must say which website each cookie belongs to; export it as a Netscape cookies.txt file"; + } + if (/^the explicitly selected cookie file contained no usable cookies/u.test(message)) { + return "the cookie file has no cookies for this website"; + } + return null; +} + +/** + * The `ghostget pdf` entry for sign-in options: runs the download, shows the + * keychain notice the same way `read` does (a JSON line for agents, a short + * notice on a terminal), and turns failures into plain sentences with a next + * step. Returns the process exit code. + */ +export async function runSignedInPdfCommand( + pdfArguments: readonly string[], + output: PdfAuthOutput, + options: { + readonly environment: Readonly>; + readonly runWordcellPdf: RunWordcellPdf; + readonly stdinIsTTY?: boolean; + readonly stderrIsTTY?: boolean; + readonly overrides?: Partial; + }, +): Promise { + const json = pdfArguments.includes("--json"); + const quiet = json || pdfArguments.includes("--quiet"); + const { cliSentence, cliStyle, renderCliError } = await import("./cli-style"); + const cookieAccess = await import("./cookie-access"); + const stderrIsTTY = options.stderrIsTTY ?? process.stderr.isTTY === true; + const style = cliStyle(options.environment, stderrIsTTY); + const fail = (code: string, message: string, next: string, exitCode: number): number => { + const sentence = cliSentence(terminalSafe(message)); + if (json) { + output.stdout(`${JSON.stringify({ ok: false, error: { code, message: sentence, next } })}\n`); + } else { + output.stderr(renderCliError(style, sentence, next)); + } + return exitCode; + }; + const { stateKeychainNoticeRecord } = await import("./keychain-notice-record"); + cookieAccess.configureCookieAccessNotice({ + environment: options.environment, + stdinIsTTY: options.stdinIsTTY ?? process.stdin.isTTY === true, + stderrIsTTY, + write: (text) => output.stderr(text), + readKey: (timeoutSeconds) => cookieAccess.terminalReadKey(timeoutSeconds), + confirm: false, + record: stateKeychainNoticeRecord(options.environment), + }); + const openFirst = "Open the link in your browser first, then run this again"; + try { + return await runSignedInPdf(pdfArguments, { + environment: options.environment, + runWordcellPdf: options.runWordcellPdf, + beforeCookieRead: () => { + if (!quiet) output.stderr("Downloading the PDF with your browser sign-in ...\n"); + }, + ...(await (async () => { + const { loadAuth } = await import("./auth"); + return { loadAuth: (id: string) => loadAuth(id, options.environment) }; + })()), + ...options.overrides, + }); + } catch (error) { + if (error instanceof cookieAccess.CookieAccessSkippedError) { + return fail("permission-skipped", error.message, "ghostget pdf --cookies-file ", 3); + } + const denied = cookieAccess.findCookieAccessError(error); + if (denied !== null) { + return fail("permission-denied", `${denied.message}. ${cookieAccess.cookieAccessRemedy(denied)}`, "ghostget pdf --help", 3); + } + if (error instanceof PdfAuthError) { + if (error.code === "usage") { + const next = error.message.startsWith("there is no connected account") ? "ghostget auth list" : "ghostget pdf --help"; + return fail("usage", error.message, next, 2); + } + if (error.code === "access") return fail("no-access", error.message, openFirst, 1); + if (error.code === "no-sign-in") return fail("no-sign-in", error.message, "ghostget browsers", 1); + return fail("download-failed", error.message, "ghostget pdf --help", 1); + } + const message = error instanceof Error ? error.message : "the signed-in PDF download failed"; + const plain = plainCookieFailure(message); + if (plain !== null) return fail("no-sign-in", plain, "ghostget pdf --help", 2); + if (/no matching cookies|no usable origin-scoped cookies|could not be read/u.test(message)) { + return fail("no-sign-in", `could not use your browser sign-in: ${message}`, openFirst, 1); + } + return fail("download-failed", `the signed-in PDF download failed: ${message}`, "ghostget pdf --help", 1); + } finally { + cookieAccess.configureCookieAccessNotice(null); + } +} diff --git a/src/performance.test.ts b/src/performance.test.ts index c6fef08d..3dd717be 100644 --- a/src/performance.test.ts +++ b/src/performance.test.ts @@ -192,7 +192,7 @@ describe("Ghostget hardening performance gates", () => { const introSource = readFileSync(join(import.meta.dir, "cli-intro.ts"), "utf8"); const styleSource = readFileSync(join(import.meta.dir, "cli-style.ts"), "utf8"); expect(runtimeImportDeclarations(cliSource)).toEqual([ - 'import { ghostgetBareUsage, ghostgetHelpRequest } from "./usage";', + 'import { ghostgetBareUsage, ghostgetHelpRequest, hasPdfSignInOptions } from "./usage";', 'import { cliStyle, renderCliError } from "./cli-style";', 'import { terminalIntro } from "./cli-intro";', 'import { GHOSTGET_VERSION } from "./version";', diff --git a/src/usage.ts b/src/usage.ts index 80326761..3d0eb613 100644 --- a/src/usage.ts +++ b/src/usage.ts @@ -367,6 +367,15 @@ Commands adapters [--json] List page-capture adapters Most commands take --root to pick the notes folder. + +Paywalled PDFs + Add the browser you use to open the paper, and Ghostget downloads it + with that sign-in: + ghostget pdf --cookie-source chrome + Also: --browser-profile , --cookie-profile , + --auth or --cookies-file . Each site gets only its own cookies. + If the site sends a sign-in page instead, open the link in your browser + first, then run it again. `; const supportHelp = `Usage: ghostget support [--json] @@ -548,6 +557,26 @@ function resolveTopic(name: string, next: string | undefined): GhostgetHelpReque return { kind: "text", text: topic.text }; } +/** + * Browser sign-in options `ghostget pdf` handles itself instead of passing to + * Wordcell. Kept here, in a module the CLI already loads, so an ordinary + * `pdf` run never imports the signed-in download code. + */ +export const PDF_SIGN_IN_OPTIONS: ReadonlySet = new Set([ + "--auth", + "--browser-profile", + "--cookie-source", + "--cookie-profile", + "--cookies-file", +]); + +/** True when `pdf` arguments (everything after `pdf`) carry a sign-in option. */ +export function hasPdfSignInOptions(pdfArguments: readonly string[]): boolean { + const separator = pdfArguments.indexOf("--"); + const scanned = separator === -1 ? pdfArguments : pdfArguments.slice(0, separator); + return scanned.some((argument) => PDF_SIGN_IN_OPTIONS.has(argument.split("=", 1)[0] ?? "")); +} + /** * Classify a help request without loading any command. Returns null when the * arguments are not a help request, so the command runs normally.