diff --git a/.changeset/url-documents-any-size.md b/.changeset/url-documents-any-size.md new file mode 100644 index 00000000..d1f483a1 --- /dev/null +++ b/.changeset/url-documents-any-size.md @@ -0,0 +1,7 @@ +--- +'@simplepdf/embed': patch +'@simplepdf/react-embed-pdf': patch +'@simplepdf/web-embed-pdf': patch +--- + +**PDFs over 50 MB now load from a URL like any other.** They used to skip the fetch from your page and go to the editor by URL, which failed for documents behind your login or only readable from your site. Documents of every size are now fetched from your page first, as smaller ones always were. diff --git a/embed/README.md b/embed/README.md index 4787cbf9..3b5f0d1f 100644 --- a/embed/README.md +++ b/embed/README.md @@ -151,7 +151,7 @@ createEmbed({ target, companyIdentifier: 'acme', document: { dataUrl: 'data:appl createEmbed({ target, companyIdentifier: 'acme', document: { file: pdfFileOrBlob } }) ``` -- **`url`**: any `http(s)` URL. Fetched from your page first (50 MB cap); on CORS / size / network failure it falls back to the editor's `?open` loader, so CORS-restricted public URLs still load. `user:pass@` credentials are allowed (they route via `?open`). A **SimplePDF documents URL** on your base-domain family (e.g. `https://acme.simplepdf.com/documents/?prefill=`) is navigated to directly, so the editor loads + prefills the stored document itself (your `context` is carried through). +- **`url`**: any `http(s)` URL. Fetched from your page first; on a CORS, HTTP or network failure, or a file too large for the browser to pass on, it falls back to the editor's `?open` loader, so CORS-restricted public URLs still load. `user:pass@` credentials are allowed (they route via `?open`). A **SimplePDF documents URL** on your base-domain family (e.g. `https://acme.simplepdf.com/documents/?prefill=`) is navigated to directly, so the editor loads + prefills the stored document itself (your `context` is carried through). - **`file`**: a `File` (e.g. from ``) or any `Blob`. Converted for you, no `FileReader`. - **`dataUrl`**: a `data:` URL string. diff --git a/embed/src/mount.ts b/embed/src/mount.ts index bf19eeb4..3f713620 100644 --- a/embed/src/mount.ts +++ b/embed/src/mount.ts @@ -25,7 +25,6 @@ export class EmbedConfigError extends Error { const DNS_LABEL = /^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/ const DEFAULT_BASE_DOMAIN = 'simplepdf.com' -const DOCUMENT_SIZE_CAP_BYTES = 50 * 1024 * 1024 // Local dev domains (a *.nil checkout host or localhost) are served over http; // every real domain is https. @@ -361,40 +360,22 @@ export const buildEditorURL = ({ return url.href } +// Past the engine's maximum string length (~384 MiB of input in Chromium), FileReader +// fires `load` with an empty result instead of `error`. const blobToDataUrl = (blob: Blob): Promise => new Promise((resolve, reject) => { const reader = new FileReader() - reader.onload = () => resolve(typeof reader.result === 'string' ? reader.result : '') + reader.onload = () => { + if (typeof reader.result !== 'string' || reader.result === '') { + reject(new Error('The browser could not encode the document as a data URL')) + return + } + resolve(reader.result) + } reader.onerror = () => reject(reader.error ?? new Error('Failed to read document')) reader.readAsDataURL(blob) }) -// Read the body stream with a running byte cap so an over-sized (or -// Content-Length-less) response is aborted mid-stream instead of buffered whole. -const readStreamCapped = async ( - body: ReadableStream, - capBytes: number, -): Promise => { - const reader = body.getReader() - const chunks: BlobPart[] = [] - // Streaming accumulation: the running counter must mutate as chunks arrive. - let receivedBytes = 0 - for (;;) { - const { done, value } = await reader.read() - if (done) { - return chunks - } - receivedBytes += value.byteLength - if (receivedBytes > capBytes) { - await reader.cancel() - return null - } - // Copy into a fresh ArrayBuffer-backed view so it is a valid BlobPart - // (the reader yields ArrayBufferLike-backed chunks). - chunks.push(new Uint8Array(value)) - } -} - const fetchDocumentAsDataUrl = async (url: string, signal: AbortSignal): Promise => { try { const response = await fetch(url, { method: 'GET', credentials: 'same-origin', signal }) @@ -402,25 +383,8 @@ const fetchDocumentAsDataUrl = async (url: string, signal: AbortSignal): Promise await response.body?.cancel().catch(() => {}) return null } - const contentLength = response.headers.get('content-length') - if (contentLength !== null && Number(contentLength) > DOCUMENT_SIZE_CAP_BYTES) { - await response.body?.cancel().catch(() => {}) - return null - } - const body = response.body - if (body === null) { - // No readable stream (e.g. an opaque response): we can't enforce the cap - // without buffering the whole body, so decline and let the editor's ?open - // loader fetch it instead. - return null - } - // readStreamCapped cancels the reader (stopping the download) if the cap is hit. - const chunks = await readStreamCapped(body, DOCUMENT_SIZE_CAP_BYTES) - if (chunks === null) { - return null - } const contentType = response.headers.get('content-type') ?? 'application/pdf' - return await blobToDataUrl(new Blob(chunks, { type: contentType })) + return await blobToDataUrl(new Blob([await response.arrayBuffer()], { type: contentType })) } catch { return null } @@ -475,8 +439,8 @@ const makeReadinessGate = (): { } // Loads `document` into the editor once it is ready. `onHostFetchFail` is the -// create path's ?open fallback (re-navigate the iframe we own); the attach path -// passes none, because it must never touch an iframe the consumer rendered. +// create path's ?open fallback for a url document (re-navigate the iframe we own); +// the attach path passes none, because it must never touch an iframe the consumer rendered. const loadDocumentWhenReady = (params: { embed: Embed embedDocument: EmbedDocument @@ -500,7 +464,8 @@ const loadDocumentWhenReady = (params: { } safeLogger.error('load_document_failed', { code: 'unexpected:unknown', - message: 'could not resolve the document to a data URL; ?open fallback is unavailable on an attached iframe', + message: + 'could not resolve the document to a data URL; the ?open fallback only applies to a url document in an iframe createEmbed created', }) return } @@ -670,25 +635,27 @@ const mountIntoContainer = ( // A documents URL is loaded by the navigation above; only the PDF / data-URL / // file arms need the async host-fetch + LOAD_DOCUMENT (with the ?open fallback). if (mountDocument !== undefined && documentsUrl === null) { + const documentUrl = 'url' in mountDocument ? mountDocument.url : null loadDocumentWhenReady({ embed, embedDocument: mountDocument, signal: documentFetchController.signal, safeLogger, whenReady: gate.whenReady, - // Host-fetch failed (CORS/size/network): re-navigate the iframe we own - // through the editor's ?open loader, which fetches the URL inside the editor. - onHostFetchFail: () => { - if ('url' in mountDocument) { - iframe.src = buildEditorURL({ - editorOrigin, - locale, - encodedContext, - hasDocumentUrl: false, - openFallbackUrl: mountDocument.url, - }) - } - }, + // Host-fetch failed (CORS, HTTP error, network, or too large to encode): re-navigate + // the iframe we own through the editor's ?open loader, which fetches the URL inside the editor. + onHostFetchFail: + documentUrl === null + ? undefined + : () => { + iframe.src = buildEditorURL({ + editorOrigin, + locale, + encodedContext, + hasDocumentUrl: false, + openFallbackUrl: documentUrl, + }) + }, }) } diff --git a/embed/test/mount.test.ts b/embed/test/mount.test.ts index 8770d2ea..79111758 100644 --- a/embed/test/mount.test.ts +++ b/embed/test/mount.test.ts @@ -105,6 +105,9 @@ describe(createEmbed.name, () => { } mounted.length = 0 document.body.innerHTML = '' + vi.restoreAllMocks() + vi.unstubAllGlobals() + vi.useRealTimers() }) // Defaults companyIdentifier so the document/target tests stay terse; pass companyIdentifier to override. @@ -114,6 +117,45 @@ describe(createEmbed.name, () => { return embed } + // Mounts under #root and answers the first readiness probe, so readiness is reached with NO + // EDITOR_READY / DOCUMENT_LOADED event (only the probe). Returns every message posted to the editor. + const mountUntilReady = async ( + args: Omit[0], 'companyIdentifier' | 'target'>, + ): Promise<{ iframe: HTMLIFrameElement; posted: { type: string; request_id: string }[] }> => { + vi.useFakeTimers() + document.body.innerHTML = '
' + mount({ target: '#root', ...args }) + const iframe = document.querySelector('#root iframe') + if (!(iframe instanceof HTMLIFrameElement) || iframe.contentWindow === null) { + throw new Error('expected an iframe with a contentWindow') + } + const contentWindow = iframe.contentWindow + const posted: { type: string; request_id: string }[] = [] + vi.spyOn(contentWindow, 'postMessage').mockImplementation((message: unknown) => { + if (typeof message === 'string') { + posted.push(JSON.parse(message)) + } + }) + await vi.advanceTimersByTimeAsync(500) + const probe = posted.find((message) => message.type === 'GET_FIELDS') + if (probe === undefined) { + throw new Error('expected a readiness probe') + } + window.dispatchEvent( + new MessageEvent('message', { + data: JSON.stringify({ + type: 'REQUEST_RESULT', + data: { request_id: probe.request_id, result: { success: true, data: { fields: [] } } }, + }), + origin: 'https://acme.simplepdf.com', + source: contentWindow, + }), + ) + // Flush the gate's microtask + the async data-URL resolution. + await vi.advanceTimersByTimeAsync(0) + return { iframe, posted } + } + it('throws EmbedConfigError when the target selector matches nothing', () => { expect(() => mount({ target: '#missing' })).toThrow(EmbedConfigError) }) @@ -289,42 +331,25 @@ describe(createEmbed.name, () => { }) it('loads the document once readiness is reached via the probe (gate posts LOAD_DOCUMENT)', async () => { - vi.useFakeTimers() - document.body.innerHTML = '
' - mount({ target: '#root', document: { dataUrl: 'data:application/pdf;base64,AAAA' } }) - const iframe = document.querySelector('#root iframe') - if (!(iframe instanceof HTMLIFrameElement) || iframe.contentWindow === null) { - throw new Error('expected an iframe with a contentWindow') - } - const contentWindow = iframe.contentWindow - const posted: { type: string; request_id: string }[] = [] - vi.spyOn(contentWindow, 'postMessage').mockImplementation((message: unknown) => { - if (typeof message === 'string') { - posted.push(JSON.parse(message)) - } - }) - // Advance to a probe tick: readiness reached with NO EDITOR_READY / DOCUMENT_LOADED - // event — only the probe. The readiness gate must still fire the deferred load. - await vi.advanceTimersByTimeAsync(500) - const probe = posted.find((message) => message.type === 'GET_FIELDS') - if (probe === undefined) { - throw new Error('expected a readiness probe') - } - window.dispatchEvent( - new MessageEvent('message', { - data: JSON.stringify({ - type: 'REQUEST_RESULT', - data: { request_id: probe.request_id, result: { success: true, data: { fields: [] } } }, - }), - origin: 'https://acme.simplepdf.com', - source: contentWindow, - }), + const { posted } = await mountUntilReady({ document: { dataUrl: 'data:application/pdf;base64,AAAA' } }) + expect(posted.some((message) => message.type === 'LOAD_DOCUMENT')).toBe(true) + }) + + it('hands a url document to the editor whatever its size, the editor being the one owner of a size ceiling', async () => { + const declaredBytes = 500 * 1024 * 1024 + vi.stubGlobal( + 'fetch', + vi.fn(() => + Promise.resolve( + new Response(new Uint8Array([37, 80, 68, 70]), { + headers: { 'content-length': String(declaredBytes), 'content-type': 'application/pdf' }, + }), + ), + ), ) - // Flush the gate's microtask + the async data-URL resolution. - await vi.advanceTimersByTimeAsync(0) + const { iframe, posted } = await mountUntilReady({ document: { url: 'https://example.com/large.pdf' } }) expect(posted.some((message) => message.type === 'LOAD_DOCUMENT')).toBe(true) - vi.restoreAllMocks() - vi.useRealTimers() + expect(recoverDocumentUrlFromOpen(iframe.src)).toBeNull() }) it('attaches to an existing