diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 5fbaf4eb9..e10e481a6 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -72,6 +72,7 @@ jobs: - '@percy/cli-exec' - '@percy/cli-snapshot' - '@percy/cli-upload' + - '@percy/cli-pdf' - '@percy/cli-build' - '@percy/cli-config' - '@percy/sdk-utils' @@ -302,6 +303,15 @@ jobs: run: yarn test:regression:config - name: Run functional discovery tests (token-free) run: yarn test:regression:functional + # Compares rendered PDF pages byte-for-byte against the linux-x64 + # goldens. PNG bytes are only reproducible for one platform + Chromium + # build (Percy pins a different Chromium snapshot per platform, and glyph + # rasterization goes through CoreText on macOS vs FreeType on Linux), so + # these goldens were generated on this runner and only this job asserts on + # them. Regenerate after a Chromium bump by re-running this step with + # UPDATE_PDF_GOLDENS=1 and committing the result. + - name: Run PDF rasterization byte tests (token-free) + run: yarn test:regression:pdf # Visual track runs last and is the ONLY step that creates a Percy build, # so the PR's single build carries all visual snapshots (no stray build # superseding it on the same commit). diff --git a/.semgrepignore b/.semgrepignore index 42057c32f..d78ff2b60 100644 --- a/.semgrepignore +++ b/.semgrepignore @@ -61,3 +61,16 @@ packages/cli-command/src/intelliStory.js # to assert no `require()` bindings leak in; the traversal roots are static # literals, no user input flows here. semgrep flags the path.join() anyway. packages/cli-command/test/noRequireBinding.test.js + +# The PDF byte-comparison regression track (test/regression/pdf-render.test.js +# and its helper) joins paths from three local sources only: `platformKey()`, +# which is `${process.platform}-${process.arch}`; a slug that slugify() has +# already reduced to [a-z0-9-] via basename(); and PDF filenames read straight +# out of the committed fixture directory with fs.readdirSync. No request or +# user input reaches these joins — the track runs offline against checked-in +# fixtures and creates no Percy build. semgrep's +# javascript.lang.security.audit.path-traversal.path-join-resolve-traversal +# rule flags the joins regardless, and inline `// nosemgrep` is not honored by +# the CI semgrep version — suppress at the file level with this rationale. +test/regression/lib/pdf-render.js +test/regression/pdf-render.test.js diff --git a/package.json b/package.json index 6626c0f8b..a0baae4e2 100644 --- a/package.json +++ b/package.json @@ -26,7 +26,8 @@ "global:unlink": "lerna exec -- yarn unlink", "test:regression": "node test/regression/regression.test.js", "test:regression:config": "node test/regression/config-validation.test.js", - "test:regression:functional": "node test/regression/functional.test.js" + "test:regression:functional": "node test/regression/functional.test.js", + "test:regression:pdf": "node test/regression/pdf-render.test.js" }, "devDependencies": { "@babel/cli": "^7.11.6", diff --git a/packages/cli-pdf/package.json b/packages/cli-pdf/package.json new file mode 100644 index 000000000..7f4341821 --- /dev/null +++ b/packages/cli-pdf/package.json @@ -0,0 +1,34 @@ +{ + "name": "@percy/cli-pdf", + "version": "1.32.10-beta.0", + "license": "MIT", + "description": "Renders PDF documents into per-page images for Percy snapshots", + "repository": { + "type": "git", + "url": "https://github.com/percy/cli", + "directory": "packages/cli-pdf" + }, + "publishConfig": { + "access": "public", + "tag": "beta" + }, + "engines": { + "node": ">=14" + }, + "files": [ + "dist" + ], + "main": "./dist/index.js", + "type": "module", + "exports": "./dist/index.js", + "scripts": { + "build": "node ../../scripts/build", + "lint": "eslint --ignore-path ../../.gitignore .", + "test": "node ../../scripts/test", + "test:coverage": "yarn test --coverage" + }, + "dependencies": { + "@percy/logger": "1.32.10-beta.0", + "pdfjs-dist": "^2.16.105" + } +} diff --git a/packages/cli-pdf/src/assets.js b/packages/cli-pdf/src/assets.js new file mode 100644 index 000000000..32e8f37e2 --- /dev/null +++ b/packages/cli-pdf/src/assets.js @@ -0,0 +1,17 @@ +import path from 'path'; +import { createRequire } from 'module'; + +const cjsRequire = createRequire(import.meta.url); + +export function pdfjsAssets() { + let root = path.dirname(cjsRequire.resolve('pdfjs-dist/package.json')); + + return { + root, + buildDir: path.join(root, 'legacy/build'), + standardFontsDir: path.join(root, 'standard_fonts'), + cmapsDir: path.join(root, 'cmaps'), + libPath: path.join(root, 'legacy/build/pdf.js'), + workerFile: 'pdf.worker.js' + }; +} diff --git a/packages/cli-pdf/src/browser-scripts.js b/packages/cli-pdf/src/browser-scripts.js new file mode 100644 index 000000000..813c33a33 --- /dev/null +++ b/packages/cli-pdf/src/browser-scripts.js @@ -0,0 +1,101 @@ +export const MIN_DIMENSION = 10; +export const MAX_DIMENSION = 2000; + +export const DEFAULT_SCALE = 2; +export const MAX_SCALE = 5; + +// Wall-clock ceiling for a single page's in-page work (open, measure, render). +// Page#eval resolves off `Runtime.callFunctionOn` with `awaitPromise: true`, +// which has no timeout of its own -- Page.TIMEOUT only covers navigation. A PDF +// that wedges pdf.js would otherwise hang the HTTP request forever while +// holding a browser page and a listening asset server. +export const PAGE_RENDER_TIMEOUT = 30000; + +export function fitScale(requestedScale, { width, height }) { + return Math.min(requestedScale, MAX_DIMENSION / width, MAX_DIMENSION / height); +} + +export function assertRasterDimensions(pageNumber, width, height) { + if (width < MIN_DIMENSION || height < MIN_DIMENSION) { + throw new Error( + `Page ${pageNumber} rasterized to ${width}x${height}px, below Percy's ` + + `${MIN_DIMENSION}px minimum. Increase \`scale\`.` + ); + } +} + +export async function openDocument(_, { origin }) { + let lib = window['pdfjs-dist/build/pdf'] || window.pdfjsLib; + + if (!lib) { + throw new Error('pdf.js did not initialise in the page'); + } + + lib.GlobalWorkerOptions.workerSrc = `${origin}/pdfjs/pdf.worker.js`; + + let doc = await lib.getDocument({ + url: `${origin}/doc.pdf`, + isEvalSupported: false, + standardFontDataUrl: `${origin}/standard_fonts/`, + cMapUrl: `${origin}/cmaps/`, + cMapPacked: true + }).promise; + + window.__percyPdf = { lib, doc }; + + return { pageCount: doc.numPages }; +} + +export async function measurePages(_, { pageNumbers }) { + let { doc } = window.__percyPdf; + let sizes = []; + + for (let pageNumber of pageNumbers) { + let page = await doc.getPage(pageNumber); + let { width, height } = page.getViewport({ scale: 1 }); + sizes.push({ pageNumber, width, height }); + page.cleanup(); + } + + return sizes; +} + +export async function renderPage(_, { pageNumber, scale }) { + let { doc } = window.__percyPdf; + let page = await doc.getPage(pageNumber); + + try { + let viewport = page.getViewport({ scale }); + let width = Math.ceil(viewport.width); + let height = Math.ceil(viewport.height); + + let canvas = document.createElement('canvas'); + canvas.width = width; + canvas.height = height; + + let context = canvas.getContext('2d'); + context.fillStyle = '#ffffff'; + context.fillRect(0, 0, width, height); + + await page.render({ canvasContext: context, viewport }).promise; + + return { + width, + height, + dataUrl: canvas.toDataURL('image/png') + }; + } finally { + page.cleanup(); + } +} + +export async function destroyDocument() { + let state = window.__percyPdf; + + if (state?.doc) { + await state.doc.destroy(); + delete window.__percyPdf; + } + + return true; +} diff --git a/packages/cli-pdf/src/index.js b/packages/cli-pdf/src/index.js new file mode 100644 index 000000000..9a9927d15 --- /dev/null +++ b/packages/cli-pdf/src/index.js @@ -0,0 +1,15 @@ +export { pdfjsAssets } from './assets.js'; +export { resolvePages, MAX_PAGES } from './pages.js'; +export { + openDocument, + measurePages, + renderPage, + destroyDocument, + fitScale, + assertRasterDimensions, + DEFAULT_SCALE, + MAX_SCALE, + PAGE_RENDER_TIMEOUT, + MIN_DIMENSION, + MAX_DIMENSION +} from './browser-scripts.js'; diff --git a/packages/cli-pdf/src/pages.js b/packages/cli-pdf/src/pages.js new file mode 100644 index 000000000..d0f8f792c --- /dev/null +++ b/packages/cli-pdf/src/pages.js @@ -0,0 +1,89 @@ +// Ceiling on how many pages one request may rasterize. Every page is held in +// memory as a PNG twice over -- once in the rasterizer's result array, once in +// the resource closure the snapshot queue keeps -- and each additionally crosses +// CDP as a base64 data URL. A 50MB PDF can carry thousands of pages, so without +// a cap a single request can exhaust the heap. Callers who genuinely want more +// can narrow with `pages` and issue several requests. +export const MAX_PAGES = 250; + +function parseSelection(value, pageCount) { + if (value == null) return range(1, pageCount); + if (typeof value === 'number') return [toPageNumber(value)]; + if (Array.isArray(value)) return value.map(toPageNumber); + + if (typeof value !== 'string') { + throw new Error(`Invalid page selection: expected a number, array or string, got ${typeof value}`); + } + + let selected = []; + + for (let part of value.split(',')) { + part = part.trim(); + if (!part) continue; + + let match = /^(\d+)\s*-\s*(\d+)?$/.exec(part); + + if (match) { + let from = toPageNumber(match[1]); + let to = match[2] == null ? pageCount : toPageNumber(match[2]); + if (to < from) throw new Error(`Invalid page range "${part}": end page is before start page`); + selected.push(...range(from, to)); + } else if (/^\d+$/.test(part)) { + selected.push(toPageNumber(part)); + } else { + throw new Error(`Invalid page selection "${part}": expected a page number or a range like "2-5"`); + } + } + + return selected; +} + +function toPageNumber(value) { + let n = Number(value); + if (!Number.isInteger(n) || n < 1) { + throw new Error(`Invalid page number "${value}": page numbers are 1-based integers`); + } + return n; +} + +function range(from, to) { + let out = []; + for (let i = from; i <= to; i++) out.push(i); + return out; +} + +export function resolvePages({ pages, excludePages } = {}, pageCount) { + if (!Number.isInteger(pageCount) || pageCount < 1) { + throw new Error(`Invalid page count: ${pageCount}`); + } + + let selected = parseSelection(pages, pageCount); + let excluded = new Set(excludePages == null ? [] : parseSelection(excludePages, pageCount)); + + let outOfRange = [...new Set(selected.filter(p => p > pageCount))]; + if (outOfRange.length) { + throw new Error( + `Requested page${outOfRange.length > 1 ? 's' : ''} ${outOfRange.join(', ')} ` + + `but the document has only ${pageCount} page${pageCount > 1 ? 's' : ''}` + ); + } + + let resolved = [...new Set(selected)] + .filter(p => !excluded.has(p)) + .sort((a, b) => a - b); + + if (!resolved.length) { + throw new Error('No pages left to snapshot after applying `pages` and `excludePages`'); + } + + if (resolved.length > MAX_PAGES) { + throw new Error( + `Requested ${resolved.length} pages but the maximum per request is ${MAX_PAGES}. ` + + 'Narrow the selection with `pages` (for example "1-100") and issue several requests.' + ); + } + + return resolved; +} + +export { parseSelection as _parseSelection }; diff --git a/packages/cli-pdf/test/.eslintrc b/packages/cli-pdf/test/.eslintrc new file mode 100644 index 000000000..d4d123f4e --- /dev/null +++ b/packages/cli-pdf/test/.eslintrc @@ -0,0 +1,6 @@ +env: + jasmine: true +rules: + import/no-extraneous-dependencies: off + no-return-assign: off + no-sequences: off diff --git a/packages/cli-pdf/test/assets.test.js b/packages/cli-pdf/test/assets.test.js new file mode 100644 index 000000000..3ccd410e2 --- /dev/null +++ b/packages/cli-pdf/test/assets.test.js @@ -0,0 +1,29 @@ +import fs from 'fs'; +import path from 'path'; +import { pdfjsAssets } from '../src/assets.js'; + +describe('@percy/cli-pdf assets', () => { + let assets = pdfjsAssets(); + + it('resolves the installed pdfjs-dist root', () => { + expect(fs.existsSync(path.join(assets.root, 'package.json'))).toBe(true); + }); + + it('points at the legacy build directory', () => { + expect(fs.existsSync(assets.buildDir)).toBe(true); + expect(fs.existsSync(path.join(assets.buildDir, 'pdf.js'))).toBe(true); + expect(fs.existsSync(path.join(assets.buildDir, assets.workerFile))).toBe(true); + }); + + it('points at the font and cmap data pdf.js fetches at runtime', () => { + expect(fs.existsSync(assets.standardFontsDir)).toBe(true); + expect(fs.existsSync(assets.cmapsDir)).toBe(true); + expect(fs.readdirSync(assets.standardFontsDir).length).toBeGreaterThan(0); + expect(fs.readdirSync(assets.cmapsDir).length).toBeGreaterThan(0); + }); + + it('exposes the injectable pdf.js library file', () => { + expect(fs.existsSync(assets.libPath)).toBe(true); + expect(fs.readFileSync(assets.libPath, 'utf-8')).toContain('getDocument'); + }); +}); diff --git a/packages/cli-pdf/test/browser-scripts.test.js b/packages/cli-pdf/test/browser-scripts.test.js new file mode 100644 index 000000000..515724d95 --- /dev/null +++ b/packages/cli-pdf/test/browser-scripts.test.js @@ -0,0 +1,243 @@ +import { + fitScale, assertRasterDimensions, + MIN_DIMENSION, MAX_DIMENSION, DEFAULT_SCALE, MAX_SCALE, + openDocument, measurePages, renderPage, destroyDocument +} from '../src/browser-scripts.js'; + +describe('@percy/cli-pdf browser scripts', () => { + describe('fitScale', () => { + it('returns the requested scale when the page fits', () => { + expect(fitScale(2, { width: 612, height: 792 })).toBe(2); + }); + + it('clamps on the constraining axis', () => { + expect(fitScale(2, { width: 612, height: 1008 })).toBeCloseTo(MAX_DIMENSION / 1008, 6); + expect(fitScale(4, { width: 1000, height: 100 })).toBeCloseTo(MAX_DIMENSION / 1000, 6); + }); + }); + + describe('assertRasterDimensions', () => { + it('accepts dimensions at or above the minimum', () => { + expect(() => assertRasterDimensions(1, MIN_DIMENSION, MIN_DIMENSION)).not.toThrow(); + }); + + it('rejects a raster below the minimum on either axis', () => { + expect(() => assertRasterDimensions(3, 4, 400)) + .toThrowError(/Page 3 rasterized to 4x400px, below Percy's 10px minimum/); + expect(() => assertRasterDimensions(1, 400, 4)) + .toThrowError(/below Percy's 10px minimum/); + }); + }); + + it('exposes the limits the rasterizer enforces', () => { + expect(MIN_DIMENSION).toBe(10); + expect(MAX_DIMENSION).toBe(2000); + expect(DEFAULT_SCALE).toBe(2); + expect(MAX_SCALE).toBe(5); + }); + + // These four run inside the browser page, reaching pdf.js and the canvas + // through `window` / `document`. Standing those globals up here exercises the + // real logic in Node -- the fetch URLs pdf.js is handed, the white pre-fill, + // and the page-handle cleanup -- rather than leaving it to an integration run. + describe('page-context scripts', () => { + let doc, pages, renderCalls, createdCanvases; + + function fakePage(pageNumber, { width = 612, height = 792, renderError } = {}) { + let page = { + pageNumber, + cleanedUp: false, + getViewport: ({ scale }) => ({ width: width * scale, height: height * scale }), + render: (opts) => { + renderCalls.push({ pageNumber, opts }); + return { promise: renderError ? Promise.reject(renderError) : Promise.resolve() }; + }, + cleanup: () => { page.cleanedUp = true; } + }; + return page; + } + + function stubPageGlobals({ numPages = 3, pageOptions = {}, lib } = {}) { + pages = new Map(); + renderCalls = []; + createdCanvases = []; + + doc = { + numPages, + destroyed: false, + getPage: async (n) => { + let page = fakePage(n, pageOptions[n] || {}); + pages.set(n, page); + return page; + }, + destroy: async () => { doc.destroyed = true; } + }; + + global.window = lib === null ? {} : { 'pdfjs-dist/build/pdf': lib || stubLib() }; + global.document = { + createElement: (tag) => { + let canvas = { + tag, + width: 0, + height: 0, + fills: [], + getContext: () => ({ + set fillStyle(v) { canvas.fillStyle = v; }, + get fillStyle() { return canvas.fillStyle; }, + fillRect: (...args) => canvas.fills.push(args) + }), + toDataURL: (type) => `data:${type};base64,UE5H` + }; + createdCanvases.push(canvas); + return canvas; + } + }; + } + + function stubLib() { + return { + GlobalWorkerOptions: {}, + getDocumentCalls: [], + getDocument(options) { + this.getDocumentCalls.push(options); + return { promise: Promise.resolve(doc) }; + } + }; + } + + afterEach(() => { + delete global.window; + delete global.document; + }); + + describe('openDocument', () => { + it('points pdf.js at the served worker, document, fonts and cmaps', async () => { + stubPageGlobals({ numPages: 4 }); + let lib = global.window['pdfjs-dist/build/pdf']; + + await expectAsync(openDocument(null, { origin: 'http://localhost:9999' })) + .toBeResolvedTo({ pageCount: 4 }); + + expect(lib.GlobalWorkerOptions.workerSrc) + .toBe('http://localhost:9999/pdfjs/pdf.worker.js'); + + let [options] = lib.getDocumentCalls; + expect(options.url).toBe('http://localhost:9999/doc.pdf'); + expect(options.standardFontDataUrl).toBe('http://localhost:9999/standard_fonts/'); + expect(options.cMapUrl).toBe('http://localhost:9999/cmaps/'); + expect(options.cMapPacked).toBe(true); + // the PDF is untrusted input; this must never be enabled + expect(options.isEvalSupported).toBe(false); + }); + + it('stashes the handle for the later scripts', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:1' }); + + expect(global.window.__percyPdf.doc).toBe(doc); + }); + + it('falls back to the window.pdfjsLib global', async () => { + stubPageGlobals({ lib: null }); + let lib = stubLib(); + global.window.pdfjsLib = lib; + + await expectAsync(openDocument(null, { origin: 'http://localhost:2' })) + .toBeResolvedTo({ pageCount: 3 }); + expect(lib.getDocumentCalls.length).toBe(1); + }); + + it('throws when pdf.js did not initialise', async () => { + stubPageGlobals({ lib: null }); + + await expectAsync(openDocument(null, { origin: 'http://localhost:3' })) + .toBeRejectedWithError('pdf.js did not initialise in the page'); + }); + }); + + describe('measurePages', () => { + it('returns each page unscaled and releases the handles', async () => { + stubPageGlobals({ pageOptions: { 2: { width: 200, height: 400 } } }); + await openDocument(null, { origin: 'http://localhost:4' }); + + await expectAsync(measurePages(null, { pageNumbers: [1, 2] })).toBeResolvedTo([ + { pageNumber: 1, width: 612, height: 792 }, + { pageNumber: 2, width: 200, height: 400 } + ]); + + expect(pages.get(1).cleanedUp).toBe(true); + expect(pages.get(2).cleanedUp).toBe(true); + }); + }); + + describe('renderPage', () => { + it('renders at the given scale and returns a PNG data URL', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:5' }); + + let out = await renderPage(null, { pageNumber: 1, scale: 2 }); + + expect(out.width).toBe(1224); + expect(out.height).toBe(1584); + expect(out.dataUrl).toBe('data:image/png;base64,UE5H'); + expect(createdCanvases[0].tag).toBe('canvas'); + }); + + it('rounds fractional viewports up', async () => { + stubPageGlobals({ pageOptions: { 1: { width: 100.2, height: 100.6 } } }); + await openDocument(null, { origin: 'http://localhost:6' }); + + let out = await renderPage(null, { pageNumber: 1, scale: 1 }); + expect(out.width).toBe(101); + expect(out.height).toBe(101); + }); + + it('pre-fills the canvas white', async () => { + // PDF pages have no intrinsic background; without this, transparent + // regions rasterize to alpha-0 black and diff against anything. + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:7' }); + await renderPage(null, { pageNumber: 1, scale: 1 }); + + expect(createdCanvases[0].fillStyle).toBe('#ffffff'); + expect(createdCanvases[0].fills).toEqual([[0, 0, 612, 792]]); + }); + + it('passes the canvas context and viewport to pdf.js', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:8' }); + await renderPage(null, { pageNumber: 3, scale: 1 }); + + expect(renderCalls.length).toBe(1); + expect(renderCalls[0].pageNumber).toBe(3); + expect(renderCalls[0].opts.canvasContext).toBeDefined(); + expect(renderCalls[0].opts.viewport).toEqual({ width: 612, height: 792 }); + }); + + it('releases the page handle even when rendering fails', async () => { + stubPageGlobals({ pageOptions: { 1: { renderError: new Error('render blew up') } } }); + await openDocument(null, { origin: 'http://localhost:9' }); + + await expectAsync(renderPage(null, { pageNumber: 1, scale: 1 })) + .toBeRejectedWithError('render blew up'); + expect(pages.get(1).cleanedUp).toBe(true); + }); + }); + + describe('destroyDocument', () => { + it('destroys the document and clears the handle', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:10' }); + + await expectAsync(destroyDocument()).toBeResolvedTo(true); + expect(doc.destroyed).toBe(true); + expect(global.window.__percyPdf).toBeUndefined(); + }); + + it('is a no-op when no document is open', async () => { + global.window = {}; + await expectAsync(destroyDocument()).toBeResolvedTo(true); + }); + }); + }); +}); diff --git a/packages/cli-pdf/test/fixture.js b/packages/cli-pdf/test/fixture.js new file mode 100644 index 000000000..6f70f2e10 --- /dev/null +++ b/packages/cli-pdf/test/fixture.js @@ -0,0 +1,42 @@ +export function buildPdf({ pageCount = 1, width = 200, height = 300 } = {}) { + let objects = []; + let pageIds = []; + + for (let i = 0; i < pageCount; i++) { + pageIds.push(3 + i * 2); + } + + objects[1] = '<< /Type /Catalog /Pages 2 0 R >>'; + objects[2] = `<< /Type /Pages /Kids [${pageIds.map(id => `${id} 0 R`).join(' ')}] /Count ${pageCount} >>`; + + for (let i = 0; i < pageCount; i++) { + let pageId = pageIds[i]; + let contentId = pageId + 1; + let inset = 10 + i * 15; + let stream = `${inset} ${inset} ${width - inset * 2} ${height - inset * 2} re f`; + + objects[pageId] = + `<< /Type /Page /Parent 2 0 R /MediaBox [0 0 ${width} ${height}] ` + + `/Contents ${contentId} 0 R /Resources << >> >>`; + objects[contentId] = `<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`; + } + + let out = '%PDF-1.4\n'; + let offsets = []; + + for (let i = 1; i < objects.length; i++) { + offsets[i] = out.length; + out += `${i} 0 obj\n${objects[i]}\nendobj\n`; + } + + let xrefStart = out.length; + out += `xref\n0 ${objects.length}\n0000000000 65535 f \n`; + for (let i = 1; i < objects.length; i++) { + out += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + } + out += `trailer\n<< /Size ${objects.length} /Root 1 0 R >>\nstartxref\n${xrefStart}\n%%EOF\n`; + + return Buffer.from(out, 'latin1'); +} + +export const NOT_A_PDF = Buffer.from('this is definitely not a pdf', 'utf8'); diff --git a/packages/cli-pdf/test/pages.test.js b/packages/cli-pdf/test/pages.test.js new file mode 100644 index 000000000..e8c74ef4f --- /dev/null +++ b/packages/cli-pdf/test/pages.test.js @@ -0,0 +1,120 @@ +import { resolvePages, MAX_PAGES } from '../src/pages.js'; + +describe('@percy/cli-pdf page selection', () => { + it('selects every page when nothing is specified', () => { + expect(resolvePages({}, 4)).toEqual([1, 2, 3, 4]); + expect(resolvePages(undefined, 2)).toEqual([1, 2]); + }); + + it('accepts a single page number', () => { + expect(resolvePages({ pages: 3 }, 5)).toEqual([3]); + }); + + it('accepts an array of pages, sorted and de-duplicated', () => { + expect(resolvePages({ pages: [3, 1, 1, 2] }, 5)).toEqual([1, 2, 3]); + }); + + it('accepts closed ranges', () => { + expect(resolvePages({ pages: '2-4' }, 6)).toEqual([2, 3, 4]); + }); + + it('accepts open-ended ranges', () => { + expect(resolvePages({ pages: '3-' }, 5)).toEqual([3, 4, 5]); + }); + + it('accepts mixed lists of pages and ranges', () => { + expect(resolvePages({ pages: '1,3-5,8' }, 10)).toEqual([1, 3, 4, 5, 8]); + }); + + it('tolerates whitespace in string selections', () => { + expect(resolvePages({ pages: ' 1 , 3 - 4 ' }, 5)).toEqual([1, 3, 4]); + }); + + it('skips empty segments in a string selection', () => { + expect(resolvePages({ pages: '1,,3' }, 5)).toEqual([1, 3]); + expect(resolvePages({ pages: '2,' }, 5)).toEqual([2]); + }); + + it('throws when a string selection resolves to nothing', () => { + expect(() => resolvePages({ pages: ',' }, 5)) + .toThrowError('No pages left to snapshot after applying `pages` and `excludePages`'); + }); + + it('applies excludePages after pages', () => { + expect(resolvePages({ pages: '1-5', excludePages: [2, 4] }, 5)).toEqual([1, 3, 5]); + }); + + it('can exclude page 1', () => { + expect(resolvePages({ excludePages: [1] }, 3)).toEqual([2, 3]); + }); + + it('can exclude the second-to-last page', () => { + expect(resolvePages({ excludePages: [3] }, 4)).toEqual([1, 2, 4]); + }); + + it('ignores excluded pages that were never selected', () => { + expect(resolvePages({ pages: [1, 2], excludePages: [5] }, 5)).toEqual([1, 2]); + }); + + it('throws when a requested page is out of range', () => { + expect(() => resolvePages({ pages: '9' }, 5)) + .toThrowError('Requested page 9 but the document has only 5 pages'); + expect(() => resolvePages({ pages: [7, 8] }, 5)) + .toThrowError('Requested pages 7, 8 but the document has only 5 pages'); + }); + + it('uses the singular in the out-of-range message for a 1-page document', () => { + expect(() => resolvePages({ pages: '2' }, 1)) + .toThrowError('Requested page 2 but the document has only 1 page'); + }); + + it('throws on a malformed selection', () => { + expect(() => resolvePages({ pages: 'abc' }, 5)) + .toThrowError(/Invalid page selection "abc"/); + expect(() => resolvePages({ pages: '5-2' }, 5)) + .toThrowError('Invalid page range "5-2": end page is before start page'); + expect(() => resolvePages({ pages: 0 }, 5)) + .toThrowError(/Invalid page number "0"/); + expect(() => resolvePages({ pages: 1.5 }, 5)) + .toThrowError(/Invalid page number "1.5"/); + expect(() => resolvePages({ pages: {} }, 5)) + .toThrowError(/expected a number, array or string, got object/); + }); + + it('throws when every page has been excluded', () => { + expect(() => resolvePages({ excludePages: '1-3' }, 3)) + .toThrowError('No pages left to snapshot after applying `pages` and `excludePages`'); + }); + + it('throws on an invalid page count', () => { + expect(() => resolvePages({}, 0)).toThrowError('Invalid page count: 0'); + }); + + describe('the per-request page cap', () => { + // Each page is held in memory as a PNG twice over and crosses CDP as a + // base64 data URL, so an uncapped document can exhaust the heap. + it('allows exactly MAX_PAGES', () => { + expect(resolvePages({}, MAX_PAGES).length).toBe(MAX_PAGES); + }); + + it('throws one page past MAX_PAGES', () => { + expect(() => resolvePages({}, MAX_PAGES + 1)).toThrowError( + new RegExp(`Requested ${MAX_PAGES + 1} pages but the maximum per request is ${MAX_PAGES}`)); + }); + + it('counts what is actually selected, not the document length', () => { + // A long document is fine so long as the request narrows it. + expect(resolvePages({ pages: '1-10' }, MAX_PAGES * 10)).toEqual( + [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]); + }); + + it('counts after exclusions are applied', () => { + expect(resolvePages({ excludePages: '1' }, MAX_PAGES + 1).length).toBe(MAX_PAGES); + }); + + it('suggests how to get under the cap', () => { + expect(() => resolvePages({}, MAX_PAGES + 1)) + .toThrowError(/Narrow the selection with `pages`/); + }); + }); +}); diff --git a/packages/cli-upload/src/utils.js b/packages/cli-upload/src/utils.js index 6ddb27d47..6a63e3485 100644 --- a/packages/cli-upload/src/utils.js +++ b/packages/cli-upload/src/utils.js @@ -1,16 +1,17 @@ import fs from 'fs'; import { - createResource, - createRootResource + createImageSnapshotResources } from '@percy/cli-command/utils'; export { yieldAll } from '@percy/cli-command/utils'; -// Returns root resource and image resource objects based on image properties. The root resource is -// a generated DOM designed to display an image at it's native size without margins or padding. +// Returns root resource and image resource objects based on image properties. The wrapper DOM +// itself lives in @percy/core's utils as createImageSnapshotResources: its exact shape is what +// percy-api matches to extract the image and skip the renderer, and the PDF snapshot path builds +// the same pair, so there must only be one definition of it. export async function getImageResources({ name, type, @@ -19,29 +20,13 @@ export async function getImageResources({ relativePath, absolutePath }) { - let rootUrl = `http://local/${encodeURIComponent(name)}`; - let imageUrl = `http://local/${encodeURIComponent(relativePath)}`; - let content = await fs.promises.readFile(absolutePath); - let mimetype = `image/${type}`; - - return [ - createRootResource(rootUrl, ` - - -
- -