diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 5fbaf4eb9..e10e481a6 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -72,6 +72,7 @@ jobs: - '@percy/cli-exec' - '@percy/cli-snapshot' - '@percy/cli-upload' + - '@percy/cli-pdf' - '@percy/cli-build' - '@percy/cli-config' - '@percy/sdk-utils' @@ -302,6 +303,15 @@ jobs: run: yarn test:regression:config - name: Run functional discovery tests (token-free) run: yarn test:regression:functional + # Compares rendered PDF pages byte-for-byte against the linux-x64 + # goldens. PNG bytes are only reproducible for one platform + Chromium + # build (Percy pins a different Chromium snapshot per platform, and glyph + # rasterization goes through CoreText on macOS vs FreeType on Linux), so + # these goldens were generated on this runner and only this job asserts on + # them. Regenerate after a Chromium bump by re-running this step with + # UPDATE_PDF_GOLDENS=1 and committing the result. + - name: Run PDF rasterization byte tests (token-free) + run: yarn test:regression:pdf # Visual track runs last and is the ONLY step that creates a Percy build, # so the PR's single build carries all visual snapshots (no stray build # superseding it on the same commit). diff --git a/.semgrepignore b/.semgrepignore index 42057c32f..d78ff2b60 100644 --- a/.semgrepignore +++ b/.semgrepignore @@ -61,3 +61,16 @@ packages/cli-command/src/intelliStory.js # to assert no `require()` bindings leak in; the traversal roots are static # literals, no user input flows here. semgrep flags the path.join() anyway. packages/cli-command/test/noRequireBinding.test.js + +# The PDF byte-comparison regression track (test/regression/pdf-render.test.js +# and its helper) joins paths from three local sources only: `platformKey()`, +# which is `${process.platform}-${process.arch}`; a slug that slugify() has +# already reduced to [a-z0-9-] via basename(); and PDF filenames read straight +# out of the committed fixture directory with fs.readdirSync. No request or +# user input reaches these joins — the track runs offline against checked-in +# fixtures and creates no Percy build. semgrep's +# javascript.lang.security.audit.path-traversal.path-join-resolve-traversal +# rule flags the joins regardless, and inline `// nosemgrep` is not honored by +# the CI semgrep version — suppress at the file level with this rationale. +test/regression/lib/pdf-render.js +test/regression/pdf-render.test.js diff --git a/package.json b/package.json index 6626c0f8b..a0baae4e2 100644 --- a/package.json +++ b/package.json @@ -26,7 +26,8 @@ "global:unlink": "lerna exec -- yarn unlink", "test:regression": "node test/regression/regression.test.js", "test:regression:config": "node test/regression/config-validation.test.js", - "test:regression:functional": "node test/regression/functional.test.js" + "test:regression:functional": "node test/regression/functional.test.js", + "test:regression:pdf": "node test/regression/pdf-render.test.js" }, "devDependencies": { "@babel/cli": "^7.11.6", diff --git a/packages/cli-pdf/package.json b/packages/cli-pdf/package.json new file mode 100644 index 000000000..7f4341821 --- /dev/null +++ b/packages/cli-pdf/package.json @@ -0,0 +1,34 @@ +{ + "name": "@percy/cli-pdf", + "version": "1.32.10-beta.0", + "license": "MIT", + "description": "Renders PDF documents into per-page images for Percy snapshots", + "repository": { + "type": "git", + "url": "https://github.com/percy/cli", + "directory": "packages/cli-pdf" + }, + "publishConfig": { + "access": "public", + "tag": "beta" + }, + "engines": { + "node": ">=14" + }, + "files": [ + "dist" + ], + "main": "./dist/index.js", + "type": "module", + "exports": "./dist/index.js", + "scripts": { + "build": "node ../../scripts/build", + "lint": "eslint --ignore-path ../../.gitignore .", + "test": "node ../../scripts/test", + "test:coverage": "yarn test --coverage" + }, + "dependencies": { + "@percy/logger": "1.32.10-beta.0", + "pdfjs-dist": "^2.16.105" + } +} diff --git a/packages/cli-pdf/src/assets.js b/packages/cli-pdf/src/assets.js new file mode 100644 index 000000000..32e8f37e2 --- /dev/null +++ b/packages/cli-pdf/src/assets.js @@ -0,0 +1,17 @@ +import path from 'path'; +import { createRequire } from 'module'; + +const cjsRequire = createRequire(import.meta.url); + +export function pdfjsAssets() { + let root = path.dirname(cjsRequire.resolve('pdfjs-dist/package.json')); + + return { + root, + buildDir: path.join(root, 'legacy/build'), + standardFontsDir: path.join(root, 'standard_fonts'), + cmapsDir: path.join(root, 'cmaps'), + libPath: path.join(root, 'legacy/build/pdf.js'), + workerFile: 'pdf.worker.js' + }; +} diff --git a/packages/cli-pdf/src/browser-scripts.js b/packages/cli-pdf/src/browser-scripts.js new file mode 100644 index 000000000..813c33a33 --- /dev/null +++ b/packages/cli-pdf/src/browser-scripts.js @@ -0,0 +1,101 @@ +export const MIN_DIMENSION = 10; +export const MAX_DIMENSION = 2000; + +export const DEFAULT_SCALE = 2; +export const MAX_SCALE = 5; + +// Wall-clock ceiling for a single page's in-page work (open, measure, render). +// Page#eval resolves off `Runtime.callFunctionOn` with `awaitPromise: true`, +// which has no timeout of its own -- Page.TIMEOUT only covers navigation. A PDF +// that wedges pdf.js would otherwise hang the HTTP request forever while +// holding a browser page and a listening asset server. +export const PAGE_RENDER_TIMEOUT = 30000; + +export function fitScale(requestedScale, { width, height }) { + return Math.min(requestedScale, MAX_DIMENSION / width, MAX_DIMENSION / height); +} + +export function assertRasterDimensions(pageNumber, width, height) { + if (width < MIN_DIMENSION || height < MIN_DIMENSION) { + throw new Error( + `Page ${pageNumber} rasterized to ${width}x${height}px, below Percy's ` + + `${MIN_DIMENSION}px minimum. Increase \`scale\`.` + ); + } +} + +export async function openDocument(_, { origin }) { + let lib = window['pdfjs-dist/build/pdf'] || window.pdfjsLib; + + if (!lib) { + throw new Error('pdf.js did not initialise in the page'); + } + + lib.GlobalWorkerOptions.workerSrc = `${origin}/pdfjs/pdf.worker.js`; + + let doc = await lib.getDocument({ + url: `${origin}/doc.pdf`, + isEvalSupported: false, + standardFontDataUrl: `${origin}/standard_fonts/`, + cMapUrl: `${origin}/cmaps/`, + cMapPacked: true + }).promise; + + window.__percyPdf = { lib, doc }; + + return { pageCount: doc.numPages }; +} + +export async function measurePages(_, { pageNumbers }) { + let { doc } = window.__percyPdf; + let sizes = []; + + for (let pageNumber of pageNumbers) { + let page = await doc.getPage(pageNumber); + let { width, height } = page.getViewport({ scale: 1 }); + sizes.push({ pageNumber, width, height }); + page.cleanup(); + } + + return sizes; +} + +export async function renderPage(_, { pageNumber, scale }) { + let { doc } = window.__percyPdf; + let page = await doc.getPage(pageNumber); + + try { + let viewport = page.getViewport({ scale }); + let width = Math.ceil(viewport.width); + let height = Math.ceil(viewport.height); + + let canvas = document.createElement('canvas'); + canvas.width = width; + canvas.height = height; + + let context = canvas.getContext('2d'); + context.fillStyle = '#ffffff'; + context.fillRect(0, 0, width, height); + + await page.render({ canvasContext: context, viewport }).promise; + + return { + width, + height, + dataUrl: canvas.toDataURL('image/png') + }; + } finally { + page.cleanup(); + } +} + +export async function destroyDocument() { + let state = window.__percyPdf; + + if (state?.doc) { + await state.doc.destroy(); + delete window.__percyPdf; + } + + return true; +} diff --git a/packages/cli-pdf/src/index.js b/packages/cli-pdf/src/index.js new file mode 100644 index 000000000..9a9927d15 --- /dev/null +++ b/packages/cli-pdf/src/index.js @@ -0,0 +1,15 @@ +export { pdfjsAssets } from './assets.js'; +export { resolvePages, MAX_PAGES } from './pages.js'; +export { + openDocument, + measurePages, + renderPage, + destroyDocument, + fitScale, + assertRasterDimensions, + DEFAULT_SCALE, + MAX_SCALE, + PAGE_RENDER_TIMEOUT, + MIN_DIMENSION, + MAX_DIMENSION +} from './browser-scripts.js'; diff --git a/packages/cli-pdf/src/pages.js b/packages/cli-pdf/src/pages.js new file mode 100644 index 000000000..d0f8f792c --- /dev/null +++ b/packages/cli-pdf/src/pages.js @@ -0,0 +1,89 @@ +// Ceiling on how many pages one request may rasterize. Every page is held in +// memory as a PNG twice over -- once in the rasterizer's result array, once in +// the resource closure the snapshot queue keeps -- and each additionally crosses +// CDP as a base64 data URL. A 50MB PDF can carry thousands of pages, so without +// a cap a single request can exhaust the heap. Callers who genuinely want more +// can narrow with `pages` and issue several requests. +export const MAX_PAGES = 250; + +function parseSelection(value, pageCount) { + if (value == null) return range(1, pageCount); + if (typeof value === 'number') return [toPageNumber(value)]; + if (Array.isArray(value)) return value.map(toPageNumber); + + if (typeof value !== 'string') { + throw new Error(`Invalid page selection: expected a number, array or string, got ${typeof value}`); + } + + let selected = []; + + for (let part of value.split(',')) { + part = part.trim(); + if (!part) continue; + + let match = /^(\d+)\s*-\s*(\d+)?$/.exec(part); + + if (match) { + let from = toPageNumber(match[1]); + let to = match[2] == null ? pageCount : toPageNumber(match[2]); + if (to < from) throw new Error(`Invalid page range "${part}": end page is before start page`); + selected.push(...range(from, to)); + } else if (/^\d+$/.test(part)) { + selected.push(toPageNumber(part)); + } else { + throw new Error(`Invalid page selection "${part}": expected a page number or a range like "2-5"`); + } + } + + return selected; +} + +function toPageNumber(value) { + let n = Number(value); + if (!Number.isInteger(n) || n < 1) { + throw new Error(`Invalid page number "${value}": page numbers are 1-based integers`); + } + return n; +} + +function range(from, to) { + let out = []; + for (let i = from; i <= to; i++) out.push(i); + return out; +} + +export function resolvePages({ pages, excludePages } = {}, pageCount) { + if (!Number.isInteger(pageCount) || pageCount < 1) { + throw new Error(`Invalid page count: ${pageCount}`); + } + + let selected = parseSelection(pages, pageCount); + let excluded = new Set(excludePages == null ? [] : parseSelection(excludePages, pageCount)); + + let outOfRange = [...new Set(selected.filter(p => p > pageCount))]; + if (outOfRange.length) { + throw new Error( + `Requested page${outOfRange.length > 1 ? 's' : ''} ${outOfRange.join(', ')} ` + + `but the document has only ${pageCount} page${pageCount > 1 ? 's' : ''}` + ); + } + + let resolved = [...new Set(selected)] + .filter(p => !excluded.has(p)) + .sort((a, b) => a - b); + + if (!resolved.length) { + throw new Error('No pages left to snapshot after applying `pages` and `excludePages`'); + } + + if (resolved.length > MAX_PAGES) { + throw new Error( + `Requested ${resolved.length} pages but the maximum per request is ${MAX_PAGES}. ` + + 'Narrow the selection with `pages` (for example "1-100") and issue several requests.' + ); + } + + return resolved; +} + +export { parseSelection as _parseSelection }; diff --git a/packages/cli-pdf/test/.eslintrc b/packages/cli-pdf/test/.eslintrc new file mode 100644 index 000000000..d4d123f4e --- /dev/null +++ b/packages/cli-pdf/test/.eslintrc @@ -0,0 +1,6 @@ +env: + jasmine: true +rules: + import/no-extraneous-dependencies: off + no-return-assign: off + no-sequences: off diff --git a/packages/cli-pdf/test/assets.test.js b/packages/cli-pdf/test/assets.test.js new file mode 100644 index 000000000..3ccd410e2 --- /dev/null +++ b/packages/cli-pdf/test/assets.test.js @@ -0,0 +1,29 @@ +import fs from 'fs'; +import path from 'path'; +import { pdfjsAssets } from '../src/assets.js'; + +describe('@percy/cli-pdf assets', () => { + let assets = pdfjsAssets(); + + it('resolves the installed pdfjs-dist root', () => { + expect(fs.existsSync(path.join(assets.root, 'package.json'))).toBe(true); + }); + + it('points at the legacy build directory', () => { + expect(fs.existsSync(assets.buildDir)).toBe(true); + expect(fs.existsSync(path.join(assets.buildDir, 'pdf.js'))).toBe(true); + expect(fs.existsSync(path.join(assets.buildDir, assets.workerFile))).toBe(true); + }); + + it('points at the font and cmap data pdf.js fetches at runtime', () => { + expect(fs.existsSync(assets.standardFontsDir)).toBe(true); + expect(fs.existsSync(assets.cmapsDir)).toBe(true); + expect(fs.readdirSync(assets.standardFontsDir).length).toBeGreaterThan(0); + expect(fs.readdirSync(assets.cmapsDir).length).toBeGreaterThan(0); + }); + + it('exposes the injectable pdf.js library file', () => { + expect(fs.existsSync(assets.libPath)).toBe(true); + expect(fs.readFileSync(assets.libPath, 'utf-8')).toContain('getDocument'); + }); +}); diff --git a/packages/cli-pdf/test/browser-scripts.test.js b/packages/cli-pdf/test/browser-scripts.test.js new file mode 100644 index 000000000..515724d95 --- /dev/null +++ b/packages/cli-pdf/test/browser-scripts.test.js @@ -0,0 +1,243 @@ +import { + fitScale, assertRasterDimensions, + MIN_DIMENSION, MAX_DIMENSION, DEFAULT_SCALE, MAX_SCALE, + openDocument, measurePages, renderPage, destroyDocument +} from '../src/browser-scripts.js'; + +describe('@percy/cli-pdf browser scripts', () => { + describe('fitScale', () => { + it('returns the requested scale when the page fits', () => { + expect(fitScale(2, { width: 612, height: 792 })).toBe(2); + }); + + it('clamps on the constraining axis', () => { + expect(fitScale(2, { width: 612, height: 1008 })).toBeCloseTo(MAX_DIMENSION / 1008, 6); + expect(fitScale(4, { width: 1000, height: 100 })).toBeCloseTo(MAX_DIMENSION / 1000, 6); + }); + }); + + describe('assertRasterDimensions', () => { + it('accepts dimensions at or above the minimum', () => { + expect(() => assertRasterDimensions(1, MIN_DIMENSION, MIN_DIMENSION)).not.toThrow(); + }); + + it('rejects a raster below the minimum on either axis', () => { + expect(() => assertRasterDimensions(3, 4, 400)) + .toThrowError(/Page 3 rasterized to 4x400px, below Percy's 10px minimum/); + expect(() => assertRasterDimensions(1, 400, 4)) + .toThrowError(/below Percy's 10px minimum/); + }); + }); + + it('exposes the limits the rasterizer enforces', () => { + expect(MIN_DIMENSION).toBe(10); + expect(MAX_DIMENSION).toBe(2000); + expect(DEFAULT_SCALE).toBe(2); + expect(MAX_SCALE).toBe(5); + }); + + // These four run inside the browser page, reaching pdf.js and the canvas + // through `window` / `document`. Standing those globals up here exercises the + // real logic in Node -- the fetch URLs pdf.js is handed, the white pre-fill, + // and the page-handle cleanup -- rather than leaving it to an integration run. + describe('page-context scripts', () => { + let doc, pages, renderCalls, createdCanvases; + + function fakePage(pageNumber, { width = 612, height = 792, renderError } = {}) { + let page = { + pageNumber, + cleanedUp: false, + getViewport: ({ scale }) => ({ width: width * scale, height: height * scale }), + render: (opts) => { + renderCalls.push({ pageNumber, opts }); + return { promise: renderError ? Promise.reject(renderError) : Promise.resolve() }; + }, + cleanup: () => { page.cleanedUp = true; } + }; + return page; + } + + function stubPageGlobals({ numPages = 3, pageOptions = {}, lib } = {}) { + pages = new Map(); + renderCalls = []; + createdCanvases = []; + + doc = { + numPages, + destroyed: false, + getPage: async (n) => { + let page = fakePage(n, pageOptions[n] || {}); + pages.set(n, page); + return page; + }, + destroy: async () => { doc.destroyed = true; } + }; + + global.window = lib === null ? {} : { 'pdfjs-dist/build/pdf': lib || stubLib() }; + global.document = { + createElement: (tag) => { + let canvas = { + tag, + width: 0, + height: 0, + fills: [], + getContext: () => ({ + set fillStyle(v) { canvas.fillStyle = v; }, + get fillStyle() { return canvas.fillStyle; }, + fillRect: (...args) => canvas.fills.push(args) + }), + toDataURL: (type) => `data:${type};base64,UE5H` + }; + createdCanvases.push(canvas); + return canvas; + } + }; + } + + function stubLib() { + return { + GlobalWorkerOptions: {}, + getDocumentCalls: [], + getDocument(options) { + this.getDocumentCalls.push(options); + return { promise: Promise.resolve(doc) }; + } + }; + } + + afterEach(() => { + delete global.window; + delete global.document; + }); + + describe('openDocument', () => { + it('points pdf.js at the served worker, document, fonts and cmaps', async () => { + stubPageGlobals({ numPages: 4 }); + let lib = global.window['pdfjs-dist/build/pdf']; + + await expectAsync(openDocument(null, { origin: 'http://localhost:9999' })) + .toBeResolvedTo({ pageCount: 4 }); + + expect(lib.GlobalWorkerOptions.workerSrc) + .toBe('http://localhost:9999/pdfjs/pdf.worker.js'); + + let [options] = lib.getDocumentCalls; + expect(options.url).toBe('http://localhost:9999/doc.pdf'); + expect(options.standardFontDataUrl).toBe('http://localhost:9999/standard_fonts/'); + expect(options.cMapUrl).toBe('http://localhost:9999/cmaps/'); + expect(options.cMapPacked).toBe(true); + // the PDF is untrusted input; this must never be enabled + expect(options.isEvalSupported).toBe(false); + }); + + it('stashes the handle for the later scripts', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:1' }); + + expect(global.window.__percyPdf.doc).toBe(doc); + }); + + it('falls back to the window.pdfjsLib global', async () => { + stubPageGlobals({ lib: null }); + let lib = stubLib(); + global.window.pdfjsLib = lib; + + await expectAsync(openDocument(null, { origin: 'http://localhost:2' })) + .toBeResolvedTo({ pageCount: 3 }); + expect(lib.getDocumentCalls.length).toBe(1); + }); + + it('throws when pdf.js did not initialise', async () => { + stubPageGlobals({ lib: null }); + + await expectAsync(openDocument(null, { origin: 'http://localhost:3' })) + .toBeRejectedWithError('pdf.js did not initialise in the page'); + }); + }); + + describe('measurePages', () => { + it('returns each page unscaled and releases the handles', async () => { + stubPageGlobals({ pageOptions: { 2: { width: 200, height: 400 } } }); + await openDocument(null, { origin: 'http://localhost:4' }); + + await expectAsync(measurePages(null, { pageNumbers: [1, 2] })).toBeResolvedTo([ + { pageNumber: 1, width: 612, height: 792 }, + { pageNumber: 2, width: 200, height: 400 } + ]); + + expect(pages.get(1).cleanedUp).toBe(true); + expect(pages.get(2).cleanedUp).toBe(true); + }); + }); + + describe('renderPage', () => { + it('renders at the given scale and returns a PNG data URL', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:5' }); + + let out = await renderPage(null, { pageNumber: 1, scale: 2 }); + + expect(out.width).toBe(1224); + expect(out.height).toBe(1584); + expect(out.dataUrl).toBe('data:image/png;base64,UE5H'); + expect(createdCanvases[0].tag).toBe('canvas'); + }); + + it('rounds fractional viewports up', async () => { + stubPageGlobals({ pageOptions: { 1: { width: 100.2, height: 100.6 } } }); + await openDocument(null, { origin: 'http://localhost:6' }); + + let out = await renderPage(null, { pageNumber: 1, scale: 1 }); + expect(out.width).toBe(101); + expect(out.height).toBe(101); + }); + + it('pre-fills the canvas white', async () => { + // PDF pages have no intrinsic background; without this, transparent + // regions rasterize to alpha-0 black and diff against anything. + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:7' }); + await renderPage(null, { pageNumber: 1, scale: 1 }); + + expect(createdCanvases[0].fillStyle).toBe('#ffffff'); + expect(createdCanvases[0].fills).toEqual([[0, 0, 612, 792]]); + }); + + it('passes the canvas context and viewport to pdf.js', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:8' }); + await renderPage(null, { pageNumber: 3, scale: 1 }); + + expect(renderCalls.length).toBe(1); + expect(renderCalls[0].pageNumber).toBe(3); + expect(renderCalls[0].opts.canvasContext).toBeDefined(); + expect(renderCalls[0].opts.viewport).toEqual({ width: 612, height: 792 }); + }); + + it('releases the page handle even when rendering fails', async () => { + stubPageGlobals({ pageOptions: { 1: { renderError: new Error('render blew up') } } }); + await openDocument(null, { origin: 'http://localhost:9' }); + + await expectAsync(renderPage(null, { pageNumber: 1, scale: 1 })) + .toBeRejectedWithError('render blew up'); + expect(pages.get(1).cleanedUp).toBe(true); + }); + }); + + describe('destroyDocument', () => { + it('destroys the document and clears the handle', async () => { + stubPageGlobals(); + await openDocument(null, { origin: 'http://localhost:10' }); + + await expectAsync(destroyDocument()).toBeResolvedTo(true); + expect(doc.destroyed).toBe(true); + expect(global.window.__percyPdf).toBeUndefined(); + }); + + it('is a no-op when no document is open', async () => { + global.window = {}; + await expectAsync(destroyDocument()).toBeResolvedTo(true); + }); + }); + }); +}); diff --git a/packages/cli-pdf/test/fixture.js b/packages/cli-pdf/test/fixture.js new file mode 100644 index 000000000..6f70f2e10 --- /dev/null +++ b/packages/cli-pdf/test/fixture.js @@ -0,0 +1,42 @@ +export function buildPdf({ pageCount = 1, width = 200, height = 300 } = {}) { + let objects = []; + let pageIds = []; + + for (let i = 0; i < pageCount; i++) { + pageIds.push(3 + i * 2); + } + + objects[1] = '<< /Type /Catalog /Pages 2 0 R >>'; + objects[2] = `<< /Type /Pages /Kids [${pageIds.map(id => `${id} 0 R`).join(' ')}] /Count ${pageCount} >>`; + + for (let i = 0; i < pageCount; i++) { + let pageId = pageIds[i]; + let contentId = pageId + 1; + let inset = 10 + i * 15; + let stream = `${inset} ${inset} ${width - inset * 2} ${height - inset * 2} re f`; + + objects[pageId] = + `<< /Type /Page /Parent 2 0 R /MediaBox [0 0 ${width} ${height}] ` + + `/Contents ${contentId} 0 R /Resources << >> >>`; + objects[contentId] = `<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`; + } + + let out = '%PDF-1.4\n'; + let offsets = []; + + for (let i = 1; i < objects.length; i++) { + offsets[i] = out.length; + out += `${i} 0 obj\n${objects[i]}\nendobj\n`; + } + + let xrefStart = out.length; + out += `xref\n0 ${objects.length}\n0000000000 65535 f \n`; + for (let i = 1; i < objects.length; i++) { + out += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + } + out += `trailer\n<< /Size ${objects.length} /Root 1 0 R >>\nstartxref\n${xrefStart}\n%%EOF\n`; + + return Buffer.from(out, 'latin1'); +} + +export const NOT_A_PDF = Buffer.from('this is definitely not a pdf', 'utf8'); diff --git a/packages/cli-pdf/test/pages.test.js b/packages/cli-pdf/test/pages.test.js new file mode 100644 index 000000000..e8c74ef4f --- /dev/null +++ b/packages/cli-pdf/test/pages.test.js @@ -0,0 +1,120 @@ +import { resolvePages, MAX_PAGES } from '../src/pages.js'; + +describe('@percy/cli-pdf page selection', () => { + it('selects every page when nothing is specified', () => { + expect(resolvePages({}, 4)).toEqual([1, 2, 3, 4]); + expect(resolvePages(undefined, 2)).toEqual([1, 2]); + }); + + it('accepts a single page number', () => { + expect(resolvePages({ pages: 3 }, 5)).toEqual([3]); + }); + + it('accepts an array of pages, sorted and de-duplicated', () => { + expect(resolvePages({ pages: [3, 1, 1, 2] }, 5)).toEqual([1, 2, 3]); + }); + + it('accepts closed ranges', () => { + expect(resolvePages({ pages: '2-4' }, 6)).toEqual([2, 3, 4]); + }); + + it('accepts open-ended ranges', () => { + expect(resolvePages({ pages: '3-' }, 5)).toEqual([3, 4, 5]); + }); + + it('accepts mixed lists of pages and ranges', () => { + expect(resolvePages({ pages: '1,3-5,8' }, 10)).toEqual([1, 3, 4, 5, 8]); + }); + + it('tolerates whitespace in string selections', () => { + expect(resolvePages({ pages: ' 1 , 3 - 4 ' }, 5)).toEqual([1, 3, 4]); + }); + + it('skips empty segments in a string selection', () => { + expect(resolvePages({ pages: '1,,3' }, 5)).toEqual([1, 3]); + expect(resolvePages({ pages: '2,' }, 5)).toEqual([2]); + }); + + it('throws when a string selection resolves to nothing', () => { + expect(() => resolvePages({ pages: ',' }, 5)) + .toThrowError('No pages left to snapshot after applying `pages` and `excludePages`'); + }); + + it('applies excludePages after pages', () => { + expect(resolvePages({ pages: '1-5', excludePages: [2, 4] }, 5)).toEqual([1, 3, 5]); + }); + + it('can exclude page 1', () => { + expect(resolvePages({ excludePages: [1] }, 3)).toEqual([2, 3]); + }); + + it('can exclude the second-to-last page', () => { + expect(resolvePages({ excludePages: [3] }, 4)).toEqual([1, 2, 4]); + }); + + it('ignores excluded pages that were never selected', () => { + expect(resolvePages({ pages: [1, 2], excludePages: [5] }, 5)).toEqual([1, 2]); + }); + + it('throws when a requested page is out of range', () => { + expect(() => resolvePages({ pages: '9' }, 5)) + .toThrowError('Requested page 9 but the document has only 5 pages'); + expect(() => resolvePages({ pages: [7, 8] }, 5)) + .toThrowError('Requested pages 7, 8 but the document has only 5 pages'); + }); + + it('uses the singular in the out-of-range message for a 1-page document', () => { + expect(() => resolvePages({ pages: '2' }, 1)) + .toThrowError('Requested page 2 but the document has only 1 page'); + }); + + it('throws on a malformed selection', () => { + expect(() => resolvePages({ pages: 'abc' }, 5)) + .toThrowError(/Invalid page selection "abc"/); + expect(() => resolvePages({ pages: '5-2' }, 5)) + .toThrowError('Invalid page range "5-2": end page is before start page'); + expect(() => resolvePages({ pages: 0 }, 5)) + .toThrowError(/Invalid page number "0"/); + expect(() => resolvePages({ pages: 1.5 }, 5)) + .toThrowError(/Invalid page number "1.5"/); + expect(() => resolvePages({ pages: {} }, 5)) + .toThrowError(/expected a number, array or string, got object/); + }); + + it('throws when every page has been excluded', () => { + expect(() => resolvePages({ excludePages: '1-3' }, 3)) + .toThrowError('No pages left to snapshot after applying `pages` and `excludePages`'); + }); + + it('throws on an invalid page count', () => { + expect(() => resolvePages({}, 0)).toThrowError('Invalid page count: 0'); + }); + + describe('the per-request page cap', () => { + // Each page is held in memory as a PNG twice over and crosses CDP as a + // base64 data URL, so an uncapped document can exhaust the heap. + it('allows exactly MAX_PAGES', () => { + expect(resolvePages({}, MAX_PAGES).length).toBe(MAX_PAGES); + }); + + it('throws one page past MAX_PAGES', () => { + expect(() => resolvePages({}, MAX_PAGES + 1)).toThrowError( + new RegExp(`Requested ${MAX_PAGES + 1} pages but the maximum per request is ${MAX_PAGES}`)); + }); + + it('counts what is actually selected, not the document length', () => { + // A long document is fine so long as the request narrows it. + expect(resolvePages({ pages: '1-10' }, MAX_PAGES * 10)).toEqual( + [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]); + }); + + it('counts after exclusions are applied', () => { + expect(resolvePages({ excludePages: '1' }, MAX_PAGES + 1).length).toBe(MAX_PAGES); + }); + + it('suggests how to get under the cap', () => { + expect(() => resolvePages({}, MAX_PAGES + 1)) + .toThrowError(/Narrow the selection with `pages`/); + }); + }); +}); diff --git a/packages/cli-upload/src/utils.js b/packages/cli-upload/src/utils.js index 6ddb27d47..6a63e3485 100644 --- a/packages/cli-upload/src/utils.js +++ b/packages/cli-upload/src/utils.js @@ -1,16 +1,17 @@ import fs from 'fs'; import { - createResource, - createRootResource + createImageSnapshotResources } from '@percy/cli-command/utils'; export { yieldAll } from '@percy/cli-command/utils'; -// Returns root resource and image resource objects based on image properties. The root resource is -// a generated DOM designed to display an image at it's native size without margins or padding. +// Returns root resource and image resource objects based on image properties. The wrapper DOM +// itself lives in @percy/core's utils as createImageSnapshotResources: its exact shape is what +// percy-api matches to extract the image and skip the renderer, and the PDF snapshot path builds +// the same pair, so there must only be one definition of it. export async function getImageResources({ name, type, @@ -19,29 +20,13 @@ export async function getImageResources({ relativePath, absolutePath }) { - let rootUrl = `http://local/${encodeURIComponent(name)}`; - let imageUrl = `http://local/${encodeURIComponent(relativePath)}`; - let content = await fs.promises.readFile(absolutePath); - let mimetype = `image/${type}`; - - return [ - createRootResource(rootUrl, ` - - - - - ${name} - - - - - - - `), - createResource(imageUrl, content, mimetype) - ]; + return createImageSnapshotResources({ + name, + rootUrl: `http://local/${encodeURIComponent(name)}`, + imageUrl: `http://local/${encodeURIComponent(relativePath)}`, + width, + height, + content: await fs.promises.readFile(absolutePath), + mimetype: `image/${type}` + }); } diff --git a/packages/core/package.json b/packages/core/package.json index 151fc497b..e379972dd 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -67,6 +67,7 @@ "yaml": "^2.4.1" }, "optionalDependencies": { - "@percy/cli-doctor": "1.32.10-beta.0" + "@percy/cli-doctor": "1.32.10-beta.0", + "@percy/cli-pdf": "1.32.10-beta.0" } } diff --git a/packages/core/src/api.js b/packages/core/src/api.js index ca92e8902..ac40508a7 100644 --- a/packages/core/src/api.js +++ b/packages/core/src/api.js @@ -8,6 +8,7 @@ import WebdriverUtils from '@percy/webdriver-utils'; import { handleSyncJob } from './snapshot.js'; import { getMaestroHierarchyDrift } from './maestro-hierarchy.js'; import { handleComparisonUpload } from './comparison-upload.js'; +import { handlePdfSnapshot } from './pdf-snapshot.js'; import { handleMaestroScreenshot } from './maestro-screenshot.js'; // Previously, we used `createRequire(import.meta.url).resolve` to resolve the path to the module. // This approach relied on `createRequire`, which is Node.js-specific and less compatible with modern ESM (ECMAScript Module) standards. @@ -278,6 +279,7 @@ export function createPercyServer(percy, port) { }) // post a comparison via multipart file upload .route('post', '/percy/comparison/upload', /* istanbul ignore next */ (req, res) => handleComparisonUpload(req, res, percy)) + .route('post', '/percy/pdf/snapshot', (req, res) => handlePdfSnapshot(req, res, percy)) // post a comparison by reading a Maestro screenshot from disk .route('post', '/percy/maestro-screenshot', (req, res) => handleMaestroScreenshot(req, res, percy)) // flushes one or more snapshots from the internal queue diff --git a/packages/core/src/config.js b/packages/core/src/config.js index 797e9d4ac..e46b5711a 100644 --- a/packages/core/src/config.js +++ b/packages/core/src/config.js @@ -1047,11 +1047,56 @@ export const comparisonSchema = { } }; +// Shape shared by `pages` and `excludePages`, defined once so the two cannot +// drift apart. The string pattern constrains only the character set: the +// grammar and every semantic rule (ranges, ordering, bounds, emptiness) belong +// to parseSelection() in @percy/cli-pdf, which is what actually reads the value +// and raises a precise error. Spelling the grammar out here as well meant the +// two disagreed — the old pattern rejected "1,,3" and "2,", which the parser +// accepts by skipping empty parts. +const pageSelection = { + oneOf: [ + { type: 'integer', minimum: 1 }, + { type: 'array', items: { type: 'integer', minimum: 1 } }, + { type: 'string', pattern: '^[\\d\\s,-]*\\d[\\d\\s,-]*$' } + ] +}; + // Grouped schemas for easier registration +export const pdfSnapshotSchema = { + $id: '/pdf-snapshot', + type: 'object', + $ref: '/snapshot#/$defs/common', + required: ['name'], + unevaluatedProperties: false, + properties: { + name: { + type: 'string', + description: 'Base snapshot name; each page becomes " | Page N"' + }, + pages: { + description: 'Pages to snapshot: 3, [1,2,5], "1-5", "1,3,8" or "2-" (to the end)', + ...pageSelection + }, + excludePages: { + description: 'Pages to omit, applied after `pages`. Same forms as `pages`.', + ...pageSelection + }, + scale: { + type: 'number', + exclusiveMinimum: 0, + maximum: 5, + default: 2, + description: 'Rasterization scale. Reduced automatically if a page would exceed 2000px.' + } + } +}; + export const schemas = [ configSchema, snapshotSchema, - comparisonSchema + comparisonSchema, + pdfSnapshotSchema ]; // Config migrate function diff --git a/packages/core/src/page.js b/packages/core/src/page.js index 97f8e7a59..b0ae2485b 100644 --- a/packages/core/src/page.js +++ b/packages/core/src/page.js @@ -71,6 +71,46 @@ function serializeDomCapture(_, options) { return { domSnapshot: PercyDOM.serialize(options), url: document.URL }; } +// Builds a real Error from a CDP exception. +// +// `exception.description` is the remote stack trace as one string, opening with +// the remote error's own `Name: message` and followed by its frames. This used +// to `throw` that bare string, so every caller saw `error.message === undefined` +// and an in-page failure surfaced as literally "undefined". +// +// `name` is deliberately blank. @percy/logger renders a thrown Error as +// `Error.prototype.toString.call(err)` and only falls back to `stack` at debug +// level (logger.js `log()`), so keeping the default name would both double the +// prefix -- "Error: Error: test error" -- and drop the remote frames from what +// the user sees. With an empty name, toString returns `message` verbatim, so the +// logged text stays byte-identical to what the thrown string produced, frames +// and all. That output is pinned by snapshot.test.js "logs execute errors and +// does not snapshot": those `at execute (:4:17)` lines are how a user +// debugs their own execute script. +// +// The trade is that `message` carries the frames too. Callers that want just the +// summary line -- an HTTP error body, say -- should take `message.split('\n')[0]`. +export function remoteError(exceptionDetails) { + let { exception, text } = exceptionDetails ?? {}; + let description = exception?.description; + + if (!description) { + // A non-Error was thrown in the page, so CDP reports the thrown value + // rather than a stack. `text` is CDP's own summary of the exception. + let value = exception?.value; + + description = value == null + ? (text || 'Unknown page error') + : (typeof value === 'object' ? JSON.stringify(value) : String(value)); + } + + let error = new Error(description); + error.name = ''; + error.stack = description; + + return error; +} + export class Page { static TIMEOUT = undefined; @@ -216,7 +256,7 @@ export class Page { }); if (exceptionDetails) { - throw exceptionDetails.exception.description; + throw remoteError(exceptionDetails); } else { return result.value; } diff --git a/packages/core/src/pdf-rasterize.js b/packages/core/src/pdf-rasterize.js new file mode 100644 index 000000000..8c3f8b932 --- /dev/null +++ b/packages/core/src/pdf-rasterize.js @@ -0,0 +1,162 @@ +import fs from 'fs'; +import logger from '@percy/logger'; +import { Server } from './server.js'; + +async function createAssetServer(pdfBuffer, assets) { + // Loopback only. This origin serves the customer's PDF with no auth, and the + // sole client is the discovery browser running on this machine -- unlike the + // API server it has no reason to be reachable off-box. + let server = Server.createServer({ port: 0, host: '127.0.0.1' }); + + server.serve('/pdfjs', assets.buildDir); + server.serve('/standard_fonts', assets.standardFontsDir); + server.serve('/cmaps', assets.cmapsDir); + + server.route('get', '/doc.pdf', (req, res) => ( + res.send(200, 'application/pdf', pdfBuffer) + )); + + server.route('get', '/', (req, res) => ( + res.send(200, 'text/html', '') + )); + + await server.listen(); + return server; +} + +// Marks an error the caller can fix by changing the request -- a page selection +// out of range, too many pages, an unusable scale, a page that rasterizes below +// Percy's minimum. handlePdfSnapshot answers 400 for these and 500 for +// everything else, so a browser launch failure or a CDP disconnect is no longer +// reported to the SDK as if the caller sent a bad request. +function asInputError(error) { + return Object.assign(error, { status: 400 }); +} + +// Page#eval has no timeout of its own -- Page.TIMEOUT only covers navigation, +// and Runtime.callFunctionOn with awaitPromise waits forever -- so every in-page +// call is raced against one. The timer is unref'd so a pending race never holds +// the process open. +// +// A losing promise needs no catch of its own: Promise.race attaches handlers to +// every input, so a late rejection from the timed-out eval is already handled. +export function withTimeout(promise, ms, description) { + let timer; + + let timeout = new Promise((resolve, reject) => { + timer = setTimeout(() => reject(new Error( + `Timed out after ${ms}ms ${description}` + )), ms); + timer.unref(); + }); + + return Promise.race([promise, timeout]).finally(() => clearTimeout(timer)); +} + +export async function rasterizePdf(percy, pdfBuffer, options) { + let log = logger('core:pdf-rasterize'); + + let { + pdfjsAssets, resolvePages, fitScale, assertRasterDimensions, + openDocument, measurePages, renderPage, destroyDocument, + DEFAULT_SCALE, MAX_SCALE, PAGE_RENDER_TIMEOUT + } = await import('@percy/cli-pdf'); + + let scale = options.scale == null ? DEFAULT_SCALE : Number(options.scale); + + if (!Number.isFinite(scale) || scale <= 0 || scale > MAX_SCALE) { + throw asInputError(new Error( + `Invalid scale ${options.scale}: expected a number between 0 and ${MAX_SCALE}` + )); + } + + let assets = pdfjsAssets(); + let server = await createAssetServer(pdfBuffer, assets); + let origin = server.address(); + let page; + + try { + await percy.browser.launch(); + page = await percy.browser.page({ meta: { snapshot: { name: 'pdf' } } }); + await page.goto(`${origin}/`); + + let pdfjsSource = await fs.promises.readFile(assets.libPath, 'utf-8'); + + await withTimeout( + /* eslint-disable-next-line no-new-func */ + page.eval(new Function(pdfjsSource)), + PAGE_RENDER_TIMEOUT, 'injecting pdf.js'); + + let { pageCount } = await withTimeout( + page.eval(openDocument, { origin }), + PAGE_RENDER_TIMEOUT, 'opening the PDF'); + + let selected; + + try { + selected = resolvePages(options, pageCount); + } catch (error) { + throw asInputError(error); + } + + log.debug(`Rendering ${selected.length} of ${pageCount} page(s) at scale ${scale}`); + + let sizes = await withTimeout( + page.eval(measurePages, { pageNumbers: selected }), + PAGE_RENDER_TIMEOUT, 'measuring the PDF pages'); + + let pages = []; + + for (let { pageNumber, width, height } of sizes) { + let effectiveScale = fitScale(scale, { width, height }); + + if (effectiveScale < scale) { + log.warn( + `Page ${pageNumber} is ${Math.round(width)}x${Math.round(height)}pt; ` + + `scale reduced from ${scale} to ${effectiveScale.toFixed(3)} to stay within ` + + 'Percy\'s 2000px limit' + ); + } + + let rendered = await withTimeout( + page.eval(renderPage, { pageNumber, scale: effectiveScale }), + PAGE_RENDER_TIMEOUT, `rendering page ${pageNumber}`); + + try { + assertRasterDimensions(pageNumber, rendered.width, rendered.height); + } catch (error) { + throw asInputError(error); + } + + let png = Buffer.from(rendered.dataUrl.slice(rendered.dataUrl.indexOf(',') + 1), 'base64'); + + log.debug(`Page ${pageNumber}: ${rendered.width}x${rendered.height}px, ${png.length} bytes`); + + pages.push({ + page: pageNumber, + width: rendered.width, + height: rendered.height, + scale: effectiveScale, + png + }); + } + + await withTimeout( + page.eval(destroyDocument), + PAGE_RENDER_TIMEOUT, 'closing the PDF'); + + return { pageCount, scale, pages }; + } finally { + // Settle both regardless: a timed-out page is exactly the case where close() + // is liable to reject, and letting that escape here would leak the asset + // server -- a listening socket still holding the customer's PDF. Report + // what failed rather than discarding it: a socket that will not drain is + // still holding that PDF, and nothing else would record it. + for (let result of await Promise.allSettled([page?.close(), server.close()])) { + /* istanbul ignore next: both closes resolve in every reachable test path */ + if (result.status === 'rejected') { + log.debug(`PDF cleanup failed: ${result.reason?.message ?? result.reason}`); + } + } + } +} diff --git a/packages/core/src/pdf-snapshot.js b/packages/core/src/pdf-snapshot.js new file mode 100644 index 000000000..fa809f276 --- /dev/null +++ b/packages/core/src/pdf-snapshot.js @@ -0,0 +1,233 @@ +import logger from '@percy/logger'; +import PercyConfig from '@percy/config'; +import { ServerError } from './server.js'; +import { handleSyncJob } from './snapshot.js'; +import { rasterizePdf } from './pdf-rasterize.js'; +import { + createImageSnapshotResources, + getPackageJSON, + normalizeOptions +} from './utils.js'; + +const MAX_PDF_BYTES = 50 * 1024 * 1024; +// base64 inflates by 4/3 and pads to a multiple of 4. +const MAX_PDF_BASE64_CHARS = Math.ceil(MAX_PDF_BYTES / 3) * 4; +const PDF_MAGIC = Buffer.from('%PDF-', 'latin1'); + +// Tagged onto the build's User-Agent so percy-api routes these pages down the +// extraction path instead of the renderer. Comparison#upload_snapshot? gates on +// `user_agent&.include?('@percy/cli-upload')` plus a root resource URL under +// `http://local/`, then recovers the image straight from the wrapper HTML. +// Measured ~1s per page extracted versus ~9-19s rendered, which is the whole +// point: the CLI already produced the exact PNG, so re-rendering it in the +// renderer fleet buys nothing. +// +// Because that check is a substring match, naming @percy/cli-pdf alongside it +// keeps the User-Agent honest about which code actually ran rather than +// impersonating the upload command. +export const UPLOAD_CLIENT_INFO = (() => { + let { version } = getPackageJSON(import.meta.url); + return [`@percy/cli-pdf/${version}`, `@percy/cli-upload/${version}`]; +})(); + +export function pageSnapshotName(name, pageNumber) { + return `${name} | Page ${pageNumber}`; +} + +function pageUrls(name, pageNumber) { + let base = `http://local/${encodeURIComponent(name)}/page-${pageNumber}`; + return { rootUrl: base, imageUrl: `${base}.png` }; +} + +export function decodePdf(pdf) { + if (!pdf || typeof pdf !== 'object') { + throw new ServerError(400, 'Missing required `pdf` object'); + } + + let { content } = pdf; + + if (typeof content !== 'string' || !content.length) { + throw new ServerError(400, 'Missing required `pdf.content` (base64-encoded PDF)'); + } + + // Reject on the ENCODED length first. Buffer.from() would otherwise allocate + // the full decode before we ever reach the size check, and the server buffers + // request bodies with no cap of its own (see IncomingMessage in server.js), so + // a 1GB body would be buffered, JSON-parsed and decoded before the 413. + if (content.length > MAX_PDF_BASE64_CHARS) { + throw new ServerError(413, `PDF exceeds the maximum size of ${MAX_PDF_BYTES / 1024 / 1024}MB`); + } + + let buffer = Buffer.from(content, 'base64'); + + if (buffer.length < PDF_MAGIC.length) { + throw new ServerError(400, '`pdf.content` is too short to be a PDF'); + } + + if (buffer.length > MAX_PDF_BYTES) { + throw new ServerError(413, `PDF exceeds the maximum size of ${MAX_PDF_BYTES / 1024 / 1024}MB`); + } + + if (!buffer.subarray(0, PDF_MAGIC.length).equals(PDF_MAGIC)) { + throw new ServerError(400, '`pdf.content` does not decode to a PDF (missing %PDF- header)'); + } + + return buffer; +} + +export async function loadPdfModule(load = () => import('@percy/cli-pdf')) { + try { + return await load(); + } catch (error) { + throw new ServerError(501, [ + 'PDF snapshots require the @percy/cli-pdf package, which is not installed.', + 'Install it with: npm install --save-dev @percy/cli-pdf', + `(underlying error: ${error.message})` + ].join(' ')); + } +} + +function validatePdfSnapshotOptions(options) { + let log = logger('core:pdf-snapshot'); + let normalized = normalizeOptions(options); + let { clientInfo, environmentInfo, pdf, ...validatable } = normalized; + + let errors = PercyConfig.validate(validatable, '/pdf-snapshot'); + + if (errors?.length > 0) { + log.warn('Invalid PDF snapshot options:'); + for (let e of errors) log.warn(`- ${e.path}: ${e.message}`); + } + + return normalized; +} + +function queuePages(percy, { name, rendered, sync, snapshotOptions }) { + return rendered.map(({ page, width, height, png }) => { + let snapshotName = pageSnapshotName(name, page); + let { rootUrl, imageUrl } = pageUrls(name, page); + + let options = { + ...snapshotOptions, + name: snapshotName, + widths: snapshotOptions.widths || [width], + minHeight: snapshotOptions.minHeight || height, + resources: () => buildPageResources({ rootUrl, imageUrl, snapshotName, width, height, png }) + }; + + if (sync) options.sync = true; + + let promise = null; + + if (sync) { + promise = new Promise((resolve, reject) => { + percy.upload(options, { resolve, reject }).catch(reject); + }); + } else { + percy.upload(options); + } + + return { page, snapshotName, promise }; + }); +} + +function buildPageResources({ rootUrl, imageUrl, snapshotName, width, height, png }) { + return createImageSnapshotResources({ + name: snapshotName, + rootUrl, + imageUrl, + width, + height, + content: png, + mimetype: 'image/png' + }); +} + +export async function handlePdfSnapshot(req, res, percy) { + let log = logger('core:pdf-snapshot'); + let body = req.body; + + if (!body || typeof body !== 'object' || Array.isArray(body) || Buffer.isBuffer(body)) { + throw new ServerError(400, 'Expected a JSON object body'); + } + + let { name, pdf, pages, excludePages, scale, ...rest } = validatePdfSnapshotOptions(body); + + if (typeof name !== 'string' || !name.trim()) { + throw new ServerError(400, 'Missing required `name`'); + } + + let buffer = decodePdf(pdf); + await loadPdfModule(); + + let sync = percy.syncMode(rest); + + let rasterized; + + try { + rasterized = await rasterizePdf(percy, buffer, { pages, excludePages, scale }); + } catch (error) { + // Only errors the caller can act on are 400s -- rasterizePdf tags those with + // `status`. A browser launch failure, an OOM, an asset-server bind error or + // a CDP disconnect are ours, not theirs, and must not tell the SDK that its + // request was malformed. + // remoteError puts the remote frames in `message` so the logger prints them + // (see page.js); the HTTP body wants only the summary line. + let message = (error?.message ?? String(error)).split('\n')[0]; + + log.error(`Failed to rasterize PDF "${name}": ${message}`); + throw new ServerError(error?.status ?? 500, `Could not rasterize PDF: ${message}`); + } + + let { pageCount, pages: rendered } = rasterized; + + percy.client.addClientInfo(rest.clientInfo); + percy.client.addClientInfo(UPLOAD_CLIENT_INFO); + percy.client.addEnvironmentInfo(rest.environmentInfo); + + let { clientInfo, environmentInfo, sync: _sync, ...snapshotOptions } = rest; + + log.info( + `PDF "${name}": snapshotting ${rendered.length} of ${pageCount} page(s)` + + (sync ? ' (waiting for comparison results)' : '') + ); + + let queued = queuePages(percy, { name, rendered, sync, snapshotOptions }); + + if (!sync) { + return res.json(200, { + success: true, + data: { + 'pdf-name': name, + 'page-count': pageCount, + 'pages-snapshotted': queued.length, + status: 'queued', + pages: queued.map(({ page, snapshotName }) => ({ + page, + 'snapshot-name': snapshotName + })) + } + }); + } + + let results = await Promise.all( + queued.map(async ({ page, snapshotName, promise }) => ({ + ...await handleSyncJob(promise, percy, 'snapshot'), + page, + 'snapshot-name': snapshotName + })) + ); + + let failed = results.filter(r => r.error || r.status === 'failure'); + + return res.json(200, { + success: true, + data: { + 'pdf-name': name, + 'page-count': pageCount, + 'pages-snapshotted': results.length, + status: failed.length ? 'failure' : 'success', + pages: results + } + }); +} diff --git a/packages/core/src/server.js b/packages/core/src/server.js index d06586a58..616f875a7 100644 --- a/packages/core/src/server.js +++ b/packages/core/src/server.js @@ -117,10 +117,12 @@ export class ServerError extends Error { export class Server extends http.Server { #sockets = new Set(); #defaultPort; + #host; - constructor({ port } = {}) { + constructor({ port, host } = {}) { super({ IncomingMessage, ServerResponse }); this.#defaultPort = port; + this.#host = host; // handle requests on end this.on('request', (req, res) => { @@ -134,8 +136,13 @@ export class Server extends http.Server { } // return host bind address - defaults to "::" + // + // An explicit `host` wins over the environment: a server that must not be + // reachable off-box (the PDF asset server, which serves the customer's + // document unauthenticated) has to stay on loopback even where an operator + // has widened PERCY_SERVER_HOST for the API server. get host() { - return process.env.PERCY_SERVER_HOST || '::'; + return this.#host || process.env.PERCY_SERVER_HOST || '::'; } // return the listening port or any default port @@ -422,8 +429,8 @@ function parseByteRange(range, size) { // shorthand function for creating a new server with specific options export function createServer(options = {}) { - let { serve, port, baseUrl = '/', ...opts } = options; - let server = new Server({ port }); + let { serve, port, host, baseUrl = '/', ...opts } = options; + let server = new Server({ port, host }); return serve ? ( server.serve(baseUrl, serve, opts) diff --git a/packages/core/src/utils.js b/packages/core/src/utils.js index c35121eff..2bea8fed0 100644 --- a/packages/core/src/utils.js +++ b/packages/core/src/utils.js @@ -488,6 +488,97 @@ export function createRootResource(url, content, attrs = {}) { return createResource(normalizeURL(url), content, 'text/html', { ...attrs, root: true }); } +// Escapes HTML-special characters for TEXT context. Snapshot names reach the +// wrapper's , and a name containing `', + imageUrl: 'http://local/a"b.png' + }); + + expect(html).not.toContain('