fix(pdf): hand pdfjs a file:// standardFontDataUrl; repair Windows CI tests

pdfjs validates standardFontDataUrl as a URL ending in a forward slash —
our raw path with a trailing path.sep is invalid on Windows (C:\...\),
failing extractText/extractImages (and every test that verifies through
them) with 'Invalid factory url: must include trailing slash'. Linux and
macOS passed only because / is also a valid URL slash. Convert with
pathToFileURL() so every platform sends file:///.../standard_fonts/.

Also escape path.sep in PDFBatchOperations' sanitizer test regex — a bare
backslash made new RegExp() a syntax error on Windows.
This commit is contained in:
2026-09-05 23:33:58 +05:30
parent cfe134931f
commit 1b2ab7b55c
2 changed files with 11 additions and 3 deletions
+8 -2
View File
@@ -441,10 +441,16 @@ async function loadPdfjs() {
// Points pdfjs-dist at its bundled standard font metrics so it doesn't warn
// (and degrade text-extraction fidelity) when a PDF uses a standard font.
// pdfjs validates this as a URL that must end with a forward slash — a raw
// Windows path (C:\...\standard_fonts\) fails that check and breaks
// extractText/extractImages on Windows, so always hand pdfjs a file:// URL.
function getStandardFontDataUrl() {
return (
path.join(path.dirname(require.resolve('pdfjs-dist/package.json')), 'standard_fonts') + path.sep
const dir = path.join(
path.dirname(require.resolve('pdfjs-dist/package.json')),
'standard_fonts'
);
const url = require('url').pathToFileURL(dir);
return url.href.endsWith('/') ? url.href : url.href + '/';
}
async function pdfExtractText(data) {