mirror of
https://github.com/amitwh/markdown-converter.git
synced 2026-08-02 10:00:17 +05:30
fix(word): harden DOCX preprocessing and temp-file cleanup
- Make the HTML preprocessor code-block and inline-code aware so code examples containing <style> / <div> / comments are preserved. - Strip all raw <div> tags (not just alignment attributes) to avoid malformed output from unmatched closing tags. - Handle uppercase tags and single/unquoted attributes. - Create temporary DOCX input files inside private mkdtemp directories instead of predictable names in the shared temp directory. - Wrap batch DOCX preprocessing in try/catch so one unreadable file does not abort the entire batch. - Add regression tests for the edge cases above.
This commit is contained in:
+42
-15
@@ -2563,17 +2563,16 @@ function performExportWithOptions(format, options) {
|
|||||||
// Use pandoc for export with advanced options
|
// Use pandoc for export with advanced options
|
||||||
|
|
||||||
let inputFile = currentFile;
|
let inputFile = currentFile;
|
||||||
|
let tempInputDir = null;
|
||||||
let tempInputFile = null;
|
let tempInputFile = null;
|
||||||
|
|
||||||
// Pre-process markdown for Word output so that <style> blocks, HTML
|
// Pre-process markdown for Word output so that <style> blocks, HTML
|
||||||
// comments, and alignment <div> tags do not appear as visible text.
|
// comments, and <div> tags do not appear as visible text.
|
||||||
if (format === 'docx') {
|
if (format === 'docx') {
|
||||||
const content = fs.readFileSync(currentFile, 'utf-8');
|
const content = fs.readFileSync(currentFile, 'utf-8');
|
||||||
const cleanedContent = WordTemplateExporter.preprocessMarkdownForWordExport(content);
|
const cleanedContent = WordTemplateExporter.preprocessMarkdownForWordExport(content);
|
||||||
tempInputFile = path.join(
|
tempInputDir = fs.mkdtempSync(path.join(require('os').tmpdir(), 'mc_export_'));
|
||||||
require('os').tmpdir(),
|
tempInputFile = path.join(tempInputDir, path.basename(currentFile));
|
||||||
`mc_export_${Date.now()}_${path.basename(currentFile)}`
|
|
||||||
);
|
|
||||||
fs.writeFileSync(tempInputFile, cleanedContent, 'utf-8');
|
fs.writeFileSync(tempInputFile, cleanedContent, 'utf-8');
|
||||||
inputFile = tempInputFile;
|
inputFile = tempInputFile;
|
||||||
}
|
}
|
||||||
@@ -2681,6 +2680,13 @@ function performExportWithOptions(format, options) {
|
|||||||
// Ignore cleanup errors
|
// Ignore cleanup errors
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if (tempInputDir) {
|
||||||
|
try {
|
||||||
|
fs.rmdirSync(tempInputDir);
|
||||||
|
} catch {
|
||||||
|
// Ignore cleanup errors
|
||||||
|
}
|
||||||
|
}
|
||||||
});
|
});
|
||||||
} else if (format === 'pptx') {
|
} else if (format === 'pptx') {
|
||||||
// Add PowerPoint footer if enabled
|
// Add PowerPoint footer if enabled
|
||||||
@@ -3777,19 +3783,33 @@ async function performBatchConversion(inputFolder, outputFolder, format, options
|
|||||||
|
|
||||||
// Build pandoc command for other formats
|
// Build pandoc command for other formats
|
||||||
let pandocInputFile = inputFile;
|
let pandocInputFile = inputFile;
|
||||||
|
let batchTempInputDir = null;
|
||||||
let batchTempInputFile = null;
|
let batchTempInputFile = null;
|
||||||
|
|
||||||
// Pre-process markdown for Word output so that <style> blocks, HTML
|
// Pre-process markdown for Word output so that <style> blocks, HTML
|
||||||
// comments, and alignment <div> tags do not appear as visible text.
|
// comments, and <div> tags do not appear as visible text.
|
||||||
if (format === 'docx') {
|
if (format === 'docx') {
|
||||||
const content = fs.readFileSync(inputFile, 'utf-8');
|
try {
|
||||||
const cleanedContent = WordTemplateExporter.preprocessMarkdownForWordExport(content);
|
const content = fs.readFileSync(inputFile, 'utf-8');
|
||||||
batchTempInputFile = path.join(
|
const cleanedContent = WordTemplateExporter.preprocessMarkdownForWordExport(content);
|
||||||
require('os').tmpdir(),
|
batchTempInputDir = fs.mkdtempSync(path.join(require('os').tmpdir(), 'mc_batch_export_'));
|
||||||
`mc_batch_export_${Date.now()}_${path.basename(inputFile)}`
|
batchTempInputFile = path.join(batchTempInputDir, path.basename(inputFile));
|
||||||
);
|
fs.writeFileSync(batchTempInputFile, cleanedContent, 'utf-8');
|
||||||
fs.writeFileSync(batchTempInputFile, cleanedContent, 'utf-8');
|
pandocInputFile = batchTempInputFile;
|
||||||
pandocInputFile = batchTempInputFile;
|
} catch (preprocessError) {
|
||||||
|
console.error(
|
||||||
|
`Batch: Failed to pre-process ${path.basename(inputFile)}:`,
|
||||||
|
preprocessError.message
|
||||||
|
);
|
||||||
|
mainWindow.webContents.send('batch-progress', {
|
||||||
|
completed: index + 1,
|
||||||
|
total: totalCount,
|
||||||
|
currentFile: path.basename(inputFile),
|
||||||
|
success: false,
|
||||||
|
});
|
||||||
|
processNextFile(index + 1);
|
||||||
|
return;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
let pandocCmd = `${getPandocPath()} "${pandocInputFile}" -o "${outputFile}"`;
|
let pandocCmd = `${getPandocPath()} "${pandocInputFile}" -o "${outputFile}"`;
|
||||||
@@ -3893,7 +3913,7 @@ async function performBatchConversion(inputFolder, outputFolder, format, options
|
|||||||
|
|
||||||
// Execute conversion (using runPandocCmd for safety)
|
// Execute conversion (using runPandocCmd for safety)
|
||||||
runPandocCmd(pandocCmd, async (error, _stdout, stderr) => {
|
runPandocCmd(pandocCmd, async (error, _stdout, stderr) => {
|
||||||
// Clean up temporary pre-processed input file
|
// Clean up temporary pre-processed input file and directory
|
||||||
if (batchTempInputFile) {
|
if (batchTempInputFile) {
|
||||||
try {
|
try {
|
||||||
fs.unlinkSync(batchTempInputFile);
|
fs.unlinkSync(batchTempInputFile);
|
||||||
@@ -3901,6 +3921,13 @@ async function performBatchConversion(inputFolder, outputFolder, format, options
|
|||||||
// Ignore cleanup errors
|
// Ignore cleanup errors
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if (batchTempInputDir) {
|
||||||
|
try {
|
||||||
|
fs.rmdirSync(batchTempInputDir);
|
||||||
|
} catch {
|
||||||
|
// Ignore cleanup errors
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if (error) {
|
if (error) {
|
||||||
console.error(`Batch: Failed to convert ${path.basename(inputFile)}:`, error.message, stderr);
|
console.error(`Batch: Failed to convert ${path.basename(inputFile)}:`, error.message, stderr);
|
||||||
|
|||||||
Binary file not shown.
@@ -58,4 +58,60 @@ Centered content
|
|||||||
const output = WordTemplateExporter.preprocessMarkdownForWordExport(input);
|
const output = WordTemplateExporter.preprocessMarkdownForWordExport(input);
|
||||||
expect(output).toBe(input);
|
expect(output).toBe(input);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('preserves HTML artifacts inside fenced code blocks', () => {
|
||||||
|
const input = `# Title
|
||||||
|
|
||||||
|
\`\`\`
|
||||||
|
<style>body{}</style>
|
||||||
|
<!-- comment -->
|
||||||
|
<div align="center">text</div>
|
||||||
|
\`\`\`
|
||||||
|
|
||||||
|
After code.
|
||||||
|
`;
|
||||||
|
const output = WordTemplateExporter.preprocessMarkdownForWordExport(input);
|
||||||
|
expect(output).toContain('<style>body{}</style>');
|
||||||
|
expect(output).toContain('<!-- comment -->');
|
||||||
|
expect(output).toContain('<div align="center">text</div>');
|
||||||
|
expect(output).toContain('After code.');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('preserves HTML artifacts inside inline code', () => {
|
||||||
|
const input = 'Use `<div align="center">` for alignment.';
|
||||||
|
const output = WordTemplateExporter.preprocessMarkdownForWordExport(input);
|
||||||
|
expect(output).toContain('`<div align="center">`');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('handles uppercase tags and unquoted attributes', () => {
|
||||||
|
const input = `<DIV align=center>
|
||||||
|
Centered
|
||||||
|
</DIV>
|
||||||
|
`;
|
||||||
|
const output = WordTemplateExporter.preprocessMarkdownForWordExport(input);
|
||||||
|
expect(output).not.toContain('<DIV');
|
||||||
|
expect(output).not.toContain('</DIV>');
|
||||||
|
expect(output).toContain('Centered');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('handles single-quoted attributes', () => {
|
||||||
|
const input = `<div align='right'>Right</div>`;
|
||||||
|
const output = WordTemplateExporter.preprocessMarkdownForWordExport(input);
|
||||||
|
expect(output).not.toContain('<div');
|
||||||
|
expect(output).not.toContain('</div>');
|
||||||
|
expect(output).toContain('Right');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('removes non-alignment div tags without leaving malformed HTML', () => {
|
||||||
|
const input = `<div class="note">Note text</div>`;
|
||||||
|
const output = WordTemplateExporter.preprocessMarkdownForWordExport(input);
|
||||||
|
expect(output).not.toContain('<div');
|
||||||
|
expect(output).not.toContain('</div>');
|
||||||
|
expect(output).toContain('Note text');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('returns non-string input unchanged', () => {
|
||||||
|
expect(WordTemplateExporter.preprocessMarkdownForWordExport(null)).toBeNull();
|
||||||
|
expect(WordTemplateExporter.preprocessMarkdownForWordExport(123)).toBe(123);
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
|||||||
Reference in New Issue
Block a user