|
|
|
@ -24,6 +24,19 @@ const getBlockText = block => String(block?.plainText || '') |
|
|
|
.replace(/\n+/g, ' ') |
|
|
|
.trim(); |
|
|
|
|
|
|
|
const getStructureInfo = block => { |
|
|
|
if (block?.structureInfo && typeof block.structureInfo === 'object') return block.structureInfo; |
|
|
|
if (typeof block?.structureInfo === 'string') { |
|
|
|
try { |
|
|
|
const parsed = JSON.parse(block.structureInfo); |
|
|
|
return parsed && typeof parsed === 'object' ? parsed : {}; |
|
|
|
} catch (error) { |
|
|
|
return {}; |
|
|
|
} |
|
|
|
} |
|
|
|
return {}; |
|
|
|
}; |
|
|
|
|
|
|
|
const normalizeXmlText = value => String(value || '') |
|
|
|
.replace(/<w:tab\s*\/?>(?:<\/w:tab>)?/g, ' ') |
|
|
|
.replace(/<w:br\s*\/?>(?:<\/w:br>)?/g, '\n') |
|
|
|
@ -166,32 +179,31 @@ const buildBlockXmlMapping = async ({ sourceBuffer, blocks }) => { |
|
|
|
* 编辑/删除后 blockIndex 可能已经重排,不能用它推断原 DOCX 段落位置。 |
|
|
|
* 用编辑前保存的原文校验旧映射,失效时按原文顺序重新定位,避免导出静默回退为原文件。 |
|
|
|
*/ |
|
|
|
const repairParagraphMappings = (documentXml, blocks) => { |
|
|
|
const repairParagraphMappings = (documentXml, blocks, deletedParagraphIndexes = []) => { |
|
|
|
const paragraphs = getParagraphs(documentXml); |
|
|
|
const deleted = new Set(deletedParagraphIndexes.map(Number)); |
|
|
|
let cursor = 0; |
|
|
|
return blocks.map(block => { |
|
|
|
if (block?.structureInfo?.xmlTableIndex !== undefined) return block; |
|
|
|
const structureInfo = block?.structureInfo && typeof block.structureInfo === 'object' |
|
|
|
? { ...block.structureInfo } |
|
|
|
: {}; |
|
|
|
const originalStructureInfo = getStructureInfo(block); |
|
|
|
if (originalStructureInfo.xmlTableIndex !== undefined) { |
|
|
|
return { ...block, structureInfo: originalStructureInfo }; |
|
|
|
} |
|
|
|
const structureInfo = { ...originalStructureInfo }; |
|
|
|
const anchorText = normalizeXmlText( |
|
|
|
structureInfo.originalPlainText || block?.plainText || '' |
|
|
|
); |
|
|
|
if (!anchorText) return block; |
|
|
|
|
|
|
|
const currentIndex = Number(structureInfo.xmlParagraphIndex); |
|
|
|
const currentParagraph = Number.isInteger(currentIndex) ? paragraphs[currentIndex] : null; |
|
|
|
const currentMatches = currentParagraph?.text && ( |
|
|
|
currentParagraph.text === anchorText || |
|
|
|
currentParagraph.text.includes(anchorText) || |
|
|
|
anchorText.includes(currentParagraph.text) |
|
|
|
); |
|
|
|
if (currentMatches) { |
|
|
|
// 原始 DOCX 没有增删段落时,段落索引是稳定的。即使当前文字已经被编辑,
|
|
|
|
// 也必须信任首次解析保存的位置,不能再按相似文字搜索,否则重复段落会错配。
|
|
|
|
if (Number.isInteger(currentIndex) && currentIndex >= 0 |
|
|
|
&& currentIndex < paragraphs.length && !deleted.has(currentIndex)) { |
|
|
|
cursor = Math.max(cursor, currentIndex + 1); |
|
|
|
return block; |
|
|
|
} |
|
|
|
|
|
|
|
const foundIndex = paragraphs.findIndex((paragraph, index) => index >= cursor && ( |
|
|
|
const foundIndex = paragraphs.findIndex((paragraph, index) => index >= cursor && !deleted.has(index) && ( |
|
|
|
paragraph.text === anchorText || |
|
|
|
paragraph.text.includes(anchorText) || |
|
|
|
anchorText.includes(paragraph.text) |
|
|
|
@ -300,6 +312,28 @@ const replaceParagraphWithHtmlRuns = (paragraphXml, html) => { |
|
|
|
return `<w:p>${paragraphProperties}${runXml}</w:p>`; |
|
|
|
}; |
|
|
|
|
|
|
|
const hasBlockContentChanges = block => { |
|
|
|
const structureInfo = getStructureInfo(block); |
|
|
|
if (structureInfo.edited) return true; |
|
|
|
const currentText = normalizeXmlText(block?.plainText || ''); |
|
|
|
const originalText = structureInfo.originalPlainText === undefined |
|
|
|
? null |
|
|
|
: normalizeXmlText(structureInfo.originalPlainText); |
|
|
|
const currentHtml = String(block?.htmlContent || '').trim(); |
|
|
|
const originalHtml = structureInfo.originalHtmlContent === undefined |
|
|
|
? null |
|
|
|
: String(structureInfo.originalHtmlContent || '').trim(); |
|
|
|
return (originalText !== null && originalText !== currentText) |
|
|
|
|| (originalHtml !== null && originalHtml !== currentHtml); |
|
|
|
}; |
|
|
|
|
|
|
|
const replaceParagraphWithPlainText = (paragraphXml, text) => { |
|
|
|
const paragraphProperties = (paragraphXml.match(/<w:pPr(?:\s[^>]*)?>[\s\S]*?<\/w:pPr>/) || [''])[0]; |
|
|
|
const firstRunProperties = (paragraphXml.match(/<w:r(?:\s[^>]*)?>\s*(<w:rPr[\s\S]*?<\/w:rPr>)/) || ['', ''])[1]; |
|
|
|
const visibleText = escapeXml(String(text || '')); |
|
|
|
return `<w:p>${paragraphProperties}<w:r>${firstRunProperties}<w:t xml:space="preserve">${visibleText}</w:t></w:r></w:p>`; |
|
|
|
}; |
|
|
|
|
|
|
|
/** |
|
|
|
* 按原 run 的文字长度分配新文本 |
|
|
|
* 无新增 Tiptap mark 时保留原 run 属性,避免普通编辑丢失原文格式 |
|
|
|
@ -338,12 +372,13 @@ const replaceTextNodes = (paragraphXml, text) => { |
|
|
|
}; |
|
|
|
|
|
|
|
const replaceParagraphText = (paragraphXml, block) => { |
|
|
|
if (!hasBlockContentChanges(block)) return paragraphXml; |
|
|
|
if (hasTiptapInlineFormatting(block?.htmlContent)) { |
|
|
|
return replaceParagraphWithHtmlRuns(paragraphXml, block.htmlContent); |
|
|
|
} |
|
|
|
const text = getBlockText(block); |
|
|
|
if (!text) return paragraphXml; |
|
|
|
return replaceTextNodes(paragraphXml, text); |
|
|
|
return replaceParagraphWithPlainText(paragraphXml, text); |
|
|
|
}; |
|
|
|
|
|
|
|
const replaceTableText = (tableXml, block) => { |
|
|
|
@ -401,7 +436,7 @@ const replaceTableText = (tableXml, block) => { |
|
|
|
}); |
|
|
|
}; |
|
|
|
|
|
|
|
const updateDocumentXml = (documentXml, blocks, deletedParagraphIndexes = [], imageWriter = null) => { |
|
|
|
const updateDocumentXml = (documentXml, blocks, deletedParagraphIndexes = [], imageWriter = null, stats = {}) => { |
|
|
|
const editableBlocks = blocks |
|
|
|
.filter(block => ['paragraph', 'heading', 'table'].includes(String(block?.blockType || '').toLowerCase())) |
|
|
|
.sort((a, b) => Number(a.blockIndex || 0) - Number(b.blockIndex || 0)); |
|
|
|
@ -411,20 +446,29 @@ const updateDocumentXml = (documentXml, blocks, deletedParagraphIndexes = [], im |
|
|
|
let nextXml = documentXml.replace(/<w:tbl(?:\s[^>]*)?>[\s\S]*?<\/w:tbl>/g, tableXml => { |
|
|
|
const block = tableBlocks.find(item => Number(item.structureInfo.xmlTableIndex) === tableIndex); |
|
|
|
tableIndex += 1; |
|
|
|
return block ? replaceTableText(tableXml, block) : tableXml; |
|
|
|
if (!block) return tableXml; |
|
|
|
const replacedTable = replaceTableText(tableXml, block); |
|
|
|
if (replacedTable !== tableXml) stats.tableRewrites = (stats.tableRewrites || 0) + 1; |
|
|
|
return replacedTable; |
|
|
|
}); |
|
|
|
let paragraphIndex = 0; |
|
|
|
return nextXml.replace(/<w:p(?:\s[^>]*)?>[\s\S]*?<\/w:p>/g, paragraphXml => { |
|
|
|
const currentIndex = paragraphIndex; |
|
|
|
paragraphIndex += 1; |
|
|
|
if (deleted.has(currentIndex)) return ''; |
|
|
|
if (deleted.has(currentIndex)) { |
|
|
|
stats.deletedParagraphs = (stats.deletedParagraphs || 0) + 1; |
|
|
|
return ''; |
|
|
|
} |
|
|
|
const block = editableBlocks.find(item => Number(item?.structureInfo?.xmlParagraphIndex) === currentIndex); |
|
|
|
if (!block) return paragraphXml; |
|
|
|
if (imageWriter) { |
|
|
|
const image = getImageDataFromBlock(block); |
|
|
|
if (image) return imageWriter(paragraphXml, image); |
|
|
|
} |
|
|
|
return replaceParagraphText(paragraphXml, block); |
|
|
|
if (!hasBlockContentChanges(block)) return paragraphXml; |
|
|
|
const replacedParagraph = replaceParagraphText(paragraphXml, block); |
|
|
|
if (replacedParagraph !== paragraphXml) stats.paragraphRewrites = (stats.paragraphRewrites || 0) + 1; |
|
|
|
return replacedParagraph; |
|
|
|
}); |
|
|
|
}; |
|
|
|
|
|
|
|
@ -432,12 +476,30 @@ const updateDocumentXml = (documentXml, blocks, deletedParagraphIndexes = [], im |
|
|
|
* 使用原始 DOCX 作为模板回写 block 文本 |
|
|
|
* 映射按解析块顺序进行,确保不编辑时原文件可以原样重新打包 |
|
|
|
*/ |
|
|
|
const exportDocxFromOriginal = async ({ sourceBuffer, blocks, deletedParagraphIndexes = [] }) => { |
|
|
|
const exportDocxFromOriginal = async ({ sourceBuffer, blocks, deletedParagraphIndexes = [], logger = null }) => { |
|
|
|
const zip = await JSZip.loadAsync(sourceBuffer); |
|
|
|
const documentFile = zip.file(XML_PATH); |
|
|
|
if (!documentFile) throw new Error('DOCX 缺少 word/document.xml'); |
|
|
|
const documentXml = await documentFile.async('string'); |
|
|
|
const mappedBlocks = repairParagraphMappings(documentXml, blocks); |
|
|
|
const mappedBlocks = repairParagraphMappings(documentXml, blocks, deletedParagraphIndexes) |
|
|
|
.map(block => ({ ...block, structureInfo: getStructureInfo(block) })); |
|
|
|
const unmappedEditedBlocks = mappedBlocks.filter(block => { |
|
|
|
const blockType = String(block?.blockType || '').toLowerCase(); |
|
|
|
if (!['paragraph', 'heading', 'table'].includes(blockType)) return false; |
|
|
|
const structureInfo = getStructureInfo(block); |
|
|
|
const hasContentChange = structureInfo.edited || ( |
|
|
|
structureInfo.originalPlainText !== undefined && |
|
|
|
normalizeXmlText(structureInfo.originalPlainText) !== normalizeXmlText(block?.plainText || '') |
|
|
|
); |
|
|
|
if (!hasContentChange) return false; |
|
|
|
return blockType === 'table' |
|
|
|
? structureInfo.xmlTableIndex === undefined |
|
|
|
: structureInfo.xmlParagraphIndex === undefined; |
|
|
|
}); |
|
|
|
if (unmappedEditedBlocks.length) { |
|
|
|
const indexes = unmappedEditedBlocks.map(block => block.blockIndex).join(', '); |
|
|
|
throw new Error(`导出失败:已修改段落未找到原文档位置(blockIndex: ${indexes}),已阻止导出原文件`); |
|
|
|
} |
|
|
|
const hasXmlMapping = mappedBlocks.some(block => |
|
|
|
block?.structureInfo?.xmlParagraphIndex !== undefined || |
|
|
|
block?.structureInfo?.xmlTableIndex !== undefined |
|
|
|
@ -471,7 +533,19 @@ const exportDocxFromOriginal = async ({ sourceBuffer, blocks, deletedParagraphIn |
|
|
|
const paragraphProperties = (paragraphXml.match(/<w:pPr(?:\s[^>]*)?>[\s\S]*?<\/w:pPr>/) || [''])[0]; |
|
|
|
return `<w:p>${paragraphProperties}${buildImageDrawingXml({ relationshipId, width, height })}</w:p>`; |
|
|
|
} : null; |
|
|
|
const nextXml = updateDocumentXml(documentXml, mappedBlocks, deletedParagraphIndexes, imageWriter); |
|
|
|
const rewriteStats = {}; |
|
|
|
const nextXml = updateDocumentXml(documentXml, mappedBlocks, deletedParagraphIndexes, imageWriter, rewriteStats); |
|
|
|
logger?.info?.(`[plagiarism] export rewriteStats ${JSON.stringify(rewriteStats)}`); |
|
|
|
const changedBlocks = mappedBlocks.filter(block => { |
|
|
|
const structureInfo = getStructureInfo(block); |
|
|
|
return structureInfo.edited || ( |
|
|
|
structureInfo.originalPlainText !== undefined && |
|
|
|
normalizeXmlText(structureInfo.originalPlainText) !== normalizeXmlText(block?.plainText || '') |
|
|
|
); |
|
|
|
}); |
|
|
|
if (changedBlocks.length && !rewriteStats.paragraphRewrites && !rewriteStats.tableRewrites && !imageBlocks.length) { |
|
|
|
throw new Error(`导出失败:检测到 ${changedBlocks.length} 个已修改 block,但 DOCX 没有实际回写`); |
|
|
|
} |
|
|
|
const hasVisibleText = /<w:t(?:\s[^>]*)?>[^<]+<\/w:t>/.test(nextXml); |
|
|
|
if (!hasVisibleText && !imageBlocks.length) return sourceBuffer; |
|
|
|
zip.file(XML_PATH, nextXml); |
|
|
|
|