ai-query对接新版freesun-agent接口的分支
You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
 
 
 

488 lines
21 KiB

/**
* 原始DOCX XML导出层
* 只替换已有段落中的文字并重新打包
*/
'use strict';
const JSZip = require('jszip');
const cheerio = require('cheerio');
const imageSize = require('image-size');
const XML_PATH = 'word/document.xml';
const RELS_PATH = 'word/_rels/document.xml.rels';
const CONTENT_TYPES_PATH = '[Content_Types].xml';
const escapeXml = value => String(value || '')
.replace(/&/g, '&')
.replace(/</g, '&lt;')
.replace(/>/g, '&gt;')
.replace(/"/g, '&quot;')
.replace(/'/g, '&apos;');
const getBlockText = block => String(block?.plainText || '')
.replace(/\r/g, '')
.replace(/\n+/g, ' ')
.trim();
const normalizeXmlText = value => String(value || '')
.replace(/<w:tab\s*\/?>(?:<\/w:tab>)?/g, ' ')
.replace(/<w:br\s*\/?>(?:<\/w:br>)?/g, '\n')
.replace(/<[^>]+>/g, '')
.replace(/&amp;/g, '&')
.replace(/&lt;/g, '<')
.replace(/&gt;/g, '>')
.replace(/&quot;/g, '"')
.replace(/&#39;|&apos;/g, "'")
.replace(/\s+/g, '')
.trim();
const getParagraphs = documentXml => {
const paragraphs = [];
documentXml.replace(/<w:p(?:\s[^>]*)?>[\s\S]*?<\/w:p>/g, paragraphXml => {
paragraphs.push({ xml: paragraphXml, text: normalizeXmlText(paragraphXml) });
return paragraphXml;
});
return paragraphs;
};
const stripHtml = value => String(value || '').replace(/<[^>]+>/g, '').replace(/&nbsp;/g, ' ').trim();
const getTableText = tableXml => normalizeXmlText(tableXml);
const getHtmlTableCells = html => {
const $ = cheerio.load(String(html || ''), { decodeEntities: false });
return $('tr').toArray().flatMap(row => $(row).find('th,td').toArray().map(cell => {
const cellHtml = $(cell).html() || '';
return { html: cellHtml, text: stripHtml(cellHtml) };
}));
};
const getOriginalTableCells = block => (block?.structureInfo?.rows || [])
.flatMap(row => row || [])
.map(cell => String(cell?.text || '').trim());
const getXmlTableCells = tableXml => {
const cells = [];
tableXml.replace(/<w:tc(?:\s[^>]*)?>[\s\S]*?<\/w:tc>/g, cellXml => {
cells.push(cellXml);
return cellXml;
});
return cells;
};
const getImageDataFromBlock = block => {
const html = String(block?.htmlContent || '');
const match = html.match(/data:image\/(png|jpeg|jpg|gif|bmp);base64,([A-Za-z0-9+/=\s]+)/i);
if (!match) return null;
const extension = match[1].toLowerCase() === 'jpg' ? 'jpeg' : match[1].toLowerCase();
return {
extension,
mimeType: `image/${extension}`,
buffer: Buffer.from(match[2].replace(/\s+/g, ''), 'base64'),
};
};
const getNextRelationshipId = relationships => {
const ids = [...relationships.matchAll(/Id="rId(\d+)"/g)].map(match => Number(match[1]));
return `rId${Math.max(0, ...ids) + 1}`;
};
const getNextMediaIndex = zip => {
const indexes = Object.keys(zip.files)
.map(name => name.match(/^word\/media\/image(\d+)\.[^/]+$/i))
.filter(Boolean)
.map(match => Number(match[1]));
return Math.max(0, ...indexes) + 1;
};
const buildImageDrawingXml = ({ relationshipId, width, height }) => {
const cx = Math.max(1, Math.round(width * 9525));
const cy = Math.max(1, Math.round(height * 9525));
return `<w:r><w:drawing><wp:inline distT="0" distB="0" distL="0" distR="0"><wp:extent cx="${cx}" cy="${cy}"/><wp:docPr id="1" name="Mermaid diagram"/><a:graphic xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main"><a:graphicData uri="http://schemas.openxmlformats.org/drawingml/2006/picture"><pic:pic xmlns:pic="http://schemas.openxmlformats.org/drawingml/2006/picture"><pic:nvPicPr><pic:cNvPr id="0" name="diagram.png"/><pic:cNvPicPr/></pic:nvPicPr><pic:blipFill><a:blip r:embed="${relationshipId}"/><a:stretch><a:fillRect/></a:stretch></pic:blipFill><pic:spPr><a:xfrm><a:off x="0" y="0"/><a:ext cx="${cx}" cy="${cy}"/></a:xfrm><a:prstGeom prst="rect"><a:avLst/></a:prstGeom></pic:spPr></pic:pic></a:graphicData></a:graphic></wp:inline></w:drawing></w:r>`;
};
const appendImageRelationship = (relationships, relationshipId, target) => {
const relation = `<Relationship Id="${relationshipId}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/image" Target="${target}"/>`;
if (relationships.includes('</Relationships>')) {
return relationships.replace('</Relationships>', `${relation}</Relationships>`);
}
return `<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">${relation}</Relationships>`;
};
const registerImageContentType = (contentTypes, extension, mimeType) => {
if (new RegExp(`<Default[^>]+Extension="${extension}"`, 'i').test(contentTypes)) return contentTypes;
const node = `<Default Extension="${extension}" ContentType="${mimeType}"/>`;
return contentTypes.replace('</Types>', `${node}</Types>`);
};
const buildBlockXmlMapping = async ({ sourceBuffer, blocks }) => {
const zip = await JSZip.loadAsync(sourceBuffer);
const documentFile = zip.file(XML_PATH);
if (!documentFile) throw new Error('DOCX 缺少 word/document.xml');
const documentXml = await documentFile.async('string');
const paragraphs = getParagraphs(documentXml);
const tables = [];
documentXml.replace(/<w:tbl(?:\s[^>]*)?>[\s\S]*?<\/w:tbl>/g, tableXml => {
tables.push(tableXml);
return tableXml;
});
let cursor = 0;
let tableCursor = 0;
return blocks.map(block => {
if (String(block?.blockType || '').toLowerCase() === 'table') {
const sourceText = normalizeXmlText(getBlockText(block));
const tableIndex = tables.findIndex((tableXml, index) => index >= tableCursor && getTableText(tableXml).includes(sourceText));
if (tableIndex >= 0) {
tableCursor = tableIndex + 1;
return {
...block,
structureInfo: { ...(block.structureInfo || {}), xmlTableIndex: tableIndex },
};
}
}
const sourceText = normalizeXmlText(getBlockText(block));
if (!sourceText) return block;
let foundIndex = -1;
for (let index = cursor; index < paragraphs.length; index += 1) {
const xmlText = paragraphs[index].text;
if (xmlText && (xmlText === sourceText || xmlText.includes(sourceText) || sourceText.includes(xmlText))) {
foundIndex = index;
break;
}
}
if (foundIndex < 0) return block;
cursor = foundIndex + 1;
return {
...block,
structureInfo: {
...(block.structureInfo || {}),
xmlParagraphIndex: foundIndex,
},
};
});
};
/**
* 编辑/删除后 blockIndex 可能已经重排,不能用它推断原 DOCX 段落位置。
* 用编辑前保存的原文校验旧映射,失效时按原文顺序重新定位,避免导出静默回退为原文件。
*/
const repairParagraphMappings = (documentXml, blocks) => {
const paragraphs = getParagraphs(documentXml);
let cursor = 0;
return blocks.map(block => {
if (block?.structureInfo?.xmlTableIndex !== undefined) return block;
const structureInfo = block?.structureInfo && typeof block.structureInfo === 'object'
? { ...block.structureInfo }
: {};
const anchorText = normalizeXmlText(
structureInfo.originalPlainText || block?.plainText || ''
);
if (!anchorText) return block;
const currentIndex = Number(structureInfo.xmlParagraphIndex);
const currentParagraph = Number.isInteger(currentIndex) ? paragraphs[currentIndex] : null;
const currentMatches = currentParagraph?.text && (
currentParagraph.text === anchorText ||
currentParagraph.text.includes(anchorText) ||
anchorText.includes(currentParagraph.text)
);
if (currentMatches) {
cursor = Math.max(cursor, currentIndex + 1);
return block;
}
const foundIndex = paragraphs.findIndex((paragraph, index) => index >= cursor && (
paragraph.text === anchorText ||
paragraph.text.includes(anchorText) ||
anchorText.includes(paragraph.text)
));
if (foundIndex < 0) return block;
cursor = foundIndex + 1;
return {
...block,
structureInfo: { ...structureInfo, xmlParagraphIndex: foundIndex },
};
});
};
const getRunText = runXml => normalizeXmlText(runXml);
const normalizeColor = value => {
const color = String(value || '').trim().replace(/^#/, '');
if (/^[0-9a-f]{6}$/i.test(color)) return color.toUpperCase();
if (/^[0-9a-f]{3}$/i.test(color)) return color.split('').map(char => char + char).join('').toUpperCase();
return '';
};
const parseInlineStyle = styleText => {
const style = {};
String(styleText || '').split(';').forEach(item => {
const separator = item.indexOf(':');
if (separator < 0) return;
const name = item.slice(0, separator).trim().toLowerCase();
const value = item.slice(separator + 1).trim();
if (name) style[name] = value;
});
return style;
};
const buildRunProperties = marks => {
const properties = [];
if (marks.bold) properties.push('<w:b/>');
if (marks.italic) properties.push('<w:i/>');
if (marks.underline) properties.push('<w:u w:val="single"/>');
if (marks.strike) properties.push('<w:strike/>');
if (marks.color) properties.push(`<w:color w:val="${marks.color}"/>`);
if (marks.fontFamily) {
const font = escapeXml(marks.fontFamily.replace(/["']/g, '').split(',')[0].trim());
if (font) properties.push(`<w:rFonts w:ascii="${font}" w:hAnsi="${font}" w:eastAsia="${font}"/>`);
}
if (marks.fontSize) {
const match = String(marks.fontSize).match(/([0-9]+(?:\.[0-9]+)?)\s*(px|pt)?/i);
if (match) {
const points = match[2]?.toLowerCase() === 'px' ? Number(match[1]) * 0.75 : Number(match[1]);
if (Number.isFinite(points) && points > 0) {
const halfPoints = Math.round(points * 2);
properties.push(`<w:sz w:val="${halfPoints}"/><w:szCs w:val="${halfPoints}"/>`);
}
}
}
return properties.length ? `<w:rPr>${properties.join('')}</w:rPr>` : '';
};
const collectHtmlRuns = html => {
const $ = cheerio.load(String(html || ''), { decodeEntities: false }, false);
const runs = [];
const visit = (node, inheritedMarks = {}) => {
if (node.type === 'text') {
if (node.data) runs.push({ text: node.data, marks: inheritedMarks });
return;
}
if (node.type !== 'tag' && node.type !== 'root') return;
const tag = String(node.name || '').toLowerCase();
const attributes = node.attribs || {};
const style = parseInlineStyle(attributes.style);
const marks = {
...inheritedMarks,
bold: inheritedMarks.bold || tag === 'strong' || tag === 'b' || /bold|[5-9]00/i.test(style['font-weight'] || ''),
italic: inheritedMarks.italic || tag === 'em' || tag === 'i' || /italic|oblique/i.test(style['font-style'] || ''),
underline: inheritedMarks.underline || tag === 'u' || /underline/i.test(style['text-decoration'] || ''),
strike: inheritedMarks.strike || tag === 's' || tag === 'del' || /line-through/i.test(style['text-decoration'] || ''),
color: normalizeColor(style.color || attributes.color) || inheritedMarks.color || '',
fontFamily: style['font-family'] || inheritedMarks.fontFamily || '',
fontSize: style['font-size'] || inheritedMarks.fontSize || '',
};
if (tag === 'br') {
runs.push({ text: '\n', marks: inheritedMarks });
return;
}
(node.children || []).forEach(child => visit(child, marks));
};
const root = $.root().get(0);
(root?.children || []).forEach(node => visit(node));
return runs;
};
const hasTiptapInlineFormatting = html => /<(strong|b|em|i|u|s|del|span|br)\b/i.test(String(html || ''));
const replaceParagraphWithHtmlRuns = (paragraphXml, html) => {
const runs = collectHtmlRuns(html).filter(run => run.text.length > 0);
if (!runs.length) return paragraphXml;
const paragraphProperties = (paragraphXml.match(/<w:pPr(?:\s[^>]*)?>[\s\S]*?<\/w:pPr>/) || [''])[0];
const runXml = runs.map(run => {
const textParts = String(run.text).replace(/\r\n?/g, '\n').split('\n');
const content = textParts.map((part, index) => {
const line = part ? `<w:t xml:space="preserve">${escapeXml(part)}</w:t>` : '';
return index === 0 ? line : `<w:br/>${line}`;
}).join('');
return `<w:r>${buildRunProperties(run.marks)}${content}</w:r>`;
}).join('');
return `<w:p>${paragraphProperties}${runXml}</w:p>`;
};
/**
* 按原 run 的文字长度分配新文本
* 无新增 Tiptap mark 时保留原 run 属性,避免普通编辑丢失原文格式
*/
const replaceTextNodes = (paragraphXml, text) => {
const runs = [];
paragraphXml.replace(/<w:r(?:\s[^>]*)?>[\s\S]*?<\/w:r>/g, runXml => {
runs.push({ runXml, length: getRunText(runXml).length });
return runXml;
});
if (!runs.length) return paragraphXml;
const totalOriginalLength = runs.reduce((sum, run) => sum + run.length, 0);
const nextText = String(text || '');
let offset = 0;
const chunks = runs.map((run, index) => {
if (index === runs.length - 1) return nextText.slice(offset);
const share = totalOriginalLength > 0
? Math.round(nextText.length * run.length / totalOriginalLength)
: (index === 0 ? nextText.length : 0);
const chunk = nextText.slice(offset, offset + share);
offset += share;
return chunk;
});
let runCursor = 0;
return paragraphXml.replace(/<w:r(?:\s[^>]*)?>[\s\S]*?<\/w:r>/g, runXml => {
const index = runCursor++;
const chunk = chunks[index] || '';
const withoutText = runXml.replace(/<w:t(?:\s[^>]*)?>[\s\S]*?<\/w:t>/g, '');
if (!chunk) return withoutText;
const textNode = `<w:t xml:space="preserve">${escapeXml(chunk)}</w:t>`;
const runEnd = withoutText.lastIndexOf('</w:r>');
return `${withoutText.slice(0, runEnd)}${textNode}${withoutText.slice(runEnd)}`;
});
};
const replaceParagraphText = (paragraphXml, block) => {
if (hasTiptapInlineFormatting(block?.htmlContent)) {
return replaceParagraphWithHtmlRuns(paragraphXml, block.htmlContent);
}
const text = getBlockText(block);
if (!text) return paragraphXml;
return replaceTextNodes(paragraphXml, text);
};
const replaceTableText = (tableXml, block) => {
const cellTexts = getHtmlTableCells(block?.htmlContent);
if (!cellTexts.length) return tableXml;
const originalCells = getOriginalTableCells(block);
const xmlCells = getXmlTableCells(tableXml);
const replacements = new Map();
let searchStart = 0;
cellTexts.forEach((cell, htmlIndex) => {
const currentText = String(cell.text || '').trim();
const originalText = String(originalCells[htmlIndex] || '').trim();
// 未变化的单元格不重建,保留原有 run、换行和单元格内部结构
if (!currentText || currentText === originalText) return;
let targetIndex = -1;
if (!originalCells.length) {
// 兼容早期任务没有保存 structureInfo.rows 的数据
targetIndex = htmlIndex < xmlCells.length ? htmlIndex : -1;
} else if (originalText) {
targetIndex = xmlCells.findIndex((xmlCell, index) => index >= searchStart
&& normalizeXmlText(xmlCell) === normalizeXmlText(originalText));
} else if (htmlIndex < xmlCells.length && !normalizeXmlText(xmlCells[htmlIndex])) {
// 空单元格没有文字锚点,只允许写入同位置的 XML 空单元格
targetIndex = htmlIndex;
}
if (targetIndex >= 0) {
replacements.set(targetIndex, cell);
searchStart = targetIndex + 1;
}
});
if (!replacements.size) return tableXml;
let xmlCellCursor = 0;
return tableXml.replace(/<w:tc(?:\s[^>]*)?>[\s\S]*?<\/w:tc>/g, cellXml => {
const targetIndex = xmlCellCursor++;
const cell = replacements.get(targetIndex);
if (!cell) return cellXml;
const cellText = cell.text;
const paragraphs = cellXml.match(/<w:p(?:\s[^>]*)?>[\s\S]*?<\/w:p>/g) || [];
if (!paragraphs.length) return cellXml;
const lines = String(cellText).split(/\r?\n/);
let paragraphCursor = 0;
return cellXml.replace(/<w:p(?:\s[^>]*)?>[\s\S]*?<\/w:p>/g, paragraphXml => {
const line = lines[paragraphCursor] === undefined
? ''
: lines[paragraphCursor];
paragraphCursor += 1;
if (hasTiptapInlineFormatting(cell.html)) {
return replaceParagraphWithHtmlRuns(paragraphXml, cell.html);
}
return replaceTextNodes(paragraphXml, line);
});
});
};
const updateDocumentXml = (documentXml, blocks, deletedParagraphIndexes = [], imageWriter = null) => {
const editableBlocks = blocks
.filter(block => ['paragraph', 'heading', 'table'].includes(String(block?.blockType || '').toLowerCase()))
.sort((a, b) => Number(a.blockIndex || 0) - Number(b.blockIndex || 0));
const deleted = new Set(deletedParagraphIndexes.map(Number));
const tableBlocks = editableBlocks.filter(block => block?.structureInfo?.xmlTableIndex !== undefined);
let tableIndex = 0;
let nextXml = documentXml.replace(/<w:tbl(?:\s[^>]*)?>[\s\S]*?<\/w:tbl>/g, tableXml => {
const block = tableBlocks.find(item => Number(item.structureInfo.xmlTableIndex) === tableIndex);
tableIndex += 1;
return block ? replaceTableText(tableXml, block) : tableXml;
});
let paragraphIndex = 0;
return nextXml.replace(/<w:p(?:\s[^>]*)?>[\s\S]*?<\/w:p>/g, paragraphXml => {
const currentIndex = paragraphIndex;
paragraphIndex += 1;
if (deleted.has(currentIndex)) return '';
const block = editableBlocks.find(item => Number(item?.structureInfo?.xmlParagraphIndex) === currentIndex);
if (!block) return paragraphXml;
if (imageWriter) {
const image = getImageDataFromBlock(block);
if (image) return imageWriter(paragraphXml, image);
}
return replaceParagraphText(paragraphXml, block);
});
};
/**
* 使用原始 DOCX 作为模板回写 block 文本
* 映射按解析块顺序进行,确保不编辑时原文件可以原样重新打包
*/
const exportDocxFromOriginal = async ({ sourceBuffer, blocks, deletedParagraphIndexes = [] }) => {
const zip = await JSZip.loadAsync(sourceBuffer);
const documentFile = zip.file(XML_PATH);
if (!documentFile) throw new Error('DOCX 缺少 word/document.xml');
const documentXml = await documentFile.async('string');
const mappedBlocks = repairParagraphMappings(documentXml, blocks);
const hasXmlMapping = mappedBlocks.some(block =>
block?.structureInfo?.xmlParagraphIndex !== undefined ||
block?.structureInfo?.xmlTableIndex !== undefined
);
// 历史任务没有 XML 映射,原样返回可避免错误按顺序改写导致正文丢失。
if (!hasXmlMapping && !deletedParagraphIndexes.length) return sourceBuffer;
const imageBlocks = mappedBlocks.filter(block => getImageDataFromBlock(block));
let relationships = zip.file(RELS_PATH)
? await zip.file(RELS_PATH).async('string')
: '<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships"></Relationships>';
let contentTypes = zip.file(CONTENT_TYPES_PATH)
? await zip.file(CONTENT_TYPES_PATH).async('string')
: '<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"></Types>';
let mediaIndex = getNextMediaIndex(zip);
const imageWriter = imageBlocks.length ? (paragraphXml, image) => {
const relationshipId = getNextRelationshipId(relationships);
const mediaName = `image${mediaIndex}.${image.extension}`;
mediaIndex += 1;
zip.file(`word/media/${mediaName}`, image.buffer);
relationships = appendImageRelationship(relationships, relationshipId, `media/${mediaName}`);
contentTypes = registerImageContentType(contentTypes, image.extension, image.mimeType);
let width = 600;
let height = 400;
try {
const dimensions = imageSize.imageSize(image.buffer);
width = Math.min(600, dimensions.width || width);
height = Math.max(1, width * (dimensions.height || height) / (dimensions.width || width));
} catch (error) {
// 图片尺寸损坏时使用稳定兜底尺寸,避免导出整体失败
}
const paragraphProperties = (paragraphXml.match(/<w:pPr(?:\s[^>]*)?>[\s\S]*?<\/w:pPr>/) || [''])[0];
return `<w:p>${paragraphProperties}${buildImageDrawingXml({ relationshipId, width, height })}</w:p>`;
} : null;
const nextXml = updateDocumentXml(documentXml, mappedBlocks, deletedParagraphIndexes, imageWriter);
const hasVisibleText = /<w:t(?:\s[^>]*)?>[^<]+<\/w:t>/.test(nextXml);
if (!hasVisibleText && !imageBlocks.length) return sourceBuffer;
zip.file(XML_PATH, nextXml);
if (imageBlocks.length) {
zip.file(RELS_PATH, relationships);
zip.file(CONTENT_TYPES_PATH, contentTypes);
}
return zip.generateAsync({ type: 'nodebuffer', compression: 'DEFLATE' });
};
module.exports = {
buildBlockXmlMapping,
exportDocxFromOriginal,
};