'use strict'; const fs = require('fs'); const fsPromises = require('fs').promises; const os = require('os'); const path = require('path'); const { spawn } = require('child_process'); const { randomUUID } = require('crypto'); const qiniu = require('qiniu'); const superagent = require('superagent'); const mammoth = require('mammoth'); const cheerio = require('cheerio'); const iconv = require('iconv-lite'); const { convertHTMLToDOCX } = require('./tools'); const { reportBusinessCall, reportFastgptResponse } = require('../services/dashboardReporter'); const MAX_FILE_COUNT = 5; const MIN_FILE_COUNT = 2; const MAX_FILE_SIZE = 50 * 1024 * 1024; const MIN_THRESHOLD = 5; const MAX_THRESHOLD = 100; const SUPPORTED_EXTS = new Set(['doc', 'docx', 'pdf', 'txt']); const ENCRYPTED_MAGIC = Buffer.from([0x00, 0x00, 0x5b, 0x00, 0xe5]); const SOFFICE = process.env.SOFFICE_PATH || 'soffice'; const MAX_COMPARE_BLOCK_CHARS = 5000; const HASH_BASE = 911382323n; const HASH_MOD = 2305843009213693951n; const EDIT_SOURCES = new Set(['manual', 'llm']); const TEXT_BLOCK_TYPES = new Set(['heading', 'paragraph', 'table']); const NON_EDITABLE_BLOCK_TYPES = new Set(['image']); const STATUS_TEXT = { created: '待解析', parsing: '解析中', parsed: '解析完成', parse_failed: '解析失败', checking: '查重中', checked: '查重完成', check_failed: '查重失败', }; const FILE_STATUS_TEXT = { pending: '待解析', parsing: '解析中', parsed: '解析完成', parse_failed: '解析失败', }; const now = () => new Date(); const toInt = (value) => { const parsed = Number.parseInt(String(value), 10); return Number.isFinite(parsed) ? parsed : null; }; const toBuffer = (value) => { if (!value) return Buffer.alloc(0); if (Buffer.isBuffer(value)) return value; if (value instanceof ArrayBuffer) return Buffer.from(value); if (ArrayBuffer.isView(value)) { return Buffer.from(value.buffer, value.byteOffset, value.byteLength); } return Buffer.from(value); }; const normalizeExt = (fileName = '') => { const ext = path.extname(String(fileName || '')).replace(/^\./, '').toLowerCase(); return ext; }; const escapeHtml = (text = '') => String(text || '') .replace(/&/g, '&') .replace(//g, '>') .replace(/"/g, '"') .replace(/'/g, '''); const normalizeHtmlFragment = (html = '') => String(html || '') .replace(/\r/g, '') .replace(/\n{2,}/g, '\n') .trim(); const normalizePlainText = (text = '') => String(text || '') .replace(/\r/g, '\n') .replace(/[ \t]+\n/g, '\n') .replace(/\n{3,}/g, '\n\n') .trim(); const normalizeFastGptContent = (content) => { if (typeof content === 'string') return content.trim(); if (Array.isArray(content)) { return content .map(item => { if (typeof item === 'string') return item; if (typeof item?.text === 'string') return item.text; return ''; }) .filter(Boolean) .join('\n') .trim(); } if (content && typeof content === 'object' && typeof content.text === 'string') { return content.text.trim(); } return ''; }; const normalizeEditSource = (value) => { const source = String(value || '').trim().toLowerCase(); return EDIT_SOURCES.has(source) ? source : 'manual'; }; const normalizeCompareText = (text = '') => { const rawToClean = []; let compareText = ''; String(text || '').split('').forEach((char, rawIndex) => { if (/\s/.test(char)) return; rawToClean.push(rawIndex); compareText += char; }); return { compareText, rawToClean }; }; const getRawRange = (normalized, start, end) => { if (!normalized.rawToClean.length) return { rawStart: 0, rawEnd: 0 }; const safeStart = Math.max(0, Math.min(start, normalized.rawToClean.length - 1)); const safeEnd = Math.max(safeStart, Math.min(end - 1, normalized.rawToClean.length - 1)); return { rawStart: normalized.rawToClean[safeStart] ?? 0, rawEnd: (normalized.rawToClean[safeEnd] ?? 0) + 1, }; }; const hashText = (text) => { let hash = 0n; for (const char of text) { hash = (hash * HASH_BASE + BigInt(char.codePointAt(0) || 0)) % HASH_MOD; } return hash; }; const buildRollingHashes = (text, size) => { const hashes = new Map(); if (text.length < size) return hashes; for (let index = 0; index <= text.length - size; index += 1) { const seed = text.slice(index, index + size); const hash = hashText(seed).toString(); if (!hashes.has(hash)) hashes.set(hash, []); hashes.get(hash).push(index); } return hashes; }; const rangesOverlap = (aStart, aEnd, bStart, bEnd) => aStart < bEnd && bStart < aEnd; const dedupeBlockPairMatches = (matches) => { const sorted = [...matches].sort((a, b) => (b.matchLength - a.matchLength) || (a.leftStart - b.leftStart) || (a.rightStart - b.rightStart) ); const kept = []; for (const item of sorted) { const overlaps = kept.some(existing => rangesOverlap(item.leftStart, item.leftEnd, existing.leftStart, existing.leftEnd) || rangesOverlap(item.rightStart, item.rightEnd, existing.rightStart, existing.rightEnd) ); if (!overlaps) kept.push(item); } return kept.sort((a, b) => (a.leftStart - b.leftStart) || (a.rightStart - b.rightStart)); }; const findBlockMatches = (leftBlock, rightBlock, threshold) => { const left = normalizeCompareText(leftBlock.plainText); const right = normalizeCompareText(rightBlock.plainText); if (left.compareText.length < threshold || right.compareText.length < threshold) return []; if (left.compareText.length > MAX_COMPARE_BLOCK_CHARS || right.compareText.length > MAX_COMPARE_BLOCK_CHARS) return []; const rightHashes = buildRollingHashes(right.compareText, threshold); const matches = []; const seen = new Set(); for (let leftIndex = 0; leftIndex <= left.compareText.length - threshold; leftIndex += 1) { const seed = left.compareText.slice(leftIndex, leftIndex + threshold); const positions = rightHashes.get(hashText(seed).toString()) || []; for (const rightIndex of positions) { if (right.compareText.slice(rightIndex, rightIndex + threshold) !== seed) continue; let length = threshold; while ( leftIndex + length < left.compareText.length && rightIndex + length < right.compareText.length && left.compareText[leftIndex + length] === right.compareText[rightIndex + length] ) { length += 1; } const key = `${leftIndex}:${rightIndex}:${length}`; if (seen.has(key)) continue; seen.add(key); const leftRaw = getRawRange(left, leftIndex, leftIndex + length); const rightRaw = getRawRange(right, rightIndex, rightIndex + length); matches.push({ leftStart: leftIndex, leftEnd: leftIndex + length, rightStart: rightIndex, rightEnd: rightIndex + length, leftRawStart: leftRaw.rawStart, leftRawEnd: leftRaw.rawEnd, rightRawStart: rightRaw.rawStart, rightRawEnd: rightRaw.rawEnd, matchText: left.compareText.slice(leftIndex, leftIndex + length), matchLength: length, }); } } return dedupeBlockPairMatches(matches); }; const mergeIntervals = (intervals) => { if (!intervals.length) return []; const sorted = [...intervals].sort((a, b) => a.start - b.start || a.end - b.end); const merged = [{ ...sorted[0] }]; for (let i = 1; i < sorted.length; i += 1) { const last = merged[merged.length - 1]; const current = sorted[i]; if (current.start <= last.end) { last.end = Math.max(last.end, current.end); } else { merged.push({ ...current }); } } return merged; }; const sumIntervals = (intervals) => mergeIntervals(intervals).reduce((sum, item) => sum + Math.max(0, item.end - item.start), 0); const toPercent = (value) => Number((Math.max(0, value) * 100).toFixed(2)); const parseJsonArray = (value, label) => { if (Array.isArray(value)) return value; if (typeof value !== 'string') throw new Error(`缺少参数: ${label}`); try { const parsed = JSON.parse(value); if (!Array.isArray(parsed)) throw new Error(); return parsed; } catch (error) { throw new Error(`${label} 必须是 JSON 数组`); } }; const getUploadedFiles = (ctx) => { const files = ctx.files || ctx.request.files || []; if (Array.isArray(files)) return files; if (Array.isArray(files.files)) return files.files; return Object.values(files).flat().filter(Boolean); }; const ensureBasicFile = (file, index) => { const originalName = String(file?.originalname || file?.originalName || file?.name || '').trim(); if (!originalName) throw new Error(`第 ${index + 1} 个文件缺少文件名`); const fileExt = normalizeExt(originalName); if (!SUPPORTED_EXTS.has(fileExt)) { throw new Error(`文件 ${originalName} 格式不支持,仅支持 doc/docx/pdf/txt`); } const fileSize = Number(file?.size || 0); if (!fileSize) throw new Error(`文件 ${originalName} 为空`); if (fileSize > MAX_FILE_SIZE) { throw new Error(`文件 ${originalName} 超过 50MB 限制`); } return { originalName, fileExt, fileSize }; }; const isEncryptedBuffer = (buffer) => Buffer.isBuffer(buffer) && buffer.length >= ENCRYPTED_MAGIC.length && buffer.subarray(0, ENCRYPTED_MAGIC.length).equals(ENCRYPTED_MAGIC); const readFileBuffer = async (file) => { if (file?.buffer) return toBuffer(file.buffer); if (!file?.path) throw new Error('上传文件缺少临时路径'); return fsPromises.readFile(file.path); }; const validateMagicByExt = (buffer, ext, fileName) => { if (!Buffer.isBuffer(buffer) || !buffer.length) { throw new Error(`文件 ${fileName} 为空`); } if (ext === 'docx') { const isZip = buffer.length >= 4 && buffer[0] === 0x50 && buffer[1] === 0x4b && [0x03, 0x05, 0x07].includes(buffer[2]); if (!isZip) throw new Error(`文件 ${fileName} 不是有效的 docx 文件`); } if (ext === 'pdf') { const header = buffer.subarray(0, 5).toString('ascii'); if (header !== '%PDF-') throw new Error(`文件 ${fileName} 不是有效的 pdf 文件`); } if (ext === 'doc') { const oleMagic = Buffer.from([0xd0, 0xcf, 0x11, 0xe0, 0xa1, 0xb1, 0x1a, 0xe1]); const hasOle = buffer.length >= oleMagic.length && buffer.subarray(0, oleMagic.length).equals(oleMagic); if (!hasOle) throw new Error(`文件 ${fileName} 不是有效的 doc 文件`); } }; const decryptIfNeeded = async (ctx, buffer, fileName) => { if (!isEncryptedBuffer(buffer)) { return { buffer, decrypted: false }; } const host = String(ctx.config?.xunruan?.host || ctx.app?.fs?.config?.xunruan?.host || '').replace(/\/+$/, ''); if (!host) throw new Error(`文件 ${fileName} 已加密,但未配置迅软解密服务`); const res = await superagent .post(`${host}/uploadSecret`) .set('Content-Type', 'application/octet-stream') .send(buffer) .responseType('arraybuffer') .timeout({ response: 30000, deadline: 120000 }); const decrypted = toBuffer(res.body); if (!decrypted.length) throw new Error(`文件 ${fileName} 解密结果为空`); return { buffer: decrypted, decrypted: true }; }; const ensureQiniuConfig = (ctx) => { const conf = ctx.config?.qiniu || ctx.app?.fs?.config?.qiniu || {}; const bucket = String(conf.bkt || '').trim(); const accessKey = String(conf.ak || '').trim(); const secretKey = String(conf.sk || '').trim(); const domain = String(conf.dmn || process.env.FS_QINIU_DOMAIN || '').trim().replace(/\/+$/, ''); if (!bucket || !accessKey || !secretKey) { throw new Error('七牛云配置不完整'); } return { bucket, accessKey, secretKey, domain }; }; const getMimeByExt = (ext) => { const map = { doc: 'application/msword', docx: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', pdf: 'application/pdf', txt: 'text/plain', }; return map[ext] || 'application/octet-stream'; }; const createQiniuUploader = (ctx) => { const { bucket, accessKey, secretKey, domain } = ensureQiniuConfig(ctx); const mac = new qiniu.auth.digest.Mac(accessKey, secretKey); const putPolicy = new qiniu.rs.PutPolicy({ scope: bucket }); const uploadToken = putPolicy.uploadToken(mac); const config = new qiniu.conf.Config(); config.zone = qiniu.zone.Zone_z0; const formUploader = new qiniu.form_up.FormUploader(config); return async ({ key, buffer, mimeType }) => { const putExtra = new qiniu.form_up.PutExtra(); putExtra.mimeType = mimeType || 'application/octet-stream'; const result = await new Promise((resolve, reject) => { formUploader.put(uploadToken, key, buffer, putExtra, (respErr, respBody, respInfo) => { if (respErr) { reject(respErr); return; } if (respInfo?.statusCode && respInfo.statusCode >= 300) { reject(new Error(`七牛上传失败: ${respInfo.statusCode}`)); return; } resolve(respBody || {}); }); }); const finalKey = String(result.key || key); return { qiniuKey: finalKey, fileUrl: domain ? `${domain}/${finalKey}` : finalKey, }; }; }; const safeQiniuKeyPart = (value = '') => String(value || '') .replace(/\\/g, '/') .replace(/[<>:"|?*\x00-\x1f]/g, '_') .split('/') .filter(Boolean) .join('_') .slice(0, 180); const runCommand = (command, args, options = {}) => new Promise((resolve, reject) => { const child = spawn(command, args, { ...options, windowsHide: true }); let stdout = ''; let stderr = ''; child.stdout?.on('data', chunk => { stdout += chunk.toString(); }); child.stderr?.on('data', chunk => { stderr += chunk.toString(); }); child.on('error', reject); child.on('close', code => { if (code === 0) { resolve({ stdout, stderr }); return; } reject(new Error(`${command} exited with code ${code}${stderr ? `: ${stderr}` : ''}`)); }); }); const writeTempFile = async (tempDir, fileName, buffer) => { const safeName = `${Date.now()}-${randomUUID()}-${safeQiniuKeyPart(fileName) || 'input'}`; const filePath = path.join(tempDir, safeName); await fsPromises.writeFile(filePath, buffer); return filePath; }; const convertDocToDocx = async (tempDir, fileName, buffer) => { const inputPath = await writeTempFile(tempDir, fileName, buffer); await runCommand(SOFFICE, ['--headless', '--convert-to', 'docx', '--outdir', tempDir, inputPath]); const baseName = path.basename(inputPath, path.extname(inputPath)); const outputPath = path.join(tempDir, `${baseName}.docx`); await fsPromises.access(outputPath); return fsPromises.readFile(outputPath); }; const uploadDocxImage = async (ctx, image) => { const upload = createQiniuUploader(ctx); const contentType = String(image.contentType || 'application/octet-stream'); const imageBuffer = toBuffer(await image.read()); if (!imageBuffer.length) throw new Error('docx 中存在空图片'); const extMap = { 'image/png': 'png', 'image/jpeg': 'jpg', 'image/jpg': 'jpg', 'image/webp': 'webp', 'image/gif': 'gif', 'image/bmp': 'bmp', 'image/tiff': 'tiff', 'image/svg+xml': 'svg', }; const ext = extMap[contentType.toLowerCase()] || 'bin'; const datePart = new Date().toISOString().slice(0, 10).replace(/-/g, ''); const key = `ai-query/plagiarism-check/docx-images/${datePart}/${randomUUID()}.${ext}`; const uploaded = await upload({ key, buffer: imageBuffer, mimeType: contentType }); return { src: uploaded.fileUrl }; }; const convertDocxToHtml = async (ctx, buffer) => { const result = await mammoth.convertToHtml( { buffer }, { convertImage: mammoth.images.imgElement(image => uploadDocxImage(ctx, image)), } ); return String(result?.value || '').trim(); }; const blocksFromHtml = (html) => { const $ = cheerio.load(`
${String(html || '')}
`, { decodeEntities: false }); const blocks = []; const pushTextBlock = (tagName, htmlContent, text, extra = {}) => { const normalizedText = normalizePlainText(text); if (!normalizedText) return; const heading = String(tagName || '').toLowerCase().match(/^h([1-6])$/); blocks.push({ blockIndex: blocks.length, blockType: heading ? 'heading' : 'paragraph', htmlContent: normalizeHtmlFragment(htmlContent), plainText: normalizedText, textLength: normalizedText.length, structureInfo: heading ? { level: Number(heading[1]), tagName, ...extra } : { tagName: tagName || 'p', ...extra }, }); }; const pushImageBlock = (element, parentTagName = 'img') => { const image = $(element); const src = String(image.attr('src') || '').trim(); if (!src) return; const alt = String(image.attr('alt') || '').trim() || null; const width = toInt(image.attr('width')); const height = toInt(image.attr('height')); blocks.push({ blockIndex: blocks.length, blockType: 'image', htmlContent: normalizeHtmlFragment($.html(element) || ''), plainText: alt || '', textLength: 0, structureInfo: { tagName: 'img', parentTagName, src, alt, width, height, contentType: String(image.attr('data-content-type') || '').trim() || null, }, }); }; const pushTableBlock = (element, parentTagName = 'table') => { const table = $(element); const rows = []; table.find('tr').each((rowIndex, rowElement) => { const cells = []; $(rowElement).find('th,td').each((cellIndex, cellElement) => { cells.push({ rowIndex, cellIndex, tagName: String(cellElement.name || '').toLowerCase() || 'td', text: normalizePlainText($(cellElement).text()), colspan: toInt($(cellElement).attr('colspan')) || 1, rowspan: toInt($(cellElement).attr('rowspan')) || 1, }); }); if (cells.length) rows.push(cells); }); const plainText = normalizePlainText(table.text()); blocks.push({ blockIndex: blocks.length, blockType: 'table', htmlContent: normalizeHtmlFragment($.html(element) || ''), plainText, textLength: 0, structureInfo: { tagName: 'table', parentTagName, rowCount: rows.length, rows, hasMergedCells: rows.some(row => row.some(cell => cell.colspan > 1 || cell.rowspan > 1)), }, }); }; const visitNode = (element, parentTagName = 'root') => { if (!element || element.type === 'comment') return; if (element.type === 'text') { const text = normalizePlainText($(element).text()); if (!text) return; pushTextBlock('p', `

${escapeHtml(text)}

`, text, { source: 'inline_text', parentTagName }); return; } const tagName = String(element.name || element.tagName || '').toLowerCase(); if (!tagName) return; if (tagName === 'img') { pushImageBlock(element, parentTagName); return; } if (tagName === 'table') { pushTableBlock(element, parentTagName); return; } if (tagName === 'p') { const directImages = $(element).children('img'); const directTables = $(element).children('table'); const paragraphText = normalizePlainText($(element).text()); const childElementCount = $(element).children().length; if (directImages.length && !paragraphText) { directImages.each((_, imgElement) => pushImageBlock(imgElement, tagName)); return; } if (directTables.length && !paragraphText) { directTables.each((_, tableElement) => pushTableBlock(tableElement, tagName)); return; } if (childElementCount === directImages.length && directImages.length > 0) { directImages.each((_, imgElement) => pushImageBlock(imgElement, tagName)); return; } if (childElementCount === directTables.length && directTables.length > 0) { directTables.each((_, tableElement) => pushTableBlock(tableElement, tagName)); return; } } if (/^h[1-6]$/.test(tagName) || tagName === 'p') { pushTextBlock(tagName, $.html(element) || '', $(element).text(), { parentTagName }); return; } if (tagName === 'li') { const htmlContent = `

${normalizeHtmlFragment($(element).html() || $(element).text())}

`; pushTextBlock('p', htmlContent, $(element).text(), { source: 'list_item', parentTagName }); return; } const directImages = $(element).children('img'); if (directImages.length) { directImages.each((_, imgElement) => pushImageBlock(imgElement, tagName)); } const directTables = $(element).children('table'); if (directTables.length) { directTables.each((_, tableElement) => pushTableBlock(tableElement, tagName)); } const childNodes = $(element).contents().toArray().filter(child => { const childTag = String(child?.name || child?.tagName || '').toLowerCase(); if (!childTag) return child?.type === 'text'; return !['img', 'table'].includes(childTag); }); const hasStructuredChild = childNodes.some(child => { const childTag = String(child?.name || child?.tagName || '').toLowerCase(); return /^h[1-6]$/.test(childTag) || ['p', 'li', 'div', 'section', 'article', 'ul', 'ol'].includes(childTag); }); if (!hasStructuredChild) { const text = normalizePlainText($(element).text()); if (text) { pushTextBlock(tagName === 'span' ? 'p' : tagName, $.html(element) || '', text, { parentTagName }); } return; } childNodes.forEach(child => visitNode(child, tagName)); }; $('#root').contents().toArray().forEach(element => visitNode(element)); return blocks; }; const htmlToPlainText = (html = '') => { const $ = cheerio.load(`
${String(html || '')}
`, { decodeEntities: false }); const text = normalizePlainText($('#root').text()); return text; }; const blocksFromPlainText = (text, source = 'text') => { const normalized = normalizePlainText(text); if (!normalized) return []; const parts = normalized .split(/\n\s*\n+/) .map(item => item.replace(/\n+/g, ' ').trim()) .filter(Boolean); return parts.map((plainText, index) => { const isHeading = plainText.length <= 30 && !/[。!?.!?;;]/.test(plainText.slice(-1)); const tag = isHeading ? 'h2' : 'p'; return { blockIndex: index, blockType: isHeading ? 'heading' : 'paragraph', htmlContent: `<${tag}>${escapeHtml(plainText)}`, plainText, textLength: plainText.length, structureInfo: { source, tagName: tag }, }; }); }; const parsePdf = async (tempDir, fileName, buffer) => { const inputPath = await writeTempFile(tempDir, fileName, buffer); const outputPath = path.join(tempDir, `${path.basename(inputPath)}.txt`); await runCommand('pdftotext', ['-layout', '-enc', 'UTF-8', inputPath, outputPath]); const text = await fsPromises.readFile(outputPath, 'utf8'); return blocksFromPlainText(text, 'pdf'); }; const parseTxt = (buffer) => { let text = buffer.toString('utf8'); if (text.includes('�') || /[\x00-\x08\x0e-\x1f]/.test(text.slice(0, 200))) { text = iconv.decode(buffer, 'gbk'); } return blocksFromPlainText(text, 'txt'); }; const parseDocumentToBlocks = async (ctx, { buffer, fileExt, originalName }) => { const tempDir = await fsPromises.mkdtemp(path.join(os.tmpdir(), 'plagiarism-parse-')); try { if (fileExt === 'txt') return parseTxt(buffer); if (fileExt === 'pdf') return await parsePdf(tempDir, originalName, buffer); let docxBuffer = buffer; if (fileExt === 'doc') { docxBuffer = await convertDocToDocx(tempDir, originalName, buffer); } const html = await convertDocxToHtml(ctx, docxBuffer); return blocksFromHtml(html); } finally { await fsPromises.rm(tempDir, { recursive: true, force: true }).catch(() => { }); } }; const resolveDocTypes = async (models, metas) => { const docs = await models.PlagiarismCheckDocTypes.findAll({ where: { enabled: true }, raw: true, }); const byId = new Map(docs.map(item => [Number(item.id), item])); const byCode = new Map(docs.map(item => [item.typeCode, item])); for (const meta of metas) { const typeId = toInt(meta?.typeId); const typeCode = String(meta?.typeCode || '').trim(); if (typeId && !byId.has(typeId)) throw new Error(`文档类型不存在或已停用: ${typeId}`); if (!typeId && typeCode && !byCode.has(typeCode)) throw new Error(`文档类型不存在或已停用: ${typeCode}`); } const defaultType = byCode.get('TypeA') || docs[0]; if (!defaultType) throw new Error('没有可用的文档类型'); return { defaultType, byId, byCode }; }; const getDocTypeForMeta = (meta, typeMaps) => { const typeId = toInt(meta?.typeId); if (typeId) return typeMaps.byId.get(typeId); const typeCode = String(meta?.typeCode || '').trim(); if (typeCode) return typeMaps.byCode.get(typeCode); return typeMaps.defaultType; }; const normalizeCreator = (ctx) => { const body = ctx.request.body || {}; return String(body.creator || body.createdBy || ctx.fs?.user?.name || ctx.state?.user?.name || '').trim() || null; }; const normalizeUserId = (ctx) => { const body = ctx.request.body || {}; return toInt(body.userId || body.externalUserId || ctx.state?.externalUserId || ctx.state?.user?.id); }; const resolveAnalyticsUserId = (ctx) => { const user = ctx?.fs?.curUser?.userInfo || {}; const pepUserId = user.pepUserId || user.pep_user_id || user.pepId || user.pep_id; return String(ctx?.fs?.userIdMapping?.internalUserId || pepUserId || user.id || user.userId || '').trim().slice(0, 64) || null; }; const getTaskBlock = async (ctx, { taskId, fileId, blockId }) => { const { models } = ctx.app.fs.dc; const block = await models.PlagiarismCheckFileBlocks.findOne({ where: { id: blockId, taskId, fileId }, }); if (!block) throw new Error('解析块不存在或不属于当前任务文件'); return block; }; const ensureTaskCanRecheck = async (ctx, taskId) => { const { models } = ctx.app.fs.dc; const task = await models.PlagiarismCheckTasks.findByPk(taskId, { raw: true }); if (!task) throw new Error('查重任务不存在'); const files = await models.PlagiarismCheckFiles.findAll({ where: { taskId }, order: [['sortOrder', 'ASC'], ['id', 'ASC']], raw: true, }); if (!files.length) throw new Error('查重任务不存在有效文件'); const invalidFiles = files.filter(file => file.parseStatus !== 'parsed'); if (invalidFiles.length) { throw new Error(`任务存在未完成解析文件: ${invalidFiles.map(file => file.originalName).join(', ')}`); } return { task, files }; }; const ensureTextBlockEditable = (block) => { const blockType = String(block?.blockType || '').trim().toLowerCase(); if (!TEXT_BLOCK_TYPES.has(blockType)) { throw new Error(`当前块类型 ${blockType || 'unknown'} 不支持编辑或降重`); } }; const recalculateFileStats = async (ctx, { taskId, fileId, transaction }) => { const { models } = ctx.app.fs.dc; const blocks = await models.PlagiarismCheckFileBlocks.findAll({ where: { taskId, fileId }, order: [['blockIndex', 'ASC'], ['id', 'ASC']], raw: true, transaction, }); await models.PlagiarismCheckFiles.update({ blockCount: blocks.length, textLength: blocks.reduce((sum, block) => sum + Number(block.textLength || 0), 0), updatedAt: now(), }, { where: { id: fileId, taskId }, transaction }); }; const clearTaskCheckResults = async (ctx, taskId, transaction) => { const { models } = ctx.app.fs.dc; await models.PlagiarismCheckMatches.destroy({ where: { taskId }, transaction }); await models.PlagiarismCheckFileSummaries.destroy({ where: { taskId }, transaction }); }; const triggerTaskRecheck = async (ctx, taskId) => { setImmediate(() => { runPlagiarismCheck(ctx, taskId).catch(error => { ctx.logger?.log?.(error); }); }); }; const buildEditableBlockResponse = (blockLike) => { const structureInfo = blockLike?.structureInfo && typeof blockLike.structureInfo === 'object' ? blockLike.structureInfo : {}; return { id: blockLike.id, taskId: blockLike.taskId, fileId: blockLike.fileId, blockIndex: blockLike.blockIndex, blockType: blockLike.blockType, htmlContent: blockLike.htmlContent, plainText: blockLike.plainText, textLength: Number(blockLike.textLength || 0), structureInfo, edited: Boolean(structureInfo.edited), editSource: structureInfo.editSource || null, editedAt: structureInfo.editedAt || null, originalHtmlContent: structureInfo.originalHtmlContent || null, originalPlainText: structureInfo.originalPlainText || null, createdAt: blockLike.createdAt, updatedAt: blockLike.updatedAt, }; }; const buildExportDocumentHtml = (file, blocks) => { const title = escapeHtml(String(file?.originalName || '文档').replace(/\.[^.]+$/, '') || '文档'); const bodyHtml = blocks.map(block => { const blockType = String(block?.blockType || '').trim().toLowerCase(); if (blockType === 'image') { const src = String(block?.structureInfo?.src || '').trim(); if (!src) return ''; const alt = escapeHtml(block?.structureInfo?.alt || ''); const width = toInt(block?.structureInfo?.width); const height = toInt(block?.structureInfo?.height); const attrs = [ `src="${src}"`, alt ? `alt="${alt}"` : '', width ? `width="${width}"` : '', height ? `height="${height}"` : '', 'style="display:block;max-width:100%;margin:12px 0;"', ].filter(Boolean).join(' '); return `

`; } if (blockType === 'table') { return normalizeHtmlFragment(block.htmlContent); } return normalizeHtmlFragment(block.htmlContent); }).filter(Boolean).join('\n'); return [ '', '', '', '', `${title}`, '', '', `${bodyHtml}`, '', ].join(''); }; const callDeDuplicationFastGpt = async (ctx, { text, prompt = '' }) => { const config = ctx.app.fs.config.fastGpt || {}; if (!config.apiUrl) throw new Error('未配置 fastGpt.apiUrl'); if (!config.deDuplicationAppKey) throw new Error('未配置 fastGpt.deDuplicationAppKey'); const normalizedText = normalizePlainText(text); if (!normalizedText) throw new Error('缺少可降重的文本内容'); const extraPrompt = String(prompt || '').trim(); const requestText = [ '原文:', normalizedText, ].filter(Boolean).join('\n'); const res = await superagent .post(`${config.apiUrl}/api/v1/chat/completions`) .send({ stream: false, detail: true, variables: { extraPrompt, }, messages: [ { role: 'user', content: [ { type: 'text', text: requestText, } ], } ], }) .set({ Authorization: `Bearer ${config.deDuplicationAppKey}`, 'Content-Type': 'application/json', }); const content = normalizeFastGptContent(res?.body?.choices?.[0]?.message?.content); if (!content) throw new Error('fastgpt 返回内容为空'); return { content, responseBody: res?.body }; }; const prepareSubmission = async (ctx) => { const files = getUploadedFiles(ctx); if (files.length < MIN_FILE_COUNT || files.length > MAX_FILE_COUNT) { throw new Error(`请上传 ${MIN_FILE_COUNT}-${MAX_FILE_COUNT} 个文件`); } const threshold = toInt(ctx.request.body?.threshold); if (!threshold || threshold < MIN_THRESHOLD || threshold > MAX_THRESHOLD) { throw new Error(`查重阈值必须在 ${MIN_THRESHOLD}-${MAX_THRESHOLD} 字之间`); } const metas = parseJsonArray(ctx.request.body?.fileMetas, 'fileMetas'); if (metas.length !== files.length) { throw new Error('fileMetas 数量必须与上传文件数量一致'); } const typeMaps = await resolveDocTypes(ctx.app.fs.dc.models, metas); const prepared = []; for (let index = 0; index < files.length; index += 1) { const file = files[index]; const meta = metas[index] || {}; const basic = ensureBasicFile(file, index); const rawBuffer = await readFileBuffer(file); const { buffer, decrypted } = await decryptIfNeeded(ctx, rawBuffer, basic.originalName); if (buffer.length > MAX_FILE_SIZE) { throw new Error(`文件 ${basic.originalName} 解密后超过 50MB 限制`); } validateMagicByExt(buffer, basic.fileExt, basic.originalName); const docType = getDocTypeForMeta(meta, typeMaps); prepared.push({ ...basic, mimeType: file.mimetype || getMimeByExt(basic.fileExt), clientFileId: meta.clientFileId ? String(meta.clientFileId) : file.uid || null, docType, buffer, decrypted, }); } return { files: prepared, threshold }; }; const cleanupUploadedTempFiles = async (files) => { await Promise.all(files.map(file => { if (!file?.path) return Promise.resolve(); return fsPromises.unlink(file.path).catch(() => { }); })); }; const createTask = async (ctx) => { const uploadFiles = getUploadedFiles(ctx); let prepared; try { prepared = await prepareSubmission(ctx); } finally { await cleanupUploadedTempFiles(uploadFiles); } const { models } = ctx.app.fs.dc; const transaction = await ctx.app.fs.dc.orm.transaction(); try { const upload = createQiniuUploader(ctx); const task = await models.PlagiarismCheckTasks.create({ threshold: prepared.threshold, status: 'parsing', statusText: STATUS_TEXT.parsing, fileCount: prepared.files.length, parsedFileCount: 0, failedFileCount: 0, creator: normalizeCreator(ctx), userId: normalizeUserId(ctx), analyticsUserId: resolveAnalyticsUserId(ctx), startedAt: now(), extra: { phase: 'parse', }, }, { transaction, returning: true }); const createdFiles = []; for (let index = 0; index < prepared.files.length; index += 1) { const item = prepared.files[index]; const key = [ 'ai-query/plagiarism-check/source', new Date().toISOString().slice(0, 10).replace(/-/g, ''), String(task.id), `${index + 1}-${randomUUID()}-${safeQiniuKeyPart(item.originalName)}`, ].join('/'); const uploaded = await upload({ key, buffer: item.buffer, mimeType: getMimeByExt(item.fileExt), }); const created = await models.PlagiarismCheckFiles.create({ taskId: task.id, docTypeId: item.docType?.id || null, clientFileId: item.clientFileId, originalName: item.originalName, fileExt: item.fileExt, mimeType: item.mimeType, fileSize: item.fileSize, qiniuKey: uploaded.qiniuKey, fileUrl: uploaded.fileUrl, parseStatus: 'pending', sortOrder: index, metadata: { decrypted: item.decrypted, docTypeCode: item.docType?.typeCode || null, docTypeName: item.docType?.typeName || null, }, }, { transaction, returning: true }); createdFiles.push({ dbFile: created, buffer: item.buffer, fileExt: item.fileExt, originalName: item.originalName, }); } await transaction.commit(); setImmediate(() => { parseTaskFiles(ctx, Number(task.id), createdFiles).catch(error => { ctx.logger?.log?.(error); }); }); return { taskId: task.id, status: 'parsing' }; } catch (error) { await transaction.rollback(); throw error; } }; const updateTaskProgress = async (ctx, taskId) => { const { models } = ctx.app.fs.dc; const files = await models.PlagiarismCheckFiles.findAll({ where: { taskId }, raw: true, }); const parsed = files.filter(item => item.parseStatus === 'parsed').length; const failed = files.filter(item => item.parseStatus === 'parse_failed').length; const finished = parsed + failed === files.length; const status = !finished ? 'parsing' : (failed > 0 ? 'parse_failed' : 'parsed'); const errorMessage = failed ? files.filter(item => item.parseStatus === 'parse_failed') .map(item => `${item.originalName}: ${item.parseError || '解析失败'}`) .join('\n') : null; await models.PlagiarismCheckTasks.update({ status, statusText: STATUS_TEXT[status], parsedFileCount: parsed, failedFileCount: failed, errorMessage, finishedAt: finished ? now() : null, updatedAt: now(), }, { where: { id: taskId } }); return { status, parsed, failed, finished }; }; const loadTaskCheckData = async (ctx, taskId) => { const { models } = ctx.app.fs.dc; const task = await models.PlagiarismCheckTasks.findByPk(taskId, { raw: true }); if (!task) throw new Error('task not found'); const files = await models.PlagiarismCheckFiles.findAll({ where: { taskId }, order: [['sortOrder', 'ASC'], ['id', 'ASC']], raw: true, }); const blocks = await models.PlagiarismCheckFileBlocks.findAll({ where: { taskId }, order: [['fileId', 'ASC'], ['blockIndex', 'ASC']], raw: true, }); const blocksByFile = new Map(); blocks.forEach(block => { const fileId = Number(block.fileId); if (!blocksByFile.has(fileId)) blocksByFile.set(fileId, []); blocksByFile.get(fileId).push(block); }); return { task, files, blocksByFile }; }; const buildMatchRows = ({ taskId, files, blocksByFile, threshold }) => { const rows = []; for (let leftFileIndex = 0; leftFileIndex < files.length - 1; leftFileIndex += 1) { for (let rightFileIndex = leftFileIndex + 1; rightFileIndex < files.length; rightFileIndex += 1) { const leftFile = files[leftFileIndex]; const rightFile = files[rightFileIndex]; const leftBlocks = blocksByFile.get(Number(leftFile.id)) || []; const rightBlocks = blocksByFile.get(Number(rightFile.id)) || []; const sameType = Number(leftFile.docTypeId || 0) === Number(rightFile.docTypeId || 0); const matchType = sameType ? 'same_type' : 'cross_type'; const color = sameType ? 'red' : 'yellow'; for (const leftBlock of leftBlocks) { if (!TEXT_BLOCK_TYPES.has(String(leftBlock.blockType || '').trim().toLowerCase())) continue; for (const rightBlock of rightBlocks) { if (!TEXT_BLOCK_TYPES.has(String(rightBlock.blockType || '').trim().toLowerCase())) continue; const blockMatches = findBlockMatches(leftBlock, rightBlock, threshold); blockMatches.forEach(match => { rows.push({ taskId, leftFileId: leftFile.id, rightFileId: rightFile.id, leftBlockId: leftBlock.id, rightBlockId: rightBlock.id, leftBlockIndex: leftBlock.blockIndex, rightBlockIndex: rightBlock.blockIndex, leftStart: match.leftStart, leftEnd: match.leftEnd, rightStart: match.rightStart, rightEnd: match.rightEnd, leftRawStart: match.leftRawStart, leftRawEnd: match.leftRawEnd, rightRawStart: match.rightRawStart, rightRawEnd: match.rightRawEnd, matchText: match.matchText, matchLength: match.matchLength, matchType, color, metadata: { leftFileIndex, rightFileIndex, }, }); }); } } } } return rows; }; const createEmptySummaryState = (files, blocksByFile) => { const state = new Map(); files.forEach(file => { const blocks = (blocksByFile.get(Number(file.id)) || []).filter(block => TEXT_BLOCK_TYPES.has(String(block.blockType || '').trim().toLowerCase()) ); const totalChars = blocks.reduce((sum, block) => sum + normalizeCompareText(block.plainText).compareText.length, 0); state.set(Number(file.id), { file, totalChars, intervals: [], repeatedBlocks: new Set(), matchCount: 0, sameTypeMatchCount: 0, crossTypeMatchCount: 0, }); }); return state; }; const buildSummaryRows = ({ taskId, files, blocksByFile, matches }) => { const state = createEmptySummaryState(files, blocksByFile); matches.forEach(match => { const left = state.get(Number(match.leftFileId)); const right = state.get(Number(match.rightFileId)); const sameType = match.matchType === 'same_type'; if (left) { left.intervals.push({ start: match.leftStart, end: match.leftEnd, blockId: Number(match.leftBlockId) }); left.repeatedBlocks.add(Number(match.leftBlockId)); left.matchCount += 1; if (sameType) left.sameTypeMatchCount += 1; else left.crossTypeMatchCount += 1; } if (right) { right.intervals.push({ start: match.rightStart, end: match.rightEnd, blockId: Number(match.rightBlockId) }); right.repeatedBlocks.add(Number(match.rightBlockId)); right.matchCount += 1; if (sameType) right.sameTypeMatchCount += 1; else right.crossTypeMatchCount += 1; } }); return Array.from(state.values()).map(item => { const intervalsByBlock = new Map(); item.intervals.forEach(interval => { if (!intervalsByBlock.has(interval.blockId)) intervalsByBlock.set(interval.blockId, []); intervalsByBlock.get(interval.blockId).push(interval); }); let repeatedChars = 0; intervalsByBlock.forEach(intervals => { repeatedChars += sumIntervals(intervals); }); const similarityPercent = item.totalChars ? toPercent(repeatedChars / item.totalChars) : 0; return { taskId, fileId: item.file.id, totalChars: item.totalChars, repeatedChars, repeatedBlockCount: item.repeatedBlocks.size, matchCount: item.matchCount, sameTypeMatchCount: item.sameTypeMatchCount, crossTypeMatchCount: item.crossTypeMatchCount, similarityPercent, metadata: {}, }; }); }; const saveCheckResults = async (ctx, taskId, matchRows, summaryRows) => { const { models } = ctx.app.fs.dc; const transaction = await ctx.app.fs.dc.orm.transaction(); try { await models.PlagiarismCheckMatches.destroy({ where: { taskId }, transaction }); await models.PlagiarismCheckFileSummaries.destroy({ where: { taskId }, transaction }); if (matchRows.length) await models.PlagiarismCheckMatches.bulkCreate(matchRows, { transaction }); if (summaryRows.length) await models.PlagiarismCheckFileSummaries.bulkCreate(summaryRows, { transaction }); const totalChars = summaryRows.reduce((sum, row) => sum + row.totalChars, 0); const repeatedChars = summaryRows.reduce((sum, row) => sum + row.repeatedChars, 0); const repeatedBlockCount = summaryRows.reduce((sum, row) => sum + row.repeatedBlockCount, 0); const similarityPercent = totalChars ? toPercent(repeatedChars / totalChars) : 0; await models.PlagiarismCheckTasks.update({ status: 'checked', statusText: STATUS_TEXT.checked, errorMessage: null, finishedAt: now(), updatedAt: now(), extra: { phase: 'check', matchCount: matchRows.length, repeatedBlockCount, repeatedChars, totalChars, similarityPercent, }, }, { where: { id: taskId }, transaction }); await transaction.commit(); return { matchCount: matchRows.length, repeatedBlockCount, repeatedChars, totalChars, similarityPercent }; } catch (error) { await transaction.rollback(); throw error; } }; const runPlagiarismCheck = async (ctx, taskId) => { const { models } = ctx.app.fs.dc; await models.PlagiarismCheckTasks.update({ status: 'checking', statusText: STATUS_TEXT.checking, errorMessage: null, updatedAt: now(), }, { where: { id: taskId } }); try { const { task, files, blocksByFile } = await loadTaskCheckData(ctx, taskId); if (!files.length) throw new Error('no files to check'); const notParsed = files.filter(file => file.parseStatus !== 'parsed'); if (notParsed.length) { throw new Error(`files not parsed: ${notParsed.map(file => file.originalName).join(', ')}`); } const threshold = Number(task.threshold); const matchRows = buildMatchRows({ taskId, files, blocksByFile, threshold }); const summaryRows = buildSummaryRows({ taskId, files, blocksByFile, matches: matchRows }); const result = await saveCheckResults(ctx, taskId, matchRows, summaryRows); await reportBusinessCall({ ctx, applicationId: 'beta-plagiarism-check', eventId: `plagiarism-check:${taskId}:completed`, traceId: `plagiarism-check:${taskId}`, userId: task.analyticsUserId || '', reportContext: { taskId }, }); return result; } catch (error) { await models.PlagiarismCheckTasks.update({ status: 'check_failed', statusText: STATUS_TEXT.check_failed, errorMessage: error?.message || String(error), finishedAt: now(), updatedAt: now(), }, { where: { id: taskId } }); throw error; } }; const updateBlockContent = async (ctx, { taskId, fileId, blockId, htmlContent, plainText, editSource }) => { const normalizedHtml = String(htmlContent || '').trim(); if (!normalizedHtml) throw new Error('缺少参数: htmlContent'); const block = await getTaskBlock(ctx, { taskId, fileId, blockId }); ensureTextBlockEditable(block); await ensureTaskCanRecheck(ctx, taskId); const nextPlainText = normalizePlainText( plainText === undefined || plainText === null || plainText === '' ? htmlToPlainText(normalizedHtml) : plainText ); if (!nextPlainText) throw new Error('编辑后的内容不能为空'); const sameContent = normalizedHtml === String(block.htmlContent || '').trim() && nextPlainText === normalizePlainText(block.plainText); if (sameContent) { return { block: buildEditableBlockResponse(block), task: { taskId, status: 'checked', rechecked: false, }, }; } const structureInfo = block.structureInfo && typeof block.structureInfo === 'object' ? { ...block.structureInfo } : {}; if (!structureInfo.originalHtmlContent) { structureInfo.originalHtmlContent = block.htmlContent; } if (!structureInfo.originalPlainText) { structureInfo.originalPlainText = block.plainText; } structureInfo.edited = true; structureInfo.editedAt = now().toISOString(); structureInfo.editSource = normalizeEditSource(editSource); const transaction = await ctx.app.fs.dc.orm.transaction(); try { await clearTaskCheckResults(ctx, taskId, transaction); await block.update({ htmlContent: normalizedHtml, plainText: nextPlainText, textLength: nextPlainText.length, structureInfo, updatedAt: now(), }, { transaction }); await ctx.app.fs.dc.models.PlagiarismCheckTasks.update({ status: 'checking', statusText: STATUS_TEXT.checking, errorMessage: null, finishedAt: null, updatedAt: now(), extra: { phase: 'check', }, }, { where: { id: taskId }, transaction }); await transaction.commit(); } catch (error) { await transaction.rollback(); throw error; } triggerTaskRecheck(ctx, taskId); const reloadedBlock = await getTaskBlock(ctx, { taskId, fileId, blockId }); return { block: buildEditableBlockResponse(reloadedBlock), task: { taskId, status: 'checking', rechecked: true, }, }; }; const deleteFileBlock = async (ctx, { taskId, fileId, blockId }) => { const block = await getTaskBlock(ctx, { taskId, fileId, blockId }); await ensureTaskCanRecheck(ctx, taskId); const transaction = await ctx.app.fs.dc.orm.transaction(); try { await clearTaskCheckResults(ctx, taskId, transaction); await ctx.app.fs.dc.models.PlagiarismCheckFileBlocks.destroy({ where: { id: blockId, taskId, fileId }, transaction, }); await ctx.app.fs.dc.orm.query( ` UPDATE plagiarism_check_file_blocks SET block_index = block_index + 1000000, updated_at = CURRENT_TIMESTAMP WHERE task_id = :taskId AND file_id = :fileId AND block_index > :blockIndex `, { replacements: { taskId, fileId, blockIndex: Number(block.blockIndex), }, type: ctx.app.fs.dc.orm.QueryTypes.UPDATE, transaction, } ); await ctx.app.fs.dc.orm.query( ` UPDATE plagiarism_check_file_blocks SET block_index = block_index - 1000001, updated_at = CURRENT_TIMESTAMP WHERE task_id = :taskId AND file_id = :fileId AND block_index >= :shiftStart `, { replacements: { taskId, fileId, shiftStart: Number(block.blockIndex) + 1000001, }, type: ctx.app.fs.dc.orm.QueryTypes.UPDATE, transaction, } ); await recalculateFileStats(ctx, { taskId, fileId, transaction }); await ctx.app.fs.dc.models.PlagiarismCheckTasks.update({ status: 'checking', statusText: STATUS_TEXT.checking, errorMessage: null, finishedAt: null, updatedAt: now(), extra: { phase: 'check', }, }, { where: { id: taskId }, transaction }); await transaction.commit(); } catch (error) { await transaction.rollback(); throw error; } triggerTaskRecheck(ctx, taskId); }; const updateTaskThreshold = async (ctx, { taskId, threshold }) => { const nextThreshold = toInt(threshold); if (!nextThreshold || nextThreshold < MIN_THRESHOLD || nextThreshold > MAX_THRESHOLD) { throw new Error(`查重阈值必须在 ${MIN_THRESHOLD}-${MAX_THRESHOLD} 字之间`); } const { task } = await ensureTaskCanRecheck(ctx, taskId); const currentThreshold = toInt(task.threshold); if (currentThreshold === nextThreshold) { return { taskId, threshold: currentThreshold, status: task.status, }; } const transaction = await ctx.app.fs.dc.orm.transaction(); try { await clearTaskCheckResults(ctx, taskId, transaction); await ctx.app.fs.dc.models.PlagiarismCheckTasks.update({ threshold: nextThreshold, status: 'checking', statusText: STATUS_TEXT.checking, errorMessage: null, finishedAt: null, updatedAt: now(), extra: { phase: 'check', }, }, { where: { id: taskId }, transaction }); await transaction.commit(); } catch (error) { await transaction.rollback(); throw error; } triggerTaskRecheck(ctx, taskId); return { taskId, threshold: nextThreshold, status: 'checking', }; }; const generateDeduplicatedBlock = async (ctx, { taskId, fileId, blockId, htmlContent, plainText, prompt }) => { const block = await getTaskBlock(ctx, { taskId, fileId, blockId }); ensureTextBlockEditable(block); const sourceHtml = String(htmlContent || '').trim() || String(block.htmlContent || '').trim(); const sourcePlainText = normalizePlainText( plainText === undefined || plainText === null || plainText === '' ? (sourceHtml ? htmlToPlainText(sourceHtml) : block.plainText) : plainText ); if (!sourcePlainText) throw new Error('缺少可降重的文本内容'); const fastgptResult = await callDeDuplicationFastGpt(ctx, { text: sourcePlainText, prompt, }); const finalPlainText = normalizePlainText(fastgptResult.content); if (!finalPlainText) throw new Error('降重结果为空'); const task = await ctx.app.fs.dc.models.PlagiarismCheckTasks.findByPk(taskId, { raw: true }); await reportFastgptResponse({ ctx, applicationId: 'beta-plagiarism-check', actionId: `plagiarism-deduplicate:${taskId}:${fileId}:${blockId}`, responseBody: fastgptResult.responseBody, appKey: ctx.app.fs.config.fastGpt?.deDuplicationAppKey, userId: task?.analyticsUserId || '', reportContext: { taskId, fileId, blockId }, }); const tagName = String(block?.structureInfo?.tagName || (block.blockType === 'heading' ? 'h2' : 'p')).toLowerCase(); const safeTag = /^h[1-6]$/.test(tagName) || tagName === 'p' ? tagName : 'p'; return { taskId, fileId, blockId, htmlContent: `<${safeTag}>${escapeHtml(finalPlainText)}`, plainText: finalPlainText, textLength: finalPlainText.length, editSource: 'llm', basedOnUpdatedAt: block.updatedAt, }; }; const exportTaskFileDocx = async (ctx, { taskId, fileId }) => { const { models } = ctx.app.fs.dc; const file = await models.PlagiarismCheckFiles.findOne({ where: { id: fileId, taskId }, raw: true, }); if (!file) throw new Error('文件不存在或不属于当前任务'); const blocks = await models.PlagiarismCheckFileBlocks.findAll({ where: { taskId, fileId }, order: [['blockIndex', 'ASC'], ['id', 'ASC']], raw: true, }); if (!blocks.length) throw new Error('当前文件暂无可导出内容'); const html = buildExportDocumentHtml(file, blocks); const buffer = await convertHTMLToDOCX(html); return { fileName: `${String(file.originalName || 'document').replace(/\.[^.]+$/, '') || 'document'}-重建版.docx`, buffer, }; }; const parseOneFile = async (ctx, taskId, item) => { const { models } = ctx.app.fs.dc; const fileId = Number(item.dbFile.id); await models.PlagiarismCheckFiles.update({ parseStatus: 'parsing', parseError: null, updatedAt: now(), }, { where: { id: fileId } }); try { const blocks = await parseDocumentToBlocks(ctx, item); if (!blocks.length) throw new Error('未解析到有效文本内容'); const rows = blocks.map(block => ({ taskId, fileId, blockIndex: block.blockIndex, blockType: block.blockType, htmlContent: block.htmlContent, plainText: block.plainText, textLength: block.textLength, structureInfo: block.structureInfo, })); const transaction = await ctx.app.fs.dc.orm.transaction(); try { await models.PlagiarismCheckFileBlocks.destroy({ where: { fileId }, transaction }); await models.PlagiarismCheckFileBlocks.bulkCreate(rows, { transaction }); await models.PlagiarismCheckFiles.update({ parseStatus: 'parsed', parseError: null, blockCount: rows.length, textLength: rows.reduce((sum, row) => sum + row.textLength, 0), updatedAt: now(), }, { where: { id: fileId }, transaction }); await transaction.commit(); } catch (error) { await transaction.rollback(); throw error; } } catch (error) { await models.PlagiarismCheckFiles.update({ parseStatus: 'parse_failed', parseError: error?.message || String(error), updatedAt: now(), }, { where: { id: fileId } }); } }; const parseTaskFiles = async (ctx, taskId, files) => { let latestProgress = null; for (const item of files) { await parseOneFile(ctx, taskId, item); latestProgress = await updateTaskProgress(ctx, taskId); } if (latestProgress?.finished && latestProgress.failed === 0) { await runPlagiarismCheck(ctx, taskId); } }; module.exports = { FILE_STATUS_TEXT, MAX_FILE_SIZE, STATUS_TEXT, blocksFromHtml, blocksFromPlainText, createTask, findBlockMatches, generateDeduplicatedBlock, htmlToPlainText, parseDocumentToBlocks, runPlagiarismCheck, updateBlockContent, updateTaskThreshold, deleteFileBlock, buildEditableBlockResponse, exportTaskFileDocx };