feat(workshop): AI创作弹窗v4-上传需求附件(doc/docx/txt/pdf)→AI打磨携带+创作上下文
* attachment-parser.js 零依赖(node:zlib)txt/docx/pdf纯文本提取;.doc明确400报错
* server.js 新增 POST /api/req/parse-attachment(20MB/30MB/3万字符限);
/api/ai/clarify 入参增 attach 拼【附件素材】段(2万字符)
* data-pages.js confirmAiScript:req-input-box内左侧回形针(新ICON_PATHS paperclip)
+ 隐藏 file input + attach chip 行(paperclip+文件名+字数+×);
空输入但有附件自动发引导语;sendTurn 携带 attach;
buildCreatePrompt 增第5参 attachText→【附件素材】段并入创作 prompt
* style.css:req-attach-row/req-chip/chip-x/req-paper 系列样式;req-input-box padding-left 调
* 端到端 4 路径 API 验证 OK;CDP 9224 冒烟(纯Runtime+DataTransfer 注入 + 持久WS)全绿;
真实 Dify 一轮打磨确认附件诉求已吸收进需求清单
* 测试样例 tmp/attach-tests/{req.txt,req.docx,old.doc}.b64(make_samples.py)
9-02 16:31 已改中立内容,避免冒烟带真实账号名
This commit is contained in:
1 parent
047f2d0b9c
commit
a92d84dcc9
5 files changed
+294
-13
No files matched your search
@@ -0,0 +1,173 @@
|
||||
/* attachment-parser.js — 需求附件解析(零依赖 Node)
|
||||
* 支持:.txt(UTF-8)/ .docx(zip → word/document.xml)/ .pdf(FlateDecode 内容流文本提取)
|
||||
* .doc 老格式(OLE2 复合文档)不支持,返回明确提示
|
||||
* 返回:{ text, chars, warn? } 或 { error }
|
||||
* 说明:附件仅用于需求澄清/创作上下文,非安全执行环境——只做文本提取,不落盘、不渲染
|
||||
*/
|
||||
'use strict';
|
||||
const zlib = require('node:zlib');
|
||||
|
||||
const XML_ENT = { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'" };
|
||||
|
||||
function xmlUnescape(s) {
|
||||
return String(s).replace(/&(#x?[0-9a-fA-F]+|[a-zA-Z]+);/g, (all, code) => {
|
||||
if (code[0] === '#') {
|
||||
const n = code[1] === 'x' || code[1] === 'X' ? parseInt(code.slice(2), 16) : parseInt(code.slice(1), 10);
|
||||
try { return String.fromCodePoint(n); } catch (e) { return ''; }
|
||||
}
|
||||
return XML_ENT[code] !== undefined ? XML_ENT[code] : all;
|
||||
});
|
||||
}
|
||||
|
||||
/* ---------- .txt ---------- */
|
||||
function parseTxt(buf) {
|
||||
const raw = buf.toString('utf8');
|
||||
// 粗略检测:若大量 U+FFFD 替换符,说明不是 UTF-8(常见 GBK 中文 txt)
|
||||
const bad = (raw.match(/\uFFFD/g) || []).length;
|
||||
const warn = bad > 0
|
||||
? '检测到非 UTF-8 编码内容,可能乱码——建议将文档另存为 UTF-8 编码后重传(当前按 UTF-8 读取)'
|
||||
: '';
|
||||
return { text: raw, chars: raw.length, warn };
|
||||
}
|
||||
|
||||
/* ---------- .docx:zip 中央目录定位 → 找 word/document.xml → inflateRaw → 抽 <w:t> ---------- */
|
||||
function parseDocx(buf) {
|
||||
// 1) 找 EOCD(从尾部向前 64KB 内)
|
||||
let eocd = -1;
|
||||
const from = Math.max(0, buf.length - 22 - 65536);
|
||||
for (let i = buf.length - 22; i >= from; i--) {
|
||||
if (buf.readUInt32LE(i) === 0x06054b50) { eocd = i; break; }
|
||||
}
|
||||
if (eocd < 0) return { error: 'docx 解析失败:不是有效的压缩包文件' };
|
||||
const total = buf.readUInt16LE(eocd + 10);
|
||||
let cd = buf.readUInt32LE(eocd + 16);
|
||||
let xml = null;
|
||||
for (let i = 0; i < total && cd + 46 <= buf.length; i++) {
|
||||
if (buf.readUInt32LE(cd) !== 0x02014b50) break; // central directory header 签名
|
||||
const compSize = buf.readUInt32LE(cd + 20);
|
||||
const nameLen = buf.readUInt16LE(cd + 28);
|
||||
const extraLen = buf.readUInt16LE(cd + 30);
|
||||
const commentLen = buf.readUInt16LE(cd + 32);
|
||||
const localOff = buf.readUInt32LE(cd + 42);
|
||||
const fname = buf.slice(cd + 46, cd + 46 + nameLen).toString('utf8');
|
||||
if (fname === 'word/document.xml') {
|
||||
const lhNameLen = buf.readUInt16LE(localOff + 26);
|
||||
const lhExtraLen = buf.readUInt16LE(localOff + 28);
|
||||
const dataStart = localOff + 30 + lhNameLen + lhExtraLen;
|
||||
if (dataStart + compSize > buf.length) return { error: 'docx 解析失败:正文数据越界' };
|
||||
try {
|
||||
xml = zlib.inflateRawSync(buf.slice(dataStart, dataStart + compSize)).toString('utf8');
|
||||
} catch (e) {
|
||||
return { error: 'docx 解析失败:正文解压出错(文件可能损坏)' };
|
||||
}
|
||||
break;
|
||||
}
|
||||
cd += 46 + nameLen + extraLen + commentLen;
|
||||
}
|
||||
if (!xml) return { error: 'docx 解析失败:未找到正文(word/document.xml)' };
|
||||
// 2) 抽文本:段尾 → 换行;w:t 内容保留;其余标签剔除;解码实体
|
||||
const text = xmlUnescape(
|
||||
xml
|
||||
.replace(/<w:tab[^>]*\/>/g, '\t')
|
||||
.replace(/<w:br[^>]*\/>/g, '\n')
|
||||
.replace(/<\/w:p\s*>/g, '\n')
|
||||
.replace(/<w:t(?:\s[^>]*)?>([\s\S]*?)<\/w:t>/g, '$1')
|
||||
.replace(/<[^>]+>/g, '')
|
||||
)
|
||||
.replace(/\n{3,}/g, '\n\n') // 压掉连续空行
|
||||
.replace(/[ \t]+\n/g, '\n')
|
||||
.trim();
|
||||
if (!text) return { error: 'docx 未提取到文本内容(文档可能为空或全为图片)' };
|
||||
return { text, chars: text.length };
|
||||
}
|
||||
|
||||
/* ---------- .pdf:对象流扫描 → FlateDecode 解压 → BT..ET 内 Tj/TJ 文本 ----------
|
||||
* 局限:扫描件(无文本层)无法提取;CID/非 Unicode 编码的中文 PDF 文本可能乱码
|
||||
*/
|
||||
function unescapePdfString(s) {
|
||||
// 处理括号字符串中的转义
|
||||
return String(s)
|
||||
.replace(/\\([nrtbf()\\])/g, (m, ch) => ({ n: '\n', r: '\r', t: '\t', b: '\b', f: '\f', '(': '(', ')': ')', '\\': '\\' })[ch])
|
||||
.replace(/\\([0-7]{1,3})/g, (m, oct) => String.fromCharCode(parseInt(oct, 8)));
|
||||
}
|
||||
|
||||
function parsePdf(buf) {
|
||||
const size = buf.length;
|
||||
const parts = [];
|
||||
const warn = [];
|
||||
let searchFrom = 0;
|
||||
// 逐对象扫描 stream 块(latin1 定位不影响二进制偏移,仅用其索引用 Buffer 切片)
|
||||
const latin = buf.toString('latin1');
|
||||
let idx = latin.indexOf('stream', 0);
|
||||
while (idx >= 0) {
|
||||
// stream 后通常是 \r\n 或 \n
|
||||
let start = idx + 6;
|
||||
while (start < size && (buf[start] === 0x0d || buf[start] === 0x0a)) start++;
|
||||
const endTag = latin.indexOf('endstream', start);
|
||||
if (endTag < 0) break;
|
||||
// 判定该 stream 是否属于内容流(往前找最近的 /Filter /FlateDecode)
|
||||
const before = latin.slice(Math.max(0, idx - 400), idx);
|
||||
const raw = buf.slice(start, endTag);
|
||||
if (/\/Filter\s*\/FlateDecode/.test(before)) {
|
||||
try {
|
||||
const dec = zlib.inflateSync(raw);
|
||||
parts.push(dec.toString('latin1'));
|
||||
} catch (e) { /* 单流失败跳过(图像流常见) */ }
|
||||
}
|
||||
searchFrom = endTag + 9;
|
||||
idx = latin.indexOf('stream', searchFrom);
|
||||
}
|
||||
if (!parts.length) {
|
||||
// 可能有未压缩文本流
|
||||
return { error: 'PDF 未提取到可解析的文本层(可能是扫描件/图片型 PDF,或需压缩流解析)' };
|
||||
}
|
||||
// 在解压内容流里抽 Tj / TJ 文本
|
||||
const out = [];
|
||||
for (const content of parts) {
|
||||
// BT ... ET 之间的文本算子
|
||||
let rest = content;
|
||||
let m;
|
||||
// 简单正则:([…]或<…>)Tj / TJ 数组
|
||||
const re = /(?:\[((?:[^\]\\]|\\.)*)\]|\(((?:[^()\\]|\\.)*)\)|<([0-9A-Fa-f\s]*)>)\s*(Tj|TJ)/g;
|
||||
while ((m = re.exec(rest)) !== null) {
|
||||
if (m[1] !== undefined) { // TJ 数组:多段拼接
|
||||
const arr = m[1];
|
||||
const pieces = [];
|
||||
const pm = /\((?:[^()\\]|\\.)*\)|<([0-9A-Fa-f\s]*)>/g;
|
||||
let q;
|
||||
while ((q = pm.exec(arr)) !== null) {
|
||||
if (q[0][0] === '(') pieces.push(unescapePdfString(q[0].slice(1, -1)));
|
||||
else pieces.push(Buffer.from(q[1].replace(/\s+/g, ''), 'hex').toString('latin1'));
|
||||
}
|
||||
out.push(pieces.join(''));
|
||||
} else if (m[2] !== undefined) { // 单括号 Tj
|
||||
out.push(unescapePdfString(m[2]));
|
||||
} else if (m[3] !== undefined) { // hex Tj
|
||||
out.push(Buffer.from(m[3].replace(/\s+/g, ''), 'hex').toString('latin1'));
|
||||
}
|
||||
}
|
||||
}
|
||||
let text = out.join('')
|
||||
.replace(/[\x00-\x08\x0b\x0c\x0e-\x1f]/g, '') // 去控制字符
|
||||
.replace(/\s{3,}/g, ' ')
|
||||
.trim();
|
||||
if (!text) return { error: 'PDF 未提取到文本(扫描件或加密文档无法读取)' };
|
||||
warn.push('PDF 提取基于文本层;若文档为扫描件/特殊编码(如部分中文 CID 字体),内容可能缺失或乱码');
|
||||
return { text, chars: text.length, warn: warn.join(';') };
|
||||
}
|
||||
|
||||
/* ---------- 主入口 ---------- */
|
||||
function parseAttachment(name, buf) {
|
||||
if (!Buffer.isBuffer(buf) || !buf.length) return { error: '文件内容为空' };
|
||||
const ext = (String(name || '').match(/\.([a-zA-Z0-9]+)$/) || [])[1];
|
||||
if (!ext) return { error: '无法识别文件类型(请使用 .txt / .docx / .pdf 扩展名)' };
|
||||
switch (ext.toLowerCase()) {
|
||||
case 'txt': return parseTxt(buf);
|
||||
case 'docx': return parseDocx(buf);
|
||||
case 'pdf': return parsePdf(buf);
|
||||
case 'doc': return { error: '暂不支持 .doc 老格式(Word 97-2003 二进制),请用 Word/WPS 另存为 .docx 或 .txt 后重传' };
|
||||
default: return { error: '不支持的文件类型 .' + ext.toLowerCase() + '(支持 .txt / .docx / .pdf)' };
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = { parseAttachment };
|
||||
Reference in new issue
Block a user