账号分析产物入库闭环:工作台入库接口+选题读视频解析

- server.js 新增 POST /api/import/account:扫描账号产出目录直写 mcn-plugin.db(视频分析→analysis/source、账号设定→persona、账号数据分析→analysis、短视频表格.xlsx→videos upsert)
- 新增 scripts/export_video_map.py:python 读 xlsx 13列输出 JSON 权威 aweme_id 映射
- /api/ai/topics 增加读最新3条视频解析 JOIN account_videos 拼入 prompt,选题基于真实视频内容
- 修复 aweme_id 匹配:normTitle 删全角标点(xlsx「!」「”」vs 文件夹无标点,4/6 失败根因)+下划线转空格;入库前清 null 脏数据
- mcn-dou-analysis SKILL.md:功能五补充工作台入库方式+注意事项第6条匹配红线
- 验证:俊希 6 analysis+6 source+137 videos 入库,JOIN 全关联,选题端到端返回贴合视频解析
This commit is contained in:
maogeigei committed 2026-09-01 10:12:36 +08:00
1 parent 4e29ddb881
commit 92c93cf66a
5 files changed
+228 -5

No files matched your search

@@ -0,0 +1,59 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
export_video_map.py - 短视频表格.xlsx → 视频映射 JSON(工作台入库接口配套,09-01 新增)
背景:短视频表格.xlsx(功能二产出,13列:序号/视频ID/视频标题/...)包含视频的权威 aweme_id,
工作台 server.js 入库接口用它匹配 视频分析/ 目录下的解析文件,避免仅靠 account_videos 表标题模糊匹配。
用法:
python export_video_map.py <短视频表格.xlsx>
→ stdout 输出 JSON 数组:[{"aweme_id": "...", "title": "...", "url": "...",
"like_count": 123, "comment_count": 4, "share_count": 5, "collect_count": 6,
"play_count": 7, "duration": 8, "publish_time": "...", "tags": "..."}]
"""
import json
import sys
def main():
if len(sys.argv) < 2:
print("usage: export_video_map.py <短视频表格.xlsx>", file=sys.stderr)
sys.exit(1)
xlsx = sys.argv[1]
try:
import openpyxl
except ImportError:
print("[]", file=sys.stderr)
print("[error] openpyxl 不可用", file=sys.stderr)
sys.exit(2)
wb = openpyxl.load_workbook(xlsx, read_only=True, data_only=True)
ws = wb.active
out = []
for row in ws.iter_rows(values_only=True):
if not row or len(row) < 3:
continue
aweme_id, title = row[1], row[2]
if aweme_id is None or not title:
continue
rec = {
"aweme_id": str(aweme_id).strip(),
"title": str(title).strip(),
"url": str(row[3] or "").strip(),
"like_count": row[4],
"comment_count": row[6],
"share_count": row[7],
"collect_count": row[8],
"play_count": row[9],
"duration": row[10],
"publish_time": str(row[11] or "").strip(),
"tags": str(row[12] or "").strip(),
}
out.append(rec)
wb.close()
print(json.dumps(out, ensure_ascii=False))
if __name__ == "__main__":
main()
@@ -7,6 +7,7 @@ const http = require('http');
const fs = require('fs');
const path = require('path');
const os = require('os');
const { execFileSync } = require('child_process');
const { URL } = require('url');
const dsh = require('./dsh-data');
@@ -57,6 +58,129 @@ function resolveSafe(rawPath) {
return null;
}
// ---- 分析产物入库(09-01 新增):扫描账号产出目录 → 写工作台库 mcn-plugin.db ----
// 背景:账号分析技能产出只落盘文件系统未入库(俊希解析/人设卡全缺),工作台选题等读不到视频内容。
// 实现参照 dsh-plugin-mcn lib/imports.js 的 analyses 入库逻辑(dsh 接口 3080/3081 不可用时直写库)。
function resolveAccountDir(name, root) {
const safeName = String(name || '').replace(/[\\/:*?"<>|]/g, '_').trim();
if (!safeName) return null;
const roots = root ? [root] : allowedRoots();
for (const r of roots) {
const dir = path.join(r, safeName);
try { if (fs.statSync(dir).isDirectory()) return dir; } catch (e) { /* 忽略 */ }
}
return null;
}
function importAccountIntoDb(db, account, dir) {
const stats = { video_analysis: 0, video_source: 0, persona: 0, account_analysis: 0, videos_upserted: 0, skipped: [] };
const now = new Date().toISOString();
const insAnalysis = db.prepare(`INSERT INTO account_video_analysis (video_id, aweme_id, analysis_time, content_json, summary) VALUES (?,?,?,?,?)`);
const insSource = db.prepare(`INSERT INTO account_video_source (video_id, aweme_id, analysis_time, content_json, summary) VALUES (?,?,?,?,?)`);
const insPersona = db.prepare(`INSERT INTO account_persona (account_id, analysis_time, content_json, summary) VALUES (?,?,?,?)`);
const insAccAna = db.prepare(`INSERT INTO account_analysis (account_id, analysis_time, content_json, summary) VALUES (?,?,?,?)`);
const acc = db.prepare('SELECT id FROM hot_accounts WHERE account_name=? LIMIT 1').get(account);
const accountId = acc ? acc.id : null;
// 清理历史误插脏数据:09-01 匹配缺陷(标点未规范化)产生 aweme_id=null 且 video_id=null 的无意义记录,重跑入库前先清
db.prepare("DELETE FROM account_video_analysis WHERE aweme_id IS NULL AND video_id IS NULL").run();
db.prepare("DELETE FROM account_video_source WHERE aweme_id IS NULL AND video_id IS NULL").run();
// 规范化标题:去 # 标签 → 删全角/半角标点(xlsx 标题带「!」「“”」而文件夹名无,09-01 匹配失败根因)→ 下划线转空格(对齐文件夹 _标签 结构)→ 空格归一
const normTitle = (s) => String(s || '').split('#')[0].replace(/[!!“”‘’"'`,.…。,、;::;·]/g, '').replace(/[_-]+/g, ' ').replace(/\s+/g, ' ').trim();
// 0) aweme_id 权威映射:短视频表格.xlsx(子进程 python,最高优先级)→ account_videos 表(兜底)
const xlsxList = readXlsxMap(dir);
const xlsxByNorm = {};
for (const r of xlsxList) { if (r.aweme_id && r.title) xlsxByNorm[normTitle(r.title)] = r.aweme_id; }
const titleRows = db.prepare('SELECT aweme_id, video_title FROM account_videos WHERE video_title IS NOT NULL').all();
const dbByNorm = {};
for (const r of titleRows) { if (r.aweme_id && r.video_title) dbByNorm[normTitle(r.video_title)] = String(r.aweme_id); }
// 匹配:精确优先;包含匹配要求双方标题 ≥10 字且按标题长度降序(最具体优先,防短标题误配)
const matchAweme = (folder) => {
const base = normTitle(folder);
if (!base) return null;
if (xlsxByNorm[base]) return xlsxByNorm[base];
if (dbByNorm[base]) return dbByNorm[base];
const cands = [];
for (const [t, aw] of Object.entries(xlsxByNorm)) {
if (base.length >= 10 && t.length >= 10 && (base.includes(t) || t.includes(base))) cands.push([t.length, aw]);
}
if (!cands.length) {
for (const [t, aw] of Object.entries(dbByNorm)) {
if (base.length >= 10 && t.length >= 10 && (base.includes(t) || t.includes(base))) cands.push([t.length, aw]);
}
}
cands.sort((a, b) => b[0] - a[0]);
return cands.length ? cands[0][1] : null;
};
// 0.5) 视频列表入库:短视频表格.xlsx → account_videos(按 account_id+aweme_id 去重,补上工作台视频列表数据)
if (accountId && xlsxList.length) {
const insVideo = db.prepare(`INSERT INTO account_videos (account_id, aweme_id, video_title, video_url, like_count, comment_count, share_count, collect_count, play_count, duration, publish_time, tags, collected_time) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)`);
for (const v of xlsxList) {
const ex = db.prepare('SELECT id FROM account_videos WHERE account_id=? AND aweme_id=? LIMIT 1').get(accountId, v.aweme_id);
if (!ex) {
insVideo.run(accountId, v.aweme_id, v.title, v.url || null, v.like_count ?? null, v.comment_count ?? null, v.share_count ?? null, v.collect_count ?? null, v.play_count ?? null, v.duration ?? null, v.publish_time || null, v.tags || null, now);
stats.videos_upserted++;
}
}
}
// 1) 视频分析目录(功能三 Step5 产出;功能五产出在 视频对标/ 目录,由 dsh 导入机制处理)
const vaDir = path.join(dir, '视频分析');
if (fs.existsSync(vaDir)) {
for (const folder of fs.readdirSync(vaDir)) {
const fdir = path.join(vaDir, folder);
let st; try { st = fs.statSync(fdir); } catch (e) { continue; }
if (!st.isDirectory()) continue;
const awemeId = matchAweme(folder);
const anPath = path.join(fdir, 'analysis.json');
if (fs.existsSync(anPath)) {
const content = fs.readFileSync(anPath, 'utf8');
if (awemeId) db.prepare('DELETE FROM account_video_analysis WHERE aweme_id=?').run(awemeId);
insAnalysis.run(null, awemeId, now, content, content.slice(0, 500));
stats.video_analysis++;
}
const coPath = path.join(fdir, 'content.json');
if (fs.existsSync(coPath)) {
const content = fs.readFileSync(coPath, 'utf8');
if (awemeId) db.prepare('DELETE FROM account_video_source WHERE aweme_id=?').run(awemeId);
insSource.run(null, awemeId, now, content, '');
stats.video_source++;
}
}
}
// 2) 人设卡 + 账号分析(去重,不删旧)
if (accountId) {
const personaPath = path.join(dir, account + '账号设定.md');
if (fs.existsSync(personaPath)) {
const content = fs.readFileSync(personaPath, 'utf8');
const dup = db.prepare('SELECT id FROM account_persona WHERE account_id=? AND content_json=? LIMIT 1').get(accountId, content);
if (!dup) { insPersona.run(accountId, now, content, content.slice(0, 200)); stats.persona++; }
else stats.skipped.push('persona(重复)');
}
const anaPath = path.join(dir, account + '账号数据分析.md');
if (fs.existsSync(anaPath)) {
const content = fs.readFileSync(anaPath, 'utf8');
const dup = db.prepare('SELECT id FROM account_analysis WHERE account_id=? AND content_json=? LIMIT 1').get(accountId, content);
if (!dup) { insAccAna.run(accountId, now, content, content.slice(0, 200)); stats.account_analysis++; }
else stats.skipped.push('account_analysis(重复)');
}
}
return { account, account_id: accountId, dir, xlsx_rows: xlsxList.length, title_map_size: Object.keys(dbByNorm).length, ...stats };
}
// 读短视频表格.xlsx → 视频映射数组(子进程调 python export_video_map.py,openpyxl 读 xlsx;失败返回 [])
function readXlsxMap(dir) {
const xlsxPath = path.join(dir, '短视频表格.xlsx');
if (!fs.existsSync(xlsxPath)) return [];
const py = process.env.PYTHON || 'python';
try {
const out = execFileSync(py, [path.join(ROOT, 'scripts', 'export_video_map.py'), xlsxPath], { encoding: 'utf8', timeout: 20000, windowsHide: true });
const arr = JSON.parse(out);
return Array.isArray(arr) ? arr : [];
} catch (e) {
log('xlsx 视频映射读取失败(用表内匹配兜底): ' + e.message);
return [];
}
}
// ---- 查找账号设定卡 SVG ----
// 产出规范:{产出根}/{达人昵称}/{达人昵称}账号设定卡.svg(功能六产出,与账号设定.md 同目录)
// 查找顺序:规范路径 → 根下直挂 → 账号目录内扫描(防目录名/账号名不一致)
@@ -438,8 +562,8 @@ const server = http.createServer((req, res) => {
try {
const { accountName } = JSON.parse(body || '{}');
if (!accountName || typeof accountName !== 'string') return sendErr(400, '缺少账号名');
// 1) 查账号定位 + 人设摘要(工作台库)
let contentText = '', personaText = '';
// 1) 查账号定位 + 人设摘要 + 最新视频解析(工作台库)
let contentText = '', personaText = '', videoRefText = '';
try {
const { DatabaseSync } = require('node:sqlite');
const db = new DatabaseSync(path.join(ROOT, 'mcn-plugin.db'), { readOnly: true });
@@ -448,6 +572,13 @@ const server = http.createServer((req, res) => {
contentText = String(acc.content || '').slice(0, 500);
const p = db.prepare('SELECT content_json, summary FROM account_persona WHERE account_id=? ORDER BY id DESC LIMIT 1').get(acc.id);
if (p) personaText = String(p.summary || p.content_json || '').slice(0, 500);
// 09-01:读该账号最新视频解析(account_video_analysis 按 aweme_id 关联),让选题基于真实视频内容
try {
const vrows = db.prepare(
'SELECT va.content_json AS c FROM account_video_analysis va JOIN account_videos v ON v.aweme_id = va.aweme_id WHERE v.account_id=? ORDER BY va.analysis_time DESC LIMIT 3'
).all(acc.id);
if (vrows.length) videoRefText = vrows.map((x) => String(x.c || '')).join('\n---\n').slice(0, 3000);
} catch (e) { /* 视频解析缺失不阻塞 */ }
}
db.close();
} catch (e) { /* 库不可用不阻塞,仍可生成 */ }
@@ -462,10 +593,11 @@ const server = http.createServer((req, res) => {
} catch (e) { /* 脚本不存在则走 env */ }
}
if (!apiKey) return sendErr(500, '未配置 DIFY_MCN_CYLG_KEY(环境变量或技能脚本内置 KEY)');
// 3) 构造选题 query(参考 references/创作流程/5_生成短视频选题.md 方法论)
const query = '你是短视频选题策划师,请为达人「' + accountName + '」结合账号设定生成 3 个爆款选题方案。\n\n'
// 3) 构造选题 query(参考 references/创作流程/5_生成短视频选题.md 方法论;09-01 追加视频解析参考)
const query = '你是短视频选题策划师,请为达人「' + accountName + '」结合账号设定与最新视频内容生成 3 个爆款选题方案。\n\n'
+ '【达人信息】\n' + (contentText || '(无账号定位信息)') + '\n'
+ (personaText ? '【人设摘要】\n' + personaText + '\n' : '')
+ (videoRefText ? '【该账号最近视频解析(选题须参考的真实视频内容)】\n' + videoRefText + '\n' : '')
+ '\n要求:\n'
+ '1. 选题要跳出账号核心行为框架,但保留账号结构性符号(固定机制/口头禅等)\n'
+ '2. 每个选题一句话说清主题+切入角度,标注目标体感(如治愈感/爽感/获得感)\n'
@@ -495,6 +627,29 @@ const server = http.createServer((req, res) => {
return;
}
// 分析产物同步入库(09-01 新增):扫描账号产出目录 → 写工作台库 mcn-plugin.db
if (p === '/api/import/account' && req.method === 'POST') {
let body = '';
req.on('data', (c) => { body += c; if (body.length > 1e6) req.destroy(); });
req.on('end', async () => {
try {
const { account, root } = JSON.parse(body || '{}');
if (!account || typeof account !== 'string') return sendErr(400, '缺少账号名');
const accountDir = resolveAccountDir(account, root);
if (!accountDir) return sendErr(404, '账号产出目录不存在: ' + account);
const { DatabaseSync } = require('node:sqlite');
const db = new DatabaseSync(path.join(ROOT, 'mcn-plugin.db'));
const res = importAccountIntoDb(db, account, accountDir);
db.close();
log('分析产物入库: ' + account + ' → ' + JSON.stringify(res));
return dshOk(200, res);
} catch (e) {
return sendErr(500, '入库失败: ' + e.message);
}
});
return;
}
// ---------- 静态文件 ----------
let filePath = path.join(PUBLIC_DIR, p === '/' ? 'index.html' : p);
// 防穿越