技能三副本迭代:路径重构+输出边界铁律+开场口径区分+审计修复
- 路径重构:脚本创作 V1.0/Lite1.0 扁平化(去技能文件夹层),分镜技能迁移至 短视频提示词生成/V1.0,旧路径清理 - 输出边界铁律固化:创作流程规范.md 新增「交付物输出边界(通用铁律)」+ S9 输出模板新增「输出边界」块(验收口径=执行约束非输出内容) - 开场钩子铁律三层固化:00_开场/06_编导终审/S9 相关规则 + S5 预设钩子口径区分(前3秒=呈现窗口,3-7秒=开场段总时长) - 帮助文档:新增「从账号分析到脚本创作」跨技能章节(你这样问|我会做什么风格) - mcn-dou-analysis 内嵌副本:红线边界声明+人设卡术语更新(设定分析说明)
This commit is contained in:
1 parent
37eb75501b
commit
b3685504d7
561 files changed
+47048
-1290
No files matched your search
@@ -0,0 +1,311 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
excel_tool.py — 统一Excel工具(筛选/查询/导出)
|
||||
|
||||
三种模式:
|
||||
1. select-top6 筛选TOP视频(最近三个月 + ≤15分钟 + 点赞分享降序,默认取6条,可用 --count 指定)
|
||||
2. query 按标题关键词查询视频数据(含三比率计算)
|
||||
3. export-selected 批量导出选中视频数据(含三比率计算)
|
||||
|
||||
Excel列结构(13列标准格式):
|
||||
0:序号 1:视频ID 2:视频标题 3:视频地址 4:点赞数 5:点赞(显示)
|
||||
6:评论数 7:分享数 8:收藏数 9:播放量 10:视频时长 11:发布时间 12:标签
|
||||
|
||||
三比率定义(账号基准均值参考,来源于历史账号分析实测均值;仅作横向对比参考,不代表当前分析账号):
|
||||
享赞比 = 分享数 / 点赞数 × 100% (基准 19.6%)
|
||||
评赞比 = 评论数 / 点赞数 × 100% (基准 1.85%)
|
||||
藏赞比 = 收藏数 / 点赞数 × 100% (基准 7.5%)
|
||||
|
||||
用法示例:
|
||||
# 筛选TOP6(默认)
|
||||
python excel_tool.py select-top6 "C:/path/to/短视频表格.xlsx" [--output top6.json]
|
||||
# 指定数量
|
||||
python excel_tool.py select-top6 "C:/path/to/短视频表格.xlsx" --count 6 [--output top6.json]
|
||||
|
||||
# 按关键词查询(逗号分隔多关键词,OR逻辑)
|
||||
python excel_tool.py query "C:/path/to/短视频表格.xlsx" --keywords "咬下苹果,露水,划船上岸"
|
||||
|
||||
# 查询所有视频并按点赞降序
|
||||
python excel_tool.py query "C:/path/to/短视频表格.xlsx" --all
|
||||
|
||||
# 批量导出选中视频(从文件读取标题关键词,每行一个)
|
||||
python excel_tool.py export-selected "C:/path/to/短视频表格.xlsx" --titles-file selected.txt --output result.json
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import json
|
||||
import datetime
|
||||
import argparse
|
||||
|
||||
try:
|
||||
import openpyxl
|
||||
except ImportError:
|
||||
print("ERROR: openpyxl not installed. Run: pip install openpyxl", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 通用工具函数
|
||||
# ============================================================
|
||||
|
||||
HEADERS = ['序号', '视频ID', '视频标题', '视频地址', '点赞数', '点赞(显示)',
|
||||
'评论数', '分享数', '收藏数', '播放量', '视频时长', '发布时间', '标签']
|
||||
|
||||
# 账号基准均值
|
||||
BENCHMARK = {'享赞比': 19.6, '评赞比': 1.85, '藏赞比': 7.5}
|
||||
|
||||
|
||||
def parse_duration(val):
|
||||
"""解析视频时长字段,返回秒数。支持 HH:MM:SS / MM:SS / 秒 / 毫秒 / 中文单位(N秒 / N分 / N分M秒)格式。"""
|
||||
if isinstance(val, (int, float)):
|
||||
return float(val) / 1000 if val > 1000 else float(val)
|
||||
if isinstance(val, str):
|
||||
val = val.strip()
|
||||
# 中文单位
|
||||
if '秒' in val or '分' in val:
|
||||
total = 0.0
|
||||
if '分' in val:
|
||||
m_part = val.split('分')[0]
|
||||
total += int(m_part) * 60 if m_part.strip().isdigit() else 0
|
||||
s_part = val.split('分')[1].replace('秒', '').strip()
|
||||
total += float(s_part) if s_part else 0
|
||||
else:
|
||||
total = float(val.replace('秒', ''))
|
||||
return total
|
||||
parts = val.split(':')
|
||||
if len(parts) == 3:
|
||||
return int(parts[0]) * 3600 + int(parts[1]) * 60 + int(parts[2])
|
||||
elif len(parts) == 2:
|
||||
return int(parts[0]) * 60 + int(parts[1])
|
||||
try:
|
||||
return float(val)
|
||||
except ValueError:
|
||||
return 99999
|
||||
return 99999
|
||||
|
||||
|
||||
def compute_ratios(likes, comments, shares, saves):
|
||||
"""计算三比率,返回字典。"""
|
||||
likes = likes or 0
|
||||
return {
|
||||
'享赞比': round((shares or 0) / likes * 100, 1) if likes else 0,
|
||||
'评赞比': round((comments or 0) / likes * 100, 2) if likes else 0,
|
||||
'藏赞比': round((saves or 0) / likes * 100, 1) if likes else 0,
|
||||
}
|
||||
|
||||
|
||||
def format_ratio_comparison(ratio_name, value):
|
||||
"""格式化比率与基准的比较标记。"""
|
||||
bench = BENCHMARK.get(ratio_name, 0)
|
||||
if value > bench * 1.3:
|
||||
return f'{ratio_name}{value}%^^'
|
||||
elif value > bench * 1.1:
|
||||
return f'{ratio_name}{value}%^'
|
||||
elif value < bench * 0.7:
|
||||
return f'{ratio_name}{value}%vv'
|
||||
elif value < bench * 0.9:
|
||||
return f'{ratio_name}{value}%v'
|
||||
else:
|
||||
return f'{ratio_name}{value}%~'
|
||||
|
||||
|
||||
def load_excel_rows(excel_path):
|
||||
"""加载Excel,返回数据行列表(跳过表头)。"""
|
||||
if not os.path.exists(excel_path):
|
||||
print(f'ERROR: file not found: {excel_path}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
wb = openpyxl.load_workbook(excel_path, read_only=True)
|
||||
ws = wb.active
|
||||
rows = list(ws.iter_rows(min_row=2, values_only=True))
|
||||
wb.close()
|
||||
return rows
|
||||
|
||||
|
||||
def row_to_dict(row):
|
||||
"""将Excel行转为字典。"""
|
||||
return dict(zip(HEADERS, [str(v) if v is not None else '' for v in row]))
|
||||
|
||||
|
||||
def row_to_summary(row):
|
||||
"""将Excel行转为含三比率的摘要字典。"""
|
||||
likes = row[4] or 0
|
||||
comments = row[6] or 0
|
||||
shares = row[7] or 0
|
||||
saves = row[8] or 0
|
||||
return {
|
||||
'标题': str(row[2]) if row[2] else '',
|
||||
'点赞': likes,
|
||||
'评论': comments,
|
||||
'分享': shares,
|
||||
'收藏': saves,
|
||||
'播放': row[9] or 0,
|
||||
'时长': f'{parse_duration(row[10]):.0f}s',
|
||||
'发布': str(row[11])[:10] if row[11] else '',
|
||||
'标签': str(row[12]) if row[12] else '',
|
||||
**compute_ratios(likes, comments, shares, saves),
|
||||
}
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 1. select-top6
|
||||
# ============================================================
|
||||
|
||||
def select_top6(excel_path, count=6):
|
||||
"""筛选TOP视频(默认取6条,可用 count 指定数量)。"""
|
||||
rows = load_excel_rows(excel_path)
|
||||
|
||||
# 时间过滤:最近三个月
|
||||
three_months_ago = datetime.datetime.now() - datetime.timedelta(days=90)
|
||||
recent = [
|
||||
r for r in rows
|
||||
if r[11] and datetime.datetime.strptime(str(r[11])[:10], '%Y-%m-%d') >= three_months_ago
|
||||
]
|
||||
|
||||
# 时长过滤:≤15分钟(900秒)
|
||||
recent = [r for r in recent if parse_duration(r[10]) <= 900]
|
||||
|
||||
# 排序:点赞数 + 分享数 降序
|
||||
recent.sort(key=lambda r: (r[4] or 0) + (r[7] or 0), reverse=True)
|
||||
|
||||
# 取前 count 条
|
||||
top = recent[:count]
|
||||
return [row_to_dict(r) for r in top]
|
||||
|
||||
|
||||
def cmd_select_top6(args):
|
||||
"""select-top6 命令处理。"""
|
||||
excel_path = os.path.expanduser(args.excel_path)
|
||||
top6 = select_top6(excel_path, count=args.count)
|
||||
|
||||
print(f'筛选完成,共 {len(top6)} 条视频:')
|
||||
for i, v in enumerate(top6):
|
||||
print(f' {i+1}. {v["视频标题"][:30]}... | 赞:{v["点赞数"]} 转:{v["分享数"]} | {v["视频时长"]} | {v["发布时间"]}')
|
||||
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding='utf-8') as f:
|
||||
json.dump(top6, f, ensure_ascii=False, indent=2)
|
||||
print(f'\n已保存到: {args.output}')
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 2. query
|
||||
# ============================================================
|
||||
|
||||
def cmd_query(args):
|
||||
"""query 命令处理:按关键词查询或全量查询。"""
|
||||
excel_path = os.path.expanduser(args.excel_path)
|
||||
rows = load_excel_rows(excel_path)
|
||||
|
||||
if args.all:
|
||||
# 全量查询,按点赞降序
|
||||
matched = sorted(rows, key=lambda r: r[4] or 0, reverse=True)
|
||||
else:
|
||||
# 关键词过滤(OR逻辑)
|
||||
keywords = [k.strip() for k in args.keywords.split(',')]
|
||||
matched = [r for r in rows if r[2] and any(k in str(r[2]) for k in keywords)]
|
||||
|
||||
print(f'查询到 {len(matched)} 条视频:')
|
||||
print(f'{"序号":<4} | {"标题":<25} | {"点赞":>8} | {"评论":>6} | {"分享":>6} | {"收藏":>6} | {"时长":>6} | {"发布":>12} | 三比率')
|
||||
print('-' * 120)
|
||||
|
||||
for row in matched:
|
||||
summary = row_to_summary(row)
|
||||
ratios = f'{format_ratio_comparison("享赞比", summary["享赞比"])} / {format_ratio_comparison("评赞比", summary["评赞比"])} / {format_ratio_comparison("藏赞比", summary["藏赞比"])}'
|
||||
print(f'{row[0]:<4} | {summary["标题"][:25]:<25} | {summary["点赞"]:>8} | {summary["评论"]:>6} | {summary["分享"]:>6} | {summary["收藏"]:>6} | {summary["时长"]:>6} | {summary["发布"]:>12} | {ratios}')
|
||||
|
||||
if args.output:
|
||||
result = [row_to_summary(r) for r in matched]
|
||||
with open(args.output, 'w', encoding='utf-8') as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
print(f'\n已保存到: {args.output}')
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 3. export-selected
|
||||
# ============================================================
|
||||
|
||||
def cmd_export_selected(args):
|
||||
"""export-selected 命令处理:批量导出选中视频。"""
|
||||
excel_path = os.path.expanduser(args.excel_path)
|
||||
rows = load_excel_rows(excel_path)
|
||||
|
||||
# 从文件读取标题关键词
|
||||
with open(args.titles_file, 'r', encoding='utf-8') as f:
|
||||
keywords = [line.strip() for line in f if line.strip()]
|
||||
|
||||
# 匹配
|
||||
matched = []
|
||||
for row in rows:
|
||||
if row[2]:
|
||||
for k in keywords:
|
||||
if k in str(row[2]):
|
||||
matched.append(row)
|
||||
break
|
||||
|
||||
# 按点赞降序
|
||||
matched.sort(key=lambda r: r[4] or 0, reverse=True)
|
||||
|
||||
result = [row_to_summary(r) for r in matched]
|
||||
|
||||
print(f'导出 {len(result)} 条视频:')
|
||||
for i, s in enumerate(result):
|
||||
print(f' {i+1}. {s["标题"][:25]}... | 赞:{s["点赞"]} | {s["时长"]} | {s["发布"]}')
|
||||
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding='utf-8') as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
print(f'\n已保存到: {args.output}')
|
||||
else:
|
||||
# 默认输出到终端
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Main
|
||||
# ============================================================
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description='统一Excel工具(筛选/查询/导出)',
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter
|
||||
)
|
||||
subparsers = parser.add_subparsers(dest='action', help='操作模式')
|
||||
|
||||
# select-top6
|
||||
p_top6 = subparsers.add_parser('select-top6', help='筛选TOP视频(默认取6条)')
|
||||
p_top6.add_argument('excel_path', help='短视频表格.xlsx 路径')
|
||||
p_top6.add_argument('--count', '-n', type=int, default=6, help='筛选数量(默认6条)')
|
||||
p_top6.add_argument('--output', '-o', default=None, help='输出JSON路径')
|
||||
|
||||
# query
|
||||
p_query = subparsers.add_parser('query', help='按标题关键词查询视频数据')
|
||||
p_query.add_argument('excel_path', help='短视频表格.xlsx 路径')
|
||||
p_query.add_argument('--keywords', '-k', default=None, help='标题关键词(逗号分隔,OR逻辑)')
|
||||
p_query.add_argument('--all', action='store_true', help='查询全部视频')
|
||||
p_query.add_argument('--output', '-o', default=None, help='输出JSON路径')
|
||||
|
||||
# export-selected
|
||||
p_export = subparsers.add_parser('export-selected', help='批量导出选中视频数据')
|
||||
p_export.add_argument('excel_path', help='短视频表格.xlsx 路径')
|
||||
p_export.add_argument('--titles-file', required=True, help='标题关键词文件(每行一个)')
|
||||
p_export.add_argument('--output', '-o', default=None, help='输出JSON路径')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.action == 'select-top6':
|
||||
cmd_select_top6(args)
|
||||
elif args.action == 'query':
|
||||
if not args.all and not args.keywords:
|
||||
print('ERROR: query 需要 --keywords 或 --all', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
cmd_query(args)
|
||||
elif args.action == 'export-selected':
|
||||
cmd_export_selected(args)
|
||||
else:
|
||||
parser.print_help()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,435 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
json_tool.py — 统一JSON工具(创建/修复/检视/验证)
|
||||
|
||||
四种模式:
|
||||
1. create-content 从文本创建 content.json(选题+视频脚本,不含分镜表)
|
||||
2. create-analysis 从文本创建 analysis.json(MCP返回解析数据,含双引号自动修复)
|
||||
3. inspect 检视JSON文件结构和内容
|
||||
4. verify-folders 批量验证视频文件夹完整性
|
||||
|
||||
用法示例:
|
||||
# 从stdin创建content.json
|
||||
python json_tool.py create-content --detail-id 28207 --title "咬下苹果" --output-dir "C:/path/to/folder" < content.txt
|
||||
|
||||
# 从临时文件创建content.json
|
||||
python json_tool.py create-content --detail-id 28207 --title "咬下苹果" --output-dir "C:/path/to/folder" --input-file _tmp.txt
|
||||
|
||||
# 从stdin创建analysis.json(自动修复未转义双引号)
|
||||
python json_tool.py create-analysis --output-dir "C:/path/to/folder" --input-file _analysis_tmp.txt
|
||||
|
||||
# 检视JSON结构
|
||||
python json_tool.py inspect --json-path "C:/path/to/content.json"
|
||||
python json_tool.py inspect --json-path "C:/path/to/analysis.json" --full
|
||||
python json_tool.py inspect --json-path "C:/path/to/原视频解析.json" --keys 赛道,人设,框架
|
||||
|
||||
# 验证文件夹
|
||||
python json_tool.py verify-folders --base-dir "~/Desktop/账号分析/{达人昵称}/视频分析"
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import json
|
||||
import io
|
||||
import argparse
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 1. create-content
|
||||
# ============================================================
|
||||
|
||||
def create_content(args):
|
||||
"""从文本创建 content.json"""
|
||||
# 读取内容:stdin 或 文件
|
||||
if args.input_file:
|
||||
with open(args.input_file, 'r', encoding='utf-8') as f:
|
||||
content = f.read()
|
||||
else:
|
||||
content = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8').read()
|
||||
|
||||
# 构建 content.json
|
||||
data = {
|
||||
'detailId': args.detail_id,
|
||||
'title': args.title,
|
||||
'content': content
|
||||
}
|
||||
|
||||
# 写入
|
||||
outpath = os.path.join(args.output_dir, 'content.json')
|
||||
with open(outpath, 'w', encoding='utf-8') as f:
|
||||
json.dump(data, f, ensure_ascii=False, indent=2)
|
||||
|
||||
print(f'content.json done, size: {os.path.getsize(outpath)}')
|
||||
|
||||
# 清理临时文件
|
||||
if args.input_file and args.input_file.startswith('_'):
|
||||
try:
|
||||
os.remove(args.input_file)
|
||||
print('temp file removed')
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 2. create-analysis (含双引号自动修复)
|
||||
# ============================================================
|
||||
|
||||
def fix_unescaped_quotes(raw):
|
||||
"""
|
||||
修复JSON字符串值中的未转义英文双引号(U+0022)。
|
||||
策略:逐次定位JSONDecodeError位置,向前搜索未转义双引号,
|
||||
若其后非JSON结构字符(: , } ] 空白),则替换为中文左引号(U+201C)。
|
||||
返回 (修复后文本, 修复次数)。
|
||||
"""
|
||||
bs = chr(92) # 反斜杠
|
||||
dq = chr(34) # 英文双引号
|
||||
lq = chr(8220) # 中文左引号
|
||||
|
||||
fixed_count = 0
|
||||
attempts = 0
|
||||
while attempts < 200:
|
||||
try:
|
||||
json.loads(raw)
|
||||
break
|
||||
except json.JSONDecodeError as e:
|
||||
pos = e.pos
|
||||
search_pos = pos
|
||||
fixed = False
|
||||
while search_pos > 0:
|
||||
search_pos = raw.rfind(dq, 0, search_pos)
|
||||
if search_pos < 0:
|
||||
break
|
||||
# 检查是否已被转义
|
||||
if search_pos == 0 or raw[search_pos - 1] != bs:
|
||||
next_char = raw[search_pos + 1] if search_pos + 1 < len(raw) else ''
|
||||
if next_char not in [':', ',', '}', ']', ' ', '\n', '\t', '\r']:
|
||||
raw = raw[:search_pos] + lq + raw[search_pos + 1:]
|
||||
fixed = True
|
||||
fixed_count += 1
|
||||
break
|
||||
search_pos -= 1
|
||||
else:
|
||||
search_pos -= 1
|
||||
if not fixed:
|
||||
print(f'Cannot fix at pos {pos}', file=sys.stderr)
|
||||
break
|
||||
attempts += 1
|
||||
|
||||
return raw, fixed_count
|
||||
|
||||
|
||||
def create_analysis(args):
|
||||
"""从文本创建 analysis.json(含双引号修复)"""
|
||||
# 读取内容
|
||||
if args.input_file:
|
||||
with open(args.input_file, 'r', encoding='utf-8') as f:
|
||||
raw = f.read()
|
||||
else:
|
||||
raw = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8').read()
|
||||
|
||||
# 修复并解析
|
||||
raw, fixed_count = fix_unescaped_quotes(raw)
|
||||
|
||||
try:
|
||||
obj = json.loads(raw)
|
||||
except json.JSONDecodeError as e:
|
||||
print(f'Failed to parse JSON after fixes: {e}', file=sys.stderr)
|
||||
# 降级:直接写入原始文本
|
||||
outpath = os.path.join(args.output_dir, 'analysis.json')
|
||||
with open(outpath, 'w', encoding='utf-8') as f:
|
||||
f.write(raw)
|
||||
print(f'Fallback wrote raw, size: {os.path.getsize(outpath)}', file=sys.stderr)
|
||||
if args.input_file and args.input_file.startswith('_'):
|
||||
try:
|
||||
os.remove(args.input_file)
|
||||
except OSError:
|
||||
pass
|
||||
sys.exit(1)
|
||||
|
||||
# 写入
|
||||
outpath = os.path.join(args.output_dir, 'analysis.json')
|
||||
with open(outpath, 'w', encoding='utf-8') as f:
|
||||
json.dump(obj, f, ensure_ascii=False, indent=2)
|
||||
|
||||
print(f'analysis.json done, size: {os.path.getsize(outpath)}, fixes: {fixed_count}')
|
||||
|
||||
# 清理临时文件
|
||||
if args.input_file and args.input_file.startswith('_'):
|
||||
try:
|
||||
os.remove(args.input_file)
|
||||
print('temp file removed')
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 3. inspect
|
||||
# ============================================================
|
||||
|
||||
def inspect(args):
|
||||
"""检视JSON文件结构"""
|
||||
json_path = args.json_path
|
||||
if not os.path.exists(json_path):
|
||||
print(f'ERROR: file not found: {json_path}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
with open(json_path, 'r', encoding='utf-8') as f:
|
||||
data = json.load(f)
|
||||
|
||||
# 兼容旧格式:原视频解析.json 的 analysis 字段(可能是嵌套JSON字符串或已解析的dict)
|
||||
if 'analysis' in data:
|
||||
analysis = data['analysis']
|
||||
if isinstance(analysis, str):
|
||||
try:
|
||||
analysis = json.loads(analysis)
|
||||
except json.JSONDecodeError as e:
|
||||
print(f' analysis parse error: {e}')
|
||||
analysis = {}
|
||||
if isinstance(analysis, dict):
|
||||
print('[原视频解析.json 格式] content 字段 + analysis 字段')
|
||||
print(f' detailId: {data.get("detailId")}')
|
||||
print(f' title: {data.get("title")}')
|
||||
content = data.get('content', '')
|
||||
print(f' content length: {len(content)}')
|
||||
print(f' analysis keys: {list(analysis.keys())}')
|
||||
if args.keys:
|
||||
print('\n--- Filtered by --keys ---')
|
||||
for k in args.keys.split(','):
|
||||
k = k.strip()
|
||||
v = analysis.get(k, '<missing>')
|
||||
if isinstance(v, str):
|
||||
print(f' [{k}]: {v[:args.max_len]}')
|
||||
elif isinstance(v, (list, dict)):
|
||||
print(f' [{k}]: {type(v).__name__}[{len(v)}]')
|
||||
else:
|
||||
print(f' [{k}]: {v}')
|
||||
elif args.full:
|
||||
print('\n--- Full analysis ---')
|
||||
for k, v in analysis.items():
|
||||
if isinstance(v, str):
|
||||
print(f' [{k}]: {v[:args.max_len]}')
|
||||
elif isinstance(v, (list, dict)):
|
||||
print(f' [{k}]: {type(v).__name__}[{len(v)}]')
|
||||
else:
|
||||
print(f' [{k}]: {v}')
|
||||
return
|
||||
|
||||
# 新格式:content.json 或 analysis.json
|
||||
print(f'Keys: {list(data.keys())}')
|
||||
for k, v in data.items():
|
||||
if isinstance(v, str):
|
||||
display = v[:args.max_len] if not args.full else v
|
||||
print(f' [{k}]: {display}')
|
||||
elif isinstance(v, list):
|
||||
print(f' [{k}]: list[{len(v)}]')
|
||||
if args.full and v:
|
||||
for i, item in enumerate(v[:5]):
|
||||
if isinstance(item, dict):
|
||||
print(f' [{i}]: {dict(list(item.items())[:3])}')
|
||||
else:
|
||||
print(f' [{i}]: {str(item)[:100]}')
|
||||
elif isinstance(v, dict):
|
||||
print(f' [{k}]: dict keys={list(v.keys())[:5]}')
|
||||
else:
|
||||
print(f' [{k}]: {v}')
|
||||
|
||||
# 指定键过滤
|
||||
if args.keys and not args.full:
|
||||
print('\n--- Filtered by --keys ---')
|
||||
for k in args.keys.split(','):
|
||||
k = k.strip()
|
||||
v = data.get(k, '<missing>')
|
||||
if isinstance(v, str):
|
||||
print(f' [{k}]: {v[:args.max_len]}')
|
||||
else:
|
||||
print(f' [{k}]: {type(v).__name__}')
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 4. convert-old
|
||||
# ============================================================
|
||||
|
||||
def convert_old(args):
|
||||
"""从旧格式 原视频解析.json 转换为新双文件格式 (content.json + analysis.json)。
|
||||
|
||||
规则:
|
||||
- content: 保留「选题 + 视频脚本」,去掉「# 分镜脚本」及之后内容
|
||||
- analysis: MCP 返回的 analysis 字段(可能为 JSON 字符串或已解析 dict),
|
||||
字符串经双引号修复后解析
|
||||
用法:
|
||||
python json_tool.py convert-old --dir "C:/path/to/视频文件夹"
|
||||
或批量: python json_tool.py convert-old --base-dir "~/Desktop/账号分析/{昵称}/视频分析"
|
||||
"""
|
||||
targets = []
|
||||
if args.base_dir:
|
||||
base = os.path.expanduser(args.base_dir)
|
||||
if not os.path.isdir(base):
|
||||
print(f'ERROR: directory not found: {base}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
for dirname in sorted(os.listdir(base)):
|
||||
dirpath = os.path.join(base, dirname)
|
||||
if os.path.isdir(dirpath) and os.path.exists(os.path.join(dirpath, '原视频解析.json')):
|
||||
targets.append(dirpath)
|
||||
elif args.dir:
|
||||
targets = [os.path.expanduser(args.dir)]
|
||||
else:
|
||||
print('ERROR: 必须提供 --dir 或 --base-dir', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
done, skipped = 0, []
|
||||
for dirpath in targets:
|
||||
old_path = os.path.join(dirpath, '原视频解析.json')
|
||||
if not os.path.exists(old_path):
|
||||
skipped.append((dirpath, '无原视频解析.json'))
|
||||
continue
|
||||
# 已存在新文件则跳过(不覆盖)
|
||||
if os.path.exists(os.path.join(dirpath, 'content.json')) or os.path.exists(os.path.join(dirpath, 'analysis.json')):
|
||||
skipped.append((dirpath, '已存在新文件,跳过'))
|
||||
continue
|
||||
try:
|
||||
with open(old_path, 'r', encoding='utf-8') as f:
|
||||
od = json.load(f)
|
||||
except Exception as e:
|
||||
skipped.append((dirpath, f'旧json读取失败: {e}'))
|
||||
continue
|
||||
|
||||
# content: 去掉「# 分镜脚本」及之后
|
||||
content = od.get('content', '')
|
||||
marker = '# 分镜脚本'
|
||||
idx = content.find(marker)
|
||||
if idx >= 0:
|
||||
content = content[:idx].rstrip()
|
||||
|
||||
# analysis: 字符串则解析(含修复),dict 直接用
|
||||
analysis = od.get('analysis')
|
||||
fixed = 0
|
||||
if isinstance(analysis, str):
|
||||
raw, fixed = fix_unescaped_quotes(analysis)
|
||||
try:
|
||||
analysis = json.loads(raw)
|
||||
except json.JSONDecodeError as e:
|
||||
skipped.append((dirpath, f'analysis解析失败: {e}'))
|
||||
continue
|
||||
if not isinstance(analysis, dict):
|
||||
skipped.append((dirpath, f'analysis非dict: {type(analysis).__name__}'))
|
||||
continue
|
||||
|
||||
# 写入新文件
|
||||
with open(os.path.join(dirpath, 'content.json'), 'w', encoding='utf-8') as f:
|
||||
json.dump({'detailId': od.get('detailId'), 'title': od.get('title'), 'content': content},
|
||||
f, ensure_ascii=False, indent=2)
|
||||
with open(os.path.join(dirpath, 'analysis.json'), 'w', encoding='utf-8') as f:
|
||||
json.dump(analysis, f, ensure_ascii=False, indent=2)
|
||||
print(f'OK {os.path.basename(dirpath)} (fixes={fixed}, content={len(content)}ch, analysis={len(analysis)}键)')
|
||||
done += 1
|
||||
|
||||
print(f'\n转换完成: {done} 条 | 跳过: {len(skipped)} 条')
|
||||
for p, reason in skipped:
|
||||
print(f' 跳过 {os.path.basename(p)[:40]}: {reason}')
|
||||
|
||||
|
||||
# ============================================================
|
||||
# 5. verify-folders
|
||||
# ============================================================
|
||||
|
||||
def verify_folders(args):
|
||||
"""批量验证视频文件夹完整性"""
|
||||
base_dir = args.base_dir
|
||||
if not os.path.isdir(base_dir):
|
||||
print(f'ERROR: directory not found: {base_dir}', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
print('=== 视频文件夹状态验证 ===')
|
||||
complete = 0
|
||||
incomplete = 0
|
||||
|
||||
for dirname in sorted(os.listdir(base_dir)):
|
||||
dirpath = os.path.join(base_dir, dirname)
|
||||
if not os.path.isdir(dirpath):
|
||||
continue
|
||||
|
||||
files = os.listdir(dirpath)
|
||||
# 过滤临时文件
|
||||
files = [f for f in files if not f.startswith('_') and not f.endswith('_tmp.txt')]
|
||||
|
||||
has_md = any('拆解分析.md' in f for f in files)
|
||||
# 兼容新格式(content.json+analysis.json)和旧格式(原视频解析.json)
|
||||
has_json_new = 'content.json' in files and 'analysis.json' in files
|
||||
has_json_old = '原视频解析.json' in files
|
||||
has_json = has_json_new or has_json_old
|
||||
|
||||
md_status = 'OK' if has_md else 'MISSING'
|
||||
json_status = 'OK' if has_json else 'MISSING'
|
||||
|
||||
if has_md and has_json:
|
||||
complete += 1
|
||||
marker = '[COMPLETE]'
|
||||
else:
|
||||
incomplete += 1
|
||||
marker = '[INCOMPLETE]'
|
||||
|
||||
file_list = ', '.join(files)
|
||||
print(f'{marker} [{md_status} md][{json_status} json] {dirname}')
|
||||
print(f' files: {file_list}')
|
||||
|
||||
print(f'\nTotal: {complete + incomplete} | Complete: {complete} | Incomplete: {incomplete}')
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Main
|
||||
# ============================================================
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description='统一JSON工具(创建/修复/检视/验证)',
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter
|
||||
)
|
||||
subparsers = parser.add_subparsers(dest='action', help='操作模式')
|
||||
|
||||
# create-content
|
||||
p_content = subparsers.add_parser('create-content', help='从文本创建 content.json')
|
||||
p_content.add_argument('--detail-id', required=True, help='视频detailId')
|
||||
p_content.add_argument('--title', required=True, help='视频标题')
|
||||
p_content.add_argument('--output-dir', required=True, help='输出目录')
|
||||
p_content.add_argument('--input-file', default=None, help='输入文件路径(默认从stdin读取)')
|
||||
|
||||
# create-analysis
|
||||
p_analysis = subparsers.add_parser('create-analysis', help='从文本创建 analysis.json(含双引号修复)')
|
||||
p_analysis.add_argument('--output-dir', required=True, help='输出目录')
|
||||
p_analysis.add_argument('--input-file', default=None, help='输入文件路径(默认从stdin读取)')
|
||||
|
||||
# inspect
|
||||
p_inspect = subparsers.add_parser('inspect', help='检视JSON文件结构')
|
||||
p_inspect.add_argument('--json-path', required=True, help='JSON文件路径')
|
||||
p_inspect.add_argument('--keys', default=None, help='只显示指定键(逗号分隔)')
|
||||
p_inspect.add_argument('--full', action='store_true', help='显示完整内容(不截断)')
|
||||
p_inspect.add_argument('--max-len', type=int, default=300, help='字符串截断长度(默认300)')
|
||||
|
||||
# verify-folders
|
||||
p_verify = subparsers.add_parser('verify-folders', help='批量验证视频文件夹完整性')
|
||||
p_verify.add_argument('--base-dir', required=True, help='视频分析根目录')
|
||||
|
||||
# convert-old
|
||||
p_convert = subparsers.add_parser('convert-old', help='从旧格式 原视频解析.json 转换为新双文件格式')
|
||||
p_convert.add_argument('--dir', default=None, help='单个视频文件夹路径')
|
||||
p_convert.add_argument('--base-dir', default=None, help='视频分析根目录(批量转换)')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.action == 'create-content':
|
||||
create_content(args)
|
||||
elif args.action == 'create-analysis':
|
||||
create_analysis(args)
|
||||
elif args.action == 'inspect':
|
||||
inspect(args)
|
||||
elif args.action == 'verify-folders':
|
||||
verify_folders(args)
|
||||
elif args.action == 'convert-old':
|
||||
convert_old(args)
|
||||
else:
|
||||
parser.print_help()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,28 @@
|
||||
{
|
||||
"mcpServers": {
|
||||
"myai-mcp-production": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "myai-mcp"],
|
||||
"env": {
|
||||
"MYAI_ENV": "prod",
|
||||
"MYAI_API_BASE_URL": "https://maiya-trans.youmanvideo.com/api",
|
||||
"MYAI_FEISHU_APP_ID": "cli_a84d0a47f93e1013",
|
||||
"MYAI_VOD_URL_PREFIX": "https://maiya-ai-fileup.youmanvideo.com",
|
||||
"MYAI_VOD_SPACE_NAME": "maiya-ai"
|
||||
},
|
||||
"disabled": false
|
||||
},
|
||||
"myai-mcp-test": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "myai-mcp"],
|
||||
"env": {
|
||||
"MYAI_ENV": "test",
|
||||
"MYAI_API_BASE_URL": "http://maiya-trans.test.youmanvideo.com/api",
|
||||
"MYAI_FEISHU_APP_ID": "cli_a84d0a47f93e1013",
|
||||
"MYAI_VOD_URL_PREFIX": "https://maiya-ai-fileup.youmanvideo.com",
|
||||
"MYAI_VOD_SPACE_NAME": "maiya-ai"
|
||||
},
|
||||
"disabled": false
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
筛选 TOP 视频 — 已整合到 excel_tool.py select-top6 子命令。
|
||||
此文件保留向后兼容,实际调用 excel_tool.select_top6()(默认取6条)。
|
||||
|
||||
用法:
|
||||
python select_top6.py <excel路径> [--count 6] [--output output.json]
|
||||
|
||||
新用法(推荐):
|
||||
python excel_tool.py select-top6 <excel路径> [--count 6] [--output output.json]
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
|
||||
# 确保能找到 excel_tool
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from excel_tool import select_top6, HEADERS
|
||||
import json
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("用法: python select_top6.py <excel路径> [--count 6] [--output output.json]")
|
||||
print("推荐: python excel_tool.py select-top6 <excel路径> [--count 6] [--output output.json]")
|
||||
sys.exit(1)
|
||||
|
||||
excel_path = os.path.expanduser(sys.argv[1])
|
||||
if not os.path.exists(excel_path):
|
||||
print(f"ERROR: 文件不存在: {excel_path}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
# 解析 --output
|
||||
output = None
|
||||
if '--output' in sys.argv:
|
||||
idx = sys.argv.index('--output')
|
||||
if idx + 1 < len(sys.argv):
|
||||
output = sys.argv[idx + 1]
|
||||
elif '-o' in sys.argv:
|
||||
idx = sys.argv.index('-o')
|
||||
if idx + 1 < len(sys.argv):
|
||||
output = sys.argv[idx + 1]
|
||||
|
||||
# 解析 --count(默认6)
|
||||
count = 6
|
||||
if '--count' in sys.argv:
|
||||
idx = sys.argv.index('--count')
|
||||
if idx + 1 < len(sys.argv):
|
||||
try:
|
||||
count = int(sys.argv[idx + 1])
|
||||
except ValueError:
|
||||
pass
|
||||
elif '-n' in sys.argv:
|
||||
idx = sys.argv.index('-n')
|
||||
if idx + 1 < len(sys.argv):
|
||||
try:
|
||||
count = int(sys.argv[idx + 1])
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
top6 = select_top6(excel_path, count=count)
|
||||
|
||||
print(f"筛选完成,共 {len(top6)} 条视频:")
|
||||
for i, v in enumerate(top6):
|
||||
print(f" {i+1}. {v['视频标题'][:30]}... | 赞:{v['点赞数']} 转:{v['分享数']} | {v['视频时长']} | {v['发布时间']}")
|
||||
|
||||
if output:
|
||||
with open(output, 'w', encoding='utf-8') as f:
|
||||
json.dump(top6, f, ensure_ascii=False, indent=2)
|
||||
print(f"\n已保存到: {output}")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in new issue
Block a user