Files
mcn-short-video/project/短视频脚本创作/V1.0/subskill/mcn-dou-analysis/scripts/json_tool.py
T
maogeigei eada288999 技能三副本迭代:路径重构+输出边界铁律+开场口径区分+审计修复
- 路径重构:脚本创作 V1.0/Lite1.0 扁平化(去技能文件夹层),分镜技能迁移至 短视频提示词生成/V1.0,旧路径清理
- 输出边界铁律固化:创作流程规范.md 新增「交付物输出边界(通用铁律)」+ S9 输出模板新增「输出边界」块(验收口径=执行约束非输出内容)
- 开场钩子铁律三层固化:00_开场/06_编导终审/S9 相关规则 + S5 预设钩子口径区分(前3秒=呈现窗口,3-7秒=开场段总时长)
- 帮助文档:新增「从账号分析到脚本创作」跨技能章节(你这样问|我会做什么风格)
- mcn-dou-analysis 内嵌副本:红线边界声明+人设卡术语更新(设定分析说明)
2026-08-26 17:28:21 +08:00

436 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
json_tool.py — 统一JSON工具(创建/修复/检视/验证)
四种模式:
1. create-content 从文本创建 content.json(选题+视频脚本,不含分镜表)
2. create-analysis 从文本创建 analysis.json(MCP返回解析数据,含双引号自动修复)
3. inspect 检视JSON文件结构和内容
4. verify-folders 批量验证视频文件夹完整性
用法示例:
# 从stdin创建content.json
python json_tool.py create-content --detail-id 28207 --title "咬下苹果" --output-dir "C:/path/to/folder" < content.txt
# 从临时文件创建content.json
python json_tool.py create-content --detail-id 28207 --title "咬下苹果" --output-dir "C:/path/to/folder" --input-file _tmp.txt
# 从stdin创建analysis.json(自动修复未转义双引号)
python json_tool.py create-analysis --output-dir "C:/path/to/folder" --input-file _analysis_tmp.txt
# 检视JSON结构
python json_tool.py inspect --json-path "C:/path/to/content.json"
python json_tool.py inspect --json-path "C:/path/to/analysis.json" --full
python json_tool.py inspect --json-path "C:/path/to/原视频解析.json" --keys 赛道,人设,框架
# 验证文件夹
python json_tool.py verify-folders --base-dir "~/Desktop/账号分析/{达人昵称}/视频分析"
"""
import sys
import os
import json
import io
import argparse
# ============================================================
# 1. create-content
# ============================================================
def create_content(args):
"""从文本创建 content.json"""
# 读取内容:stdin 或 文件
if args.input_file:
with open(args.input_file, 'r', encoding='utf-8') as f:
content = f.read()
else:
content = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8').read()
# 构建 content.json
data = {
'detailId': args.detail_id,
'title': args.title,
'content': content
}
# 写入
outpath = os.path.join(args.output_dir, 'content.json')
with open(outpath, 'w', encoding='utf-8') as f:
json.dump(data, f, ensure_ascii=False, indent=2)
print(f'content.json done, size: {os.path.getsize(outpath)}')
# 清理临时文件
if args.input_file and args.input_file.startswith('_'):
try:
os.remove(args.input_file)
print('temp file removed')
except OSError:
pass
# ============================================================
# 2. create-analysis (含双引号自动修复)
# ============================================================
def fix_unescaped_quotes(raw):
"""
修复JSON字符串值中的未转义英文双引号(U+0022)。
策略:逐次定位JSONDecodeError位置,向前搜索未转义双引号,
若其后非JSON结构字符(: , } ] 空白),则替换为中文左引号(U+201C)。
返回 (修复后文本, 修复次数)。
"""
bs = chr(92) # 反斜杠
dq = chr(34) # 英文双引号
lq = chr(8220) # 中文左引号
fixed_count = 0
attempts = 0
while attempts < 200:
try:
json.loads(raw)
break
except json.JSONDecodeError as e:
pos = e.pos
search_pos = pos
fixed = False
while search_pos > 0:
search_pos = raw.rfind(dq, 0, search_pos)
if search_pos < 0:
break
# 检查是否已被转义
if search_pos == 0 or raw[search_pos - 1] != bs:
next_char = raw[search_pos + 1] if search_pos + 1 < len(raw) else ''
if next_char not in [':', ',', '}', ']', ' ', '\n', '\t', '\r']:
raw = raw[:search_pos] + lq + raw[search_pos + 1:]
fixed = True
fixed_count += 1
break
search_pos -= 1
else:
search_pos -= 1
if not fixed:
print(f'Cannot fix at pos {pos}', file=sys.stderr)
break
attempts += 1
return raw, fixed_count
def create_analysis(args):
"""从文本创建 analysis.json(含双引号修复)"""
# 读取内容
if args.input_file:
with open(args.input_file, 'r', encoding='utf-8') as f:
raw = f.read()
else:
raw = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8').read()
# 修复并解析
raw, fixed_count = fix_unescaped_quotes(raw)
try:
obj = json.loads(raw)
except json.JSONDecodeError as e:
print(f'Failed to parse JSON after fixes: {e}', file=sys.stderr)
# 降级:直接写入原始文本
outpath = os.path.join(args.output_dir, 'analysis.json')
with open(outpath, 'w', encoding='utf-8') as f:
f.write(raw)
print(f'Fallback wrote raw, size: {os.path.getsize(outpath)}', file=sys.stderr)
if args.input_file and args.input_file.startswith('_'):
try:
os.remove(args.input_file)
except OSError:
pass
sys.exit(1)
# 写入
outpath = os.path.join(args.output_dir, 'analysis.json')
with open(outpath, 'w', encoding='utf-8') as f:
json.dump(obj, f, ensure_ascii=False, indent=2)
print(f'analysis.json done, size: {os.path.getsize(outpath)}, fixes: {fixed_count}')
# 清理临时文件
if args.input_file and args.input_file.startswith('_'):
try:
os.remove(args.input_file)
print('temp file removed')
except OSError:
pass
# ============================================================
# 3. inspect
# ============================================================
def inspect(args):
"""检视JSON文件结构"""
json_path = args.json_path
if not os.path.exists(json_path):
print(f'ERROR: file not found: {json_path}', file=sys.stderr)
sys.exit(1)
with open(json_path, 'r', encoding='utf-8') as f:
data = json.load(f)
# 兼容旧格式:原视频解析.json 的 analysis 字段(可能是嵌套JSON字符串或已解析的dict)
if 'analysis' in data:
analysis = data['analysis']
if isinstance(analysis, str):
try:
analysis = json.loads(analysis)
except json.JSONDecodeError as e:
print(f' analysis parse error: {e}')
analysis = {}
if isinstance(analysis, dict):
print('[原视频解析.json 格式] content 字段 + analysis 字段')
print(f' detailId: {data.get("detailId")}')
print(f' title: {data.get("title")}')
content = data.get('content', '')
print(f' content length: {len(content)}')
print(f' analysis keys: {list(analysis.keys())}')
if args.keys:
print('\n--- Filtered by --keys ---')
for k in args.keys.split(','):
k = k.strip()
v = analysis.get(k, '<missing>')
if isinstance(v, str):
print(f' [{k}]: {v[:args.max_len]}')
elif isinstance(v, (list, dict)):
print(f' [{k}]: {type(v).__name__}[{len(v)}]')
else:
print(f' [{k}]: {v}')
elif args.full:
print('\n--- Full analysis ---')
for k, v in analysis.items():
if isinstance(v, str):
print(f' [{k}]: {v[:args.max_len]}')
elif isinstance(v, (list, dict)):
print(f' [{k}]: {type(v).__name__}[{len(v)}]')
else:
print(f' [{k}]: {v}')
return
# 新格式:content.json 或 analysis.json
print(f'Keys: {list(data.keys())}')
for k, v in data.items():
if isinstance(v, str):
display = v[:args.max_len] if not args.full else v
print(f' [{k}]: {display}')
elif isinstance(v, list):
print(f' [{k}]: list[{len(v)}]')
if args.full and v:
for i, item in enumerate(v[:5]):
if isinstance(item, dict):
print(f' [{i}]: {dict(list(item.items())[:3])}')
else:
print(f' [{i}]: {str(item)[:100]}')
elif isinstance(v, dict):
print(f' [{k}]: dict keys={list(v.keys())[:5]}')
else:
print(f' [{k}]: {v}')
# 指定键过滤
if args.keys and not args.full:
print('\n--- Filtered by --keys ---')
for k in args.keys.split(','):
k = k.strip()
v = data.get(k, '<missing>')
if isinstance(v, str):
print(f' [{k}]: {v[:args.max_len]}')
else:
print(f' [{k}]: {type(v).__name__}')
# ============================================================
# 4. convert-old
# ============================================================
def convert_old(args):
"""从旧格式 原视频解析.json 转换为新双文件格式 (content.json + analysis.json)。
规则:
- content: 保留「选题 + 视频脚本」,去掉「# 分镜脚本」及之后内容
- analysis: MCP 返回的 analysis 字段(可能为 JSON 字符串或已解析 dict),
字符串经双引号修复后解析
用法:
python json_tool.py convert-old --dir "C:/path/to/视频文件夹"
或批量: python json_tool.py convert-old --base-dir "~/Desktop/账号分析/{昵称}/视频分析"
"""
targets = []
if args.base_dir:
base = os.path.expanduser(args.base_dir)
if not os.path.isdir(base):
print(f'ERROR: directory not found: {base}', file=sys.stderr)
sys.exit(1)
for dirname in sorted(os.listdir(base)):
dirpath = os.path.join(base, dirname)
if os.path.isdir(dirpath) and os.path.exists(os.path.join(dirpath, '原视频解析.json')):
targets.append(dirpath)
elif args.dir:
targets = [os.path.expanduser(args.dir)]
else:
print('ERROR: 必须提供 --dir 或 --base-dir', file=sys.stderr)
sys.exit(1)
done, skipped = 0, []
for dirpath in targets:
old_path = os.path.join(dirpath, '原视频解析.json')
if not os.path.exists(old_path):
skipped.append((dirpath, '无原视频解析.json'))
continue
# 已存在新文件则跳过(不覆盖)
if os.path.exists(os.path.join(dirpath, 'content.json')) or os.path.exists(os.path.join(dirpath, 'analysis.json')):
skipped.append((dirpath, '已存在新文件,跳过'))
continue
try:
with open(old_path, 'r', encoding='utf-8') as f:
od = json.load(f)
except Exception as e:
skipped.append((dirpath, f'旧json读取失败: {e}'))
continue
# content: 去掉「# 分镜脚本」及之后
content = od.get('content', '')
marker = '# 分镜脚本'
idx = content.find(marker)
if idx >= 0:
content = content[:idx].rstrip()
# analysis: 字符串则解析(含修复),dict 直接用
analysis = od.get('analysis')
fixed = 0
if isinstance(analysis, str):
raw, fixed = fix_unescaped_quotes(analysis)
try:
analysis = json.loads(raw)
except json.JSONDecodeError as e:
skipped.append((dirpath, f'analysis解析失败: {e}'))
continue
if not isinstance(analysis, dict):
skipped.append((dirpath, f'analysis非dict: {type(analysis).__name__}'))
continue
# 写入新文件
with open(os.path.join(dirpath, 'content.json'), 'w', encoding='utf-8') as f:
json.dump({'detailId': od.get('detailId'), 'title': od.get('title'), 'content': content},
f, ensure_ascii=False, indent=2)
with open(os.path.join(dirpath, 'analysis.json'), 'w', encoding='utf-8') as f:
json.dump(analysis, f, ensure_ascii=False, indent=2)
print(f'OK {os.path.basename(dirpath)} (fixes={fixed}, content={len(content)}ch, analysis={len(analysis)}键)')
done += 1
print(f'\n转换完成: {done} 条 | 跳过: {len(skipped)} 条')
for p, reason in skipped:
print(f' 跳过 {os.path.basename(p)[:40]}: {reason}')
# ============================================================
# 5. verify-folders
# ============================================================
def verify_folders(args):
"""批量验证视频文件夹完整性"""
base_dir = args.base_dir
if not os.path.isdir(base_dir):
print(f'ERROR: directory not found: {base_dir}', file=sys.stderr)
sys.exit(1)
print('=== 视频文件夹状态验证 ===')
complete = 0
incomplete = 0
for dirname in sorted(os.listdir(base_dir)):
dirpath = os.path.join(base_dir, dirname)
if not os.path.isdir(dirpath):
continue
files = os.listdir(dirpath)
# 过滤临时文件
files = [f for f in files if not f.startswith('_') and not f.endswith('_tmp.txt')]
has_md = any('拆解分析.md' in f for f in files)
# 兼容新格式(content.json+analysis.json)和旧格式(原视频解析.json)
has_json_new = 'content.json' in files and 'analysis.json' in files
has_json_old = '原视频解析.json' in files
has_json = has_json_new or has_json_old
md_status = 'OK' if has_md else 'MISSING'
json_status = 'OK' if has_json else 'MISSING'
if has_md and has_json:
complete += 1
marker = '[COMPLETE]'
else:
incomplete += 1
marker = '[INCOMPLETE]'
file_list = ', '.join(files)
print(f'{marker} [{md_status} md][{json_status} json] {dirname}')
print(f' files: {file_list}')
print(f'\nTotal: {complete + incomplete} | Complete: {complete} | Incomplete: {incomplete}')
# ============================================================
# Main
# ============================================================
def main():
parser = argparse.ArgumentParser(
description='统一JSON工具(创建/修复/检视/验证)',
formatter_class=argparse.RawDescriptionHelpFormatter
)
subparsers = parser.add_subparsers(dest='action', help='操作模式')
# create-content
p_content = subparsers.add_parser('create-content', help='从文本创建 content.json')
p_content.add_argument('--detail-id', required=True, help='视频detailId')
p_content.add_argument('--title', required=True, help='视频标题')
p_content.add_argument('--output-dir', required=True, help='输出目录')
p_content.add_argument('--input-file', default=None, help='输入文件路径(默认从stdin读取)')
# create-analysis
p_analysis = subparsers.add_parser('create-analysis', help='从文本创建 analysis.json(含双引号修复)')
p_analysis.add_argument('--output-dir', required=True, help='输出目录')
p_analysis.add_argument('--input-file', default=None, help='输入文件路径(默认从stdin读取)')
# inspect
p_inspect = subparsers.add_parser('inspect', help='检视JSON文件结构')
p_inspect.add_argument('--json-path', required=True, help='JSON文件路径')
p_inspect.add_argument('--keys', default=None, help='只显示指定键(逗号分隔)')
p_inspect.add_argument('--full', action='store_true', help='显示完整内容(不截断)')
p_inspect.add_argument('--max-len', type=int, default=300, help='字符串截断长度(默认300)')
# verify-folders
p_verify = subparsers.add_parser('verify-folders', help='批量验证视频文件夹完整性')
p_verify.add_argument('--base-dir', required=True, help='视频分析根目录')
# convert-old
p_convert = subparsers.add_parser('convert-old', help='从旧格式 原视频解析.json 转换为新双文件格式')
p_convert.add_argument('--dir', default=None, help='单个视频文件夹路径')
p_convert.add_argument('--base-dir', default=None, help='视频分析根目录(批量转换)')
args = parser.parse_args()
if args.action == 'create-content':
create_content(args)
elif args.action == 'create-analysis':
create_analysis(args)
elif args.action == 'inspect':
inspect(args)
elif args.action == 'verify-folders':
verify_folders(args)
elif args.action == 'convert-old':
convert_old(args)
else:
parser.print_help()
if __name__ == '__main__':
main()