Files
mcn-short-video/project/短视频脚本创作/V1.0/subskills/mcn-dou-analysis/scripts/json_tool.py
T
maogeigei f1f6288b7b refactor(skill): 主技能 subskill 目录更名 subskills(09-02 用户要求)
- V1.0/subskill/ → V1.0/subskills/(browser-harness/mcn-dou-analysis/mcn-script-review/mcn-video-prompt,git 识别 R100 纯重命名保留历史)
- 路径说明同步:V1.0/SKILL.md(96/105行)、Lite1.0/SKILL.md(66行)、mcn-dou-analysis/SKILL.md 红线行、操作规范.md 159/175行、mcn-work-shop app.js SKILL_HINT_ACCOUNT
- 用户环境 ~/.workbuddy/skills/短视频脚本创作/ 与源仓库 V1.0 同 inode(junction),自动同步无需单独改
- mcn-dou-analysis 内部 subskills/nuwa-skill-main 为正常结构不受影响
2026-09-02 10:36:23 +08:00

436 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
json_tool.py — 统一JSON工具(创建/修复/检视/验证)
四种模式:
1. create-content 从文本创建 content.json(选题+视频脚本,不含分镜表)
2. create-analysis 从文本创建 analysis.json(MCP返回解析数据,含双引号自动修复)
3. inspect 检视JSON文件结构和内容
4. verify-folders 批量验证视频文件夹完整性
用法示例:
# 从stdin创建content.json
python json_tool.py create-content --detail-id 28207 --title "咬下苹果" --output-dir "C:/path/to/folder" < content.txt
# 从临时文件创建content.json
python json_tool.py create-content --detail-id 28207 --title "咬下苹果" --output-dir "C:/path/to/folder" --input-file _tmp.txt
# 从stdin创建analysis.json(自动修复未转义双引号)
python json_tool.py create-analysis --output-dir "C:/path/to/folder" --input-file _analysis_tmp.txt
# 检视JSON结构
python json_tool.py inspect --json-path "C:/path/to/content.json"
python json_tool.py inspect --json-path "C:/path/to/analysis.json" --full
python json_tool.py inspect --json-path "C:/path/to/原视频解析.json" --keys 赛道,人设,框架
# 验证文件夹
python json_tool.py verify-folders --base-dir "~/Desktop/账号分析/{达人昵称}/视频分析"
"""
import sys
import os
import json
import io
import argparse
# ============================================================
# 1. create-content
# ============================================================
def create_content(args):
"""从文本创建 content.json"""
# 读取内容:stdin 或 文件
if args.input_file:
with open(args.input_file, 'r', encoding='utf-8') as f:
content = f.read()
else:
content = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8').read()
# 构建 content.json
data = {
'detailId': args.detail_id,
'title': args.title,
'content': content
}
# 写入
outpath = os.path.join(args.output_dir, 'content.json')
with open(outpath, 'w', encoding='utf-8') as f:
json.dump(data, f, ensure_ascii=False, indent=2)
print(f'content.json done, size: {os.path.getsize(outpath)}')
# 清理临时文件
if args.input_file and args.input_file.startswith('_'):
try:
os.remove(args.input_file)
print('temp file removed')
except OSError:
pass
# ============================================================
# 2. create-analysis (含双引号自动修复)
# ============================================================
def fix_unescaped_quotes(raw):
"""
修复JSON字符串值中的未转义英文双引号(U+0022)。
策略:逐次定位JSONDecodeError位置,向前搜索未转义双引号,
若其后非JSON结构字符(: , } ] 空白),则替换为中文左引号(U+201C)。
返回 (修复后文本, 修复次数)。
"""
bs = chr(92) # 反斜杠
dq = chr(34) # 英文双引号
lq = chr(8220) # 中文左引号
fixed_count = 0
attempts = 0
while attempts < 200:
try:
json.loads(raw)
break
except json.JSONDecodeError as e:
pos = e.pos
search_pos = pos
fixed = False
while search_pos > 0:
search_pos = raw.rfind(dq, 0, search_pos)
if search_pos < 0:
break
# 检查是否已被转义
if search_pos == 0 or raw[search_pos - 1] != bs:
next_char = raw[search_pos + 1] if search_pos + 1 < len(raw) else ''
if next_char not in [':', ',', '}', ']', ' ', '\n', '\t', '\r']:
raw = raw[:search_pos] + lq + raw[search_pos + 1:]
fixed = True
fixed_count += 1
break
search_pos -= 1
else:
search_pos -= 1
if not fixed:
print(f'Cannot fix at pos {pos}', file=sys.stderr)
break
attempts += 1
return raw, fixed_count
def create_analysis(args):
"""从文本创建 analysis.json(含双引号修复)"""
# 读取内容
if args.input_file:
with open(args.input_file, 'r', encoding='utf-8') as f:
raw = f.read()
else:
raw = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8').read()
# 修复并解析
raw, fixed_count = fix_unescaped_quotes(raw)
try:
obj = json.loads(raw)
except json.JSONDecodeError as e:
print(f'Failed to parse JSON after fixes: {e}', file=sys.stderr)
# 降级:直接写入原始文本
outpath = os.path.join(args.output_dir, 'analysis.json')
with open(outpath, 'w', encoding='utf-8') as f:
f.write(raw)
print(f'Fallback wrote raw, size: {os.path.getsize(outpath)}', file=sys.stderr)
if args.input_file and args.input_file.startswith('_'):
try:
os.remove(args.input_file)
except OSError:
pass
sys.exit(1)
# 写入
outpath = os.path.join(args.output_dir, 'analysis.json')
with open(outpath, 'w', encoding='utf-8') as f:
json.dump(obj, f, ensure_ascii=False, indent=2)
print(f'analysis.json done, size: {os.path.getsize(outpath)}, fixes: {fixed_count}')
# 清理临时文件
if args.input_file and args.input_file.startswith('_'):
try:
os.remove(args.input_file)
print('temp file removed')
except OSError:
pass
# ============================================================
# 3. inspect
# ============================================================
def inspect(args):
"""检视JSON文件结构"""
json_path = args.json_path
if not os.path.exists(json_path):
print(f'ERROR: file not found: {json_path}', file=sys.stderr)
sys.exit(1)
with open(json_path, 'r', encoding='utf-8') as f:
data = json.load(f)
# 兼容旧格式:原视频解析.json 的 analysis 字段(可能是嵌套JSON字符串或已解析的dict)
if 'analysis' in data:
analysis = data['analysis']
if isinstance(analysis, str):
try:
analysis = json.loads(analysis)
except json.JSONDecodeError as e:
print(f' analysis parse error: {e}')
analysis = {}
if isinstance(analysis, dict):
print('[原视频解析.json 格式] content 字段 + analysis 字段')
print(f' detailId: {data.get("detailId")}')
print(f' title: {data.get("title")}')
content = data.get('content', '')
print(f' content length: {len(content)}')
print(f' analysis keys: {list(analysis.keys())}')
if args.keys:
print('\n--- Filtered by --keys ---')
for k in args.keys.split(','):
k = k.strip()
v = analysis.get(k, '<missing>')
if isinstance(v, str):
print(f' [{k}]: {v[:args.max_len]}')
elif isinstance(v, (list, dict)):
print(f' [{k}]: {type(v).__name__}[{len(v)}]')
else:
print(f' [{k}]: {v}')
elif args.full:
print('\n--- Full analysis ---')
for k, v in analysis.items():
if isinstance(v, str):
print(f' [{k}]: {v[:args.max_len]}')
elif isinstance(v, (list, dict)):
print(f' [{k}]: {type(v).__name__}[{len(v)}]')
else:
print(f' [{k}]: {v}')
return
# 新格式:content.json 或 analysis.json
print(f'Keys: {list(data.keys())}')
for k, v in data.items():
if isinstance(v, str):
display = v[:args.max_len] if not args.full else v
print(f' [{k}]: {display}')
elif isinstance(v, list):
print(f' [{k}]: list[{len(v)}]')
if args.full and v:
for i, item in enumerate(v[:5]):
if isinstance(item, dict):
print(f' [{i}]: {dict(list(item.items())[:3])}')
else:
print(f' [{i}]: {str(item)[:100]}')
elif isinstance(v, dict):
print(f' [{k}]: dict keys={list(v.keys())[:5]}')
else:
print(f' [{k}]: {v}')
# 指定键过滤
if args.keys and not args.full:
print('\n--- Filtered by --keys ---')
for k in args.keys.split(','):
k = k.strip()
v = data.get(k, '<missing>')
if isinstance(v, str):
print(f' [{k}]: {v[:args.max_len]}')
else:
print(f' [{k}]: {type(v).__name__}')
# ============================================================
# 4. convert-old
# ============================================================
def convert_old(args):
"""从旧格式 原视频解析.json 转换为新双文件格式 (content.json + analysis.json)。
规则:
- content: 保留「选题 + 视频脚本」,去掉「# 分镜脚本」及之后内容
- analysis: MCP 返回的 analysis 字段(可能为 JSON 字符串或已解析 dict),
字符串经双引号修复后解析
用法:
python json_tool.py convert-old --dir "C:/path/to/视频文件夹"
或批量: python json_tool.py convert-old --base-dir "~/Desktop/账号分析/{昵称}/视频分析"
"""
targets = []
if args.base_dir:
base = os.path.expanduser(args.base_dir)
if not os.path.isdir(base):
print(f'ERROR: directory not found: {base}', file=sys.stderr)
sys.exit(1)
for dirname in sorted(os.listdir(base)):
dirpath = os.path.join(base, dirname)
if os.path.isdir(dirpath) and os.path.exists(os.path.join(dirpath, '原视频解析.json')):
targets.append(dirpath)
elif args.dir:
targets = [os.path.expanduser(args.dir)]
else:
print('ERROR: 必须提供 --dir 或 --base-dir', file=sys.stderr)
sys.exit(1)
done, skipped = 0, []
for dirpath in targets:
old_path = os.path.join(dirpath, '原视频解析.json')
if not os.path.exists(old_path):
skipped.append((dirpath, '无原视频解析.json'))
continue
# 已存在新文件则跳过(不覆盖)
if os.path.exists(os.path.join(dirpath, 'content.json')) or os.path.exists(os.path.join(dirpath, 'analysis.json')):
skipped.append((dirpath, '已存在新文件,跳过'))
continue
try:
with open(old_path, 'r', encoding='utf-8') as f:
od = json.load(f)
except Exception as e:
skipped.append((dirpath, f'旧json读取失败: {e}'))
continue
# content: 去掉「# 分镜脚本」及之后
content = od.get('content', '')
marker = '# 分镜脚本'
idx = content.find(marker)
if idx >= 0:
content = content[:idx].rstrip()
# analysis: 字符串则解析(含修复),dict 直接用
analysis = od.get('analysis')
fixed = 0
if isinstance(analysis, str):
raw, fixed = fix_unescaped_quotes(analysis)
try:
analysis = json.loads(raw)
except json.JSONDecodeError as e:
skipped.append((dirpath, f'analysis解析失败: {e}'))
continue
if not isinstance(analysis, dict):
skipped.append((dirpath, f'analysis非dict: {type(analysis).__name__}'))
continue
# 写入新文件
with open(os.path.join(dirpath, 'content.json'), 'w', encoding='utf-8') as f:
json.dump({'detailId': od.get('detailId'), 'title': od.get('title'), 'content': content},
f, ensure_ascii=False, indent=2)
with open(os.path.join(dirpath, 'analysis.json'), 'w', encoding='utf-8') as f:
json.dump(analysis, f, ensure_ascii=False, indent=2)
print(f'OK {os.path.basename(dirpath)} (fixes={fixed}, content={len(content)}ch, analysis={len(analysis)}键)')
done += 1
print(f'\n转换完成: {done} 条 | 跳过: {len(skipped)} 条')
for p, reason in skipped:
print(f' 跳过 {os.path.basename(p)[:40]}: {reason}')
# ============================================================
# 5. verify-folders
# ============================================================
def verify_folders(args):
"""批量验证视频文件夹完整性"""
base_dir = args.base_dir
if not os.path.isdir(base_dir):
print(f'ERROR: directory not found: {base_dir}', file=sys.stderr)
sys.exit(1)
print('=== 视频文件夹状态验证 ===')
complete = 0
incomplete = 0
for dirname in sorted(os.listdir(base_dir)):
dirpath = os.path.join(base_dir, dirname)
if not os.path.isdir(dirpath):
continue
files = os.listdir(dirpath)
# 过滤临时文件
files = [f for f in files if not f.startswith('_') and not f.endswith('_tmp.txt')]
has_md = any('拆解分析.md' in f for f in files)
# 兼容新格式(content.json+analysis.json)和旧格式(原视频解析.json)
has_json_new = 'content.json' in files and 'analysis.json' in files
has_json_old = '原视频解析.json' in files
has_json = has_json_new or has_json_old
md_status = 'OK' if has_md else 'MISSING'
json_status = 'OK' if has_json else 'MISSING'
if has_md and has_json:
complete += 1
marker = '[COMPLETE]'
else:
incomplete += 1
marker = '[INCOMPLETE]'
file_list = ', '.join(files)
print(f'{marker} [{md_status} md][{json_status} json] {dirname}')
print(f' files: {file_list}')
print(f'\nTotal: {complete + incomplete} | Complete: {complete} | Incomplete: {incomplete}')
# ============================================================
# Main
# ============================================================
def main():
parser = argparse.ArgumentParser(
description='统一JSON工具(创建/修复/检视/验证)',
formatter_class=argparse.RawDescriptionHelpFormatter
)
subparsers = parser.add_subparsers(dest='action', help='操作模式')
# create-content
p_content = subparsers.add_parser('create-content', help='从文本创建 content.json')
p_content.add_argument('--detail-id', required=True, help='视频detailId')
p_content.add_argument('--title', required=True, help='视频标题')
p_content.add_argument('--output-dir', required=True, help='输出目录')
p_content.add_argument('--input-file', default=None, help='输入文件路径(默认从stdin读取)')
# create-analysis
p_analysis = subparsers.add_parser('create-analysis', help='从文本创建 analysis.json(含双引号修复)')
p_analysis.add_argument('--output-dir', required=True, help='输出目录')
p_analysis.add_argument('--input-file', default=None, help='输入文件路径(默认从stdin读取)')
# inspect
p_inspect = subparsers.add_parser('inspect', help='检视JSON文件结构')
p_inspect.add_argument('--json-path', required=True, help='JSON文件路径')
p_inspect.add_argument('--keys', default=None, help='只显示指定键(逗号分隔)')
p_inspect.add_argument('--full', action='store_true', help='显示完整内容(不截断)')
p_inspect.add_argument('--max-len', type=int, default=300, help='字符串截断长度(默认300)')
# verify-folders
p_verify = subparsers.add_parser('verify-folders', help='批量验证视频文件夹完整性')
p_verify.add_argument('--base-dir', required=True, help='视频分析根目录')
# convert-old
p_convert = subparsers.add_parser('convert-old', help='从旧格式 原视频解析.json 转换为新双文件格式')
p_convert.add_argument('--dir', default=None, help='单个视频文件夹路径')
p_convert.add_argument('--base-dir', default=None, help='视频分析根目录(批量转换)')
args = parser.parse_args()
if args.action == 'create-content':
create_content(args)
elif args.action == 'create-analysis':
create_analysis(args)
elif args.action == 'inspect':
inspect(args)
elif args.action == 'verify-folders':
verify_folders(args)
elif args.action == 'convert-old':
convert_old(args)
else:
parser.print_help()
if __name__ == '__main__':
main()