- V1.0/subskill/ → V1.0/subskills/(browser-harness/mcn-dou-analysis/mcn-script-review/mcn-video-prompt,git 识别 R100 纯重命名保留历史) - 路径说明同步:V1.0/SKILL.md(96/105行)、Lite1.0/SKILL.md(66行)、mcn-dou-analysis/SKILL.md 红线行、操作规范.md 159/175行、mcn-work-shop app.js SKILL_HINT_ACCOUNT - 用户环境 ~/.workbuddy/skills/短视频脚本创作/ 与源仓库 V1.0 同 inode(junction),自动同步无需单独改 - mcn-dou-analysis 内部 subskills/nuwa-skill-main 为正常结构不受影响
314 lines
12 KiB
Python
314 lines
12 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
excel_tool.py — 统一Excel工具(筛选/查询/导出)
|
||
|
||
三种模式:
|
||
1. select-top6 筛选TOP视频(最近三个月 + ≤15分钟 + 点赞分享降序,默认取6条,可用 --count 指定;默认数量为 SKILL.md「视频筛选标准」的执行镜像,调整该参数时须同步本处)
|
||
2. query 按标题关键词查询视频数据(含三比率计算)
|
||
3. export-selected 批量导出选中视频数据(含三比率计算)
|
||
|
||
Excel列结构(13列标准格式):
|
||
0:序号 1:视频ID 2:视频标题 3:视频地址 4:点赞数 5:点赞(显示)
|
||
6:评论数 7:分享数 8:收藏数 9:播放量 10:视频时长 11:发布时间 12:标签
|
||
|
||
三比率定义(账号基准均值参考,来源于历史账号分析实测均值;仅作横向对比参考,不代表当前分析账号):
|
||
享赞比 = 分享数 / 点赞数 × 100% (基准 19.6%)
|
||
评赞比 = 评论数 / 点赞数 × 100% (基准 1.85%)
|
||
藏赞比 = 收藏数 / 点赞数 × 100% (基准 7.5%)
|
||
|
||
用法示例:
|
||
# 筛选TOP视频(默认6条,数量与 SKILL.md「视频筛选标准」同步)
|
||
python excel_tool.py select-top6 "C:/path/to/短视频表格.xlsx" [--output top6.json]
|
||
# 指定数量
|
||
python excel_tool.py select-top6 "C:/path/to/短视频表格.xlsx" --count 6 [--output top6.json]
|
||
|
||
# 按关键词查询(逗号分隔多关键词,OR逻辑)
|
||
python excel_tool.py query "C:/path/to/短视频表格.xlsx" --keywords "咬下苹果,露水,划船上岸"
|
||
|
||
# 查询所有视频并按点赞降序
|
||
python excel_tool.py query "C:/path/to/短视频表格.xlsx" --all
|
||
|
||
# 批量导出选中视频(从文件读取标题关键词,每行一个)
|
||
python excel_tool.py export-selected "C:/path/to/短视频表格.xlsx" --titles-file selected.txt --output result.json
|
||
"""
|
||
|
||
import sys
|
||
import os
|
||
import json
|
||
import datetime
|
||
import argparse
|
||
|
||
try:
|
||
import openpyxl
|
||
except ImportError:
|
||
print("ERROR: openpyxl not installed. Run: pip install openpyxl", file=sys.stderr)
|
||
sys.exit(1)
|
||
|
||
|
||
# ============================================================
|
||
# 通用工具函数
|
||
# ============================================================
|
||
|
||
HEADERS = ['序号', '视频ID', '视频标题', '视频地址', '点赞数', '点赞(显示)',
|
||
'评论数', '分享数', '收藏数', '播放量', '视频时长', '发布时间', '标签']
|
||
|
||
# 账号基准均值
|
||
BENCHMARK = {'享赞比': 19.6, '评赞比': 1.85, '藏赞比': 7.5}
|
||
|
||
|
||
def parse_duration(val):
|
||
"""解析视频时长字段,返回秒数。支持 HH:MM:SS / MM:SS / 秒 / 毫秒 / 中文单位(N秒 / N分 / N分M秒)格式。"""
|
||
if isinstance(val, (int, float)):
|
||
return float(val) / 1000 if val > 1000 else float(val)
|
||
if isinstance(val, str):
|
||
val = val.strip()
|
||
# 中文单位
|
||
if '秒' in val or '分' in val:
|
||
total = 0.0
|
||
if '分' in val:
|
||
m_part = val.split('分')[0]
|
||
total += int(m_part) * 60 if m_part.strip().isdigit() else 0
|
||
s_part = val.split('分')[1].replace('秒', '').strip()
|
||
total += float(s_part) if s_part else 0
|
||
else:
|
||
total = float(val.replace('秒', ''))
|
||
return total
|
||
parts = val.split(':')
|
||
if len(parts) == 3:
|
||
return int(parts[0]) * 3600 + int(parts[1]) * 60 + int(parts[2])
|
||
elif len(parts) == 2:
|
||
return int(parts[0]) * 60 + int(parts[1])
|
||
try:
|
||
return float(val)
|
||
except ValueError:
|
||
return 99999
|
||
return 99999
|
||
|
||
|
||
def compute_ratios(likes, comments, shares, saves):
|
||
"""计算三比率,返回字典。"""
|
||
likes = likes or 0
|
||
return {
|
||
'享赞比': round((shares or 0) / likes * 100, 1) if likes else 0,
|
||
'评赞比': round((comments or 0) / likes * 100, 2) if likes else 0,
|
||
'藏赞比': round((saves or 0) / likes * 100, 1) if likes else 0,
|
||
}
|
||
|
||
|
||
def format_ratio_comparison(ratio_name, value):
|
||
"""格式化比率与基准的比较标记。"""
|
||
bench = BENCHMARK.get(ratio_name, 0)
|
||
if value > bench * 1.3:
|
||
return f'{ratio_name}{value}%^^'
|
||
elif value > bench * 1.1:
|
||
return f'{ratio_name}{value}%^'
|
||
elif value < bench * 0.7:
|
||
return f'{ratio_name}{value}%vv'
|
||
elif value < bench * 0.9:
|
||
return f'{ratio_name}{value}%v'
|
||
else:
|
||
return f'{ratio_name}{value}%~'
|
||
|
||
|
||
def load_excel_rows(excel_path):
|
||
"""加载Excel,返回数据行列表(跳过表头)。"""
|
||
if not os.path.exists(excel_path):
|
||
print(f'ERROR: file not found: {excel_path}', file=sys.stderr)
|
||
sys.exit(1)
|
||
wb = openpyxl.load_workbook(excel_path, read_only=True)
|
||
ws = wb.active
|
||
rows = list(ws.iter_rows(min_row=2, values_only=True))
|
||
wb.close()
|
||
return rows
|
||
|
||
|
||
def row_to_dict(row):
|
||
"""将Excel行转为字典。"""
|
||
return dict(zip(HEADERS, [str(v) if v is not None else '' for v in row]))
|
||
|
||
|
||
def row_to_summary(row):
|
||
"""将Excel行转为含三比率的摘要字典。"""
|
||
likes = row[4] or 0
|
||
comments = row[6] or 0
|
||
shares = row[7] or 0
|
||
saves = row[8] or 0
|
||
return {
|
||
'标题': str(row[2]) if row[2] else '',
|
||
'点赞': likes,
|
||
'评论': comments,
|
||
'分享': shares,
|
||
'收藏': saves,
|
||
'播放': row[9] or 0,
|
||
'时长': f'{parse_duration(row[10]):.0f}s',
|
||
'发布': str(row[11])[:10] if row[11] else '',
|
||
'标签': str(row[12]) if row[12] else '',
|
||
**compute_ratios(likes, comments, shares, saves),
|
||
}
|
||
|
||
|
||
# ============================================================
|
||
# 1. select-top6
|
||
# ============================================================
|
||
|
||
def select_top6(excel_path, count=6):
|
||
"""筛选TOP视频(默认取6条,可用 count 指定数量)。
|
||
默认数量为 SKILL.md「视频筛选标准」的执行镜像:调整 SKILL.md 该参数时必须同步本默认值。
|
||
"""
|
||
rows = load_excel_rows(excel_path)
|
||
|
||
# 时间过滤:最近三个月
|
||
three_months_ago = datetime.datetime.now() - datetime.timedelta(days=90)
|
||
recent = [
|
||
r for r in rows
|
||
if r[11] and datetime.datetime.strptime(str(r[11])[:10], '%Y-%m-%d') >= three_months_ago
|
||
]
|
||
|
||
# 时长过滤:≤15分钟(900秒)
|
||
recent = [r for r in recent if parse_duration(r[10]) <= 900]
|
||
|
||
# 排序:点赞数 + 分享数 降序
|
||
recent.sort(key=lambda r: (r[4] or 0) + (r[7] or 0), reverse=True)
|
||
|
||
# 取前 count 条
|
||
top = recent[:count]
|
||
return [row_to_dict(r) for r in top]
|
||
|
||
|
||
def cmd_select_top6(args):
|
||
"""select-top6 命令处理。"""
|
||
excel_path = os.path.expanduser(args.excel_path)
|
||
top6 = select_top6(excel_path, count=args.count)
|
||
|
||
print(f'筛选完成,共 {len(top6)} 条视频:')
|
||
for i, v in enumerate(top6):
|
||
print(f' {i+1}. {v["视频标题"][:30]}... | 赞:{v["点赞数"]} 转:{v["分享数"]} | {v["视频时长"]} | {v["发布时间"]}')
|
||
|
||
if args.output:
|
||
with open(args.output, 'w', encoding='utf-8') as f:
|
||
json.dump(top6, f, ensure_ascii=False, indent=2)
|
||
print(f'\n已保存到: {args.output}')
|
||
|
||
|
||
# ============================================================
|
||
# 2. query
|
||
# ============================================================
|
||
|
||
def cmd_query(args):
|
||
"""query 命令处理:按关键词查询或全量查询。"""
|
||
excel_path = os.path.expanduser(args.excel_path)
|
||
rows = load_excel_rows(excel_path)
|
||
|
||
if args.all:
|
||
# 全量查询,按点赞降序
|
||
matched = sorted(rows, key=lambda r: r[4] or 0, reverse=True)
|
||
else:
|
||
# 关键词过滤(OR逻辑)
|
||
keywords = [k.strip() for k in args.keywords.split(',')]
|
||
matched = [r for r in rows if r[2] and any(k in str(r[2]) for k in keywords)]
|
||
|
||
print(f'查询到 {len(matched)} 条视频:')
|
||
print(f'{"序号":<4} | {"标题":<25} | {"点赞":>8} | {"评论":>6} | {"分享":>6} | {"收藏":>6} | {"时长":>6} | {"发布":>12} | 三比率')
|
||
print('-' * 120)
|
||
|
||
for row in matched:
|
||
summary = row_to_summary(row)
|
||
ratios = f'{format_ratio_comparison("享赞比", summary["享赞比"])} / {format_ratio_comparison("评赞比", summary["评赞比"])} / {format_ratio_comparison("藏赞比", summary["藏赞比"])}'
|
||
print(f'{row[0]:<4} | {summary["标题"][:25]:<25} | {summary["点赞"]:>8} | {summary["评论"]:>6} | {summary["分享"]:>6} | {summary["收藏"]:>6} | {summary["时长"]:>6} | {summary["发布"]:>12} | {ratios}')
|
||
|
||
if args.output:
|
||
result = [row_to_summary(r) for r in matched]
|
||
with open(args.output, 'w', encoding='utf-8') as f:
|
||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||
print(f'\n已保存到: {args.output}')
|
||
|
||
|
||
# ============================================================
|
||
# 3. export-selected
|
||
# ============================================================
|
||
|
||
def cmd_export_selected(args):
|
||
"""export-selected 命令处理:批量导出选中视频。"""
|
||
excel_path = os.path.expanduser(args.excel_path)
|
||
rows = load_excel_rows(excel_path)
|
||
|
||
# 从文件读取标题关键词
|
||
with open(args.titles_file, 'r', encoding='utf-8') as f:
|
||
keywords = [line.strip() for line in f if line.strip()]
|
||
|
||
# 匹配
|
||
matched = []
|
||
for row in rows:
|
||
if row[2]:
|
||
for k in keywords:
|
||
if k in str(row[2]):
|
||
matched.append(row)
|
||
break
|
||
|
||
# 按点赞降序
|
||
matched.sort(key=lambda r: r[4] or 0, reverse=True)
|
||
|
||
result = [row_to_summary(r) for r in matched]
|
||
|
||
print(f'导出 {len(result)} 条视频:')
|
||
for i, s in enumerate(result):
|
||
print(f' {i+1}. {s["标题"][:25]}... | 赞:{s["点赞"]} | {s["时长"]} | {s["发布"]}')
|
||
|
||
if args.output:
|
||
with open(args.output, 'w', encoding='utf-8') as f:
|
||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||
print(f'\n已保存到: {args.output}')
|
||
else:
|
||
# 默认输出到终端
|
||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||
|
||
|
||
# ============================================================
|
||
# Main
|
||
# ============================================================
|
||
|
||
def main():
|
||
parser = argparse.ArgumentParser(
|
||
description='统一Excel工具(筛选/查询/导出)',
|
||
formatter_class=argparse.RawDescriptionHelpFormatter
|
||
)
|
||
subparsers = parser.add_subparsers(dest='action', help='操作模式')
|
||
|
||
# select-top6
|
||
p_top6 = subparsers.add_parser('select-top6', help='筛选TOP视频(默认取6条,与 SKILL.md「视频筛选标准」同步)')
|
||
p_top6.add_argument('excel_path', help='短视频表格.xlsx 路径')
|
||
p_top6.add_argument('--count', '-n', type=int, default=6, help='筛选数量(默认6条)')
|
||
p_top6.add_argument('--output', '-o', default=None, help='输出JSON路径')
|
||
|
||
# query
|
||
p_query = subparsers.add_parser('query', help='按标题关键词查询视频数据')
|
||
p_query.add_argument('excel_path', help='短视频表格.xlsx 路径')
|
||
p_query.add_argument('--keywords', '-k', default=None, help='标题关键词(逗号分隔,OR逻辑)')
|
||
p_query.add_argument('--all', action='store_true', help='查询全部视频')
|
||
p_query.add_argument('--output', '-o', default=None, help='输出JSON路径')
|
||
|
||
# export-selected
|
||
p_export = subparsers.add_parser('export-selected', help='批量导出选中视频数据')
|
||
p_export.add_argument('excel_path', help='短视频表格.xlsx 路径')
|
||
p_export.add_argument('--titles-file', required=True, help='标题关键词文件(每行一个)')
|
||
p_export.add_argument('--output', '-o', default=None, help='输出JSON路径')
|
||
|
||
args = parser.parse_args()
|
||
|
||
if args.action == 'select-top6':
|
||
cmd_select_top6(args)
|
||
elif args.action == 'query':
|
||
if not args.all and not args.keywords:
|
||
print('ERROR: query 需要 --keywords 或 --all', file=sys.stderr)
|
||
sys.exit(1)
|
||
cmd_query(args)
|
||
elif args.action == 'export-selected':
|
||
cmd_export_selected(args)
|
||
else:
|
||
parser.print_help()
|
||
|
||
|
||
if __name__ == '__main__':
|
||
main()
|