Files
mcn-short-video/.workbuddy/tmp_refcheck.py
T

96 lines
3.7 KiB
Python

# -*- coding: utf-8 -*-
"""引用完整性检查 v2:全角标点 + 分块N解析 + KB内部引用"""
import os, re
BASE = r"D:/AgentSkill/MCNVideo AI/project/短视频脚本创作/V1.0/脚本创作技能"
KB = os.path.join(BASE, "references", "知识库")
kb_files = {}
for d in sorted(os.listdir(KB)):
dp = os.path.join(KB, d)
if os.path.isdir(dp):
kb_files[d] = sorted(f for f in os.listdir(dp) if f.endswith(".md"))
total = sum(len(v) for v in kb_files.values())
# 分块前缀分类:文件名以"分块N_"开头的分类,允许 分块N 引用方式
block_cats = {c: {int(re.match(r'分块(\d+)_', f).group(1)): f for f in fs if re.match(r'分块(\d+)_', f)}
for c, fs in kb_files.items()}
scan_files = []
for sub in ["创作流程"]:
p = os.path.join(BASE, "references", sub)
for f in sorted(os.listdir(p)):
if f.endswith(".md"):
scan_files.append(("流程/" + f, os.path.join(p, f)))
scan_files.append(("创作流程规范.md", os.path.join(BASE, "references", "创作流程规范.md")))
scan_files.append(("SKILL.md", os.path.join(BASE, "SKILL.md")))
# 知识库内部交叉引用
for c, fs in kb_files.items():
for f in fs:
scan_files.append((f"知识库/{c}/{f}", os.path.join(KB, c, f)))
FN = r'[\w()()·::\-\\\u4e00-\u9fff/]+?\.md'
file_ref_re = re.compile(FN)
cat_name_re = re.compile(r'(\d{2}_[\w()()·\-\u4e00-\u9fff]+)')
block_re = re.compile(r'分块(\d+)')
cat_set = set(kb_files.keys())
referenced = set() # (cat, fn)
broken = []
for label, path in scan_files:
try:
lines = open(path, encoding="utf-8").read().splitlines()
except Exception as e:
print(f"READ FAIL {label}: {e}")
continue
for i, line in enumerate(lines, 1):
cats_in_line = [c for c in cat_set if c in line]
# 文件名引用
for m in file_ref_re.finditer(line):
ref = m.group(0).replace("\\", "/")
fn = ref.split("/")[-1]
# 尝试带路径匹配
hit = None
for c, fs in kb_files.items():
if fn in fs:
hit = (c, fn)
if hit:
referenced.add(hit)
elif "知识库" in line or ref.startswith("知识库") or "/" in ref:
# 可能是知识库引用但没匹配上(也排除产物/流程自身引用)
if not any(fn == x for x in ["创作流程规范.md"]) and not fn.startswith(("0", "1")) or "知识库" in ref:
if "知识库" in ref or ("知识库" in line and ("/" in ref)):
broken.append((label, i, ref, line.strip()[:100]))
# 分块N引用(限当前行有分类上下文)
for m in block_re.finditer(line):
n = int(m.group(1))
for c in cats_in_line:
if c in block_cats and n in block_cats[c]:
referenced.add((c, block_cats[c][n]))
print(f"总文件: {total}")
print("\n===== A. 疑似断链(引用了知识库路径但未命中文件) =====")
seen = set()
for b in broken:
k = (b[2],)
if k in seen: continue
seen.add(k)
print(f" {b[0]}:{b[1]} -> {b[2]}\n {b[3]}")
if not broken:
print(" 无")
print("\n===== B. 零引用文件(流程+规范+SKILL+知识库内部均未引用文件名/分块号) =====")
orphans = []
for c, fs in kb_files.items():
for fn in fs:
if (c, fn) not in referenced:
orphans.append(f"{c}/{fn}")
print(f"零引用: {len(orphans)} / {total}")
for o in orphans:
print(f" {o}")
print("\n===== C. 分类引用统计 =====")
for c, fs in kb_files.items():
used = sum(1 for fn in fs if (c, fn) in referenced)
print(f" {c}: {used}/{len(fs)}")