# -*- coding: utf-8 -*- """引用完整性检查 v2:全角标点 + 分块N解析 + KB内部引用""" import os, re BASE = r"D:/AgentSkill/MCNVideo AI/project/短视频脚本创作/V1.0/脚本创作技能" KB = os.path.join(BASE, "references", "知识库") kb_files = {} for d in sorted(os.listdir(KB)): dp = os.path.join(KB, d) if os.path.isdir(dp): kb_files[d] = sorted(f for f in os.listdir(dp) if f.endswith(".md")) total = sum(len(v) for v in kb_files.values()) # 分块前缀分类:文件名以"分块N_"开头的分类,允许 分块N 引用方式 block_cats = {c: {int(re.match(r'分块(\d+)_', f).group(1)): f for f in fs if re.match(r'分块(\d+)_', f)} for c, fs in kb_files.items()} scan_files = [] for sub in ["创作流程"]: p = os.path.join(BASE, "references", sub) for f in sorted(os.listdir(p)): if f.endswith(".md"): scan_files.append(("流程/" + f, os.path.join(p, f))) scan_files.append(("创作流程规范.md", os.path.join(BASE, "references", "创作流程规范.md"))) scan_files.append(("SKILL.md", os.path.join(BASE, "SKILL.md"))) # 知识库内部交叉引用 for c, fs in kb_files.items(): for f in fs: scan_files.append((f"知识库/{c}/{f}", os.path.join(KB, c, f))) FN = r'[\w()()·::\-\\\u4e00-\u9fff/]+?\.md' file_ref_re = re.compile(FN) cat_name_re = re.compile(r'(\d{2}_[\w()()·\-\u4e00-\u9fff]+)') block_re = re.compile(r'分块(\d+)') cat_set = set(kb_files.keys()) referenced = set() # (cat, fn) broken = [] for label, path in scan_files: try: lines = open(path, encoding="utf-8").read().splitlines() except Exception as e: print(f"READ FAIL {label}: {e}") continue for i, line in enumerate(lines, 1): cats_in_line = [c for c in cat_set if c in line] # 文件名引用 for m in file_ref_re.finditer(line): ref = m.group(0).replace("\\", "/") fn = ref.split("/")[-1] # 尝试带路径匹配 hit = None for c, fs in kb_files.items(): if fn in fs: hit = (c, fn) if hit: referenced.add(hit) elif "知识库" in line or ref.startswith("知识库") or "/" in ref: # 可能是知识库引用但没匹配上(也排除产物/流程自身引用) if not any(fn == x for x in ["创作流程规范.md"]) and not fn.startswith(("0", "1")) or "知识库" in ref: if "知识库" in ref or ("知识库" in line and ("/" in ref)): broken.append((label, i, ref, line.strip()[:100])) # 分块N引用(限当前行有分类上下文) for m in block_re.finditer(line): n = int(m.group(1)) for c in cats_in_line: if c in block_cats and n in block_cats[c]: referenced.add((c, block_cats[c][n])) print(f"总文件: {total}") print("\n===== A. 疑似断链(引用了知识库路径但未命中文件) =====") seen = set() for b in broken: k = (b[2],) if k in seen: continue seen.add(k) print(f" {b[0]}:{b[1]} -> {b[2]}\n {b[3]}") if not broken: print(" 无") print("\n===== B. 零引用文件(流程+规范+SKILL+知识库内部均未引用文件名/分块号) =====") orphans = [] for c, fs in kb_files.items(): for fn in fs: if (c, fn) not in referenced: orphans.append(f"{c}/{fn}") print(f"零引用: {len(orphans)} / {total}") for o in orphans: print(f" {o}") print("\n===== C. 分类引用统计 =====") for c, fs in kb_files.items(): used = sum(1 for fn in fs if (c, fn) in referenced) print(f" {c}: {used}/{len(fs)}")