Files
mcn-short-video/.workbuddy/skills/nuwa-skill-main/scripts/quality_check.py
T

280 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
自动检查生成的SKILL.md是否通过Phase 4质量标准。
对照通过标准表格逐项检查,输出通过/不通过和具体原因。
用法:
python3 quality_check.py <SKILL.md路径>
示例:
python3 quality_check.py .claude/skills/elon-musk-perspective/SKILL.md
"""
import sys
import re
from pathlib import Path
def check_mental_models(content: str) -> tuple[bool, str]:
"""检查心智模型数量(3-7个)"""
# 匹配 ### 模型N: 或 ### N. 等模式
models = re.findall(r'^###\s+(?:模型|Model|心智模型)\s*\d', content, re.MULTILINE)
if not models:
# fallback: 数「### 」开头的行在心智模型section中
in_section = False
count = 0
for line in content.split('\n'):
if re.match(r'^##\s+.*心智模型|Mental Model', line, re.IGNORECASE):
in_section = True
continue
if in_section and re.match(r'^##\s+', line) and '心智模型' not in line:
break
if in_section and re.match(r'^###\s+', line):
count += 1
if count > 0:
passed = 3 <= count <= 7
return passed, f"{count}个心智模型 {'✅' if passed else '❌ (应为3-7个)'}"
count = len(models)
if count == 0:
return False, "未检测到心智模型section"
passed = 3 <= count <= 7
return passed, f"{count}个心智模型 {'✅' if passed else '❌ (应为3-7个)'}"
def check_limitations(content: str) -> tuple[bool, str]:
"""检查每个模型是否有局限性"""
has_limitation = bool(re.search(r'局限|失效|不适用|盲区|limitation|blind spot', content, re.IGNORECASE))
return has_limitation, "有局限性标注 ✅" if has_limitation else "❌ 未找到局限性描述"
def check_expression_dna(content: str) -> tuple[bool, str]:
"""检查表达DNA辨识度"""
dna_section = bool(re.search(r'表达DNA|Expression DNA|表达风格', content, re.IGNORECASE))
if not dna_section:
return False, "❌ 未找到表达DNA section"
# 检查是否有具体的风格描述(句式、词汇等)
style_markers = len(re.findall(r'句式|词汇|语气|幽默|节奏|确定性|引用|口头禅', content))
passed = style_markers >= 3
return passed, f"表达DNA特征: {style_markers}项 {'✅' if passed else '❌ (应≥3项)'}"
def check_honest_boundary(content: str) -> tuple[bool, str]:
"""检查诚实边界(至少3条)"""
# 找诚实边界section
boundary_match = re.search(r'(?:##\s+.*诚实边界|## Honest Boundary)(.*?)(?=\n##\s|\Z)', content, re.DOTALL | re.IGNORECASE)
if not boundary_match:
return False, "❌ 未找到诚实边界section"
boundary_text = boundary_match.group(1)
# 计算列表项
items = re.findall(r'^[-*]\s+', boundary_text, re.MULTILINE)
count = len(items)
passed = count >= 3
return passed, f"诚实边界: {count}条 {'✅' if passed else '❌ (应≥3条)'}"
def check_tensions(content: str) -> tuple[bool, str]:
"""检查内在张力(至少2对)"""
tension_markers = len(re.findall(r'张力|矛盾|tension|paradox|一方面.*另一方面|既.*又', content, re.IGNORECASE))
passed = tension_markers >= 2
return passed, f"内在张力: {tension_markers}处 {'✅' if passed else '❌ (应≥2处)'}"
def check_primary_sources(content: str) -> tuple[bool, str]:
"""检查一手来源占比"""
# 找调研来源section
source_section = re.search(r'(?:##\s+.*来源|## Source|## Reference)(.*?)(?=\n##\s|\Z)', content, re.DOTALL | re.IGNORECASE)
if not source_section:
return True, "未找到来源section(跳过检查)"
source_text = source_section.group(1)
primary = len(re.findall(r'一手|primary|本人著作|原始', source_text, re.IGNORECASE))
secondary = len(re.findall(r'二手|secondary|转述|评论', source_text, re.IGNORECASE))
total = primary + secondary
if total == 0:
return True, "未标记来源类型(跳过检查)"
ratio = primary / total
passed = ratio > 0.5
return passed, f"一手来源占比: {primary}/{total} ({ratio:.0%}) {'✅' if passed else '❌ (应>50%)'}"
def check_terminology_adaptation(content: str) -> tuple[bool, str]:
"""检查面向用户交付物中是否有内部术语残留。
检查section标题和正文,确保没有学术黑话。"""
# SKILL.md文件是给AI agent用的,保留术语是合理的
is_skill_file = bool(re.search(r'^---\s*\nname:\s*', content, re.MULTILINE))
if is_skill_file:
return True, "SKILL.md文件(给AI agent用),术语保留 ✅"
# 完整的禁用术语列表(含正文检查)
jargon_terms = [
'心智模型', '决策启发式', '表达DNA', '诚实边界',
'内在张力', '智识谱系', '反模式',
'三重验证', '跨域复现', '生成力', '排他性'
]
# 检查section标题(## 开头的行)
section_headers = re.findall(r'^##\s+.*$', content, re.MULTILINE)
jargon_in_headers = []
for header in section_headers:
for term in jargon_terms:
if term in header:
jargon_in_headers.append(f'标题"{header.strip()}"含"{term}"')
# 检查正文中的术语残留(排除在代码块/引用块中的情况)
jargon_in_body = []
lines = content.split('\n')
in_code_block = False
for i, line in enumerate(lines, 1):
if line.strip().startswith('```'):
in_code_block = not in_code_block
continue
if in_code_block:
continue
for term in jargon_terms:
if term in line:
# 排除在"禁用词列表"中提到术语本身的情况
if '禁用' in line or '不出现' in line or '❌' in line:
continue
jargon_in_body.append(f'第{i}行含"{term}": {line.strip()[:60]}...')
all_issues = jargon_in_headers + jargon_in_body[:5] # 最多显示5条正文问题
if all_issues:
return False, f"❌ 术语残留: {'; '.join(all_issues[:5])}"
return True, "面向用户交付物,无术语残留 ✅"
def check_persona_card_format(content: str) -> tuple[bool, str]:
"""检查短视频人设卡是否符合12段标准结构。
当文件名包含'人设'或'账号卡片'时触发此检查。"""
# 判断是否是人设卡
if not re.search(r'人设|账号卡片|persona', content[:200], re.IGNORECASE):
return True, "非人设卡文件(跳过12段结构检查)"
expected_sections = [
'账号核心定位', '账号内容基因', '角色原型库', '情绪结构',
'视听风格', '五大内容打法', '内容规则', '系列内容规划',
'差异化壁垒', '红线清单', '当前需解决的问题', '人设底线'
]
found_sections = []
missing_sections = []
for section in expected_sections:
if re.search(rf'##\s+.*{section}', content):
found_sections.append(section)
else:
missing_sections.append(section)
if missing_sections:
return False, f"❌ 缺少段落: {', '.join(missing_sections)}"
return True, f"12段结构完整 ✅ ({len(found_sections)}/12)"
# 检查学术化表达
def check_academic_writing(content: str) -> tuple[bool, str]:
"""检查正文中是否有学术化表达(人设卡场景)。"""
# SKILL.md文件不检查
is_skill_file = bool(re.search(r'^---\s*\nname:\s*', content, re.MULTILINE))
if is_skill_file:
return True, "SKILL.md文件(跳过学术化检查)"
academic_patterns = [
'认知框架', '降维打击', '范式', '策略性', '结构化',
'方法论体系', '底层逻辑架构', '思维范式', '认知操作系统'
]
found = []
lines = content.split('\n')
in_code_block = False
for i, line in enumerate(lines, 1):
if line.strip().startswith('```'):
in_code_block = not in_code_block
continue
if in_code_block:
continue
for pattern in academic_patterns:
if pattern in line:
found.append(f'第{i}行: "{pattern}"')
if found:
return False, f"❌ 学术化表达: {'; '.join(found[:3])}"
return True, "无学术化表达 ✅"
def main():
if len(sys.argv) < 2:
print("用法: python3 quality_check.py <文件路径>")
sys.exit(1)
skill_path = Path(sys.argv[1])
if not skill_path.exists():
print(f"❌ 文件不存在: {skill_path}")
sys.exit(1)
content = skill_path.read_text(encoding='utf-8')
# 判断文件类型
is_skill_file = bool(re.search(r'^---\s*\nname:\s*', content, re.MULTILINE))
is_persona_card = bool(re.search(r'人设|账号卡片|persona', content[:500], re.IGNORECASE))
if is_skill_file:
# SKILL.md文件:运行全部检查(内部术语是合理的)
checks = [
("心智模型数量", check_mental_models),
("模型局限性", check_limitations),
("表达DNA辨识度", check_expression_dna),
("诚实边界", check_honest_boundary),
("内在张力", check_tensions),
("一手来源占比", check_primary_sources),
("术语通俗化", check_terminology_adaptation),
]
file_type = "SKILL.md(AI agent Skill文件)"
elif is_persona_card:
# 人设卡:只运行面向用户的检查项
checks = [
("术语通俗化", check_terminology_adaptation),
("人设卡12段结构", check_persona_card_format),
("学术化表达", check_academic_writing),
]
file_type = "人设卡(面向编导/创作者)"
else:
# 其他面向用户的文档
checks = [
("术语通俗化", check_terminology_adaptation),
("学术化表达", check_academic_writing),
]
file_type = "面向用户文档"
print(f"质量检查: {skill_path.name}")
print(f"文件类型: {file_type}")
print("=" * 50)
passed_count = 0
total = len(checks)
for name, check_fn in checks:
passed, detail = check_fn(content)
status = "✅ PASS" if passed else "❌ FAIL"
print(f" {name:<12} {status} {detail}")
if passed:
passed_count += 1
print("=" * 50)
print(f"结果: {passed_count}/{total} 通过")
if passed_count == total:
print("🎉 全部通过,可以交付")
elif passed_count >= total - 1:
print("⚠️ 基本通过,建议修复不通过项后交付")
else:
print("❌ 多项不通过,建议回到Phase 2迭代")
sys.exit(0 if passed_count == total else 1)
if __name__ == '__main__':
main()