init the project

This commit is contained in:
evyzacq
2026-05-25 15:09:42 +08:00
commit 7fc0e7852e
122 changed files with 14557 additions and 0 deletions
+474
View File
@@ -0,0 +1,474 @@
"""
Stage 3: Deterministic Merge & Completeness Audit.
- Merges IR rule fragments, deduplicating by trigger+actions similarity.
- Reassigns stable rule_ids.
- Generates an audit report covering:
1. Logic tree node coverage
2. Table enumeration coverage
3. Global switch state coverage
Outputs:
- ir_final.json (in doc_parser output per spec)
- ir_audit_report.md (in doc_parser output)
"""
import json
import hashlib
import sys
from collections import defaultdict
from pathlib import Path
import config
PASS = "[PASS]"
WARN = "[WARN]"
FAIL = "[FAIL]"
def load_fragments() -> list[dict]:
"""Load IR fragments from Stage 2."""
return config.load_json(config.IR_FRAGMENTS_JSON)
def load_semantic_index() -> dict:
"""Load semantic index from Stage 1."""
return config.load_json(config.SEMANTIC_INDEX_JSON)
def rule_signature(rule: dict) -> str:
"""Generate a dedup signature from trigger + actions.
Two rules with identical trigger conditions and actions produce
the same signature and should be merged.
"""
trigger = rule.get("trigger", {})
actions = rule.get("actions", [])
# Normalize: sort conditions by signal name for stability
conditions = sorted(trigger.get("conditions", []), key=lambda c: c.get("signal", ""))
# Sort actions by description
sorted_actions = sorted(actions, key=lambda a: a.get("description", ""))
sig_data = {
"conditions": conditions,
"actions": sorted_actions,
}
sig_json = json.dumps(sig_data, ensure_ascii=False, sort_keys=True)
return hashlib.sha256(sig_json.encode()).hexdigest()[:16]
def merge_rules(fragments: list[dict]) -> list[dict]:
"""Merge rules across all fragments, deduplicating by trigger+actions."""
signature_map: dict[str, dict] = {}
order = []
for fragment in fragments:
for rule in fragment.get("rules", []):
sig = rule_signature(rule)
if sig in signature_map:
# Merge sources
existing = signature_map[sig]
existing_sources = existing.setdefault("sources", [])
for src in rule.get("sources", []):
if src not in existing_sources:
existing_sources.append(src)
# Use the more detailed description
if len(rule.get("description", "")) > len(existing.get("description", "")):
existing["description"] = rule["description"]
else:
signature_map[sig] = dict(rule)
order.append(sig)
merged = [signature_map[sig] for sig in order]
print(f" 合并前: {sum(len(f.get('rules', [])) for f in fragments)} 条规则")
print(f" 合并后: {len(merged)} 条规则")
return merged
def assign_rule_ids(rules: list[dict], feature_id: str = "DRL-001") -> list[dict]:
"""Reassign stable rule_ids based on type and sequence."""
type_counters = defaultdict(int)
for rule in rules:
# Determine type from the first action's type
actions = rule.get("actions", [])
if any(a.get("type") == "user_interaction" for a in actions) and \
not any(a.get("type") == "system" for a in actions):
rtype = "UI"
elif any("SDK" in str(a) for a in actions):
rtype = "SDK"
else:
rtype = "SYS"
type_counters[rtype] += 1
seq = type_counters[rtype]
rule["rule_id"] = f"{feature_id}-{rtype}-FG-{seq:02d}"
# Also generate top-level feature metadata
return rules
def find_all_logic_tree_nodes(doc: dict) -> dict[str, list[dict]]:
"""Return {image_id: [all nodes]} for all logic trees."""
result = {}
for img in doc.get("image_analysis", []):
lt = img.get("logic_tree")
rid = img.get("rid", "")
if lt and rid:
result[rid] = lt.get("nodes", [])
return result
def find_referenced_nodes(rules: list[dict]) -> dict[str, set[str]]:
"""Return {image_id: {referenced node ids}} across all rules."""
referenced = defaultdict(set)
for rule in rules:
for src in rule.get("sources", []):
if src.get("type") == "logic_tree":
image_id = src.get("image_id", "")
for nid in src.get("node_ids", []):
referenced[image_id].add(nid)
return dict(referenced)
def audit_logic_tree_coverage(
doc: dict, rules: list[dict]
) -> list[dict]:
"""Generate coverage statistics for logic tree nodes."""
all_nodes = find_all_logic_tree_nodes(doc)
referenced = find_referenced_nodes(rules)
results = []
for image_id, nodes in all_nodes.items():
ref_set = referenced.get(image_id, set())
decision_nodes = [n for n in nodes if n["type"] == "decision"]
action_nodes = [n for n in nodes if n["type"] == "action"]
state_nodes = [n for n in nodes if n["type"] == "state"]
decisions_covered = [n for n in decision_nodes if n["id"] in ref_set]
actions_covered = [n for n in action_nodes if n["id"] in ref_set]
decisions_uncovered = [n for n in decision_nodes if n["id"] not in ref_set]
actions_uncovered = [n for n in action_nodes if n["id"] not in ref_set]
total_checkable = len(decision_nodes) + len(action_nodes)
total_covered = len(decisions_covered) + len(actions_covered)
coverage = (total_covered / total_checkable * 100) if total_checkable > 0 else 100
status = PASS if coverage >= 95 else (WARN if coverage >= 70 else FAIL)
detail_parts = [f"{total_covered}/{total_checkable} decision+action 节点被引用"]
if decisions_uncovered:
detail_parts.append(
f"未覆盖的 decision: {[n['id'] + ': ' + n.get('condition','')[:40] for n in decisions_uncovered]}"
)
if actions_uncovered:
detail_parts.append(
f"未覆盖的 action: {[n['id'] + ': ' + n.get('description','')[:40] for n in actions_uncovered]}"
)
results.append({
"check": f"逻辑树 {image_id} 节点覆盖率",
"status": status,
"coverage_pct": round(coverage, 1),
"detail": "; ".join(detail_parts),
"image_id": image_id,
"uncovered_decisions": decisions_uncovered,
"uncovered_actions": actions_uncovered,
})
return results
def find_table_enums(doc: dict) -> list[dict]:
"""Find enumerated values in tables (e.g., app types, limit methods)."""
enums = []
for section in doc.get("sections", []):
for block in section.get("blocks", []):
if block["type"] != "table":
continue
headers = block.get("headers", [])
if not headers:
continue
# Look for the "功能" / "功能详细说明" table pattern (key-value pairs)
if "功能" in headers and "功能详细说明" in headers:
for row in block.get("rows", []):
cols = row.get("columns", [])
key_col = next((c for c in cols if c.get("name") == "功能"), None)
val_col = next(
(c for c in cols if c.get("name") == "功能详细说明"), None
)
if key_col and val_col:
enums.append({
"section": section.get("source", ""),
"row": key_col.get("row"),
"key": key_col.get("text", ""),
"value": val_col.get("text", ""),
})
else:
# Generic table: record first column values as potential enum
first_col_name = headers[0] if headers else ""
values = []
for row in block.get("rows", []):
for col in row.get("columns", []):
if col.get("name") == first_col_name:
values.append(col.get("text", ""))
if values:
enums.append({
"section": section.get("source", ""),
"column": first_col_name,
"values": values,
})
return enums
def audit_table_enums(rules: list[dict], doc: dict) -> list[dict]:
"""Check if key enumerated values appear in rule preconditions."""
results = []
table_enums = find_table_enums(doc)
# Collect all rule precondition fields and their values
rule_preconditions = []
for rule in rules:
precond = rule.get("precondition", {})
rule_preconditions.append(precond)
# Check specific enum categories
app_types = {"系统限制", "SDK限制", "其他应用"}
switch_states = {"开启", "关闭"}
app_states = {"前台", "后台"}
# App type coverage
found_app_types = set()
for precond in rule_preconditions:
at = precond.get("app_type", "")
if at:
found_app_types.add(at)
missing_types = app_types - found_app_types
results.append({
"check": "应用类型枚举覆盖",
"status": PASS if not missing_types else WARN,
"detail": f"已覆盖: {found_app_types or ''}"
+ (f"; 未覆盖: {missing_types}" if missing_types else ""),
})
# App state coverage
found_states = set()
for precond in rule_preconditions:
st = precond.get("app_state", "")
if st:
found_states.add(st)
missing_states = app_states - found_states
results.append({
"check": "应用前后台状态覆盖",
"status": PASS if not missing_states else WARN,
"detail": f"已覆盖: {found_states or ''}"
+ (f"; 未覆盖: {missing_states}" if missing_states else ""),
})
# Trigger signal coverage (check each table enum key appears)
trigger_signals = set()
for rule in rules:
for cond in rule.get("trigger", {}).get("conditions", []):
signal = cond.get("signal", "")
if signal:
trigger_signals.add(signal)
# Check if key concepts from the doc appear in signals
key_signals = {"车速", "档位", "车速_持续时间", "应用请求启动"}
missing_signals = key_signals - trigger_signals
results.append({
"check": "触发信号覆盖(车速/档位/持续时间/启动请求)",
"status": PASS if not missing_signals else WARN,
"detail": f"已覆盖信号: {sorted(trigger_signals)}"
+ (f"; 未覆盖: {missing_signals}" if missing_signals else ""),
})
return results
def audit_switch_coverage(rules: list[dict]) -> list[dict]:
"""Check that rules cover both switch ON and OFF states."""
switch_on = False
switch_off = False
switch_rules = []
for rule in rules:
precond = rule.get("precondition", {})
sw = precond.get("switch", "")
if sw == "开启":
switch_on = True
switch_rules.append(rule.get("rule_id", "?"))
elif sw == "关闭":
switch_off = True
switch_rules.append(rule.get("rule_id", "?"))
status = PASS
detail_parts = []
if switch_on:
detail_parts.append(f"开关=开启: 有规则覆盖")
else:
detail_parts.append(f"开关=开启: 未找到规则")
status = FAIL
if switch_off:
detail_parts.append(f"开关=关闭: 有规则覆盖")
else:
detail_parts.append(f"开关=关闭: 未找到规则")
status = FAIL
return [{
"check": "开关状态完整性(开启/关闭)",
"status": status,
"detail": "; ".join(detail_parts),
}]
def generate_audit_report(
rules: list[dict],
doc: dict,
feature_name: str,
lt_results: list[dict],
enum_results: list[dict],
switch_results: list[dict],
) -> str:
"""Generate ir_audit_report.md in Markdown format."""
lines = []
lines.append(f"# IR 完整性审计报告")
lines.append(f"")
lines.append(f"**功能**: {feature_name}")
lines.append(f"**规则总数**: {len(rules)}")
lines.append(f"**生成时间**: {__import__('datetime').datetime.now().isoformat()}")
lines.append(f"")
# Human review notice
lines.append(f"> ⚠️ **重要**: 请人工审查以下 ⚠️ 和 ❌ 项,确认是文档遗漏还是 IR 提取遗漏。")
lines.append(f'> 如无需修改,在对应项后标注 **"已确认"**。')
lines.append(f"")
# ---- Logic Tree Coverage ----
lines.append(f"## 1. 逻辑树节点覆盖率")
lines.append(f"")
lines.append(f"| 图片 ID | 覆盖率 | 状态 | 详情 |")
lines.append(f"|---------|--------|------|------|")
for r in lt_results:
lines.append(
f"| {r['image_id']} | {r['coverage_pct']}% | {r['status']} | {r['detail']} |"
)
lines.append(f"")
# Uncovered node details
for r in lt_results:
if r["uncovered_decisions"] or r["uncovered_actions"]:
lines.append(f"### {r['image_id']} 未覆盖节点详情")
lines.append(f"")
for n in r.get("uncovered_decisions", []):
lines.append(f"- **Decision** `{n['id']}`: {n.get('condition', '?')}")
for n in r.get("uncovered_actions", []):
lines.append(f"- **Action** `{n['id']}`: {n.get('description', '?')}")
lines.append(f"")
# ---- Table Enumeration Coverage ----
lines.append(f"## 2. 表格枚举覆盖")
lines.append(f"")
lines.append(f"| 检查项 | 状态 | 详情 |")
lines.append(f"|--------|------|------|")
for r in enum_results:
lines.append(f"| {r['check']} | {r['status']} | {r['detail']} |")
lines.append(f"")
# ---- Switch Coverage ----
lines.append(f"## 3. 全局开关状态覆盖")
lines.append(f"")
lines.append(f"| 检查项 | 状态 | 详情 |")
lines.append(f"|--------|------|------|")
for r in switch_results:
lines.append(f"| {r['check']} | {r['status']} | {r['detail']} |")
lines.append(f"")
# ---- Rule Summary ----
lines.append(f"## 4. 规则清单")
lines.append(f"")
lines.append(f"| rule_id | Priority | 简述 |")
lines.append(f"|---------|----------|------|")
for rule in rules:
desc = rule.get("description", "")[:80]
lines.append(f"| {rule.get('rule_id', '?')} | {rule.get('priority', '?')} | {desc} |")
lines.append(f"")
return "\n".join(lines)
def main():
print("=" * 60)
print("阶段三:确定性合并与完整性校验")
print("=" * 60)
# 1. Load inputs
print(f"\n[1/5] 加载输入...")
fragments = load_fragments()
doc = config.load_input_document()
semantic_index = load_semantic_index()
feature_name = semantic_index.get("feature_name", "行车娱乐限制")
feature_id = "DRL-001"
print(f" 功能: {feature_name} ({feature_id})")
print(f" 片段数: {len(fragments)}")
# 2. Merge rules
print(f"\n[2/5] 合并去重...")
merged_rules = merge_rules(fragments)
# 3. Reassign rule IDs
print(f"\n[3/5] 重分配 rule_id...")
final_rules = assign_rule_ids(merged_rules, feature_id)
print(f" 已分配 {len(final_rules)} 个稳定 ID")
# Collect top-level metadata
ir_final = {
"feature": feature_name,
"feature_id": feature_id,
"rules": final_rules,
}
# Save ir_final.json
print(f"\n[4/5] 生成审计报告...")
lt_results = audit_logic_tree_coverage(doc, final_rules)
enum_results = audit_table_enums(final_rules, doc)
switch_results = audit_switch_coverage(final_rules)
report = generate_audit_report(
final_rules, doc, feature_name,
lt_results, enum_results, switch_results
)
# 5. Save outputs
print(f"\n[5/5] 保存输出...")
config.save_json(ir_final, config.IR_FINAL_JSON)
print(f" IR: {config.IR_FINAL_JSON}")
with open(config.IR_AUDIT_REPORT_MD, "w", encoding="utf-8") as f:
f.write(report)
print(f" 审计报告: {config.IR_AUDIT_REPORT_MD}")
# Print quick summary
print(f"\n完成!")
issue_count = sum(
1 for r in lt_results + enum_results + switch_results
if r["status"] in (WARN, FAIL)
)
print(f" 规则: {len(final_rules)}")
print(f" 审计问题: {issue_count} 个需要人工审查")
if issue_count > 0:
print(f"\n 请查看 {config.IR_AUDIT_REPORT_MD} 并审查标记项。")
if __name__ == "__main__":
main()