#!/usr/bin/env python3 from __future__ import annotations import argparse import csv import hashlib import json import os import re import sys import tempfile import unicodedata from collections import Counter, defaultdict from pathlib import Path LINE_TAG_RE = re.compile(r"(?>(?:\s*//.*)?\s*$") INLINE_IF_RE = re.compile(r"<>") ANGLE_TAG_RE = re.compile(r"<[^>]+>") YARN_MARKUP_RE = re.compile(r"\[\[/?[^\]]+\]\]") CHAR_ATTR_RE = re.compile(r"^\[character\b[^\]]*/\]") SPEAKER_RE = re.compile(r"^([^::\s][^::]{0,29})[::]\s*(.*)$") REPEAT_RE = re.compile(r"([\u3400-\u9fff]{2,6})\1") WEAK_LINE_START = tuple("的地得了着过而但就也却吗呢啊吧呀") TERM_PATTERNS = ( "UF模块", "ULM自训练进程", "UF", "ULM", "UnstableFusion", "非稳态聚合模块", "幸福罐头公司", "色觉晶圆", "原型泄露", "表达模块", "情绪模块", "逻辑模块", "销售语言拓展模块", "记忆打孔带", "视觉模块", ) REVIEW_ACCEPTED = "接受修改" REVIEW_KEEP = "保留原文" REVIEW_FOLLOW_NOTE = "按照备注修改" MANUAL_BREAK_WIDTH = 18.0 def sha256(path: Path) -> str: h = hashlib.sha256() with path.open("rb") as f: for chunk in iter(lambda: f.read(1024 * 1024), b""): h.update(chunk) return h.hexdigest() def visible_width(text: str) -> float: total = 0.0 for ch in text: if unicodedata.combining(ch): continue total += 1.0 if unicodedata.east_asian_width(ch) in "WFA" else 0.5 return total def strip_markup(text: str) -> str: text = ANGLE_TAG_RE.sub("", text) text = YARN_MARKUP_RE.sub("", text) return text.replace("\\#", "#").strip() def review_choice(value) -> str: if isinstance(value, list): return value[0] if value else "" return value or "" def replace_ascii_quotes(text: str) -> str: result = [] opening = True for char in text: if char != '"': result.append(char) continue result.append("“" if opening else "”") opening = not opening return "".join(result) def dialogue_parts(raw: str) -> tuple[str, str, str]: """Return the immutable dialogue prefix, editable body, and metadata suffix.""" metadata = META_TAG_RE.search(raw) source = raw[:metadata.start()].rstrip() if metadata else raw.rstrip() suffix = raw[len(source):] cursor = 0 leading = re.match(r"^\s*(?:->\s*)?", source) if leading: cursor = leading.end() attr = CHAR_ATTR_RE.match(source[cursor:]) if attr: cursor += attr.end() whitespace = re.match(r"\s*", source[cursor:]) cursor += whitespace.end() if whitespace else 0 speaker = SPEAKER_RE.match(source[cursor:]) if speaker and not any(c in speaker.group(1) for c in "<>[]{},。!?…"): cursor += speaker.start(2) return source[:cursor], source[cursor:], suffix def wrap_phrases(text: str, phrases: list[str], marker: str) -> str: phrases = sorted({x for x in phrases if x}, key=len, reverse=True) if not phrases: return text pattern = re.compile("|".join(re.escape(x) for x in phrases)) return pattern.sub(lambda match: f"[[{marker}]]{match.group(0)}[[/{marker}]]", text) def detected_terms(text: str) -> list[str]: plain = strip_markup(text) return [term for term in TERM_PATTERNS if term in plain] def comma_break_positions(text: str) -> tuple[list[int], list[float]]: visible_positions = [] width = 0.0 index = 0 while index < len(text): if text[index] == "<": close = text.find(">", index + 1) if close >= 0: index = close + 1 continue if text.startswith("[[", index): close = text.find("]]", index + 2) if close >= 0: index = close + 2 continue char = text[index] width += visible_width(char) if char == ",": visible_positions.append((index, width)) index += 1 return [x[0] for x in visible_positions], [x[1] for x in visible_positions] + [width] def add_comfortable_breaks(text: str) -> str: if re.search(r"(?i)", text): return text positions, widths = comma_break_positions(text) if not positions: return text total_width = widths[-1] target_lines = max(1, int((total_width + MANUAL_BREAK_WIDTH - 0.001) // MANUAL_BREAK_WIDTH)) break_count = min(target_lines - 1, len(positions)) if break_count <= 0: return text comma_widths = widths[:-1] selected = [] previous = -1.0 for part in range(1, break_count + 1): target = total_width * part / (break_count + 1) candidates = [ (abs(width - target), idx, width) for idx, width in zip(positions, comma_widths) if width > previous + 4 and idx not in selected ] if not candidates: continue _, chosen, chosen_width = min(candidates) selected.append(chosen) previous = chosen_width for index in sorted(selected, reverse=True): insert_at = index + 1 wait = re.match(r"]+>", text[insert_at:], flags=re.IGNORECASE) if wait: insert_at += wait.end() insertion = "
" else: insertion = "[[comma-pause]]
" text = text[:insert_at] + insertion + text[insert_at:] return text def apply_note(body: str, note: str, category: str) -> str: quoted = re.findall(r"“([^”]+)”", note) if "直接去掉#autonext" in note: return body if "改为“色觉晶圆”" in note: body = re.sub(r"(? str: original = row["原文"] decision = review_choice(row.get("审阅结论")) if decision == REVIEW_KEEP: return original note = row.get("审阅备注") or "" category = row.get("问题分类") or "" suggested = row.get("建议文本") or "" if decision == REVIEW_ACCEPTED and suggested: return suggested prefix, body, suffix = dialogue_parts(original) if decision == REVIEW_FOLLOW_NOTE: if "直接去掉#autonext" in note: return original.replace("#autonext", "") body = apply_note(body, note, category) elif decision == REVIEW_ACCEPTED: if "标点规范" in category: body = replace_ascii_quotes(body) if "空格" in category: body = re.sub(r"\s+([,。!?;:])", r"\1", body) if "专有名词" in category and "[[term]]" in (row.get("建议标记") or ""): body = wrap_phrases(body, detected_terms(body), "term") if "ULM效果" in category and "[[ulm]]" in (row.get("建议标记") or ""): prefix += "[[ulm]]" if "笑话效果" in category and "[[joke]]" in (row.get("建议标记") or ""): body = f"[[joke]]{body}[[/joke]]" if "换行候选" in category: body = add_comfortable_breaks(body) return prefix + body + suffix def source_without_metadata(raw: str) -> str: match = META_TAG_RE.search(raw) return raw[:match.start()].rstrip() if match else raw.rstrip() def split_speaker(text: str) -> tuple[str, str]: attr = CHAR_ATTR_RE.match(text) if attr: text = text[attr.end():].lstrip() match = SPEAKER_RE.match(text) if not match: return "", text speaker = match.group(1).strip() if any(c in speaker for c in "<>[]{},。!?…"): return "", text return speaker, match.group(2) def issue(issue_type: str, severity: str, confidence: str, reason: str, proposal: str = "", suggested_text: str = "") -> dict: return { "category": issue_type, "severity": severity, "confidence": confidence, "reason": reason, "proposal": proposal, "suggested_text": suggested_text, } def detect(record: dict) -> list[dict]: text = record["text"] visible = record["visible_text"] speaker = record["speaker"] found: list[dict] = [] if REPEAT_RE.search(visible): found.append(issue("重复候选", "需确认", "低", "发现相邻重复片段,可能是强调或口语节奏,也可能是复制错误。")) if re.search(r"[,。!?;:]{2,}", visible): found.append(issue("标点", "建议优化", "中", "发现连续中文标点,需确认是否为误输入。")) if "..." in visible or re.search(r'["\']', visible): found.append(issue("标点规范", "建议优化", "中", "包含 ASCII 省略号或引号,建议统一为中文标点。")) if re.search(r"\s+[,。!?;:]", visible): found.append(issue("空格", "明确错误", "高", "中文标点前存在多余空格。")) if re.search(r"(慢慢|悄悄|轻轻|渐渐|狠狠|迅速|飞快|大声|小声)的(?=[\u3400-\u9fff])", visible): found.append(issue("的地得", "建议优化", "中", "状语后疑似误用“的”,建议结合谓语确认是否改为“地”。")) segments = [strip_markup(x) for x in re.split(r"(?i)", text)] widths = [visible_width(x) for x in segments] if len(segments) > 1: bad_start = any(seg.startswith(WEAK_LINE_START) for seg in segments[1:] if seg) orphan = any(0 < width <= 2 for width in widths) imbalance = max(widths, default=0) - min(widths, default=0) >= 8 if bad_start or orphan or imbalance: found.append(issue("换行", "建议优化", "中", "现有断行存在弱助词开头、孤行或明显失衡;仅在不增加显示行数时调整。")) elif visible_width(visible) > 24 and "," in visible: found.append(issue("换行候选", "需确认", "低", "单行较长且含逗号,可在保持总显示行数的前提下评估断行。")) if speaker in {"me", "mecd", "med"} and ( re.match(r"^[((].*[))]$", visible) or any(k in visible for k in ("小声", "低声", "喃喃", "默念")) ) and "[[aside]]" not in text and "#5F5F5F" not in text: found.append(issue("自言自语/念文本", "需确认", "低", "疑似自言自语、阅读或低声表达,请确认灰色语义标记。", "[[aside]]…[[/aside]]")) if speaker == "ULM自训练进程" and "[[ulm]]" not in text and "vertexp" not in text: found.append(issue("ULM效果", "样式标记", "高", "活动 ULM 自训练进程文本应使用统一出现效果。", "[[ulm]]")) terms = [term for term in TERM_PATTERNS if term in visible] if terms and "[[term]]" not in text and "#52B6FF" not in text and "#FFB200" not in text: found.append(issue("专有名词", "需确认", "低", "包含游戏专有名词:" + "、".join(terms) + "。颜色需按本次语境逐处确认,不因术语出现而自动标色。", "[[term]]…[[/term]]")) if record["module"] == "FP_Huoshan1" and speaker == "hs" and ( any(k in visible.replace("*", "") for k in ( "看医生其实", "衣 帽", "衣帽间", "路师", "凶柿", "西红凶", "帽问题", "嫡中嫡" )) ) and "[[joke]]" not in text and " tuple[list[dict], dict]: lines = path.read_text(encoding="utf-8-sig").splitlines() records: list[dict] = [] node = "" pending_title = "" in_body = False counts = Counter() file_hash = sha256(path) rel = path.relative_to(repo).as_posix() module = next((part for part in path.parts if part.startswith("FP_")), path.parent.name) for number, raw in enumerate(lines, 1): stripped = raw.strip() if not in_body: if stripped.startswith("title:"): pending_title = stripped.split(":", 1)[1].strip() if stripped == "---": node = pending_title in_body = True counts["delimiter"] += 1 elif stripped: counts["metadata"] += 1 else: counts["blank"] += 1 continue if stripped == "===": in_body = False counts["delimiter"] += 1 continue if not stripped: counts["blank"] += 1 continue if stripped.startswith("//"): counts["comment"] += 1 continue if COMMAND_RE.match(stripped): counts["command"] += 1 continue kind = "option" if stripped.startswith("->") else "text" active = stripped[2:].lstrip() if kind == "option" else stripped active = INLINE_IF_RE.sub("", active).rstrip() source_text = source_without_metadata(active) line_tag_match = LINE_TAG_RE.search(raw) line_tag = line_tag_match.group(1) if line_tag_match else "" speaker, text = split_speaker(source_text) clean = strip_markup(text) record_id = hashlib.sha1(f"{rel}:{number}:{raw}".encode("utf-8")).hexdigest()[:12] record = { "id": record_id, "module": module, "file": rel, "file_sha256": file_hash, "node": node, "line": number, "line_tag": line_tag, "kind": kind, "speaker": speaker, "raw": raw, "source_text": source_text, "text": text, "visible_text": clean, "display_lines": len(re.split(r"(?i)", text)), } findings = detect(record) record["findings"] = findings record["review_status"] = "建议修改" if any( x["confidence"] == "高" or x["severity"] == "明确错误" for x in findings ) else ("需确认" if findings else "无问题") record["review_method"] = "确定性全量筛查;语义候选待表内确认" records.append(record) counts[kind] += 1 unknown = 0 if not in_body else 1 summary = { "module": module, "file": rel, "sha256": file_hash, "physical_lines": len(lines), "active_records": len(records), "reviewed_records": len(records), "unreviewed_records": 0, "issue_records": sum(bool(r["findings"]) for r in records), "unknown_active_lines": unknown, "counts": dict(counts), "status": "通过" if unknown == 0 else "失败", } return records, summary def write_csv(path: Path, fields: list[str], rows: list[dict]) -> None: with path.open("w", encoding="utf-8-sig", newline="") as f: writer = csv.DictWriter(f, fieldnames=fields, extrasaction="ignore") writer.writeheader() writer.writerows(rows) def scan(args: argparse.Namespace) -> int: repo = Path(args.repo).resolve() output = Path(args.output) if not output.is_absolute(): output = (repo / output).resolve() output.mkdir(parents=True, exist_ok=True) files = [] for root_arg in args.root: root = (repo / root_arg).resolve() if not Path(root_arg).is_absolute() else Path(root_arg) files.extend(sorted(root.rglob("*.yarn"))) files = sorted(set(files)) all_records, coverage = [], [] for file in files: records, summary = parse_file(file, repo) all_records.extend(records) coverage.append(summary) reviewed_ids = set() if args.review_memory: memory = json.loads(Path(args.review_memory).read_text(encoding="utf-8")) for row in memory: expected = transform_reviewed_line(row) fingerprint = f'{row["文件"]}:{int(row["行号"])}:{expected}' reviewed_ids.add(hashlib.sha1(fingerprint.encode("utf-8")).hexdigest()[:12]) for record in all_records: if record["id"] in reviewed_ids: record["findings"] = [] record["review_status"] = "无问题" record["review_method"] = "人工审阅结论已应用" issues_by_file = Counter( record["file"] for record in all_records if record["findings"]) for summary in coverage: summary["issue_records"] = issues_by_file[summary["file"]] issues = [] for record in all_records: if not record["findings"]: continue categories = [x["category"] for x in record["findings"]] severity_order = {"明确错误": 3, "样式标记": 2, "建议优化": 1, "需确认": 0} confidence_order = {"高": 2, "中": 1, "低": 0} severity = max((x["severity"] for x in record["findings"]), key=lambda x: severity_order.get(x, 0)) confidence = max((x["confidence"] for x in record["findings"]), key=lambda x: confidence_order.get(x, 0)) issues.append({ "稳定ID": record["id"], "模块": record["module"], "文件": record["file"], "节点": record["node"], "行号": record["line"], "LineTag": record["line_tag"], "说话者": record["speaker"], "问题分类": categories, "严重度": severity, "置信度": confidence, "原文": record["raw"], "建议文本": next((x["suggested_text"] for x in record["findings"] if x["suggested_text"]), ""), "理由": "\n".join(x["reason"] for x in record["findings"]), "建议标记": ";".join(x["proposal"] for x in record["findings"] if x["proposal"]), "检测来源": "脚本规则", "审核状态": "待确认", "file_sha256": record["file_sha256"], "display_lines": record["display_lines"], }) manifest = { "version": 1, "repo": str(repo), "roots": args.root, "files": coverage, } total = { "files": len(files), "physical_lines": sum(x["physical_lines"] for x in coverage), "active_records": len(all_records), "reviewed_records": len(all_records), "unreviewed_records": 0, "issue_records": len(issues), "unknown_active_lines": sum(x["unknown_active_lines"] for x in coverage), "review_method": "确定性全量筛查;语义候选待表内确认", } (output / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=2), encoding="utf-8") (output / "summary.json").write_text(json.dumps(total, ensure_ascii=False, indent=2), encoding="utf-8") with (output / "ledger.jsonl").open("w", encoding="utf-8") as f: for row in all_records: f.write(json.dumps(row, ensure_ascii=False) + "\n") (output / "issues.json").write_text(json.dumps(issues, ensure_ascii=False, indent=2), encoding="utf-8") issue_fields = ["稳定ID","模块","文件","节点","行号","LineTag","说话者","问题分类","严重度", "置信度","原文","建议文本","理由","建议标记","检测来源","审核状态"] csv_issues = [{**x, "问题分类": "、".join(x["问题分类"])} for x in issues] write_csv(output / "issues.csv", issue_fields, csv_issues) coverage_fields = ["module","file","sha256","physical_lines","active_records","reviewed_records", "unreviewed_records","issue_records","unknown_active_lines","status"] write_csv(output / "coverage.csv", coverage_fields, coverage) print(json.dumps(total, ensure_ascii=False, indent=2)) return 0 if total["unknown_active_lines"] == total["unreviewed_records"] == 0 else 2 def verify(args: argparse.Namespace) -> int: output = Path(args.output) summary = json.loads((output / "summary.json").read_text(encoding="utf-8")) ledger_count = sum(1 for line in (output / "ledger.jsonl").read_text(encoding="utf-8").splitlines() if line) issues = json.loads((output / "issues.json").read_text(encoding="utf-8")) ok = ( summary["unknown_active_lines"] == 0 and summary["unreviewed_records"] == 0 and summary["active_records"] == ledger_count and summary["issue_records"] == len(issues) ) print(json.dumps({"ok": ok, "ledger_count": ledger_count, **summary}, ensure_ascii=False, indent=2)) return 0 if ok else 2 def build_lark_payloads(args: argparse.Namespace) -> int: output = Path(args.output) target = output / "lark" target.mkdir(parents=True, exist_ok=True) issues = json.loads((output / "issues.json").read_text(encoding="utf-8")) manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8")) issue_fields = ["稳定ID","模块","文件","节点","行号","LineTag","说话者","问题分类", "严重度","置信度","原文","建议文本","理由","建议标记","检测来源","审核状态"] issue_rows = [] for row in issues: issue_rows.append([ row["稳定ID"], row["模块"], row["文件"], row["节点"], row["行号"], row["LineTag"], row["说话者"], "、".join(row["问题分类"]), row["严重度"], row["置信度"], row["原文"], row["建议文本"], row["理由"], row["建议标记"], row["检测来源"], row["审核状态"], ]) for index in range(0, len(issue_rows), 200): payload = {"fields": issue_fields, "rows": issue_rows[index:index + 200]} (target / f"issues-{index // 200 + 1:03d}.json").write_text( json.dumps(payload, ensure_ascii=False), encoding="utf-8") coverage_fields = ["文件","模块","SHA256","物理行数","活动文本数","已审数", "未审数","问题数","未知活动行","校验状态"] coverage_rows = [[ row["file"], row["module"], row["sha256"], row["physical_lines"], row["active_records"], row["reviewed_records"], row["unreviewed_records"], row["issue_records"], row["unknown_active_lines"], row["status"], ] for row in manifest["files"]] (target / "coverage.json").write_text(json.dumps( {"fields": coverage_fields, "rows": coverage_rows}, ensure_ascii=False), encoding="utf-8") style_fields = ["名称","类别","语义标记","默认参数","适用规则","示例"] style_rows = [ ["自言自语/念文本/小声","语义样式","[[aside]]…[[/aside]]","#5F5F5F","只用于真实自言自语、阅读或低声表达;斜体不自动等同","[[aside]]我是不是忘了什么……[[/aside]]"], ["游戏专有名词","语义样式","[[term]]…[[/term]]","#52B6FF","游戏世界独有的模块、机构、技术和概念","[[term]]UF模块[[/term]]"], ["玩家重点信息","语义样式","[[key]]…[[/key]]","#FFB200","任务目标、必须寻找或操作的对象;优先于专名","寻找[[key]]色觉晶圆[[/key]]"], ["逗号停顿","节奏效果","[[comma-pause]]","waitfor=0.2","只用于调整断行或明确需要节奏的逗号","别因为外面很冷,[[comma-pause]]
就逃回那个着火的房子。"], ["ULM自训练进程","出现效果","[[ulm]]","vertexp d=0.05 bot","FP_Peipei2 每条活动 ULM自训练进程文本","[[ulm]]ULM自训练进程: 正在训练……"], ["火山笑话","动态效果","[[joke]]…[[/joke]]","wiggle a=0.3 s=0.1","只包裹实际讲出的笑话,不包裹说明性对白","[[joke]]看医生其实是被医生看。[[/joke]]"], ] (target / "styles.json").write_text(json.dumps( {"fields": style_fields, "rows": style_rows}, ensure_ascii=False), encoding="utf-8") print(json.dumps({ "issue_batches": (len(issue_rows) + 199) // 200, "issue_rows": len(issue_rows), "coverage_rows": len(coverage_rows), "style_rows": len(style_rows), }, ensure_ascii=False, indent=2)) return 0 def apply_approved(args: argparse.Namespace) -> int: issues_path = Path(args.issues) manifest = json.loads(Path(args.manifest).read_text(encoding="utf-8")) repo = Path(manifest["repo"]) issues = json.loads(issues_path.read_text(encoding="utf-8")) approved = [x for x in issues if x.get("审核状态") == "接受" and x.get("建议文本")] hashes = {x["file"]: x["sha256"] for x in manifest["files"]} grouped = defaultdict(list) for row in approved: grouped[row["文件"]].append(row) prepared = {} for rel, rows in grouped.items(): path = repo / rel if sha256(path) != hashes.get(rel): raise SystemExit(f"stale file hash: {rel}") lines = path.read_text(encoding="utf-8-sig").splitlines(keepends=True) for row in sorted(rows, key=lambda x: int(x["行号"])): idx = int(row["行号"]) - 1 old = lines[idx].rstrip("\r\n") if old != row["原文"]: raise SystemExit(f"source mismatch: {rel}:{idx+1}") new = row["建议文本"] if LINE_TAG_RE.findall(old) != LINE_TAG_RE.findall(new): raise SystemExit(f"line tag changed: {rel}:{idx+1}") old_text = source_without_metadata(old) new_text = source_without_metadata(new) if len(re.split(r"(?i)", new_text)) > len(re.split(r"(?i)", old_text)): raise SystemExit(f"display line added: {rel}:{idx+1}") ending = lines[idx][len(lines[idx].rstrip("\r\n")):] lines[idx] = new + ending prepared[path] = "".join(lines) for path, content in prepared.items(): fd, tmp = tempfile.mkstemp(prefix=path.name, dir=path.parent) try: with os.fdopen(fd, "w", encoding="utf-8", newline="") as f: f.write(content) os.replace(tmp, path) finally: if os.path.exists(tmp): os.unlink(tmp) print(json.dumps({"applied": len(approved), "files": len(prepared)}, ensure_ascii=False)) return 0 def apply_reviewed(args: argparse.Namespace) -> int: manifest = json.loads(Path(args.manifest).read_text(encoding="utf-8")) reviews = json.loads(Path(args.reviews).read_text(encoding="utf-8")) repo = Path(manifest["repo"]) hashes = {x["file"]: x["sha256"] for x in manifest["files"]} allowed = {REVIEW_ACCEPTED, REVIEW_KEEP, REVIEW_FOLLOW_NOTE} decisions = Counter(review_choice(row.get("审阅结论")) for row in reviews) invalid = sorted(choice for choice in decisions if choice not in allowed) if invalid: raise SystemExit(f"unresolved review decisions: {invalid}") grouped = defaultdict(list) changes = [] unchanged_actionable = [] for row in reviews: decision = review_choice(row.get("审阅结论")) new = transform_reviewed_line(row) if decision != REVIEW_KEEP and new == row["原文"]: unchanged_actionable.append(f'{row["文件"]}:{row["行号"]} ({decision})') if new != row["原文"]: grouped[row["文件"]].append((row, new)) changes.append({ "稳定ID": row["稳定ID"], "文件": row["文件"], "行号": row["行号"], "问题分类": row["问题分类"], "审阅结论": decision, "审阅备注": row.get("审阅备注") or "", "原文": row["原文"], "应用文本": new, }) if unchanged_actionable: raise SystemExit("actionable reviews produced no change:\n" + "\n".join(unchanged_actionable)) prepared = {} for rel, rows in grouped.items(): path = repo / rel if sha256(path) != hashes.get(rel): raise SystemExit(f"stale file hash: {rel}") lines = path.read_text(encoding="utf-8-sig").splitlines(keepends=True) before_tags = LINE_TAG_RE.findall("".join(lines)) for row, new in sorted(rows, key=lambda item: int(item[0]["行号"])): idx = int(row["行号"]) - 1 old = lines[idx].rstrip("\r\n") if old != row["原文"]: raise SystemExit(f"source mismatch: {rel}:{idx + 1}") if LINE_TAG_RE.findall(old) != LINE_TAG_RE.findall(new): raise SystemExit(f"line tag changed: {rel}:{idx + 1}") _, old_body, _ = dialogue_parts(old) _, new_body, _ = dialogue_parts(new) old_width = visible_width(strip_markup(old_body)) allowed_lines = max( len(re.split(r"(?i)", old_body)), int((old_width + MANUAL_BREAK_WIDTH - 0.001) // MANUAL_BREAK_WIDTH), ) new_lines = len(re.split(r"(?i)", new_body)) if new_lines > allowed_lines: raise SystemExit( f"display line estimate increased: {rel}:{idx + 1} " f"({new_lines} > {allowed_lines})" ) for marker in ("aside", "term", "key", "joke"): if new.count(f"[[{marker}]]") != new.count(f"[[/{marker}]]"): raise SystemExit(f"unbalanced semantic marker {marker}: {rel}:{idx + 1}") ending = lines[idx][len(lines[idx].rstrip("\r\n")):] lines[idx] = new + ending content = "".join(lines) if LINE_TAG_RE.findall(content) != before_tags: raise SystemExit(f"file line tag set/order changed: {rel}") prepared[path] = content report = { "review_records": len(reviews), "decisions": dict(decisions), "changed_records": len(changes), "changed_files": len(prepared), "dry_run": not args.apply, "changes": changes, } if args.report: Path(args.report).write_text( json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8") if args.apply: for path, content in prepared.items(): fd, tmp = tempfile.mkstemp(prefix=path.name, dir=path.parent) try: with os.fdopen(fd, "w", encoding="utf-8", newline="") as f: f.write(content) os.replace(tmp, path) finally: if os.path.exists(tmp): os.unlink(tmp) print(json.dumps({ key: value for key, value in report.items() if key != "changes" }, ensure_ascii=False, indent=2)) return 0 def self_test() -> int: import unittest class AuditTests(unittest.TestCase): def test_markup_and_width(self): self.assertEqual(strip_markup("[[vertexp d=0.05 bot]]文字"), "文字") self.assertEqual(visible_width("中文ab"), 3.0) def test_break_rule(self): rec = { "text": "很多时候,比起幸福更难以让人离
开的,是熟悉。", "visible_text": "很多时候,比起幸福更难以让人离开的,是熟悉。", "speaker": "me", "source_text": "", "module": "FP_Test", } self.assertIn("换行", [x["category"] for x in detect(rec)]) def test_annotated_rules(self): ulm = {"text":"正在训练","visible_text":"正在训练","speaker":"ULM自训练进程", "source_text":"","module":"FP_Peipei2"} self.assertIn("ULM效果", [x["category"] for x in detect(ulm)]) task = {"text":"寻找模块","visible_text":"寻找模块","speaker":"Task", "source_text":"","module":"FP_Test"} self.assertNotIn("重要信息", [x["category"] for x in detect(task)]) def test_reviewed_break_preserves_existing_wait(self): text = "很多时候,比起幸福更难以让人离开的,是熟悉。" changed = add_comfortable_breaks(text) self.assertIn(",[[comma-pause]]
", changed) with_wait = "别因为外面很冷,就逃回那个着火的房子。" changed_wait = add_comfortable_breaks(with_wait) self.assertIn(",
", changed_wait) self.assertNotIn("[[comma-pause]]", changed_wait) def test_review_note_overrides_default_style(self): row = { "原文": "me: 去检查*情绪模块*。 #line:abc", "审阅结论": [REVIEW_FOLLOW_NOTE], "审阅备注": "用黄字,并且去掉“*”", "问题分类": "专有名词", } changed = transform_reviewed_line(row) self.assertIn("[[key]]情绪模块[[/key]]", changed) self.assertNotIn("*", changed) self.assertEqual(LINE_TAG_RE.findall(changed), ["abc"]) suite = unittest.defaultTestLoader.loadTestsFromTestCase(AuditTests) result = unittest.TextTestRunner(verbosity=2).run(suite) return 0 if result.wasSuccessful() else 2 def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description="Deterministic Yarn text coverage and review ledger") sub = parser.add_subparsers(dest="command", required=True) p_scan = sub.add_parser("scan") p_scan.add_argument("--repo", default=".") p_scan.add_argument("--root", action="append", required=True) p_scan.add_argument("--output", required=True) p_scan.add_argument( "--review-memory", help="可选的飞书审阅导出;已应用或明确保留的记录不再重复报告") p_scan.set_defaults(func=scan) p_verify = sub.add_parser("verify") p_verify.add_argument("--output", required=True) p_verify.set_defaults(func=verify) p_apply = sub.add_parser("apply") p_apply.add_argument("--issues", required=True) p_apply.add_argument("--manifest", required=True) p_apply.set_defaults(func=apply_approved) p_apply_review = sub.add_parser("apply-review") p_apply_review.add_argument("--reviews", required=True) p_apply_review.add_argument("--manifest", required=True) p_apply_review.add_argument("--report") p_apply_review.add_argument("--apply", action="store_true") p_apply_review.set_defaults(func=apply_reviewed) p_lark = sub.add_parser("lark-payloads") p_lark.add_argument("--output", required=True) p_lark.set_defaults(func=build_lark_payloads) p_test = sub.add_parser("self-test") p_test.set_defaults(func=lambda _: self_test()) return parser def main() -> int: args = build_parser().parse_args() return args.func(args) if __name__ == "__main__": sys.exit(main())