#!/usr/bin/env python3 # -*- coding: utf-8 -*- """M11b-2b 双向列集 diff:world_sync 消费方代码引用的列 vs pbl_runtime_ext 权威列集。 PM 裁决 B 的执行要求:「引用后跑双向列集 diff 留证(world_sync 消费方代码若依赖旧结构, 适配到权威结构)」。 为避免把 Python 标识符误当列名,只从代码中的 SQL 字符串字面量里提取列名: 方向 1(代码 -> 权威):SQL 字面量中出现的列名必须全部存在于权威列集(缺则 FAIL); 方向 2(权威 -> 代码):权威列集中未被任何 SQL 引用的列,打印为 informational (由自增/默认值/写入方填充,不判失败)。 用法:python3 modules/world_sync/scripts/m11b2b_column_diff.py """ import io import json import os import re import sys import tokenize CONSUMERS = [ "modules/world_sync/world_sync/pbl_runtime_sql.py", "modules/world_sync/world_sync/pbl_runtime_tx.py", "modules/world_sync/world_sync/pbl_runtime_schema.py", ] AUTHORITY = { "pbl_runtime_event": "modules/pbl_runtime_ext/models/pbl_runtime_event.json", "pbl_entity_state": "modules/pbl_runtime_ext/models/pbl_entity_state.json", } SQL_HINT = re.compile(r"\b(select|insert\s+into|update|delete\s+from|on\s+duplicate)\b", re.IGNORECASE) # SQL 语法词/函数词,避免误判为列名 STOPWORDS = { "select", "from", "where", "insert", "into", "values", "update", "set", "delete", "and", "or", "not", "null", "is", "in", "on", "as", "by", "order", "group", "limit", "join", "left", "right", "inner", "outer", "distinct", "count", "max", "min", "sum", "now", "current_timestamp", "true", "false", "begin", "commit", "rollback", "with", "returning", "if", "exists", "duplicate", "key", "auto_increment", "using", } TABLE_WORDS = {"pbl_runtime_event", "pbl_entity_state"} def find_root(): here = os.path.dirname(os.path.abspath(__file__)) cur = here for _ in range(10): if os.path.isdir(os.path.join(cur, "modules")) and os.path.isdir(os.path.join(cur, "apps")): return cur parent = os.path.dirname(cur) if parent == cur: break cur = parent return os.path.abspath(os.path.join(here, os.pardir, os.pardir, os.pardir)) def authority_columns(root, table): with open(os.path.join(root, AUTHORITY[table]), "r", encoding="utf-8") as fh: data = json.load(fh) return [f.get("name") for f in (data.get("fields") or []) if isinstance(f, dict) and f.get("name")] def sql_literals(path): """用 tokenize 取出文件中的字符串字面量,仅保留含 SQL 关键字者。""" with open(path, "rb") as fh: raw = fh.read() out = [] try: for tok in tokenize.tokenize(io.BytesIO(raw).readline): if tok.type != tokenize.STRING: continue try: val = ast_literal_eval(tok.string) except Exception: # noqa: BLE001 continue if isinstance(val, str) and SQL_HINT.search(val): out.append(val) except (tokenize.TokenError, IndentationError, SyntaxError): pass return out def ast_literal_eval(src): """轻量解析字符串字面量(含 f-string 前缀),失败返回 None。""" s = src # 去掉 f/r/b 等前缀 m = re.match(r"^([a-zA-Z]*)", s) prefix = m.group(1).lower() if m else "" body = s[len(prefix):] if "f" in prefix: # f-string:把 {expr} 替换为占位,仅取字面部分 body = re.sub(r"\{[^{}]*\}", " ", body) if "b" in prefix: body = body.encode("latin-1", "ignore").decode("unicode_escape") try: import ast as _ast return _ast.literal_eval(body) if body[:1] in "\"'" else body except Exception: # noqa: BLE001 return body try: import ast as _ast return _ast.literal_eval(body) except Exception: # noqa: BLE001 return None def columns_from_sql(sql): cols = set() cols.update(re.findall(r"`([a-zA-Z_][a-zA-Z_0-9]*)`", sql)) for grp in re.findall(r"INSERT\s+INTO\s+[\w`.]+\s*\(([^)]*)\)", sql, re.IGNORECASE): for tok in grp.split(","): tok = tok.strip().strip("`") if re.fullmatch(r"[a-zA-Z_][a-zA-Z_0-9]*", tok or ""): cols.add(tok) for grp in re.findall(r"\bSET\s+(.+?)(?:\bWHERE\b|$)", sql, re.IGNORECASE | re.DOTALL): cols.update(re.findall(r"([a-zA-Z_][a-zA-Z_0-9]*)\s*=", grp)) for grp in re.findall(r"\bWHERE\s+(.+?)$", sql, re.IGNORECASE | re.DOTALL): for tok in re.findall(r"([a-zA-Z_][a-zA-Z_0-9]*)\s*(?:=|<|>|!|in\b|is\b)", grp, re.IGNORECASE): cols.add(tok) for grp in re.findall(r"\b(?:ORDER\s+BY|GROUP\s+BY)\s+([\w`,\s.]+)", sql, re.IGNORECASE): for tok in grp.split(","): tok = tok.strip().strip("`").split()[-1] if tok.strip() else "" if re.fullmatch(r"[a-zA-Z_][a-zA-Z_0-9]*", tok or ""): cols.add(tok) for tok in re.findall(r"\b(?:pbl_runtime_event|pbl_entity_state)\.([a-zA-Z_][a-zA-Z_0-9]*)", sql): cols.add(tok) return {c for c in cols if c.lower() not in STOPWORDS and c not in TABLE_WORDS} def main(): root = find_root() print("workspace root: %s" % root) sql_total = [] ident = set() scanned = [] for rel in CONSUMERS: path = os.path.join(root, rel) if not os.path.isfile(path): print(" SKIP (absent): %s" % rel) continue scanned.append(rel) lits = sql_literals(path) sql_total.extend(lits) for s in lits: ident |= columns_from_sql(s) print("consumer files scanned: %d -> %s" % (len(scanned), scanned)) print("SQL literals found: %d, distinct column identifiers: %d" % (len(sql_total), len(ident))) auth_cols = {} for table in AUTHORITY: auth_cols[table] = authority_columns(root, table) print("\nauthority columns [%s] (%d cols): %s" % (table, len(auth_cols[table]), auth_cols[table])) print("\n--- 方向 1: 代码 SQL 引用列 ⊆ 权威列集(必须全部成立)---") bad = [] for name in sorted(ident): hit = [t for t in auth_cols if name in auth_cols[t]] print(" %-22s %s %s" % (name, "OK " if hit else "MISS", ("->" + ",".join(hit)) if hit else "(不在任何权威列集)")) if not hit: bad.append(name) print("\n--- 方向 2: 权威列集中未被 SQL 显式引用者(informational)---") for table in AUTHORITY: unused = [c for c in auth_cols[table] if c not in ident] print(" [%s] unreferenced=%d %s" % (table, len(unused), unused)) print("\ncode columns checked: %d, missing from authority: %d" % (len(ident), len(bad))) if bad: print("MISSING: %s" % bad) print("RESULT: FAIL") return 1 print("RESULT: PASS") return 0 if __name__ == "__main__": sys.exit(main())