Files
llm/scripts/analyze_1c_saved_state_object_details.py

377 lines
15 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Analyze saved-state object changes beyond storage bytes."""
from __future__ import annotations
import argparse
import difflib
import importlib.util
import json
import re
import sys
from pathlib import Path
from typing import Any
WORD_RE = re.compile(r"[\wА-Яа-яЁё]{3,}", re.UNICODE)
BASE64ISH_RE = re.compile(r"^[A-Za-z0-9+/=_-]{24,}$")
REPO_ROOT = Path(__file__).resolve().parents[1]
def load_payload_to_text():
module_path = REPO_ROOT / "plugins" / "1c" / "parser" / "payload.py"
spec = importlib.util.spec_from_file_location("onec_payload", module_path)
if spec is None or spec.loader is None:
raise RuntimeError(f"Cannot load payload parser: {module_path}")
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module.payload_to_text
payload_to_text = load_payload_to_text()
def load_json(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8-sig"))
def write_json(path: Path, data: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(data, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def safe_join(root: Path, file_name: str) -> Path:
relative = Path(file_name.replace("\\", "/"))
if relative.is_absolute() or ".." in relative.parts or not str(relative):
raise ValueError(file_name)
return root / relative
def words(text: str) -> list[str]:
result: list[str] = []
seen: set[str] = set()
for match in WORD_RE.finditer(text):
value = match.group(0)
key = value.casefold()
if key not in seen:
result.append(value)
seen.add(key)
return result
def semantic_words(values: list[str], *, limit: int = 40) -> list[str]:
result: list[str] = []
seen: set[str] = set()
for value in values:
if not value:
continue
if BASE64ISH_RE.match(value):
continue
if value.isdigit():
continue
if len(value) > 80:
continue
has_cyrillic = any("А" <= char <= "я" or char in "Ёё" for char in value)
has_1c_shape = any(marker in value for marker in ("Форма", "Команда", "Реквизит", "Модуль", "Область", "Процедура", "Функция"))
if not has_cyrillic and not has_1c_shape:
continue
key = value.casefold()
if key in seen:
continue
result.append(value)
seen.add(key)
if len(result) >= limit:
break
return result
def classify_payload_part(item: dict[str, Any], file_name: str) -> str:
kind = str(item.get("kind") or "")
suffix = ""
if "." in file_name:
suffix = file_name.rsplit(".", 1)[1]
if kind in {"CommonModule", "ObjectModule", "ManagerModule"} and suffix == "0":
return "bsl_module_text"
if kind == "Form" and not suffix:
return "form_descriptor"
if kind == "Form" and suffix == "0":
return "form_body"
if suffix == "0":
return "primary_payload"
return "metadata_payload"
def textish_from_payload(decoded: dict[str, Any]) -> tuple[str | None, str | None]:
text = decoded.get("text")
if text:
return text, str(decoded.get("encoding") or "")
payload = decoded.get("payload")
if not isinstance(payload, bytes):
return None, None
candidates: list[tuple[str, str, int]] = []
for encoding in ("utf-8-sig", "utf-8", "cp1251", "utf-16-le"):
try:
candidate = payload.decode(encoding, errors="ignore").replace("\x00", "").replace("\ufeff", "")
except Exception:
continue
word_count = len(words(candidate))
cyrillic_count = sum(1 for char in candidate if "А" <= char <= "я" or char in "Ёё")
known_1c_terms = sum(
candidate.count(term)
for term in ("Процедура", "Функция", "Конец", "Если", "Тогда", "Область", "Перем", "Экспорт", "пример")
)
mojibake_penalty = candidate.count("Р") * 8 + candidate.count("С") * 4
score = word_count * 5 + cyrillic_count + known_1c_terms * 500 - mojibake_penalty
if word_count:
candidates.append((candidate, f"{encoding}:lossy", score))
if not candidates:
return None, None
candidates.sort(key=lambda item: item[2], reverse=True)
return candidates[0][0], candidates[0][1]
def common_edges(left: str, right: str) -> tuple[int, int]:
prefix = 0
for a, b in zip(left, right):
if a != b:
break
prefix += 1
suffix = 0
left_tail = left[prefix:]
right_tail = right[prefix:]
for a, b in zip(reversed(left_tail), reversed(right_tail)):
if a != b:
break
suffix += 1
return prefix, suffix
def text_window(text: str, center: int, size: int = 500) -> str:
start = max(center - size // 2, 0)
end = min(center + size // 2, len(text))
return text[start:end].replace("\x00", "")
def line_diff(left: str, right: str, *, limit: int) -> list[str]:
left_lines = left.splitlines()
right_lines = right.splitlines()
diff = list(difflib.unified_diff(left_lines, right_lines, fromfile="active", tofile="saved", lineterm=""))
if len(diff) > limit:
return [*diff[:limit], f"... truncated {len(diff) - limit} lines ..."]
return diff
def analyze_payload(active_path: Path | None, saved_path: Path) -> dict[str, Any]:
saved_raw = saved_path.read_bytes()
saved_decoded = payload_to_text(saved_raw)
active_decoded: dict[str, Any] | None = None
if active_path and active_path.exists():
active_decoded = payload_to_text(active_path.read_bytes())
result: dict[str, Any] = {
"saved_path": str(saved_path),
"active_path": str(active_path) if active_path else None,
"saved": {
"raw_bytes": saved_decoded.get("raw_bytes"),
"payload_bytes": saved_decoded.get("payload_bytes"),
"compression": saved_decoded.get("compression"),
"encoding": saved_decoded.get("encoding"),
},
"active": None,
"text_comparable": False,
}
if active_decoded:
result["active"] = {
"raw_bytes": active_decoded.get("raw_bytes"),
"payload_bytes": active_decoded.get("payload_bytes"),
"compression": active_decoded.get("compression"),
"encoding": active_decoded.get("encoding"),
}
saved_text, saved_text_mode = textish_from_payload(saved_decoded)
active_text, active_text_mode = textish_from_payload(active_decoded) if active_decoded else (None, None)
result["saved"]["text_mode"] = saved_text_mode
if result["active"] is not None:
result["active"]["text_mode"] = active_text_mode
if saved_text is None:
result["summary"] = "Saved payload is not text-decodable."
return result
result["saved_strings_sample"] = words(saved_text)[:80]
if active_text is None:
result["summary"] = "Saved text payload has no active counterpart."
result["text_comparable"] = False
result["saved_text_sample"] = text_window(saved_text, 0)
return result
result["text_comparable"] = True
prefix, suffix = common_edges(active_text, saved_text)
active_words = {value.casefold(): value for value in words(active_text)}
saved_words = {value.casefold(): value for value in words(saved_text)}
added_keys = [key for key in saved_words if key not in active_words]
removed_keys = [key for key in active_words if key not in saved_words]
result["text_diff"] = {
"active_chars": len(active_text),
"saved_chars": len(saved_text),
"delta_chars": len(saved_text) - len(active_text),
"common_prefix_chars": prefix,
"common_suffix_chars": suffix,
"added_words": [saved_words[key] for key in added_keys[:80]],
"removed_words": [active_words[key] for key in removed_keys[:80]],
"active_window": text_window(active_text, prefix),
"saved_window": text_window(saved_text, prefix),
"unified_diff": line_diff(active_text, saved_text, limit=120),
}
result["semantic_hints"] = {
"added_terms": semantic_words(result["text_diff"]["added_words"]),
"removed_terms": semantic_words(result["text_diff"]["removed_words"]),
}
result["summary"] = "Text payload differs." if active_text != saved_text else "Text payload matches."
return result
def storage_root(saved_table: str, active_table: str, roots: dict[str, Path]) -> tuple[Path | None, Path | None]:
saved_root = roots.get(saved_table)
active_root = roots.get(active_table)
return saved_root, active_root
def build_extension_cas_map(summary_path: Path | None) -> dict[tuple[str, str], str]:
if not summary_path or not summary_path.exists():
return {}
data = load_json(summary_path)
result: dict[tuple[str, str], str] = {}
for extension in data.get("extensions") or []:
extension_name = str(extension.get("extension_name") or "")
for obj in extension.get("sample_objects") or []:
for part in obj.get("parts") or []:
object_id = str(part.get("object_id") or "").casefold()
cas_key = str(part.get("cas_key") or "")
if extension_name and object_id and cas_key:
result[(extension_name.casefold(), object_id)] = cas_key
return result
def extension_object_id_from_saved_file(file_name: str) -> str | None:
if "__" not in file_name:
return None
object_id = file_name.split("__", 1)[1]
if object_id == "configinfo":
return None
return object_id.casefold()
def active_extension_path(item: dict[str, Any], file_name: str, extension_cas_map: dict[tuple[str, str], str], config_cas_all_dir: Path | None) -> tuple[Path | None, str | None]:
if not config_cas_all_dir:
return None, None
extension = str(item.get("extension") or "").casefold()
object_id = extension_object_id_from_saved_file(file_name)
if not extension or not object_id:
return None, None
cas_key = extension_cas_map.get((extension, object_id))
if not cas_key:
return None, None
path = config_cas_all_dir / cas_key
return (path if path.exists() else None), cas_key
def analyze(comparison: dict[str, Any], roots: dict[str, Path], *, extension_manifest_summary: Path | None = None, config_cas_all_dir: Path | None = None) -> dict[str, Any]:
extension_cas_map = build_extension_cas_map(extension_manifest_summary)
objects = []
for item in comparison.get("object_changes") or []:
details = []
for storage in item.get("storage") or []:
saved_root, active_root = storage_root(str(storage.get("saved_table")), str(storage.get("active_table")), roots)
file_name = str(storage.get("file_name") or "")
if not saved_root:
details.append({"file_name": file_name, "error": f"Missing saved root for {storage.get('saved_table')}"})
continue
try:
saved_path = safe_join(saved_root, file_name)
active_path = safe_join(active_root, file_name) if active_root else None
except ValueError:
details.append({"file_name": file_name, "error": "Unsafe storage file name."})
continue
if not saved_path.exists():
details.append({"file_name": file_name, "error": f"Saved payload file is missing: {saved_path}"})
continue
active_cas_key = None
if not (active_path and active_path.exists()) and storage.get("saved_table") == "ConfigCASSave":
active_path, active_cas_key = active_extension_path(item, file_name, extension_cas_map, config_cas_all_dir)
payload_detail = analyze_payload(active_path if active_path and active_path.exists() else None, saved_path)
details.append({
"file_name": file_name,
"payload_role": classify_payload_part(item, file_name),
"saved_table": storage.get("saved_table"),
"active_table": storage.get("active_table"),
"active_exists": storage.get("active_exists"),
"active_cas_key": active_cas_key,
"payload": payload_detail,
})
objects.append({
"full_name": item.get("full_name"),
"layer": item.get("layer"),
"extension": item.get("extension"),
"kind": item.get("kind"),
"kind_ru": item.get("kind_ru"),
"name": item.get("name"),
"synonym": item.get("synonym"),
"change_state": item.get("change_state"),
"details": details,
})
return {
"schema": "onec_saved_state_object_detail.v1",
"source_schema": comparison.get("schema"),
"database": comparison.get("database"),
"view": comparison.get("view"),
"object_details": objects,
"counts": {
"objects": len(objects),
"details": sum(len(item.get("details") or []) for item in objects),
},
"safety": {
"read_only": True,
"sql_write_performed": False,
"public_terms_are_1c_objects": True,
},
"active_extension_resolution": {
"extension_manifest_summary": str(extension_manifest_summary) if extension_manifest_summary else None,
"config_cas_all_dir": str(config_cas_all_dir) if config_cas_all_dir else None,
"mapped_parts": len(extension_cas_map),
},
}
def main() -> int:
parser = argparse.ArgumentParser(description="Analyze saved-state object changes beyond storage bytes.")
parser.add_argument("--comparison", type=Path, required=True)
parser.add_argument("--config-save-dir", type=Path, required=True)
parser.add_argument("--config-dir", type=Path, required=True)
parser.add_argument("--config-cas-save-dir", type=Path, required=True)
parser.add_argument("--config-cas-dir", type=Path, required=True)
parser.add_argument("--extension-manifest-summary", type=Path)
parser.add_argument("--config-cas-all-dir", type=Path)
parser.add_argument("--output", type=Path)
args = parser.parse_args()
roots = {
"ConfigSave": args.config_save_dir,
"Config": args.config_dir,
"ConfigCASSave": args.config_cas_save_dir,
"ConfigCAS": args.config_cas_dir,
}
result = analyze(
load_json(args.comparison),
roots,
extension_manifest_summary=args.extension_manifest_summary,
config_cas_all_dir=args.config_cas_all_dir,
)
if args.output:
write_json(args.output, result)
print(json.dumps({"output": str(args.output) if args.output else None, "schema": result["schema"], "counts": result["counts"]}, ensure_ascii=False))
return 0
if __name__ == "__main__":
raise SystemExit(main())