Files
llm/scripts/build_1c_xml_guid_index.py
T

225 lines
8.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Build a GUID index for a 1C XML configuration dump.
The index is intentionally mechanical: it records GUID occurrences in XML
attributes and element text, plus the top metadata object declared by each
file. It does not infer SQL table names or storage roles.
"""
from __future__ import annotations
import argparse
import json
import os
import re
from pathlib import Path
from typing import Any
GUID_RE = re.compile(
r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}"
)
TOP_OBJECT_RE = re.compile(
r"<(?P<kind>[A-Za-z][A-Za-z0-9]*)\s+[^>]*uuid=\"(?P<guid>"
+ GUID_RE.pattern
+ r")\"",
re.S,
)
NAME_RE = re.compile(r"<Name>(?P<name>.*?)</Name>", re.S)
SYNONYM_RE = re.compile(r"<(?:[A-Za-z0-9]+:)?content>(?P<content>.*?)</(?:[A-Za-z0-9]+:)?content>", re.S)
def clean_xml_text(value: str) -> str:
return repair_mojibake(
value.replace("&quot;", '"')
.replace("&apos;", "'")
.replace("&lt;", "<")
.replace("&gt;", ">")
.replace("&amp;", "&")
.strip()
)
def repair_mojibake(value: str) -> str:
if not value:
return value
for source_encoding in ("gbk", "cp1255", "cp1252", "latin1"):
try:
candidate = value.encode(source_encoding).decode("cp1251")
except UnicodeError:
continue
candidate_cyrillic = sum(1 for char in candidate if "А" <= char <= "я" or char == "ё" or char == "Ё")
value_cyrillic = sum(1 for char in value if "А" <= char <= "я" or char == "ё" or char == "Ё")
if candidate_cyrillic > value_cyrillic:
return repair_mojibake(candidate)
cjk_count = sum(1 for char in value if "\u4e00" <= char <= "\u9fff")
if cjk_count:
try:
candidate = value.encode("gbk").decode("cp1251")
return repair_mojibake(candidate)
except UnicodeError:
pass
cyrillic_count = sum(1 for char in value if "А" <= char <= "я" or char == "ё" or char == "Ё")
suspicious_count = sum(1 for char in value if char in "ÐÑÂÃÄÅÆÇÈÉÊËÌÍÎÏÒÓÔÕÖרÙÚÛÜÝÞßàáâãäåæçèéêëìíîïòóôõö÷øùúûüýþÿ")
if suspicious_count <= cyrillic_count:
return value
try:
repaired = value.encode("latin1").decode("cp1251")
except UnicodeError:
return value
repaired_cyrillic = sum(1 for char in repaired if "А" <= char <= "я" or char == "ё" or char == "Ё")
return repaired if repaired_cyrillic > cyrillic_count else value
def inspect_xml_file(path: Path, root_dir: Path, *, max_occurrences_per_file: int) -> dict[str, Any]:
item: dict[str, Any] = {
"path": repair_mojibake(str(path)),
"relative_path": repair_mojibake(str(path.relative_to(root_dir))),
"status": "ok",
"root_tag": "",
"top_object": None,
"occurrences": [],
}
try:
text = path.read_text(encoding="utf-8", errors="replace")
except OSError as exc:
item["status"] = "read_error"
item["error"] = str(exc)
return item
root_match = re.search(r"<(?P<tag>[A-Za-z][A-Za-z0-9]*)[\s>]", text)
item["root_tag"] = root_match.group("tag") if root_match else ""
object_match = TOP_OBJECT_RE.search(text)
if object_match:
object_start = object_match.start()
object_end = TOP_OBJECT_RE.search(text, object_match.end())
object_fragment = text[object_start : object_end.start() if object_end else min(len(text), object_start + 500_000)]
name_match = NAME_RE.search(object_fragment)
synonym_match = SYNONYM_RE.search(object_fragment)
top_object = {
"guid": object_match.group("guid").lower(),
"xml_kind": object_match.group("kind"),
"name": clean_xml_text(name_match.group("name")) if name_match else "",
"synonym": clean_xml_text(synonym_match.group("content")) if synonym_match else "",
}
else:
top_object = None
item["top_object"] = top_object
occurrences = []
for match in GUID_RE.finditer(text):
occurrences.append({"guid": match.group(0).lower(), "offset": match.start()})
if len(occurrences) >= max_occurrences_per_file:
item["occurrences"] = occurrences
item["truncated"] = True
return item
item["occurrences"] = occurrences
item["truncated"] = False
return item
def build_guid_map(files: list[dict[str, Any]], *, max_occurrences_per_guid: int) -> dict[str, Any]:
guid_map: dict[str, Any] = {}
for file_item in files:
if file_item.get("status") != "ok":
continue
top_object = file_item.get("top_object")
if top_object:
guid = top_object["guid"]
entry = guid_map.setdefault(guid, {"total_occurrences": 0, "top_objects": [], "occurrences": []})
entry["top_objects"].append(
{
"path": file_item["path"],
"relative_path": file_item["relative_path"],
"xml_kind": top_object["xml_kind"],
"name": top_object["name"],
"synonym": top_object["synonym"],
}
)
for occurrence in file_item.get("occurrences") or []:
guid = occurrence["guid"]
entry = guid_map.setdefault(guid, {"total_occurrences": 0, "top_objects": [], "occurrences": []})
entry["total_occurrences"] += 1
if len(entry["occurrences"]) < max_occurrences_per_guid:
entry["occurrences"].append(
{
"path": file_item["path"],
"relative_path": file_item["relative_path"],
"offset": occurrence["offset"],
}
)
return dict(sorted(guid_map.items()))
def list_xml_paths(root: Path, max_relative_depth: int) -> list[Path]:
if max_relative_depth <= 0:
return sorted(root.rglob("*.xml"))
result: list[Path] = []
root_parts = len(root.parts)
for current, dirs, files in os.walk(root):
current_path = Path(current)
relative_depth = len(current_path.parts) - root_parts
if relative_depth >= max_relative_depth - 1:
dirs[:] = []
for file_name in files:
if file_name.lower().endswith(".xml"):
path = current_path / file_name
if len(path.relative_to(root).parts) <= max_relative_depth:
result.append(path)
return sorted(result)
def main() -> int:
parser = argparse.ArgumentParser(description="Build a GUID index for a 1C XML dump.")
parser.add_argument("xml_root", type=Path)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--max-occurrences-per-file", type=int, default=5000)
parser.add_argument("--max-occurrences-per-guid", type=int, default=20)
parser.add_argument(
"--max-relative-depth",
type=int,
default=0,
help="Only scan XML files whose relative path has at most this many parts; 0 scans all files.",
)
args = parser.parse_args()
xml_root = args.xml_root.resolve()
paths = list_xml_paths(xml_root, args.max_relative_depth)
files = [
inspect_xml_file(path, xml_root, max_occurrences_per_file=args.max_occurrences_per_file)
for path in paths
]
guid_map = build_guid_map(files, max_occurrences_per_guid=args.max_occurrences_per_guid)
top_object_count = sum(1 for item in files if item.get("top_object"))
parse_errors = sum(1 for item in files if item.get("status") != "ok")
report = {
"schema": "onec_xml_guid_index.v1",
"xml_root": repair_mojibake(str(xml_root)),
"file_count": len(files),
"top_object_count": top_object_count,
"parse_error_count": parse_errors,
"guid_count": len(guid_map),
"files": files,
"guid_map": guid_map,
}
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(
json.dumps(
{
"output": str(args.output),
"files": len(files),
"top_objects": top_object_count,
"guids": len(guid_map),
"parse_errors": parse_errors,
},
ensure_ascii=False,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(main())